curl https://api.naga.ac/v1/responses \
-H "Authorization: Bearer $NAGA_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "gpt-5",
"input": [
{
"role": "user",
"content": [
{
"type": "input_text",
"text": "say hi"
}
]
}
]
}'import os
from openai import OpenAI
client = OpenAI(
api_key=os.environ["NAGA_API_KEY"],
base_url="https://api.naga.ac/v1",
)
result = client.responses.create(
model="gpt-5",
input=[
{
"role": "user",
"content": [
{
"type": "input_text",
"text": "say hi"
}
]
}
],
)
print(result)
import OpenAI from "openai";
const client = new OpenAI({
apiKey: process.env.NAGA_API_KEY,
baseURL: "https://api.naga.ac/v1",
});
const result = await client.responses.create({
model: "gpt-5",
input: [
{
"role": "user",
"content": [
{
"type": "input_text",
"text": "say hi"
}
]
}
],
});
console.log(result);
package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.naga.ac/v1/responses"
payload := strings.NewReader("{\n \"model\": \"gpt-5\",\n \"input\": [\n {\n \"role\": \"user\",\n \"content\": [\n {\n \"type\": \"input_text\",\n \"text\": \"say hi\"\n }\n ]\n }\n ]\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}{
"id": "resp_01H8Zt4Qk9",
"object": "response",
"created_at": 1774000000,
"status": "completed",
"model": "gpt-5",
"output": [
{
"type": "message",
"id": "msg_1",
"status": "completed",
"role": "assistant",
"content": [
{
"type": "output_text",
"text": "hello world"
}
]
}
],
"error": null,
"incomplete_details": null,
"usage": {
"input_tokens": 3,
"input_tokens_details": {
"cached_tokens": 0
},
"output_tokens": 2,
"output_tokens_details": {
"reasoning_tokens": 0
},
"total_tokens": 5
},
"instructions": null,
"tools": [],
"tool_choice": "auto",
"temperature": 1,
"top_p": 1,
"parallel_tool_calls": true,
"metadata": null,
"text": {
"format": {
"type": "text"
}
},
"truncation": "disabled",
"max_output_tokens": null
}Create a response
Creates a model response in the OpenAI Responses dialect and charges the account for the tokens it used. Requires an API key. /v1/chat/completions and /v1/messages serve the same models in the other two dialects.
curl https://api.naga.ac/v1/responses \
-H "Authorization: Bearer $NAGA_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "gpt-5",
"input": [
{
"role": "user",
"content": [
{
"type": "input_text",
"text": "say hi"
}
]
}
]
}'import os
from openai import OpenAI
client = OpenAI(
api_key=os.environ["NAGA_API_KEY"],
base_url="https://api.naga.ac/v1",
)
result = client.responses.create(
model="gpt-5",
input=[
{
"role": "user",
"content": [
{
"type": "input_text",
"text": "say hi"
}
]
}
],
)
print(result)
import OpenAI from "openai";
const client = new OpenAI({
apiKey: process.env.NAGA_API_KEY,
baseURL: "https://api.naga.ac/v1",
});
const result = await client.responses.create({
model: "gpt-5",
input: [
{
"role": "user",
"content": [
{
"type": "input_text",
"text": "say hi"
}
]
}
],
});
console.log(result);
package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.naga.ac/v1/responses"
payload := strings.NewReader("{\n \"model\": \"gpt-5\",\n \"input\": [\n {\n \"role\": \"user\",\n \"content\": [\n {\n \"type\": \"input_text\",\n \"text\": \"say hi\"\n }\n ]\n }\n ]\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}{
"id": "resp_01H8Zt4Qk9",
"object": "response",
"created_at": 1774000000,
"status": "completed",
"model": "gpt-5",
"output": [
{
"type": "message",
"id": "msg_1",
"status": "completed",
"role": "assistant",
"content": [
{
"type": "output_text",
"text": "hello world"
}
]
}
],
"error": null,
"incomplete_details": null,
"usage": {
"input_tokens": 3,
"input_tokens_details": {
"cached_tokens": 0
},
"output_tokens": 2,
"output_tokens_details": {
"reasoning_tokens": 0
},
"total_tokens": 5
},
"instructions": null,
"tools": [],
"tool_choice": "auto",
"temperature": 1,
"top_p": 1,
"parallel_tool_calls": true,
"metadata": null,
"text": {
"format": {
"type": "text"
}
},
"truncation": "disabled",
"max_output_tokens": null
}Authorizations
An account API key. Send it as Authorization: Bearer <key>.
Headers
The routing strategy when the body names none. An unknown name is ignored. A NagaAI extension. The order in which NagaAI tries the providers that serve a model. A NagaAI extension.
reliability, balanced, latency, throughput, cache Body
The OpenAI-compatible Responses request. The gateway takes an unrecognized key and drops it, and keeps nothing between requests.
The model to run, named by an id or alias from NagaAI's catalog. An entry without the chat.completions capability draws a refusal.
The prompt: one string, or the conversation as typed items, oldest first. Required.
One entry of the input list, chosen by its type. Six of twenty-six tags do something.
- Option 1
- Option 2
- Option 3
- Option 4
- Option 5
- Option 6
- Option 7
- Option 8
- Option 9
- Option 10
- Option 11
- Option 12
- Option 13
- Option 14
- Option 15
- Option 16
- Option 17
- Option 18
- Option 19
- Option 20
- Option 21
- Option 22
- Option 23
- Option 24
- Option 25
- Option 26
- Option 27
Hide child attributes
Hide child attributes
Whose turn this is, and what it may carry.
user, assistant, system, developer The tag that selects this branch: type is message.
message What the turn says. The gateway refuses a message without it.
One entry of a message's content. The role decides which types it admits.
A name for the speaker. The gateway drops it.
An identifier for this item. The gateway looks nothing up by it.
The lifecycle state a replayed item carried, never acted on.
Whether an assistant turn is commentary or the final answer.
commentary, final_answer How NagaAI orders the providers for this request. It reorders them and never removes one. A NagaAI extension.
Hide child attributes
Hide child attributes
The strategy for this request. An unknown name draws a 400.
reliability, balanced, latency, throughput, cache A system instruction put ahead of the conversation.
The ceiling on generated tokens, reasoning included. At least 1, and the gateway raises it where the reasoning budget would not fit.
x >= 1How much randomness goes into the reply, between 0 and 2. A provider may still refuse a value this range allows.
0 <= x <= 2The sampling cut-off, between 0 and 1, and an alternative to temperature.
0 <= x <= 1The tools the model may call. web_search and web_search_preview are the only hosted ones the gateway runs; other types travel no further.
One entry of tools[], chosen by its type. Five of the eighteen types reach the model; the gateway drops the other thirteen.
- Option 1
- Option 2
- Option 3
- Option 4
- Option 5
- Option 6
- Option 7
- Option 8
- Option 9
- Option 10
- Option 11
- Option 12
- Option 13
- Option 14
- Option 15
- Option 16
- Option 17
- Option 18
Hide child attributes
Hide child attributes
The tag that selects this branch: type is function.
function What the model calls the function by.
What the function does, written for the model. Falls back to function.description.
Whether the model follows parameters exactly.
The same declaration written the chat-style way; the flat keys win.
Hide child attributes
Hide child attributes
What the model calls the function by. Required, and non-empty.
1What the function does, written for the model to choose by.
Whether the model follows the parameter schema exactly.
Which tool, if any, the model is to call. Naming a tool the gateway does not run reads as if the key were absent.
auto, none, required Whether the model may return several tool calls in one turn. It reaches only some providers.
What to do when the conversation outgrows the context window. It reaches only some providers.
auto, disabled How much reasoning to do, and what summary of it comes back.
Hide child attributes
Hide child attributes
How much reasoning to spend before answering. The gateway maps the step onto whatever scale the provider offers, and drops it where there is none.
none, minimal, low, medium, high, xhigh Which summary of its own reasoning to ask the model for. Absent, generate_summary stands in for it.
concise, detailed, auto The older spelling of summary, read only when summary is absent.
concise, detailed, auto The shape of the reply, and how long it should run.
Hide child attributes
Hide child attributes
The shape the reply must take.
Hide child attributes
Hide child attributes
Which of the three formats the reply takes.
text, json_object, json_schema A short name for the schema. Required if type is json_schema.
What the schema is for, written for the model.
Whether the model is held to the schema exactly.
How long the reply should run.
low, medium, high Whether the reply arrives incrementally as the model writes it. An explicit null draws a refusal.
Response
The response. application/json carries one response object; text/event-stream carries the event-typed lifecycle frames.
A response: the non-streaming body, and the snapshot every lifecycle frame carries.
The system instructions the turn ran with, or null.
The tools the model was offered, in the shape the gateway took them.
How the model was told to choose among them.
The sampling temperature the turn ran with; 1 when the request set none.
The sampling cut-off the turn ran with; 1 when the request set none.
Whether the model could call several tools at once.
The metadata the request sent, echoed back unchanged. The gateway drops it upstream.
The output-format settings the turn ran with.
What was to happen if the conversation outgrew the context window.
The ceiling on generated tokens the turn ran with, or null.
This response's identifier, repeated on every frame. No handle to resume from.
What kind of object this is. Always response.
response When the response was created, in seconds since the Unix epoch.
Where the turn stands, as the provider named it.
What the model produced, item by item, in order.
One item the model produced, addressed by output_index.
- Message item
- Reasoning item
- Synthesized reasoning item
- Function call item
- Function call output item
- Web search call item
- Image generation call item
- Image generation call, added view
Hide child attributes
Hide child attributes
What kind of item this is. Always message.
message The item's identifier, quoted by every stream frame about it.
Whether the item is still unfinished.
in_progress, completed, incomplete Who the message is from. assistant on anything the model produced.
user, assistant, system, developer, tool The message itself: output_text parts, or a refusal part.
Which pass of a two-pass answer this message is.
commentary, final_answer Why the turn stopped short, or null.
Hide child attributes
Hide child attributes
What cut the turn short.
max_output_tokens, content_filter, tool_use_required, upstream_timeout, upstream_disconnect, gateway_cancelled, other What the request cost in tokens, or null while the turn runs.
Hide child attributes
Hide child attributes
Tokens of the prompt, cached ones included.
The split of input_tokens the provider reported.
Hide child attributes
Hide child attributes
How many prompt tokens the provider served from its own cache. 0 when it reported none.
How many prompt tokens were audio. A NagaAI extension, present only when the provider split them out.
Tokens the model generated, reasoning and image tokens included.
The split of output_tokens the provider reported.
Hide child attributes
Hide child attributes
The provider's own total for the request, reported as received.
The model that answered, by its id in the NagaAI catalog.
The reasoning settings the turn ran with.