curl https://api.naga.ac/v1/messages \
-H "Authorization: Bearer $NAGA_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "claude-sonnet-4.5",
"max_tokens": 256,
"messages": [
{
"role": "user",
"content": "Say hi."
}
]
}'import os
from anthropic import Anthropic
client = Anthropic(
api_key=os.environ["NAGA_API_KEY"],
base_url="https://api.naga.ac",
)
result = client.messages.create(
model="claude-sonnet-4.5",
max_tokens=256,
messages=[
{
"role": "user",
"content": "Say hi."
}
],
)
print(result)
const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: 'claude-sonnet-4.5',
max_tokens: 256,
messages: [{role: 'user', content: 'Say hi.'}]
})
};
fetch('https://api.naga.ac/v1/messages', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.naga.ac/v1/messages"
payload := strings.NewReader("{\n \"model\": \"claude-sonnet-4.5\",\n \"max_tokens\": 256,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Say hi.\"\n }\n ]\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}{
"id": "msg_01UvWq7Kt3xZbN9",
"type": "message",
"role": "assistant",
"content": [
{
"type": "text",
"text": "Hi!",
"citations": null
}
],
"model": "claude-sonnet-4.5",
"stop_reason": "end_turn",
"stop_sequence": null,
"stop_details": null,
"container": null,
"usage": {
"input_tokens": 5,
"output_tokens": 4,
"cache_creation": null,
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"server_tool_use": null,
"service_tier": null,
"inference_geo": null,
"output_tokens_details": null
}
}Create a message
Creates a model response in the Anthropic Messages dialect and charges the account for the tokens it used. Requires an API key. /v1/chat/completions and /v1/responses serve the same models in the other two dialects.
curl https://api.naga.ac/v1/messages \
-H "Authorization: Bearer $NAGA_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "claude-sonnet-4.5",
"max_tokens": 256,
"messages": [
{
"role": "user",
"content": "Say hi."
}
]
}'import os
from anthropic import Anthropic
client = Anthropic(
api_key=os.environ["NAGA_API_KEY"],
base_url="https://api.naga.ac",
)
result = client.messages.create(
model="claude-sonnet-4.5",
max_tokens=256,
messages=[
{
"role": "user",
"content": "Say hi."
}
],
)
print(result)
const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: 'claude-sonnet-4.5',
max_tokens: 256,
messages: [{role: 'user', content: 'Say hi.'}]
})
};
fetch('https://api.naga.ac/v1/messages', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.naga.ac/v1/messages"
payload := strings.NewReader("{\n \"model\": \"claude-sonnet-4.5\",\n \"max_tokens\": 256,\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"Say hi.\"\n }\n ]\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}{
"id": "msg_01UvWq7Kt3xZbN9",
"type": "message",
"role": "assistant",
"content": [
{
"type": "text",
"text": "Hi!",
"citations": null
}
],
"model": "claude-sonnet-4.5",
"stop_reason": "end_turn",
"stop_sequence": null,
"stop_details": null,
"container": null,
"usage": {
"input_tokens": 5,
"output_tokens": 4,
"cache_creation": null,
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"server_tool_use": null,
"service_tier": null,
"inference_geo": null,
"output_tokens_details": null
}
}Authorizations
An account API key. Send it as Authorization: Bearer <key>.
Headers
The routing strategy when the body names none. An unknown name is ignored. A NagaAI extension. The order in which NagaAI tries the providers that serve a model. A NagaAI extension.
reliability, balanced, latency, throughput, cache Body
The Anthropic-compatible Messages request. The gateway takes an unrecognized key and drops it.
The model to run, named by an id or alias from NagaAI's catalog. An entry without the chat.completions capability draws a refusal.
The ceiling on generated tokens. Required, at least 1, and the gateway raises it where the reasoning budget would not fit.
x >= 1The conversation so far, oldest first: one to 100000 entries. A block sent under the wrong role draws a refusal.
1 - 100000 elementsHide child attributes
Hide child attributes
Whose turn this is, and what blocks it may carry.
user, assistant, system What was said — one string, or at least one content block.
One entry of a message's content. The role decides which tags it admits.
- Option 1
- Option 2
- Option 3
- Option 4
- Option 5
- Option 6
- Option 7
- Option 8
- Option 9
- Option 10
- Option 11
- Option 12
- Option 13
- Option 14
- Option 15
- Option 16
How NagaAI orders the providers for this request. It reorders them and never removes one. A NagaAI extension.
Hide child attributes
Hide child attributes
The strategy for this request. An unknown name draws a 400.
reliability, balanced, latency, throughput, cache Whether the reply arrives incrementally as the model writes it. An explicit null draws a refusal.
How much randomness goes into the reply, between 0 and 2. A provider may still refuse a value this range allows.
0 <= x <= 2The sampling cut-off, between 0 and 1, and an alternative to temperature.
0 <= x <= 1How many top candidates each token is drawn from. It reaches only some providers.
Strings that end generation the moment the model produces one. The reply never names which matched.
The tools the model may call. Two dated web-search stamps are the only hosted ones the gateway serves; the other eight families travel no further.
One entry of tools[], in one of the three shapes this dialect accepts.
- Client tool
- Web search tool
- Hosted tool
Hide child attributes
Hide child attributes
What the model calls the tool by, never empty.
What the tool does, written for the model to choose by.
Whether the model's arguments follow input_schema exactly.
Which tool, if any, the model is to call.
How much reasoning the model does before answering. The budget becomes a tier: under 1024 draws a refusal, and output_config.effort wins over it.
- Option 1
- Option 2
- Option 3
Hide child attributes
Hide child attributes
Configuration for the output: the reasoning tier, and a schema the reply must satisfy.
Hide child attributes
Hide child attributes
The effort tier to ask the provider for. It overrides whatever thinking asked for.
low, medium, high, xhigh, max Response
The message. application/json carries one message object; text/event-stream carries the event-typed frames of the stream.
- Option 1
- Option 2
The message. The gateway answers a request it does not forward with the second shape, which omits container and stop_details and adds created.
This answer's identifier.
What kind of object this is. Always message.
Who wrote it. Always assistant.
The answer itself, block by block.
One block of the answer's content, in the order the model produced it.
- Option 1
- Option 2
- Option 3
- Option 4
- Option 5
- Option 6
- Option 7
- Option 8
Hide child attributes
Hide child attributes
The model that answered, by its id in the NagaAI catalog.
Why the model stopped, in one of four values.
Always null. Stop sequences are not reported back, which is why stop_reason is never stop_sequence either.
Always null. stop_reason carries the whole of what is known about why the turn ended.
Always null. Code-execution containers are not served.
What the request cost in tokens.
Hide child attributes
Hide child attributes
Tokens of the prompt read fresh, cache reads excluded.
Tokens the model generated, its reasoning included.
Always null. The per-duration breakdown of cache writes is not reported; cache_creation_input_tokens carries the total.
Tokens written into the prompt cache by this request, or null.
Tokens served to this request out of the prompt cache, or null.
Always null. This gateway routes by its own catalog, so the Anthropic service tier does not apply to an answer served through it.
Always null. Where the inference ran is not reported.
Always null. The breakdown of output_tokens into thinking tokens and the rest is not surfaced. output_tokens is the total that counts.