/api/v1/gatewaySend a chat request through the gateway
Routes a chat completion through the LLM gateway (provider selection, response cache, model fallback via `options.fallbackModels`). `messages` and `model` are required; `temperature`/`maxTokens` are accepted at the top level OR inside `options`. The response-cache partition is the caller's own org/user — a body `tenantId` is ignored for scoping. Errors: 429 `quota_exceeded` when the plan's monthly gateway quota is spent; 403 `gateway_policy_denied` / 503 `gateway_policy_unavailable` when the org's egress policy refuses or cannot be read; 422 `BYOK_KEY_REQUIRED` (503 `BYOK_KEY_UNAVAILABLE` on a resolver error) when the org has no provider key and the platform key is not billable to them; 422 `NO_API_KEY` when no key exists at all. Successful responses include `billedTo` when the platform key was used.
Authentication
Send Authorization: Bearer YOUR_API_KEY on every request. Generate API keys at /dashboard/settings/api-keys.
Request body required
Example
{
"messages": [
{
"role": "string",
"content": "string"
}
],
"model": "gpt-4o",
"tenantId": "00000000-0000-0000-0000-000000000000",
"temperature": 0,
"maxTokens": 1,
"options": {
"temperature": 0,
"maxTokens": 1,
"fallbackModels": [
"string"
]
}
}Schema
{
"application/json": {
"schema": {
"type": "object",
"properties": {
"messages": {
"minItems": 1,
"maxItems": 500,
"type": "array",
"items": {
"type": "object",
"properties": {
"role": {
"type": "string",
"maxLength": 60
},
"content": {
"type": "string",
"maxLength": 1000000
}
},
"required": [
"role",
"content"
],
"additionalProperties": false
}
},
"model": {
"type": "string",
"minLength": 1,
"maxLength": 200,
"example": "gpt-4o"
},
"tenantId": {
"type": "string",
"format": "uuid",
"pattern": "^([0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[1-8][0-9a-fA-F]{3}-[89abAB][0-9a-fA-F]{3}-[0-9a-fA-F]{12}|00000000-0000-0000-0000-000000000000|ffffffff-ffff-ffff-ffff-ffffffffffff)$"
},
"temperature": {
"type": "number",
"minimum": 0,
"maximum": 2
},
"maxTokens": {
"type": "integer",
"minimum": 1,
"maximum": 1000000
},
"options": {
"type": "object",
"properties": {
"temperature": {
"type": "number",
"minimum": 0,
"maximum": 2
},
"maxTokens": {
"type": "integer",
"minimum": 1,
"maximum": 1000000
},
"fallbackModels": {
"maxItems": 20,
"type": "array",
"items": {
"type": "string",
"maxLength": 200
}
}
},
"additionalProperties": false
}
},
"required": [
"messages",
"model"
],
"additionalProperties": false
}
}
}Response
All status codes
Code samples
cURL
curl -X POST \
https://evalguard.ai/api/v1/gateway \
-H "Authorization: Bearer $EVALGUARD_API_KEY" \
-H "Content-Type: application/json" \
-d '{ "messages": [ { "role": "string", "content": "string" } ], "model": "gpt-4o", "tenantId": "00000000-0000-0000-0000-000000000000", "temperature": 0, "maxTokens": 1, "options": { "temperature": 0, "maxTokens": 1, "fallbackModels": [ "string" ] } }'TypeScript
// The TypeScript SDK (@evalguard/sdk) exposes TYPED methods — runEval,
// getEval, runSecurityScan, checkFirewall, … — not a generic request().
// For an arbitrary endpoint, call it directly:
const res = await fetch("https://evalguard.ai/api/v1/gateway", {
method: "POST",
headers: {
Authorization: `Bearer ${process.env.EVALGUARD_API_KEY}`,
"Content-Type": "application/json",
},
body: JSON.stringify({
"messages": [
{
"role": "string",
"content": "string"
}
],
"model": "gpt-4o",
"tenantId": "00000000-0000-0000-0000-000000000000",
"temperature": 0,
"maxTokens": 1,
"options": {
"temperature": 0,
"maxTokens": 1,
"fallbackModels": [
"string"
]
}
}),
});
console.log(res.status, await res.json());Python
# The Python SDK (pip install evalguardai) exposes TYPED methods on
# EvalGuardClient — run_eval, get_eval, … — not a generic request().
# For an arbitrary endpoint, call it directly:
import os
import requests
headers = {"Authorization": f"Bearer {os.environ['EVALGUARD_API_KEY']}"}
headers["Content-Type"] = "application/json"
response = requests.request(
"POST",
"https://evalguard.ai/api/v1/gateway",
headers=headers,
json={
"messages": [
{
"role": "string",
"content": "string"
}
],
"model": "gpt-4o",
"tenantId": "00000000-0000-0000-0000-000000000000",
"temperature": 0,
"maxTokens": 1,
"options": {
"temperature": 0,
"maxTokens": 1,
"fallbackModels": [
"string"
]
}
},
)
print(response.status_code, response.json())Go
package main
import (
"context"
"fmt"
"net/http"
"os"
"strings"
)
func main() {
body := strings.NewReader(`{"messages":[{"role":"string","content":"string"}],"model":"gpt-4o","tenantId":"00000000-0000-0000-0000-000000000000","temperature":0,"maxTokens":1,"options":{"temperature":0,"maxTokens":1,"fallbackModels":["string"]}}`)
req, _ := http.NewRequestWithContext(context.Background(), "POST", "https://evalguard.ai/api/v1/gateway", body)
req.Header.Set("Authorization", "Bearer "+os.Getenv("EVALGUARD_API_KEY"))
req.Header.Set("Content-Type", "application/json")
resp, err := http.DefaultClient.Do(req)
if err != nil { panic(err) }
defer resp.Body.Close()
fmt.Println(resp.Status)
}