/api/v1/gateway/semantic-cacheGet semantic-cache hit/miss + savings
Returns gateway cache statistics across both tiers: a fleet-wide exact-match tier (Redis, tenant+model-scoped keys) and an in-memory semantic tier (cosine similarity over prompt embeddings, per-runtime). Fields: `entries`, `maxEntries`, `hits`, `misses`, `hitRate`, `estimatedSavingsUsd`, `avgLatencyReductionMs` (not tracked by the core stats — always 0), `ttlSec`, `enabled`, `similarityThreshold`, and the exact-tier `exactHits`/`exactMisses` counters. Semantic and exact counters are summed so the numbers are fleet-wide rather than one replica's slice; if the in-memory cache is unreachable, the Redis exact-tier numbers are still returned with `enabled: false`. There is no per-model breakdown. The similarity threshold and TTL are deployment-wide env settings (`EVALGUARD_GATEWAY_CACHE_THRESHOLD`, default 0.92; `EVALGUARD_GATEWAY_CACHE_TTL_SEC`, default 600), not per-policy.
Authentication
Send Authorization: Bearer YOUR_API_KEY on every request. Generate API keys at /dashboard/settings/api-keys.
Response
200 example
{
"success": true
}All status codes
Code samples
cURL
curl -X GET \ https://evalguard.ai/api/v1/gateway/semantic-cache \ -H "Authorization: Bearer $EVALGUARD_API_KEY"
TypeScript
// The TypeScript SDK (@evalguard/sdk) exposes TYPED methods — runEval,
// getEval, runSecurityScan, checkFirewall, … — not a generic request().
// For an arbitrary endpoint, call it directly:
const res = await fetch("https://evalguard.ai/api/v1/gateway/semantic-cache", {
method: "GET",
headers: { Authorization: `Bearer ${process.env.EVALGUARD_API_KEY}` },
});
console.log(res.status, await res.json());Python
# The Python SDK (pip install evalguardai) exposes TYPED methods on
# EvalGuardClient — run_eval, get_eval, … — not a generic request().
# For an arbitrary endpoint, call it directly:
import os
import requests
headers = {"Authorization": f"Bearer {os.environ['EVALGUARD_API_KEY']}"}
response = requests.request("GET", "https://evalguard.ai/api/v1/gateway/semantic-cache", headers=headers)
print(response.status_code, response.json())Go
package main
import (
"context"
"fmt"
"net/http"
"os"
)
func main() {
req, _ := http.NewRequestWithContext(context.Background(), "GET", "https://evalguard.ai/api/v1/gateway/semantic-cache", nil)
req.Header.Set("Authorization", "Bearer "+os.Getenv("EVALGUARD_API_KEY"))
resp, err := http.DefaultClient.Do(req)
if err != nil { panic(err) }
defer resp.Body.Close()
fmt.Println(resp.Status)
}