Skip to content
POST/api/v1/scorers/calibrate

CLHF — evaluator/human agreement + threshold suggestion

Continuous learning from human feedback. Quantifies how well an evaluator agrees with human labels using chance-corrected Cohen's kappa, and recommends the score threshold that maximizes agreement. Supply `pairs` (human/machine pass-fail labels) and/or `scored` (humanPass + machineScore).

Authentication

Send Authorization: Bearer YOUR_API_KEY on every request. Generate API keys at /dashboard/settings/api-keys.

Request body required

Example

{
  "projectId": "00000000-0000-0000-0000-000000000000",
  "scorerId": "string",
  "pairs": [
    {
      "human": false,
      "machine": false
    }
  ],
  "scored": [
    {
      "humanPass": false,
      "machineScore": 0
    }
  ],
  "currentThreshold": 0
}
Schema
{
  "application/json": {
    "schema": {
      "type": "object",
      "properties": {
        "projectId": {
          "type": "string",
          "format": "uuid",
          "pattern": "^([0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[1-8][0-9a-fA-F]{3}-[89abAB][0-9a-fA-F]{3}-[0-9a-fA-F]{12}|00000000-0000-0000-0000-000000000000|ffffffff-ffff-ffff-ffff-ffffffffffff)$"
        },
        "scorerId": {
          "type": "string",
          "maxLength": 200
        },
        "pairs": {
          "maxItems": 100000,
          "type": "array",
          "items": {
            "type": "object",
            "properties": {
              "human": {
                "type": "boolean"
              },
              "machine": {
                "type": "boolean"
              }
            },
            "required": [
              "human",
              "machine"
            ],
            "additionalProperties": false
          }
        },
        "scored": {
          "maxItems": 100000,
          "type": "array",
          "items": {
            "type": "object",
            "properties": {
              "humanPass": {
                "type": "boolean"
              },
              "machineScore": {
                "type": "number",
                "minimum": 0,
                "maximum": 1
              }
            },
            "required": [
              "humanPass",
              "machineScore"
            ],
            "additionalProperties": false
          }
        },
        "currentThreshold": {
          "type": "number",
          "minimum": 0,
          "maximum": 1
        }
      },
      "additionalProperties": false
    }
  }
}

Response

All status codes

200Agreement report and/or threshold suggestion
400No pairs or scored data provided
401Unauthorized — no valid credential was presented (AUTH_REQUIRED).
403Forbidden — the credential is valid but lacks the API-key scope, member role, or plan entitlement this operation requires.
429Too Many Requests — the per-key or per-organization rate limit was exceeded. Honour the Retry-After header.
500Internal Server Error — an unhandled error was converted to the standard error envelope (INTERNAL_ERROR).

Code samples

cURL

curl -X POST \
  https://evalguard.ai/api/v1/scorers/calibrate \
  -H "Authorization: Bearer $EVALGUARD_API_KEY" \
  -H "Content-Type: application/json" \
  -d '{ "projectId": "00000000-0000-0000-0000-000000000000", "scorerId": "string", "pairs": [ { "human": false, "machine": false } ], "scored": [ { "humanPass": false, "machineScore": 0 } ], "currentThreshold": 0 }'

TypeScript

// The TypeScript SDK (@evalguard/sdk) exposes TYPED methods — runEval,
// getEval, runSecurityScan, checkFirewall, … — not a generic request().
// For an arbitrary endpoint, call it directly:

const res = await fetch("https://evalguard.ai/api/v1/scorers/calibrate", {
  method: "POST",
  headers: {
    Authorization: `Bearer ${process.env.EVALGUARD_API_KEY}`,
    "Content-Type": "application/json",
  },
  body: JSON.stringify({
    "projectId": "00000000-0000-0000-0000-000000000000",
    "scorerId": "string",
    "pairs": [
      {
        "human": false,
        "machine": false
      }
    ],
    "scored": [
      {
        "humanPass": false,
        "machineScore": 0
      }
    ],
    "currentThreshold": 0
  }),
});
console.log(res.status, await res.json());

Python

# The Python SDK (pip install evalguardai) exposes TYPED methods on
# EvalGuardClient — run_eval, get_eval, … — not a generic request().
# For an arbitrary endpoint, call it directly:

import os
import requests

headers = {"Authorization": f"Bearer {os.environ['EVALGUARD_API_KEY']}"}
headers["Content-Type"] = "application/json"

response = requests.request(
    "POST",
    "https://evalguard.ai/api/v1/scorers/calibrate",
    headers=headers,
    json={
    "projectId": "00000000-0000-0000-0000-000000000000",
    "scorerId": "string",
    "pairs": [
        {
            "human": False,
            "machine": False
        }
    ],
    "scored": [
        {
            "humanPass": False,
            "machineScore": 0
        }
    ],
    "currentThreshold": 0
},
)
print(response.status_code, response.json())

Go

package main

import (
	"context"
	"fmt"
	"net/http"
	"os"
	"strings"
)

func main() {
	body := strings.NewReader(`{"projectId":"00000000-0000-0000-0000-000000000000","scorerId":"string","pairs":[{"human":false,"machine":false}],"scored":[{"humanPass":false,"machineScore":0}],"currentThreshold":0}`)
	req, _ := http.NewRequestWithContext(context.Background(), "POST", "https://evalguard.ai/api/v1/scorers/calibrate", body)
	req.Header.Set("Authorization", "Bearer "+os.Getenv("EVALGUARD_API_KEY"))
	req.Header.Set("Content-Type", "application/json")
	resp, err := http.DefaultClient.Do(req)
	if err != nil { panic(err) }
	defer resp.Body.Close()
	fmt.Println(resp.Status)
}

Errors

400401403429500

Other Evaluations endpoints