curl -X POST https://api.60db.ai/judge/evaluate \
-H "Authorization: Bearer your-api-key" \
-H "Content-Type: application/json" \
-d '{
"state": "Agent: I can't refund that, it's outside the window.\nCaller: This is the third time I've called about this.",
"questions": {
"tone": {
"type": "choice",
"instructions": "How did the agent come across?",
"criteria": {
"professional": "Calm, courteous, takes ownership",
"dismissive": "Brushes the caller off",
"unknown": "Not enough of the call to tell"
}
},
"satisfaction": {
"type": "score",
"instructions": "How satisfied does the caller sound?",
"criteria": [
"Angry",
"Unhappy",
"Neutral",
"Satisfied"
]
},
"resolved": {
"type": "noul",
"instructions": "Was the problem solved?"
}
},
"label": "support-call-qa"
}'
const response = await fetch('https://api.60db.ai/judge/evaluate', {
method: 'POST',
headers: {
'Authorization': 'Bearer your-api-key',
'Content-Type': 'application/json',
},
body: JSON.stringify({
"state": "Agent: I can't refund that, it's outside the window.\nCaller: This is the third time I've called about this.",
"questions": {
"tone": {
"type": "choice",
"instructions": "How did the agent come across?",
"criteria": {
"professional": "Calm, courteous, takes ownership",
"dismissive": "Brushes the caller off",
"unknown": "Not enough of the call to tell"
}
},
"satisfaction": {
"type": "score",
"instructions": "How satisfied does the caller sound?",
"criteria": [
"Angry",
"Unhappy",
"Neutral",
"Satisfied"
]
},
"resolved": {
"type": "noul",
"instructions": "Was the problem solved?"
}
},
"label": "support-call-qa"
}),
});
const data = await response.json();
import requests
response = requests.post(
"https://api.60db.ai/judge/evaluate",
headers={
"Authorization": "Bearer your-api-key",
"Content-Type": "application/json",
},
json={
"state": "Agent: I can't refund that, it's outside the window.\nCaller: This is the third time I've called about this.",
"questions": {
"tone": {
"type": "choice",
"instructions": "How did the agent come across?",
"criteria": {
"professional": "Calm, courteous, takes ownership",
"dismissive": "Brushes the caller off",
"unknown": "Not enough of the call to tell"
}
},
"satisfaction": {
"type": "score",
"instructions": "How satisfied does the caller sound?",
"criteria": [
"Angry",
"Unhappy",
"Neutral",
"Satisfied"
]
},
"resolved": {
"type": "noul",
"instructions": "Was the problem solved?"
}
},
"label": "support-call-qa"
},
)
data = response.json()
{
"success": true,
"data": {
"id": "a1b2c3d4-e5f6-4789-abcd-ef0123456789",
"saved": true,
"rubric_id": null,
"model": "60db-decision-model-v1",
"answers": {
"tone": {
"type": "choice",
"choice": "dismissive",
"probabilities": {
"professional": 0.23492,
"dismissive": 0.418141,
"unknown": 0.346939
},
"confidence": 0.4681
},
"satisfaction": {
"type": "score",
"score": 1.1592,
"legend": {
"0": "Angry",
"1": "Unhappy",
"2": "Neutral",
"3": "Satisfied"
},
"probabilities": {
"0": 0.255292,
"1": 0.420406,
"2": 0.234124,
"3": 0.090178
},
"confidence": 0.4704
},
"resolved": {
"type": "noul",
"noul": 0.275
}
},
"summary": {
"tone": "dismissive",
"satisfaction": 1.1592,
"resolved": 0.275
},
"min_confidence": 0.4681,
"escalated_questions": 2,
"usage": {
"input_tokens": 84,
"output_tokens": 0
},
"metadata": {
"backend": "60db-local",
"probabilities": "model_softmax",
"confidence": "typesafe_adapter_metrics_uncalibrated",
"confidence_source": "typesafe-ai/system-one-adapter-python@adffc2eab300a4fa3c0e92252d4ffd6ceaa53700/src/system_one_adapter/_utils/confidence_metrics.py",
"source_model": "60db-jev-source",
"escalated_questions": 2
},
"latency_ms": 4,
"credits_charged": 1.5100000000000002e-06
}
}
Judge
Evaluate
Score content against a rubric of choice, score and true/false questions
POST
/
judge
/
evaluate
curl -X POST https://api.60db.ai/judge/evaluate \
-H "Authorization: Bearer your-api-key" \
-H "Content-Type: application/json" \
-d '{
"state": "Agent: I can't refund that, it's outside the window.\nCaller: This is the third time I've called about this.",
"questions": {
"tone": {
"type": "choice",
"instructions": "How did the agent come across?",
"criteria": {
"professional": "Calm, courteous, takes ownership",
"dismissive": "Brushes the caller off",
"unknown": "Not enough of the call to tell"
}
},
"satisfaction": {
"type": "score",
"instructions": "How satisfied does the caller sound?",
"criteria": [
"Angry",
"Unhappy",
"Neutral",
"Satisfied"
]
},
"resolved": {
"type": "noul",
"instructions": "Was the problem solved?"
}
},
"label": "support-call-qa"
}'
const response = await fetch('https://api.60db.ai/judge/evaluate', {
method: 'POST',
headers: {
'Authorization': 'Bearer your-api-key',
'Content-Type': 'application/json',
},
body: JSON.stringify({
"state": "Agent: I can't refund that, it's outside the window.\nCaller: This is the third time I've called about this.",
"questions": {
"tone": {
"type": "choice",
"instructions": "How did the agent come across?",
"criteria": {
"professional": "Calm, courteous, takes ownership",
"dismissive": "Brushes the caller off",
"unknown": "Not enough of the call to tell"
}
},
"satisfaction": {
"type": "score",
"instructions": "How satisfied does the caller sound?",
"criteria": [
"Angry",
"Unhappy",
"Neutral",
"Satisfied"
]
},
"resolved": {
"type": "noul",
"instructions": "Was the problem solved?"
}
},
"label": "support-call-qa"
}),
});
const data = await response.json();
import requests
response = requests.post(
"https://api.60db.ai/judge/evaluate",
headers={
"Authorization": "Bearer your-api-key",
"Content-Type": "application/json",
},
json={
"state": "Agent: I can't refund that, it's outside the window.\nCaller: This is the third time I've called about this.",
"questions": {
"tone": {
"type": "choice",
"instructions": "How did the agent come across?",
"criteria": {
"professional": "Calm, courteous, takes ownership",
"dismissive": "Brushes the caller off",
"unknown": "Not enough of the call to tell"
}
},
"satisfaction": {
"type": "score",
"instructions": "How satisfied does the caller sound?",
"criteria": [
"Angry",
"Unhappy",
"Neutral",
"Satisfied"
]
},
"resolved": {
"type": "noul",
"instructions": "Was the problem solved?"
}
},
"label": "support-call-qa"
},
)
data = response.json()
{
"success": true,
"data": {
"id": "a1b2c3d4-e5f6-4789-abcd-ef0123456789",
"saved": true,
"rubric_id": null,
"model": "60db-decision-model-v1",
"answers": {
"tone": {
"type": "choice",
"choice": "dismissive",
"probabilities": {
"professional": 0.23492,
"dismissive": 0.418141,
"unknown": 0.346939
},
"confidence": 0.4681
},
"satisfaction": {
"type": "score",
"score": 1.1592,
"legend": {
"0": "Angry",
"1": "Unhappy",
"2": "Neutral",
"3": "Satisfied"
},
"probabilities": {
"0": 0.255292,
"1": 0.420406,
"2": 0.234124,
"3": 0.090178
},
"confidence": 0.4704
},
"resolved": {
"type": "noul",
"noul": 0.275
}
},
"summary": {
"tone": "dismissive",
"satisfaction": 1.1592,
"resolved": 0.275
},
"min_confidence": 0.4681,
"escalated_questions": 2,
"usage": {
"input_tokens": 84,
"output_tokens": 0
},
"metadata": {
"backend": "60db-local",
"probabilities": "model_softmax",
"confidence": "typesafe_adapter_metrics_uncalibrated",
"confidence_source": "typesafe-ai/system-one-adapter-python@adffc2eab300a4fa3c0e92252d4ffd6ceaa53700/src/system_one_adapter/_utils/confidence_metrics.py",
"source_model": "60db-jev-source",
"escalated_questions": 2
},
"latency_ms": 4,
"credits_charged": 1.5100000000000002e-06
}
}
Ask up to 32 questions about one piece of content and get a probability distribution for each.
Judge is a classifier, not a generator. It never writes prose: every answer is a distribution over an answer space you defined, which is what makes results comparable across thousands of runs. A language model asked “is this caller angry?” returns a paragraph; this returns
The descriptions are what the model reads — they are the thing to tune, not a prompt. A label with a vague description produces a vague answer.
Context is 8K tokens per question, not per request. The shared
0.86 on dismissive.
Each question is asked independently — every one sees the same state and its own criteria, and nothing else. They are not a conversation.
Billed per input token. A failed run is refunded automatically — you are never charged for an answer you did not get. See Judge pricing.
Your existing 60db API key already works. Judge wraps the 60db Jev model behind the same platform credential as TTS, STT and Memory — you do not need a separate key, and you do not need to reissue the one you have. Keys created from now on carry a
judge scope; keys issued earlier are accepted on their slm scope.Request
Headers
string
required
Bearer token — your standard 60db API key (
sk_…), or a user JWT. There is no separate Judge credential.string
required
application/json
Body
string | object | array
required
The content every question is asked about — a transcript, a ticket, a document, or any JSON structure. Top-level numbers, booleans and
null are rejected.object
Map of answer key → question, 1–32 entries. The key is your handle on the answer; it is never sent to the model.Supply this or An ordered array, lowest first, max 10 levels.
rubric_id, never both.Each question is one of three shapes:choice — pick one of your options.{ "type": "choice",
"instructions": "How did the agent come across?",
"criteria": { "professional": "Calm and courteous",
"dismissive": "Brushes the caller off",
"unknown": "Not clear from the transcript" } }
criteria is a map of option name → description, max 255 options.score — rate on a ladder.{ "type": "score", "criteria": ["Angry", "Unhappy", "Neutral", "Satisfied"] }
noul — true or false.{ "type": "noul", "instructions": "Was the problem solved?" }
criteria is optional: { "true": "...", "false": "..." }.string
Run a saved rubric by id instead of inline questions. See List rubrics.
string
default:"jev-latest"
Model name. Fetch the valid list from List models rather than hardcoding it.
boolean
default:"true"
false bills the run but keeps it out of history.string
Tag for grouping runs in history. Max 120 characters.
Response
object
One answer per question, under the same key you supplied.choice —
choice (the winning option), probabilities (every option, summing to 1), confidence.score — score (a probability-weighted position on your ladder, e.g. 1.87 — not an index), legend (level index → your description), probabilities, confidence.noul — noul, a raw probability of true. There is deliberately no confidence field: the upstream is explicit that this is not a calibrated correctness estimate.object
One headline value per question — the chosen option, the score, or the probability. What the history list renders.
number | null
Lowest confidence across the answers, or
null when every question was noul. Below 0.85 a human should look — that is the upstream’s own escalation threshold.integer
How many answers the service refined with its second-stage model because confidence was low.
object
input_tokens and output_tokens. Output is always 0 — a classifier emits no text.number
What this run cost, in USD.
string | null
The stored run’s id, or
null when save: false.Notes
Always include an
unknown option on a choice. Without an escape hatch the model must pick one of your real labels even on off-topic content.state is re-encoded for every question, so a long document with many questions is fine only while each individual row fits. Over-long content is rejected with a 400 before anything is charged.
Errors
| Status | Meaning |
|---|---|
400 | Invalid rubric — the message names the offending question. Also returned when the request exceeds 32 KiB or the 8K-token context. |
402 | Insufficient credits; details.shortfall says how much is missing. |
403 | Your role cannot run the judge, or an API key lacks the judge scope. |
404 | Unknown rubric_id, or it belongs to another user. |
429 | The judge’s inference queue is full — back off and retry. |
502 / 503 | The judge service is misconfigured or unavailable. |
Example
curl -X POST https://api.60db.ai/judge/evaluate \
-H "Authorization: Bearer your-api-key" \
-H "Content-Type: application/json" \
-d '{
"state": "Agent: I can't refund that, it's outside the window.\nCaller: This is the third time I've called about this.",
"questions": {
"tone": {
"type": "choice",
"instructions": "How did the agent come across?",
"criteria": {
"professional": "Calm, courteous, takes ownership",
"dismissive": "Brushes the caller off",
"unknown": "Not enough of the call to tell"
}
},
"satisfaction": {
"type": "score",
"instructions": "How satisfied does the caller sound?",
"criteria": [
"Angry",
"Unhappy",
"Neutral",
"Satisfied"
]
},
"resolved": {
"type": "noul",
"instructions": "Was the problem solved?"
}
},
"label": "support-call-qa"
}'
const response = await fetch('https://api.60db.ai/judge/evaluate', {
method: 'POST',
headers: {
'Authorization': 'Bearer your-api-key',
'Content-Type': 'application/json',
},
body: JSON.stringify({
"state": "Agent: I can't refund that, it's outside the window.\nCaller: This is the third time I've called about this.",
"questions": {
"tone": {
"type": "choice",
"instructions": "How did the agent come across?",
"criteria": {
"professional": "Calm, courteous, takes ownership",
"dismissive": "Brushes the caller off",
"unknown": "Not enough of the call to tell"
}
},
"satisfaction": {
"type": "score",
"instructions": "How satisfied does the caller sound?",
"criteria": [
"Angry",
"Unhappy",
"Neutral",
"Satisfied"
]
},
"resolved": {
"type": "noul",
"instructions": "Was the problem solved?"
}
},
"label": "support-call-qa"
}),
});
const data = await response.json();
import requests
response = requests.post(
"https://api.60db.ai/judge/evaluate",
headers={
"Authorization": "Bearer your-api-key",
"Content-Type": "application/json",
},
json={
"state": "Agent: I can't refund that, it's outside the window.\nCaller: This is the third time I've called about this.",
"questions": {
"tone": {
"type": "choice",
"instructions": "How did the agent come across?",
"criteria": {
"professional": "Calm, courteous, takes ownership",
"dismissive": "Brushes the caller off",
"unknown": "Not enough of the call to tell"
}
},
"satisfaction": {
"type": "score",
"instructions": "How satisfied does the caller sound?",
"criteria": [
"Angry",
"Unhappy",
"Neutral",
"Satisfied"
]
},
"resolved": {
"type": "noul",
"instructions": "Was the problem solved?"
}
},
"label": "support-call-qa"
},
)
data = response.json()
{
"success": true,
"data": {
"id": "a1b2c3d4-e5f6-4789-abcd-ef0123456789",
"saved": true,
"rubric_id": null,
"model": "60db-decision-model-v1",
"answers": {
"tone": {
"type": "choice",
"choice": "dismissive",
"probabilities": {
"professional": 0.23492,
"dismissive": 0.418141,
"unknown": 0.346939
},
"confidence": 0.4681
},
"satisfaction": {
"type": "score",
"score": 1.1592,
"legend": {
"0": "Angry",
"1": "Unhappy",
"2": "Neutral",
"3": "Satisfied"
},
"probabilities": {
"0": 0.255292,
"1": 0.420406,
"2": 0.234124,
"3": 0.090178
},
"confidence": 0.4704
},
"resolved": {
"type": "noul",
"noul": 0.275
}
},
"summary": {
"tone": "dismissive",
"satisfaction": 1.1592,
"resolved": 0.275
},
"min_confidence": 0.4681,
"escalated_questions": 2,
"usage": {
"input_tokens": 84,
"output_tokens": 0
},
"metadata": {
"backend": "60db-local",
"probabilities": "model_softmax",
"confidence": "typesafe_adapter_metrics_uncalibrated",
"confidence_source": "typesafe-ai/system-one-adapter-python@adffc2eab300a4fa3c0e92252d4ffd6ceaa53700/src/system_one_adapter/_utils/confidence_metrics.py",
"source_model": "60db-jev-source",
"escalated_questions": 2
},
"latency_ms": 4,
"credits_charged": 1.5100000000000002e-06
}
}