llm judge
Create llm judge test
Requires scopes: eval:write
POST
/
api
/
llm-judge
/
test
Create llm judge test
curl --request POST \
--url https://evalgate.com/api/llm-judge/test \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"input": "<string>",
"output": "<string>",
"promptTemplate": "<string>",
"expectedOutput": "<string>",
"provider": "local",
"aggregation": "all_pass",
"context": "<string>",
"model": "<string>",
"criteria": {},
"settings": {},
"behavior": "<string>",
"taskType": "<string>",
"passThreshold": 123,
"judges": [
{
"type": "rule",
"id": "<string>",
"provider": "local",
"threshold": 123,
"model": "<string>",
"temperature": 123,
"weight": 123,
"seed": 123,
"topP": 123,
"maxTokens": 123,
"rule": "json_schema",
"top_p": 123,
"timeoutMs": 123,
"cacheEnabled": false
}
],
"instabilityThreshold": 123,
"stabilityRuns": 123
}
'import requests
url = "https://evalgate.com/api/llm-judge/test"
payload = {
"input": "<string>",
"output": "<string>",
"promptTemplate": "<string>",
"expectedOutput": "<string>",
"provider": "local",
"aggregation": "all_pass",
"context": "<string>",
"model": "<string>",
"criteria": {},
"settings": {},
"behavior": "<string>",
"taskType": "<string>",
"passThreshold": 123,
"judges": [
{
"type": "rule",
"id": "<string>",
"provider": "local",
"threshold": 123,
"model": "<string>",
"temperature": 123,
"weight": 123,
"seed": 123,
"topP": 123,
"maxTokens": 123,
"rule": "json_schema",
"top_p": 123,
"timeoutMs": 123,
"cacheEnabled": False
}
],
"instabilityThreshold": 123,
"stabilityRuns": 123
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
input: '<string>',
output: '<string>',
promptTemplate: '<string>',
expectedOutput: '<string>',
provider: 'local',
aggregation: 'all_pass',
context: '<string>',
model: '<string>',
criteria: {},
settings: {},
behavior: '<string>',
taskType: '<string>',
passThreshold: 123,
judges: [
{
type: 'rule',
id: '<string>',
provider: 'local',
threshold: 123,
model: '<string>',
temperature: 123,
weight: 123,
seed: 123,
topP: 123,
maxTokens: 123,
rule: 'json_schema',
top_p: 123,
timeoutMs: 123,
cacheEnabled: false
}
],
instabilityThreshold: 123,
stabilityRuns: 123
})
};
fetch('https://evalgate.com/api/llm-judge/test', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://evalgate.com/api/llm-judge/test",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'input' => '<string>',
'output' => '<string>',
'promptTemplate' => '<string>',
'expectedOutput' => '<string>',
'provider' => 'local',
'aggregation' => 'all_pass',
'context' => '<string>',
'model' => '<string>',
'criteria' => [
],
'settings' => [
],
'behavior' => '<string>',
'taskType' => '<string>',
'passThreshold' => 123,
'judges' => [
[
'type' => 'rule',
'id' => '<string>',
'provider' => 'local',
'threshold' => 123,
'model' => '<string>',
'temperature' => 123,
'weight' => 123,
'seed' => 123,
'topP' => 123,
'maxTokens' => 123,
'rule' => 'json_schema',
'top_p' => 123,
'timeoutMs' => 123,
'cacheEnabled' => false
]
],
'instabilityThreshold' => 123,
'stabilityRuns' => 123
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://evalgate.com/api/llm-judge/test"
payload := strings.NewReader("{\n \"input\": \"<string>\",\n \"output\": \"<string>\",\n \"promptTemplate\": \"<string>\",\n \"expectedOutput\": \"<string>\",\n \"provider\": \"local\",\n \"aggregation\": \"all_pass\",\n \"context\": \"<string>\",\n \"model\": \"<string>\",\n \"criteria\": {},\n \"settings\": {},\n \"behavior\": \"<string>\",\n \"taskType\": \"<string>\",\n \"passThreshold\": 123,\n \"judges\": [\n {\n \"type\": \"rule\",\n \"id\": \"<string>\",\n \"provider\": \"local\",\n \"threshold\": 123,\n \"model\": \"<string>\",\n \"temperature\": 123,\n \"weight\": 123,\n \"seed\": 123,\n \"topP\": 123,\n \"maxTokens\": 123,\n \"rule\": \"json_schema\",\n \"top_p\": 123,\n \"timeoutMs\": 123,\n \"cacheEnabled\": false\n }\n ],\n \"instabilityThreshold\": 123,\n \"stabilityRuns\": 123\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://evalgate.com/api/llm-judge/test")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"input\": \"<string>\",\n \"output\": \"<string>\",\n \"promptTemplate\": \"<string>\",\n \"expectedOutput\": \"<string>\",\n \"provider\": \"local\",\n \"aggregation\": \"all_pass\",\n \"context\": \"<string>\",\n \"model\": \"<string>\",\n \"criteria\": {},\n \"settings\": {},\n \"behavior\": \"<string>\",\n \"taskType\": \"<string>\",\n \"passThreshold\": 123,\n \"judges\": [\n {\n \"type\": \"rule\",\n \"id\": \"<string>\",\n \"provider\": \"local\",\n \"threshold\": 123,\n \"model\": \"<string>\",\n \"temperature\": 123,\n \"weight\": 123,\n \"seed\": 123,\n \"topP\": 123,\n \"maxTokens\": 123,\n \"rule\": \"json_schema\",\n \"top_p\": 123,\n \"timeoutMs\": 123,\n \"cacheEnabled\": false\n }\n ],\n \"instabilityThreshold\": 123,\n \"stabilityRuns\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://evalgate.com/api/llm-judge/test")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"input\": \"<string>\",\n \"output\": \"<string>\",\n \"promptTemplate\": \"<string>\",\n \"expectedOutput\": \"<string>\",\n \"provider\": \"local\",\n \"aggregation\": \"all_pass\",\n \"context\": \"<string>\",\n \"model\": \"<string>\",\n \"criteria\": {},\n \"settings\": {},\n \"behavior\": \"<string>\",\n \"taskType\": \"<string>\",\n \"passThreshold\": 123,\n \"judges\": [\n {\n \"type\": \"rule\",\n \"id\": \"<string>\",\n \"provider\": \"local\",\n \"threshold\": 123,\n \"model\": \"<string>\",\n \"temperature\": 123,\n \"weight\": 123,\n \"seed\": 123,\n \"topP\": 123,\n \"maxTokens\": 123,\n \"rule\": \"json_schema\",\n \"top_p\": 123,\n \"timeoutMs\": 123,\n \"cacheEnabled\": false\n }\n ],\n \"instabilityThreshold\": 123,\n \"stabilityRuns\": 123\n}"
response = http.request(request)
puts response.read_body{
"promptPreview": "<string>",
"judgement": {
"provider": "local",
"score": 123,
"latencyMs": 123,
"aggregation": "all_pass",
"model": "<string>",
"reasoning": "<string>",
"flags": [
"<string>"
],
"passed": true,
"signals": {},
"retries": 123,
"judgeResults": [
{
"type": "rule",
"latencyMs": 123,
"judgeId": "<string>",
"error": "<string>",
"provider": "local",
"result": {
"provider": "local",
"score": 123,
"latencyMs": 123,
"model": "<string>",
"reasoning": "<string>",
"passed": true,
"signals": {},
"retries": 123,
"judgeId": "<string>",
"parseStatus": "failed",
"confidence": 123,
"modelCallId": "<string>",
"tokenUsage": {
"input": 123,
"output": 123,
"total": 123
},
"modelProvenance": {
"provider": "local",
"model": "<string>",
"temperature": 123,
"prompt_hash": "<string>",
"top_p": 123,
"max_tokens": 123,
"seed": 123
},
"cached": false,
"judgeFingerprint": {}
},
"model": "<string>",
"weight": 123,
"modelProvenance": {
"provider": "local",
"model": "<string>",
"temperature": 123,
"prompt_hash": "<string>",
"top_p": 123,
"max_tokens": 123,
"seed": 123
}
}
],
"judgeId": "<string>",
"parseStatus": "failed",
"disagreement": {
"rate": 123,
"reasoningDivergence": 123,
"outlierJudgeIds": [
"<string>"
],
"requiresEscalation": true
},
"judgeResultId": 123,
"judgeConfigHash": "<string>",
"comparabilityStatus": "not_comparable",
"confidence": 123,
"totalCostUsd": 123,
"tokenUsage": {
"input": 123,
"output": 123,
"total": 123
},
"variance": 123
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}Authorizations
Bearer authentication header of the form Bearer <token>, where <token> is your auth token.
Body
application/json
provider
Option 1 · enum<string>Option 2 · enum<string>Option 3 · enum<string>Option 4 · enum<string>Option 5 · enum<string>Option 6 · enum<string>Option 7 · enum<string>Option 8 · enum<string>
Available options:
local aggregation
Option 1 · enum<string>Option 2 · enum<string>Option 3 · enum<string>Option 4 · enum<string>Option 5 · enum<string>
Available options:
all_pass Show child attributes
Show child attributes
Show child attributes
Show child attributes
Show child attributes
Show child attributes
⌘I
Create llm judge test
curl --request POST \
--url https://evalgate.com/api/llm-judge/test \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"input": "<string>",
"output": "<string>",
"promptTemplate": "<string>",
"expectedOutput": "<string>",
"provider": "local",
"aggregation": "all_pass",
"context": "<string>",
"model": "<string>",
"criteria": {},
"settings": {},
"behavior": "<string>",
"taskType": "<string>",
"passThreshold": 123,
"judges": [
{
"type": "rule",
"id": "<string>",
"provider": "local",
"threshold": 123,
"model": "<string>",
"temperature": 123,
"weight": 123,
"seed": 123,
"topP": 123,
"maxTokens": 123,
"rule": "json_schema",
"top_p": 123,
"timeoutMs": 123,
"cacheEnabled": false
}
],
"instabilityThreshold": 123,
"stabilityRuns": 123
}
'import requests
url = "https://evalgate.com/api/llm-judge/test"
payload = {
"input": "<string>",
"output": "<string>",
"promptTemplate": "<string>",
"expectedOutput": "<string>",
"provider": "local",
"aggregation": "all_pass",
"context": "<string>",
"model": "<string>",
"criteria": {},
"settings": {},
"behavior": "<string>",
"taskType": "<string>",
"passThreshold": 123,
"judges": [
{
"type": "rule",
"id": "<string>",
"provider": "local",
"threshold": 123,
"model": "<string>",
"temperature": 123,
"weight": 123,
"seed": 123,
"topP": 123,
"maxTokens": 123,
"rule": "json_schema",
"top_p": 123,
"timeoutMs": 123,
"cacheEnabled": False
}
],
"instabilityThreshold": 123,
"stabilityRuns": 123
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
input: '<string>',
output: '<string>',
promptTemplate: '<string>',
expectedOutput: '<string>',
provider: 'local',
aggregation: 'all_pass',
context: '<string>',
model: '<string>',
criteria: {},
settings: {},
behavior: '<string>',
taskType: '<string>',
passThreshold: 123,
judges: [
{
type: 'rule',
id: '<string>',
provider: 'local',
threshold: 123,
model: '<string>',
temperature: 123,
weight: 123,
seed: 123,
topP: 123,
maxTokens: 123,
rule: 'json_schema',
top_p: 123,
timeoutMs: 123,
cacheEnabled: false
}
],
instabilityThreshold: 123,
stabilityRuns: 123
})
};
fetch('https://evalgate.com/api/llm-judge/test', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://evalgate.com/api/llm-judge/test",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'input' => '<string>',
'output' => '<string>',
'promptTemplate' => '<string>',
'expectedOutput' => '<string>',
'provider' => 'local',
'aggregation' => 'all_pass',
'context' => '<string>',
'model' => '<string>',
'criteria' => [
],
'settings' => [
],
'behavior' => '<string>',
'taskType' => '<string>',
'passThreshold' => 123,
'judges' => [
[
'type' => 'rule',
'id' => '<string>',
'provider' => 'local',
'threshold' => 123,
'model' => '<string>',
'temperature' => 123,
'weight' => 123,
'seed' => 123,
'topP' => 123,
'maxTokens' => 123,
'rule' => 'json_schema',
'top_p' => 123,
'timeoutMs' => 123,
'cacheEnabled' => false
]
],
'instabilityThreshold' => 123,
'stabilityRuns' => 123
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://evalgate.com/api/llm-judge/test"
payload := strings.NewReader("{\n \"input\": \"<string>\",\n \"output\": \"<string>\",\n \"promptTemplate\": \"<string>\",\n \"expectedOutput\": \"<string>\",\n \"provider\": \"local\",\n \"aggregation\": \"all_pass\",\n \"context\": \"<string>\",\n \"model\": \"<string>\",\n \"criteria\": {},\n \"settings\": {},\n \"behavior\": \"<string>\",\n \"taskType\": \"<string>\",\n \"passThreshold\": 123,\n \"judges\": [\n {\n \"type\": \"rule\",\n \"id\": \"<string>\",\n \"provider\": \"local\",\n \"threshold\": 123,\n \"model\": \"<string>\",\n \"temperature\": 123,\n \"weight\": 123,\n \"seed\": 123,\n \"topP\": 123,\n \"maxTokens\": 123,\n \"rule\": \"json_schema\",\n \"top_p\": 123,\n \"timeoutMs\": 123,\n \"cacheEnabled\": false\n }\n ],\n \"instabilityThreshold\": 123,\n \"stabilityRuns\": 123\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://evalgate.com/api/llm-judge/test")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"input\": \"<string>\",\n \"output\": \"<string>\",\n \"promptTemplate\": \"<string>\",\n \"expectedOutput\": \"<string>\",\n \"provider\": \"local\",\n \"aggregation\": \"all_pass\",\n \"context\": \"<string>\",\n \"model\": \"<string>\",\n \"criteria\": {},\n \"settings\": {},\n \"behavior\": \"<string>\",\n \"taskType\": \"<string>\",\n \"passThreshold\": 123,\n \"judges\": [\n {\n \"type\": \"rule\",\n \"id\": \"<string>\",\n \"provider\": \"local\",\n \"threshold\": 123,\n \"model\": \"<string>\",\n \"temperature\": 123,\n \"weight\": 123,\n \"seed\": 123,\n \"topP\": 123,\n \"maxTokens\": 123,\n \"rule\": \"json_schema\",\n \"top_p\": 123,\n \"timeoutMs\": 123,\n \"cacheEnabled\": false\n }\n ],\n \"instabilityThreshold\": 123,\n \"stabilityRuns\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://evalgate.com/api/llm-judge/test")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"input\": \"<string>\",\n \"output\": \"<string>\",\n \"promptTemplate\": \"<string>\",\n \"expectedOutput\": \"<string>\",\n \"provider\": \"local\",\n \"aggregation\": \"all_pass\",\n \"context\": \"<string>\",\n \"model\": \"<string>\",\n \"criteria\": {},\n \"settings\": {},\n \"behavior\": \"<string>\",\n \"taskType\": \"<string>\",\n \"passThreshold\": 123,\n \"judges\": [\n {\n \"type\": \"rule\",\n \"id\": \"<string>\",\n \"provider\": \"local\",\n \"threshold\": 123,\n \"model\": \"<string>\",\n \"temperature\": 123,\n \"weight\": 123,\n \"seed\": 123,\n \"topP\": 123,\n \"maxTokens\": 123,\n \"rule\": \"json_schema\",\n \"top_p\": 123,\n \"timeoutMs\": 123,\n \"cacheEnabled\": false\n }\n ],\n \"instabilityThreshold\": 123,\n \"stabilityRuns\": 123\n}"
response = http.request(request)
puts response.read_body{
"promptPreview": "<string>",
"judgement": {
"provider": "local",
"score": 123,
"latencyMs": 123,
"aggregation": "all_pass",
"model": "<string>",
"reasoning": "<string>",
"flags": [
"<string>"
],
"passed": true,
"signals": {},
"retries": 123,
"judgeResults": [
{
"type": "rule",
"latencyMs": 123,
"judgeId": "<string>",
"error": "<string>",
"provider": "local",
"result": {
"provider": "local",
"score": 123,
"latencyMs": 123,
"model": "<string>",
"reasoning": "<string>",
"passed": true,
"signals": {},
"retries": 123,
"judgeId": "<string>",
"parseStatus": "failed",
"confidence": 123,
"modelCallId": "<string>",
"tokenUsage": {
"input": 123,
"output": 123,
"total": 123
},
"modelProvenance": {
"provider": "local",
"model": "<string>",
"temperature": 123,
"prompt_hash": "<string>",
"top_p": 123,
"max_tokens": 123,
"seed": 123
},
"cached": false,
"judgeFingerprint": {}
},
"model": "<string>",
"weight": 123,
"modelProvenance": {
"provider": "local",
"model": "<string>",
"temperature": 123,
"prompt_hash": "<string>",
"top_p": 123,
"max_tokens": 123,
"seed": 123
}
}
],
"judgeId": "<string>",
"parseStatus": "failed",
"disagreement": {
"rate": 123,
"reasoningDivergence": 123,
"outlierJudgeIds": [
"<string>"
],
"requiresEscalation": true
},
"judgeResultId": 123,
"judgeConfigHash": "<string>",
"comparabilityStatus": "not_comparable",
"confidence": 123,
"totalCostUsd": 123,
"tokenUsage": {
"input": 123,
"output": 123,
"total": 123
},
"variance": 123
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}