evaluations
Create evaluations human review
Requires scopes: eval:write
POST
/
api
/
evaluations
/
{id}
/
human-review
Create evaluations human review
curl --request POST \
--url https://evalgate.com/api/evaluations/{id}/human-review \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"sampleSize": 123,
"runId": 123,
"seed": "<string>",
"assignedReviewerId": "<string>"
}
'import requests
url = "https://evalgate.com/api/evaluations/{id}/human-review"
payload = {
"sampleSize": 123,
"runId": 123,
"seed": "<string>",
"assignedReviewerId": "<string>"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({sampleSize: 123, runId: 123, seed: '<string>', assignedReviewerId: '<string>'})
};
fetch('https://evalgate.com/api/evaluations/{id}/human-review', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://evalgate.com/api/evaluations/{id}/human-review",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'sampleSize' => 123,
'runId' => 123,
'seed' => '<string>',
'assignedReviewerId' => '<string>'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://evalgate.com/api/evaluations/{id}/human-review"
payload := strings.NewReader("{\n \"sampleSize\": 123,\n \"runId\": 123,\n \"seed\": \"<string>\",\n \"assignedReviewerId\": \"<string>\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://evalgate.com/api/evaluations/{id}/human-review")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"sampleSize\": 123,\n \"runId\": 123,\n \"seed\": \"<string>\",\n \"assignedReviewerId\": \"<string>\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://evalgate.com/api/evaluations/{id}/human-review")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"sampleSize\": 123,\n \"runId\": 123,\n \"seed\": \"<string>\",\n \"assignedReviewerId\": \"<string>\"\n}"
response = http.request(request)
puts response.read_body{
"batch": {
"id": 123,
"evaluationId": 123,
"evaluationRunId": 123,
"sampleSize": 123,
"samplingStrategy": "<string>",
"seed": "<string>",
"assignedReviewerId": "<string>",
"createdBy": "<string>",
"createdAt": "<string>"
},
"items": [
{
"sampleId": 123,
"testCaseId": 123,
"testResultId": 123,
"snapshot": {
"testCaseId": 123,
"testResultId": 123,
"systemVerdict": "pass",
"caseName": "<string>",
"input": "<string>",
"expectedOutput": "<string>",
"output": "<string>",
"error": "<string>",
"assertions": {},
"messages": [
{
"role": "<string>",
"content": "<string>",
"timestamp": "<string>"
}
],
"toolCalls": [
{
"name": "<string>",
"arguments": {},
"output": "<string>",
"timestamp": "<string>"
}
],
"trajectory": {
"testCaseId": 123,
"status": "failed",
"trajectory": {
"id": "<string>",
"input": "<string>",
"finalOutput": "<string>",
"steps": [
{
"index": 123,
"type": "final",
"content": "<string>",
"toolName": "<string>",
"toolArgs": {},
"timestamp": 123
}
],
"totalSteps": 123,
"totalToolCalls": 123,
"durationMs": 123,
"terminatedEarly": false,
"error": "<string>"
},
"behavior": "tool_use",
"taskType": "<string>",
"idealTrajectory": {
"expectedSteps": 123,
"expectedToolCalls": 123,
"requiredTools": [
"<string>"
],
"forbiddenTools": [
"<string>"
],
"allowParallel": false,
"maxRedundantSteps": 123
},
"baseline": {
"behavior": "tool_use",
"stepDistribution": {
"p50": 123,
"p75": 123,
"p90": 123
},
"toolCallDistribution": {
"p50": 123,
"p75": 123
},
"durationDistribution": {
"p50": 123,
"p75": 123
},
"commonToolSequences": [
[
"<string>"
]
],
"redundancyProfile": {
"mean": 123,
"p90": 123
},
"sampleSize": 123,
"updatedAt": "<string>",
"taskType": "<string>",
"parallelToolGroups": [
[
"<string>"
]
],
"version": "<string>",
"model": "<string>"
},
"evaluation": {
"schemaVersion": 123,
"valid": true,
"score": 123,
"failureModes": [
"inefficient_path"
],
"metrics": {
"actualSteps": 123,
"actualToolCalls": 123,
"actualDurationMs": 123,
"redundancyRatio": 123,
"redundancySteps": 123,
"sequenceSimilarity": 123,
"efficiency": 123,
"toolUsage": 123,
"redundancy": 123,
"latency": 123,
"parallelizationPenalty": 123,
"parallelizationScore": 123,
"requiredToolsMissing": [
"<string>"
],
"forbiddenToolsUsed": [
"<string>"
],
"extraToolsUsed": [
"<string>"
],
"selectedBaselineVersion": "<string>",
"baselineStepP50": 123,
"baselineStepP75": 123,
"baselineToolCallP50": 123,
"baselineToolCallP75": 123,
"baselineDurationP50": 123,
"baselineDurationP75": 123,
"idealExpectedSteps": 123,
"idealExpectedToolCalls": 123
},
"trajectoryId": "<string>",
"behavior": "tool_use",
"taskType": "<string>",
"baselineVersion": "<string>"
}
},
"trajectoryMetrics": {
"testCaseId": 123,
"status": "failed",
"score": 123,
"failureModes": [
"inefficient_path"
],
"metrics": {
"actualSteps": 123,
"actualToolCalls": 123,
"actualDurationMs": 123,
"redundancyRatio": 123,
"redundancySteps": 123,
"sequenceSimilarity": 123,
"efficiency": 123,
"toolUsage": 123,
"redundancy": 123,
"latency": 123,
"parallelizationPenalty": 123,
"parallelizationScore": 123,
"requiredToolsMissing": [
"<string>"
],
"forbiddenToolsUsed": [
"<string>"
],
"extraToolsUsed": [
"<string>"
],
"selectedBaselineVersion": "<string>",
"baselineStepP50": 123,
"baselineStepP75": 123,
"baselineToolCallP50": 123,
"baselineToolCallP75": 123,
"baselineDurationP50": 123,
"baselineDurationP75": 123,
"idealExpectedSteps": 123,
"idealExpectedToolCalls": 123
},
"trajectoryId": "<string>",
"behavior": "tool_use",
"taskType": "<string>",
"baselineVersion": "<string>"
}
},
"reviews": [
{
"id": 123,
"batchId": 123,
"sampleId": 123,
"testCaseId": 123,
"reviewerId": "<string>",
"systemVerdict": "pass",
"reviewVerdict": "pass",
"agreedWithSystem": false,
"rating": 123,
"feedback": "<string>",
"labels": {},
"createdAt": "<string>"
}
],
"latestReview": {
"id": 123,
"batchId": 123,
"sampleId": 123,
"testCaseId": 123,
"reviewerId": "<string>",
"systemVerdict": "pass",
"reviewVerdict": "pass",
"agreedWithSystem": false,
"rating": 123,
"feedback": "<string>",
"labels": {},
"createdAt": "<string>"
}
}
],
"stats": {
"totalSamples": 123,
"reviewedSamples": 123,
"totalReviews": 123,
"comparableReviews": 123,
"agreementRate": 123,
"disagreementRate": 123,
"uncertainRate": 123,
"verdictDistribution": {
"pass": 123,
"fail": 123,
"uncertain": 123
}
},
"history": [
{
"batchId": 123,
"createdAt": "<string>",
"totalReviews": 123,
"comparableReviews": 123,
"disagreementRate": 123
}
]
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}Authorizations
Bearer authentication header of the form Bearer <token>, where <token> is your auth token.
Path Parameters
⌘I
Create evaluations human review
curl --request POST \
--url https://evalgate.com/api/evaluations/{id}/human-review \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"sampleSize": 123,
"runId": 123,
"seed": "<string>",
"assignedReviewerId": "<string>"
}
'import requests
url = "https://evalgate.com/api/evaluations/{id}/human-review"
payload = {
"sampleSize": 123,
"runId": 123,
"seed": "<string>",
"assignedReviewerId": "<string>"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({sampleSize: 123, runId: 123, seed: '<string>', assignedReviewerId: '<string>'})
};
fetch('https://evalgate.com/api/evaluations/{id}/human-review', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://evalgate.com/api/evaluations/{id}/human-review",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'sampleSize' => 123,
'runId' => 123,
'seed' => '<string>',
'assignedReviewerId' => '<string>'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://evalgate.com/api/evaluations/{id}/human-review"
payload := strings.NewReader("{\n \"sampleSize\": 123,\n \"runId\": 123,\n \"seed\": \"<string>\",\n \"assignedReviewerId\": \"<string>\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://evalgate.com/api/evaluations/{id}/human-review")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"sampleSize\": 123,\n \"runId\": 123,\n \"seed\": \"<string>\",\n \"assignedReviewerId\": \"<string>\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://evalgate.com/api/evaluations/{id}/human-review")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"sampleSize\": 123,\n \"runId\": 123,\n \"seed\": \"<string>\",\n \"assignedReviewerId\": \"<string>\"\n}"
response = http.request(request)
puts response.read_body{
"batch": {
"id": 123,
"evaluationId": 123,
"evaluationRunId": 123,
"sampleSize": 123,
"samplingStrategy": "<string>",
"seed": "<string>",
"assignedReviewerId": "<string>",
"createdBy": "<string>",
"createdAt": "<string>"
},
"items": [
{
"sampleId": 123,
"testCaseId": 123,
"testResultId": 123,
"snapshot": {
"testCaseId": 123,
"testResultId": 123,
"systemVerdict": "pass",
"caseName": "<string>",
"input": "<string>",
"expectedOutput": "<string>",
"output": "<string>",
"error": "<string>",
"assertions": {},
"messages": [
{
"role": "<string>",
"content": "<string>",
"timestamp": "<string>"
}
],
"toolCalls": [
{
"name": "<string>",
"arguments": {},
"output": "<string>",
"timestamp": "<string>"
}
],
"trajectory": {
"testCaseId": 123,
"status": "failed",
"trajectory": {
"id": "<string>",
"input": "<string>",
"finalOutput": "<string>",
"steps": [
{
"index": 123,
"type": "final",
"content": "<string>",
"toolName": "<string>",
"toolArgs": {},
"timestamp": 123
}
],
"totalSteps": 123,
"totalToolCalls": 123,
"durationMs": 123,
"terminatedEarly": false,
"error": "<string>"
},
"behavior": "tool_use",
"taskType": "<string>",
"idealTrajectory": {
"expectedSteps": 123,
"expectedToolCalls": 123,
"requiredTools": [
"<string>"
],
"forbiddenTools": [
"<string>"
],
"allowParallel": false,
"maxRedundantSteps": 123
},
"baseline": {
"behavior": "tool_use",
"stepDistribution": {
"p50": 123,
"p75": 123,
"p90": 123
},
"toolCallDistribution": {
"p50": 123,
"p75": 123
},
"durationDistribution": {
"p50": 123,
"p75": 123
},
"commonToolSequences": [
[
"<string>"
]
],
"redundancyProfile": {
"mean": 123,
"p90": 123
},
"sampleSize": 123,
"updatedAt": "<string>",
"taskType": "<string>",
"parallelToolGroups": [
[
"<string>"
]
],
"version": "<string>",
"model": "<string>"
},
"evaluation": {
"schemaVersion": 123,
"valid": true,
"score": 123,
"failureModes": [
"inefficient_path"
],
"metrics": {
"actualSteps": 123,
"actualToolCalls": 123,
"actualDurationMs": 123,
"redundancyRatio": 123,
"redundancySteps": 123,
"sequenceSimilarity": 123,
"efficiency": 123,
"toolUsage": 123,
"redundancy": 123,
"latency": 123,
"parallelizationPenalty": 123,
"parallelizationScore": 123,
"requiredToolsMissing": [
"<string>"
],
"forbiddenToolsUsed": [
"<string>"
],
"extraToolsUsed": [
"<string>"
],
"selectedBaselineVersion": "<string>",
"baselineStepP50": 123,
"baselineStepP75": 123,
"baselineToolCallP50": 123,
"baselineToolCallP75": 123,
"baselineDurationP50": 123,
"baselineDurationP75": 123,
"idealExpectedSteps": 123,
"idealExpectedToolCalls": 123
},
"trajectoryId": "<string>",
"behavior": "tool_use",
"taskType": "<string>",
"baselineVersion": "<string>"
}
},
"trajectoryMetrics": {
"testCaseId": 123,
"status": "failed",
"score": 123,
"failureModes": [
"inefficient_path"
],
"metrics": {
"actualSteps": 123,
"actualToolCalls": 123,
"actualDurationMs": 123,
"redundancyRatio": 123,
"redundancySteps": 123,
"sequenceSimilarity": 123,
"efficiency": 123,
"toolUsage": 123,
"redundancy": 123,
"latency": 123,
"parallelizationPenalty": 123,
"parallelizationScore": 123,
"requiredToolsMissing": [
"<string>"
],
"forbiddenToolsUsed": [
"<string>"
],
"extraToolsUsed": [
"<string>"
],
"selectedBaselineVersion": "<string>",
"baselineStepP50": 123,
"baselineStepP75": 123,
"baselineToolCallP50": 123,
"baselineToolCallP75": 123,
"baselineDurationP50": 123,
"baselineDurationP75": 123,
"idealExpectedSteps": 123,
"idealExpectedToolCalls": 123
},
"trajectoryId": "<string>",
"behavior": "tool_use",
"taskType": "<string>",
"baselineVersion": "<string>"
}
},
"reviews": [
{
"id": 123,
"batchId": 123,
"sampleId": 123,
"testCaseId": 123,
"reviewerId": "<string>",
"systemVerdict": "pass",
"reviewVerdict": "pass",
"agreedWithSystem": false,
"rating": 123,
"feedback": "<string>",
"labels": {},
"createdAt": "<string>"
}
],
"latestReview": {
"id": 123,
"batchId": 123,
"sampleId": 123,
"testCaseId": 123,
"reviewerId": "<string>",
"systemVerdict": "pass",
"reviewVerdict": "pass",
"agreedWithSystem": false,
"rating": 123,
"feedback": "<string>",
"labels": {},
"createdAt": "<string>"
}
}
],
"stats": {
"totalSamples": 123,
"reviewedSamples": 123,
"totalReviews": 123,
"comparableReviews": 123,
"agreementRate": 123,
"disagreementRate": 123,
"uncertainRate": 123,
"verdictDistribution": {
"pass": 123,
"fail": 123,
"uncertain": 123
}
},
"history": [
{
"batchId": 123,
"createdAt": "<string>",
"totalReviews": 123,
"comparableReviews": 123,
"disagreementRate": 123
}
]
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}