evaluations
Get evaluations human review
Requires scopes: runs:read
GET
/
api
/
evaluations
/
{id}
/
human-review
Get evaluations human review
curl --request GET \
--url https://evalgate.com/api/evaluations/{id}/human-review \
--header 'Authorization: Bearer <token>'import requests
url = "https://evalgate.com/api/evaluations/{id}/human-review"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://evalgate.com/api/evaluations/{id}/human-review', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://evalgate.com/api/evaluations/{id}/human-review",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://evalgate.com/api/evaluations/{id}/human-review"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://evalgate.com/api/evaluations/{id}/human-review")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://evalgate.com/api/evaluations/{id}/human-review")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body{
"batch": {
"id": 123,
"evaluationId": 123,
"evaluationRunId": 123,
"sampleSize": 123,
"samplingStrategy": "<string>",
"seed": "<string>",
"assignedReviewerId": "<string>",
"createdBy": "<string>",
"createdAt": "<string>"
},
"items": [
{
"sampleId": 123,
"testCaseId": 123,
"testResultId": 123,
"snapshot": {
"testCaseId": 123,
"testResultId": 123,
"systemVerdict": "pass",
"caseName": "<string>",
"input": "<string>",
"expectedOutput": "<string>",
"output": "<string>",
"error": "<string>",
"assertions": {},
"messages": [
{
"role": "<string>",
"content": "<string>",
"timestamp": "<string>"
}
],
"toolCalls": [
{
"name": "<string>",
"arguments": {},
"output": "<string>",
"timestamp": "<string>"
}
],
"trajectory": {
"testCaseId": 123,
"status": "failed",
"trajectory": {
"id": "<string>",
"input": "<string>",
"finalOutput": "<string>",
"steps": [
{
"index": 123,
"type": "final",
"content": "<string>",
"toolName": "<string>",
"toolArgs": {},
"timestamp": 123
}
],
"totalSteps": 123,
"totalToolCalls": 123,
"durationMs": 123,
"terminatedEarly": false,
"error": "<string>"
},
"behavior": "tool_use",
"taskType": "<string>",
"idealTrajectory": {
"expectedSteps": 123,
"expectedToolCalls": 123,
"requiredTools": [
"<string>"
],
"forbiddenTools": [
"<string>"
],
"allowParallel": false,
"maxRedundantSteps": 123
},
"baseline": {
"behavior": "tool_use",
"stepDistribution": {
"p50": 123,
"p75": 123,
"p90": 123
},
"toolCallDistribution": {
"p50": 123,
"p75": 123
},
"durationDistribution": {
"p50": 123,
"p75": 123
},
"commonToolSequences": [
[
"<string>"
]
],
"redundancyProfile": {
"mean": 123,
"p90": 123
},
"sampleSize": 123,
"updatedAt": "<string>",
"taskType": "<string>",
"parallelToolGroups": [
[
"<string>"
]
],
"version": "<string>",
"model": "<string>"
},
"evaluation": {
"schemaVersion": 123,
"valid": true,
"score": 123,
"failureModes": [
"inefficient_path"
],
"metrics": {
"actualSteps": 123,
"actualToolCalls": 123,
"actualDurationMs": 123,
"redundancyRatio": 123,
"redundancySteps": 123,
"sequenceSimilarity": 123,
"efficiency": 123,
"toolUsage": 123,
"redundancy": 123,
"latency": 123,
"parallelizationPenalty": 123,
"parallelizationScore": 123,
"requiredToolsMissing": [
"<string>"
],
"forbiddenToolsUsed": [
"<string>"
],
"extraToolsUsed": [
"<string>"
],
"selectedBaselineVersion": "<string>",
"baselineStepP50": 123,
"baselineStepP75": 123,
"baselineToolCallP50": 123,
"baselineToolCallP75": 123,
"baselineDurationP50": 123,
"baselineDurationP75": 123,
"idealExpectedSteps": 123,
"idealExpectedToolCalls": 123
},
"trajectoryId": "<string>",
"behavior": "tool_use",
"taskType": "<string>",
"baselineVersion": "<string>"
}
},
"trajectoryMetrics": {
"testCaseId": 123,
"status": "failed",
"score": 123,
"failureModes": [
"inefficient_path"
],
"metrics": {
"actualSteps": 123,
"actualToolCalls": 123,
"actualDurationMs": 123,
"redundancyRatio": 123,
"redundancySteps": 123,
"sequenceSimilarity": 123,
"efficiency": 123,
"toolUsage": 123,
"redundancy": 123,
"latency": 123,
"parallelizationPenalty": 123,
"parallelizationScore": 123,
"requiredToolsMissing": [
"<string>"
],
"forbiddenToolsUsed": [
"<string>"
],
"extraToolsUsed": [
"<string>"
],
"selectedBaselineVersion": "<string>",
"baselineStepP50": 123,
"baselineStepP75": 123,
"baselineToolCallP50": 123,
"baselineToolCallP75": 123,
"baselineDurationP50": 123,
"baselineDurationP75": 123,
"idealExpectedSteps": 123,
"idealExpectedToolCalls": 123
},
"trajectoryId": "<string>",
"behavior": "tool_use",
"taskType": "<string>",
"baselineVersion": "<string>"
}
},
"reviews": [
{
"id": 123,
"batchId": 123,
"sampleId": 123,
"testCaseId": 123,
"reviewerId": "<string>",
"systemVerdict": "pass",
"reviewVerdict": "pass",
"agreedWithSystem": false,
"rating": 123,
"feedback": "<string>",
"labels": {},
"createdAt": "<string>"
}
],
"latestReview": {
"id": 123,
"batchId": 123,
"sampleId": 123,
"testCaseId": 123,
"reviewerId": "<string>",
"systemVerdict": "pass",
"reviewVerdict": "pass",
"agreedWithSystem": false,
"rating": 123,
"feedback": "<string>",
"labels": {},
"createdAt": "<string>"
}
}
],
"stats": {
"totalSamples": 123,
"reviewedSamples": 123,
"totalReviews": 123,
"comparableReviews": 123,
"agreementRate": 123,
"disagreementRate": 123,
"uncertainRate": 123,
"verdictDistribution": {
"pass": 123,
"fail": 123,
"uncertain": 123
}
},
"history": [
{
"batchId": 123,
"createdAt": "<string>",
"totalReviews": 123,
"comparableReviews": 123,
"disagreementRate": 123
}
]
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}Authorizations
Bearer authentication header of the form Bearer <token>, where <token> is your auth token.
Path Parameters
⌘I
Get evaluations human review
curl --request GET \
--url https://evalgate.com/api/evaluations/{id}/human-review \
--header 'Authorization: Bearer <token>'import requests
url = "https://evalgate.com/api/evaluations/{id}/human-review"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://evalgate.com/api/evaluations/{id}/human-review', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://evalgate.com/api/evaluations/{id}/human-review",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://evalgate.com/api/evaluations/{id}/human-review"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://evalgate.com/api/evaluations/{id}/human-review")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://evalgate.com/api/evaluations/{id}/human-review")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body{
"batch": {
"id": 123,
"evaluationId": 123,
"evaluationRunId": 123,
"sampleSize": 123,
"samplingStrategy": "<string>",
"seed": "<string>",
"assignedReviewerId": "<string>",
"createdBy": "<string>",
"createdAt": "<string>"
},
"items": [
{
"sampleId": 123,
"testCaseId": 123,
"testResultId": 123,
"snapshot": {
"testCaseId": 123,
"testResultId": 123,
"systemVerdict": "pass",
"caseName": "<string>",
"input": "<string>",
"expectedOutput": "<string>",
"output": "<string>",
"error": "<string>",
"assertions": {},
"messages": [
{
"role": "<string>",
"content": "<string>",
"timestamp": "<string>"
}
],
"toolCalls": [
{
"name": "<string>",
"arguments": {},
"output": "<string>",
"timestamp": "<string>"
}
],
"trajectory": {
"testCaseId": 123,
"status": "failed",
"trajectory": {
"id": "<string>",
"input": "<string>",
"finalOutput": "<string>",
"steps": [
{
"index": 123,
"type": "final",
"content": "<string>",
"toolName": "<string>",
"toolArgs": {},
"timestamp": 123
}
],
"totalSteps": 123,
"totalToolCalls": 123,
"durationMs": 123,
"terminatedEarly": false,
"error": "<string>"
},
"behavior": "tool_use",
"taskType": "<string>",
"idealTrajectory": {
"expectedSteps": 123,
"expectedToolCalls": 123,
"requiredTools": [
"<string>"
],
"forbiddenTools": [
"<string>"
],
"allowParallel": false,
"maxRedundantSteps": 123
},
"baseline": {
"behavior": "tool_use",
"stepDistribution": {
"p50": 123,
"p75": 123,
"p90": 123
},
"toolCallDistribution": {
"p50": 123,
"p75": 123
},
"durationDistribution": {
"p50": 123,
"p75": 123
},
"commonToolSequences": [
[
"<string>"
]
],
"redundancyProfile": {
"mean": 123,
"p90": 123
},
"sampleSize": 123,
"updatedAt": "<string>",
"taskType": "<string>",
"parallelToolGroups": [
[
"<string>"
]
],
"version": "<string>",
"model": "<string>"
},
"evaluation": {
"schemaVersion": 123,
"valid": true,
"score": 123,
"failureModes": [
"inefficient_path"
],
"metrics": {
"actualSteps": 123,
"actualToolCalls": 123,
"actualDurationMs": 123,
"redundancyRatio": 123,
"redundancySteps": 123,
"sequenceSimilarity": 123,
"efficiency": 123,
"toolUsage": 123,
"redundancy": 123,
"latency": 123,
"parallelizationPenalty": 123,
"parallelizationScore": 123,
"requiredToolsMissing": [
"<string>"
],
"forbiddenToolsUsed": [
"<string>"
],
"extraToolsUsed": [
"<string>"
],
"selectedBaselineVersion": "<string>",
"baselineStepP50": 123,
"baselineStepP75": 123,
"baselineToolCallP50": 123,
"baselineToolCallP75": 123,
"baselineDurationP50": 123,
"baselineDurationP75": 123,
"idealExpectedSteps": 123,
"idealExpectedToolCalls": 123
},
"trajectoryId": "<string>",
"behavior": "tool_use",
"taskType": "<string>",
"baselineVersion": "<string>"
}
},
"trajectoryMetrics": {
"testCaseId": 123,
"status": "failed",
"score": 123,
"failureModes": [
"inefficient_path"
],
"metrics": {
"actualSteps": 123,
"actualToolCalls": 123,
"actualDurationMs": 123,
"redundancyRatio": 123,
"redundancySteps": 123,
"sequenceSimilarity": 123,
"efficiency": 123,
"toolUsage": 123,
"redundancy": 123,
"latency": 123,
"parallelizationPenalty": 123,
"parallelizationScore": 123,
"requiredToolsMissing": [
"<string>"
],
"forbiddenToolsUsed": [
"<string>"
],
"extraToolsUsed": [
"<string>"
],
"selectedBaselineVersion": "<string>",
"baselineStepP50": 123,
"baselineStepP75": 123,
"baselineToolCallP50": 123,
"baselineToolCallP75": 123,
"baselineDurationP50": 123,
"baselineDurationP75": 123,
"idealExpectedSteps": 123,
"idealExpectedToolCalls": 123
},
"trajectoryId": "<string>",
"behavior": "tool_use",
"taskType": "<string>",
"baselineVersion": "<string>"
}
},
"reviews": [
{
"id": 123,
"batchId": 123,
"sampleId": 123,
"testCaseId": 123,
"reviewerId": "<string>",
"systemVerdict": "pass",
"reviewVerdict": "pass",
"agreedWithSystem": false,
"rating": 123,
"feedback": "<string>",
"labels": {},
"createdAt": "<string>"
}
],
"latestReview": {
"id": 123,
"batchId": 123,
"sampleId": 123,
"testCaseId": 123,
"reviewerId": "<string>",
"systemVerdict": "pass",
"reviewVerdict": "pass",
"agreedWithSystem": false,
"rating": 123,
"feedback": "<string>",
"labels": {},
"createdAt": "<string>"
}
}
],
"stats": {
"totalSamples": 123,
"reviewedSamples": 123,
"totalReviews": 123,
"comparableReviews": 123,
"agreementRate": 123,
"disagreementRate": 123,
"uncertainRate": 123,
"verdictDistribution": {
"pass": 123,
"fail": 123,
"uncertain": 123
}
},
"history": [
{
"batchId": 123,
"createdAt": "<string>",
"totalReviews": 123,
"comparableReviews": 123,
"disagreementRate": 123
}
]
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}