evaluations
List evaluations evaluations
Requires scopes: eval:read
GET
/
api
/
evaluations
List evaluations evaluations
curl --request GET \
--url https://evalgate.com/api/evaluations \
--header 'Authorization: Bearer <token>'import requests
url = "https://evalgate.com/api/evaluations"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://evalgate.com/api/evaluations', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://evalgate.com/api/evaluations",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://evalgate.com/api/evaluations"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://evalgate.com/api/evaluations")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://evalgate.com/api/evaluations")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body{
"testCases": [
{
"id": 123,
"evaluationId": 123,
"name": "<string>",
"input": "<string>",
"inputHash": "<string>",
"expectedOutput": "<string>",
"metadata": {
"tags": [
"<string>"
],
"category": "policy",
"difficulty": "medium",
"source": "manual",
"behavior": "tool_use",
"failureClusterId": "<string>",
"agentWorkflow": {
"version": 1,
"workflowClass": "approval_gated_mutation",
"autonomyLevel": "manual",
"approvalMode": "human_review",
"riskLevel": "low",
"memoryMode": "session",
"backgroundExecution": true,
"requiresCheckpoint": true,
"expectedHandoff": true,
"requiredHooks": [
"session_start"
],
"activeTools": [
"<string>"
],
"permissionScope": {
"filesystem": [
"<string>"
],
"network": "open",
"externalSystems": [
"<string>"
],
"mutatesState": false
},
"expectedFailureModes": [
"<string>"
],
"scenario": "<string>",
"provenance": {
"source": "manual",
"evaluationId": 123,
"runId": 123,
"testCaseId": 123,
"traceId": "<string>"
}
},
"workflowExperimentRecipe": {
"version": 1,
"title": "<string>",
"objective": "<string>",
"workflowClass": "approval_gated_mutation",
"source": "coverage_gap",
"linkedFailureModes": [
"<string>"
],
"linkedTestCaseIds": [
123
],
"comparedVariants": [
{
"id": "<string>",
"label": "<string>",
"changes": [
"<string>"
],
"expectedImpact": "<string>",
"autonomyLevel": "manual",
"approvalMode": "human_review",
"memoryMode": "session",
"hooks": [
"session_start"
]
}
],
"successMetrics": [
"<string>"
],
"guardrails": [
"<string>"
],
"recommendedIterations": 123,
"rationale": "<string>"
},
"candidateId": "<string>",
"failureReportId": 123,
"evalCaseId": "<string>"
},
"lifecycleState": "<string>",
"originType": "<string>",
"datasetId": "<string>",
"datasetVersionId": "<string>",
"datasetRowId": "<string>",
"datasetRowContentHash": "<string>",
"lastLifecycleTransitionAt": "2023-11-07T05:31:56Z",
"archivedAt": "2023-11-07T05:31:56Z",
"supersededById": 123,
"supersededAt": "2023-11-07T05:31:56Z",
"supersessionReason": "<string>",
"createdAt": "2023-11-07T05:31:56Z"
}
],
"runs": [
{
"id": 123,
"evaluationId": 123,
"organizationId": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"idempotencyKey": "<string>",
"status": "<string>",
"totalCases": 123,
"passedCases": 123,
"failedCases": 123,
"processedCount": 123,
"traceLog": {
"messages": [
{
"role": "user",
"content": "<string>",
"timestamp": "<string>"
}
],
"originalEvaluationId": 123,
"shadowEvalType": "<string>",
"resolvedTraceIds": [
"<string>"
],
"jobId": 123,
"requestedAt": "<string>",
"enqueuedAt": "<string>",
"enqueueError": "<string>",
"lastError": "<string>",
"completedAt": "<string>",
"averageScore": 123,
"trajectoryCoverageRate": 123,
"trajectoryAverageScore": 123,
"trajectoryFailureCounts": {},
"error": "<string>",
"failedAt": "<string>",
"totalCases": 123,
"processedCount": 123,
"cancelledAt": "<string>",
"traceIds": [
"<string>"
],
"import": {
"sourceTraceId": "<string>",
"sourceOrgId": "<string>",
"importedAt": "<string>",
"source": "<string>",
"checkReport": {},
"gateCheckedAt": "<string>"
},
"type": "<string>",
"scoreImprovement": 123,
"dateRange": {
"start": "<string>",
"end": "<string>"
},
"filters": {},
"createdBy": "<string>"
},
"trajectory": {
"schemaVersion": 123,
"entries": [
{
"testCaseId": 123,
"status": "failed",
"trajectory": {
"id": "<string>",
"input": "<string>",
"finalOutput": "<string>",
"steps": [
{
"index": 123,
"type": "final",
"content": "<string>",
"toolName": "<string>",
"toolArgs": {},
"timestamp": 123
}
],
"totalSteps": 123,
"totalToolCalls": 123,
"durationMs": 123,
"terminatedEarly": false,
"error": "<string>"
},
"behavior": "tool_use",
"taskType": "<string>",
"idealTrajectory": {
"expectedSteps": 123,
"expectedToolCalls": 123,
"requiredTools": [
"<string>"
],
"forbiddenTools": [
"<string>"
],
"allowParallel": false,
"maxRedundantSteps": 123
},
"baseline": {
"behavior": "tool_use",
"stepDistribution": {
"p50": 123,
"p75": 123,
"p90": 123
},
"toolCallDistribution": {
"p50": 123,
"p75": 123
},
"durationDistribution": {
"p50": 123,
"p75": 123
},
"commonToolSequences": [
[
"<string>"
]
],
"redundancyProfile": {
"mean": 123,
"p90": 123
},
"sampleSize": 123,
"updatedAt": "<string>",
"taskType": "<string>",
"parallelToolGroups": [
[
"<string>"
]
],
"version": "<string>",
"model": "<string>"
},
"evaluation": {
"schemaVersion": 123,
"valid": true,
"score": 123,
"failureModes": [
"inefficient_path"
],
"metrics": {
"actualSteps": 123,
"actualToolCalls": 123,
"actualDurationMs": 123,
"redundancyRatio": 123,
"redundancySteps": 123,
"sequenceSimilarity": 123,
"efficiency": 123,
"toolUsage": 123,
"redundancy": 123,
"latency": 123,
"parallelizationPenalty": 123,
"parallelizationScore": 123,
"requiredToolsMissing": [
"<string>"
],
"forbiddenToolsUsed": [
"<string>"
],
"extraToolsUsed": [
"<string>"
],
"selectedBaselineVersion": "<string>",
"baselineStepP50": 123,
"baselineStepP75": 123,
"baselineToolCallP50": 123,
"baselineToolCallP75": 123,
"baselineDurationP50": 123,
"baselineDurationP75": 123,
"idealExpectedSteps": 123,
"idealExpectedToolCalls": 123
},
"trajectoryId": "<string>",
"behavior": "tool_use",
"taskType": "<string>",
"baselineVersion": "<string>"
}
}
]
},
"trajectoryMetrics": {
"schemaVersion": 123,
"totalResults": 123,
"evaluatedResults": 123,
"coverageRate": 123,
"averageTrajectoryScore": 123,
"failureCounts": {
"inefficient_path": 123,
"unnecessary_tool_call": 123,
"missed_tool_call": 123,
"looping_behavior": 123,
"non_parallel_execution": 123,
"premature_termination": 123
},
"byTestCase": [
{
"testCaseId": 123,
"status": "failed",
"score": 123,
"failureModes": [
"inefficient_path"
],
"metrics": {
"actualSteps": 123,
"actualToolCalls": 123,
"actualDurationMs": 123,
"redundancyRatio": 123,
"redundancySteps": 123,
"sequenceSimilarity": 123,
"efficiency": 123,
"toolUsage": 123,
"redundancy": 123,
"latency": 123,
"parallelizationPenalty": 123,
"parallelizationScore": 123,
"requiredToolsMissing": [
"<string>"
],
"forbiddenToolsUsed": [
"<string>"
],
"extraToolsUsed": [
"<string>"
],
"selectedBaselineVersion": "<string>",
"baselineStepP50": 123,
"baselineStepP75": 123,
"baselineToolCallP50": 123,
"baselineToolCallP75": 123,
"baselineDurationP50": 123,
"baselineDurationP75": 123,
"idealExpectedSteps": 123,
"idealExpectedToolCalls": 123
},
"trajectoryId": "<string>",
"behavior": "tool_use",
"taskType": "<string>",
"baselineVersion": "<string>"
}
]
},
"baselineVersion": "<string>",
"scoreProvenance": {
"scoringVersion": "<string>",
"assertionSetVersion": "<string>",
"taskType": "<string>",
"judgeConfigHash": "<string>",
"promptTemplateHash": "<string>",
"aggregationVersion": "<string>",
"embeddingModel": "<string>",
"behaviorVersion": "<string>",
"calibrationStudyId": "<string>"
},
"measurementStatus": "<string>",
"measurementSummary": {
"version": "evalgate.measurement.v1",
"status": "valid",
"sampleSize": 123,
"metrics": {},
"warnings": [
"<string>"
],
"computationHash": "<string>",
"uncertainty": {
"lower": 123,
"upper": 123,
"confidenceLevel": 123,
"method": "percentile-bootstrap",
"resamples": 123,
"seed": 123
}
},
"measurementCompletedAt": "2023-11-07T05:31:56Z",
"activeMeasurementRevisionId": "<string>",
"qualitySummary": {
"version": "evalgate.measurement.v1",
"status": "valid",
"sampleSize": 123,
"metrics": {},
"warnings": [
"<string>"
],
"computationHash": "<string>",
"uncertainty": {
"lower": 123,
"upper": 123,
"confidenceLevel": 123,
"method": "percentile-bootstrap",
"resamples": 123,
"seed": 123
}
},
"qualitySummaryCompletedAt": "2023-11-07T05:31:56Z",
"orchestrationGraph": {
"schemaVersion": 1,
"rootNodeId": "<string>",
"nodes": [
{
"id": "<string>",
"type": "rejected",
"label": "<string>",
"agentIndex": 123,
"strategyTrack": "<string>",
"passRate": 123,
"kept": false,
"metadata": {}
}
],
"edges": [
{
"from": "<string>",
"to": "<string>",
"label": "<string>"
}
],
"capturedAt": "<string>",
"gateDecisions": [
{
"timestamp": "<string>",
"passed": true,
"exitCode": 123,
"reasonCode": "<string>",
"thresholds": {
"minScore": 123,
"maxDrop": 123,
"warnDrop": 123
},
"reasonMessage": "<string>",
"policy": "<string>",
"score": 123,
"baselineScore": 123,
"regressionDelta": 123,
"baselineRunId": 123,
"ciRunUrl": "<string>",
"failedTestCaseIds": [
123
]
}
]
},
"startedAt": "2023-11-07T05:31:56Z",
"completedAt": "2023-11-07T05:31:56Z",
"environment": "<string>",
"playgroundId": 123,
"variantId": 123,
"promptVersionId": "<string>",
"baselineVariantId": 123,
"triggerSource": "<string>",
"createdAt": "2023-11-07T05:31:56Z"
}
],
"id": 123,
"name": "<string>",
"description": "<string>",
"type": "<string>",
"status": "<string>",
"organizationId": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"createdBy": "<string>",
"executionSettings": {
"maxRetries": 123,
"timeout": 123,
"parallel": false,
"batchSize": 123,
"measurementPolicy": {
"enabled": true,
"evaluatorReleaseId": "<string>",
"validationType": "full",
"requestedMetrics": [
"<string>"
],
"requiredEnvironments": [
"dev"
],
"gate": {
"mode": "block",
"allowedMeasurementStatuses": [
"<string>"
],
"requireConclusiveInterval": false,
"minimumSampleSize": 123,
"maximumEce": 123,
"minimumAgreement": 123,
"maximumPositionFlipRate": 123,
"requireCoverage": false
}
}
},
"modelSettings": {
"model": "<string>",
"systemPrompt": "<string>",
"temperature": 123,
"maxTokens": 123,
"topP": 123,
"provider": "<string>",
"calibrationThreshold": 123,
"judgeStrictness": 123,
"customMetrics": [
{
"name": "<string>",
"formula": "<string>",
"weight": 123,
"threshold": 123
}
],
"calibrationMeta": {}
},
"customMetrics": [
{
"name": "<string>",
"formula": "<string>",
"weight": 123,
"threshold": 123
}
],
"executorType": "<string>",
"executorConfig": {
"type": "<string>",
"endpoint": "<string>",
"headers": {}
},
"publishedRunId": 123,
"publishedVersion": 123,
"projectKey": "<string>",
"lockedBySessionId": "<string>",
"lockAcquiredAt": "2023-11-07T05:31:56Z",
"createdAt": "2023-11-07T05:31:56Z",
"updatedAt": "2023-11-07T05:31:56Z"
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}Authorizations
Bearer authentication header of the form Bearer <token>, where <token> is your auth token.
Query Parameters
Required range:
1 <= x <= 1000Required range:
x >= 0Response
Successful response
- Option 1
- Option 2
Show child attributes
Show child attributes
Show child attributes
Show child attributes
Show child attributes
Show child attributes
Show child attributes
Show child attributes
Show child attributes
Show child attributes
Show child attributes
Show child attributes
⌘I
List evaluations evaluations
curl --request GET \
--url https://evalgate.com/api/evaluations \
--header 'Authorization: Bearer <token>'import requests
url = "https://evalgate.com/api/evaluations"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://evalgate.com/api/evaluations', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://evalgate.com/api/evaluations",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://evalgate.com/api/evaluations"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://evalgate.com/api/evaluations")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://evalgate.com/api/evaluations")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body{
"testCases": [
{
"id": 123,
"evaluationId": 123,
"name": "<string>",
"input": "<string>",
"inputHash": "<string>",
"expectedOutput": "<string>",
"metadata": {
"tags": [
"<string>"
],
"category": "policy",
"difficulty": "medium",
"source": "manual",
"behavior": "tool_use",
"failureClusterId": "<string>",
"agentWorkflow": {
"version": 1,
"workflowClass": "approval_gated_mutation",
"autonomyLevel": "manual",
"approvalMode": "human_review",
"riskLevel": "low",
"memoryMode": "session",
"backgroundExecution": true,
"requiresCheckpoint": true,
"expectedHandoff": true,
"requiredHooks": [
"session_start"
],
"activeTools": [
"<string>"
],
"permissionScope": {
"filesystem": [
"<string>"
],
"network": "open",
"externalSystems": [
"<string>"
],
"mutatesState": false
},
"expectedFailureModes": [
"<string>"
],
"scenario": "<string>",
"provenance": {
"source": "manual",
"evaluationId": 123,
"runId": 123,
"testCaseId": 123,
"traceId": "<string>"
}
},
"workflowExperimentRecipe": {
"version": 1,
"title": "<string>",
"objective": "<string>",
"workflowClass": "approval_gated_mutation",
"source": "coverage_gap",
"linkedFailureModes": [
"<string>"
],
"linkedTestCaseIds": [
123
],
"comparedVariants": [
{
"id": "<string>",
"label": "<string>",
"changes": [
"<string>"
],
"expectedImpact": "<string>",
"autonomyLevel": "manual",
"approvalMode": "human_review",
"memoryMode": "session",
"hooks": [
"session_start"
]
}
],
"successMetrics": [
"<string>"
],
"guardrails": [
"<string>"
],
"recommendedIterations": 123,
"rationale": "<string>"
},
"candidateId": "<string>",
"failureReportId": 123,
"evalCaseId": "<string>"
},
"lifecycleState": "<string>",
"originType": "<string>",
"datasetId": "<string>",
"datasetVersionId": "<string>",
"datasetRowId": "<string>",
"datasetRowContentHash": "<string>",
"lastLifecycleTransitionAt": "2023-11-07T05:31:56Z",
"archivedAt": "2023-11-07T05:31:56Z",
"supersededById": 123,
"supersededAt": "2023-11-07T05:31:56Z",
"supersessionReason": "<string>",
"createdAt": "2023-11-07T05:31:56Z"
}
],
"runs": [
{
"id": 123,
"evaluationId": 123,
"organizationId": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"idempotencyKey": "<string>",
"status": "<string>",
"totalCases": 123,
"passedCases": 123,
"failedCases": 123,
"processedCount": 123,
"traceLog": {
"messages": [
{
"role": "user",
"content": "<string>",
"timestamp": "<string>"
}
],
"originalEvaluationId": 123,
"shadowEvalType": "<string>",
"resolvedTraceIds": [
"<string>"
],
"jobId": 123,
"requestedAt": "<string>",
"enqueuedAt": "<string>",
"enqueueError": "<string>",
"lastError": "<string>",
"completedAt": "<string>",
"averageScore": 123,
"trajectoryCoverageRate": 123,
"trajectoryAverageScore": 123,
"trajectoryFailureCounts": {},
"error": "<string>",
"failedAt": "<string>",
"totalCases": 123,
"processedCount": 123,
"cancelledAt": "<string>",
"traceIds": [
"<string>"
],
"import": {
"sourceTraceId": "<string>",
"sourceOrgId": "<string>",
"importedAt": "<string>",
"source": "<string>",
"checkReport": {},
"gateCheckedAt": "<string>"
},
"type": "<string>",
"scoreImprovement": 123,
"dateRange": {
"start": "<string>",
"end": "<string>"
},
"filters": {},
"createdBy": "<string>"
},
"trajectory": {
"schemaVersion": 123,
"entries": [
{
"testCaseId": 123,
"status": "failed",
"trajectory": {
"id": "<string>",
"input": "<string>",
"finalOutput": "<string>",
"steps": [
{
"index": 123,
"type": "final",
"content": "<string>",
"toolName": "<string>",
"toolArgs": {},
"timestamp": 123
}
],
"totalSteps": 123,
"totalToolCalls": 123,
"durationMs": 123,
"terminatedEarly": false,
"error": "<string>"
},
"behavior": "tool_use",
"taskType": "<string>",
"idealTrajectory": {
"expectedSteps": 123,
"expectedToolCalls": 123,
"requiredTools": [
"<string>"
],
"forbiddenTools": [
"<string>"
],
"allowParallel": false,
"maxRedundantSteps": 123
},
"baseline": {
"behavior": "tool_use",
"stepDistribution": {
"p50": 123,
"p75": 123,
"p90": 123
},
"toolCallDistribution": {
"p50": 123,
"p75": 123
},
"durationDistribution": {
"p50": 123,
"p75": 123
},
"commonToolSequences": [
[
"<string>"
]
],
"redundancyProfile": {
"mean": 123,
"p90": 123
},
"sampleSize": 123,
"updatedAt": "<string>",
"taskType": "<string>",
"parallelToolGroups": [
[
"<string>"
]
],
"version": "<string>",
"model": "<string>"
},
"evaluation": {
"schemaVersion": 123,
"valid": true,
"score": 123,
"failureModes": [
"inefficient_path"
],
"metrics": {
"actualSteps": 123,
"actualToolCalls": 123,
"actualDurationMs": 123,
"redundancyRatio": 123,
"redundancySteps": 123,
"sequenceSimilarity": 123,
"efficiency": 123,
"toolUsage": 123,
"redundancy": 123,
"latency": 123,
"parallelizationPenalty": 123,
"parallelizationScore": 123,
"requiredToolsMissing": [
"<string>"
],
"forbiddenToolsUsed": [
"<string>"
],
"extraToolsUsed": [
"<string>"
],
"selectedBaselineVersion": "<string>",
"baselineStepP50": 123,
"baselineStepP75": 123,
"baselineToolCallP50": 123,
"baselineToolCallP75": 123,
"baselineDurationP50": 123,
"baselineDurationP75": 123,
"idealExpectedSteps": 123,
"idealExpectedToolCalls": 123
},
"trajectoryId": "<string>",
"behavior": "tool_use",
"taskType": "<string>",
"baselineVersion": "<string>"
}
}
]
},
"trajectoryMetrics": {
"schemaVersion": 123,
"totalResults": 123,
"evaluatedResults": 123,
"coverageRate": 123,
"averageTrajectoryScore": 123,
"failureCounts": {
"inefficient_path": 123,
"unnecessary_tool_call": 123,
"missed_tool_call": 123,
"looping_behavior": 123,
"non_parallel_execution": 123,
"premature_termination": 123
},
"byTestCase": [
{
"testCaseId": 123,
"status": "failed",
"score": 123,
"failureModes": [
"inefficient_path"
],
"metrics": {
"actualSteps": 123,
"actualToolCalls": 123,
"actualDurationMs": 123,
"redundancyRatio": 123,
"redundancySteps": 123,
"sequenceSimilarity": 123,
"efficiency": 123,
"toolUsage": 123,
"redundancy": 123,
"latency": 123,
"parallelizationPenalty": 123,
"parallelizationScore": 123,
"requiredToolsMissing": [
"<string>"
],
"forbiddenToolsUsed": [
"<string>"
],
"extraToolsUsed": [
"<string>"
],
"selectedBaselineVersion": "<string>",
"baselineStepP50": 123,
"baselineStepP75": 123,
"baselineToolCallP50": 123,
"baselineToolCallP75": 123,
"baselineDurationP50": 123,
"baselineDurationP75": 123,
"idealExpectedSteps": 123,
"idealExpectedToolCalls": 123
},
"trajectoryId": "<string>",
"behavior": "tool_use",
"taskType": "<string>",
"baselineVersion": "<string>"
}
]
},
"baselineVersion": "<string>",
"scoreProvenance": {
"scoringVersion": "<string>",
"assertionSetVersion": "<string>",
"taskType": "<string>",
"judgeConfigHash": "<string>",
"promptTemplateHash": "<string>",
"aggregationVersion": "<string>",
"embeddingModel": "<string>",
"behaviorVersion": "<string>",
"calibrationStudyId": "<string>"
},
"measurementStatus": "<string>",
"measurementSummary": {
"version": "evalgate.measurement.v1",
"status": "valid",
"sampleSize": 123,
"metrics": {},
"warnings": [
"<string>"
],
"computationHash": "<string>",
"uncertainty": {
"lower": 123,
"upper": 123,
"confidenceLevel": 123,
"method": "percentile-bootstrap",
"resamples": 123,
"seed": 123
}
},
"measurementCompletedAt": "2023-11-07T05:31:56Z",
"activeMeasurementRevisionId": "<string>",
"qualitySummary": {
"version": "evalgate.measurement.v1",
"status": "valid",
"sampleSize": 123,
"metrics": {},
"warnings": [
"<string>"
],
"computationHash": "<string>",
"uncertainty": {
"lower": 123,
"upper": 123,
"confidenceLevel": 123,
"method": "percentile-bootstrap",
"resamples": 123,
"seed": 123
}
},
"qualitySummaryCompletedAt": "2023-11-07T05:31:56Z",
"orchestrationGraph": {
"schemaVersion": 1,
"rootNodeId": "<string>",
"nodes": [
{
"id": "<string>",
"type": "rejected",
"label": "<string>",
"agentIndex": 123,
"strategyTrack": "<string>",
"passRate": 123,
"kept": false,
"metadata": {}
}
],
"edges": [
{
"from": "<string>",
"to": "<string>",
"label": "<string>"
}
],
"capturedAt": "<string>",
"gateDecisions": [
{
"timestamp": "<string>",
"passed": true,
"exitCode": 123,
"reasonCode": "<string>",
"thresholds": {
"minScore": 123,
"maxDrop": 123,
"warnDrop": 123
},
"reasonMessage": "<string>",
"policy": "<string>",
"score": 123,
"baselineScore": 123,
"regressionDelta": 123,
"baselineRunId": 123,
"ciRunUrl": "<string>",
"failedTestCaseIds": [
123
]
}
]
},
"startedAt": "2023-11-07T05:31:56Z",
"completedAt": "2023-11-07T05:31:56Z",
"environment": "<string>",
"playgroundId": 123,
"variantId": 123,
"promptVersionId": "<string>",
"baselineVariantId": 123,
"triggerSource": "<string>",
"createdAt": "2023-11-07T05:31:56Z"
}
],
"id": 123,
"name": "<string>",
"description": "<string>",
"type": "<string>",
"status": "<string>",
"organizationId": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"createdBy": "<string>",
"executionSettings": {
"maxRetries": 123,
"timeout": 123,
"parallel": false,
"batchSize": 123,
"measurementPolicy": {
"enabled": true,
"evaluatorReleaseId": "<string>",
"validationType": "full",
"requestedMetrics": [
"<string>"
],
"requiredEnvironments": [
"dev"
],
"gate": {
"mode": "block",
"allowedMeasurementStatuses": [
"<string>"
],
"requireConclusiveInterval": false,
"minimumSampleSize": 123,
"maximumEce": 123,
"minimumAgreement": 123,
"maximumPositionFlipRate": 123,
"requireCoverage": false
}
}
},
"modelSettings": {
"model": "<string>",
"systemPrompt": "<string>",
"temperature": 123,
"maxTokens": 123,
"topP": 123,
"provider": "<string>",
"calibrationThreshold": 123,
"judgeStrictness": 123,
"customMetrics": [
{
"name": "<string>",
"formula": "<string>",
"weight": 123,
"threshold": 123
}
],
"calibrationMeta": {}
},
"customMetrics": [
{
"name": "<string>",
"formula": "<string>",
"weight": 123,
"threshold": 123
}
],
"executorType": "<string>",
"executorConfig": {
"type": "<string>",
"endpoint": "<string>",
"headers": {}
},
"publishedRunId": 123,
"publishedVersion": 123,
"projectKey": "<string>",
"lockedBySessionId": "<string>",
"lockAcquiredAt": "2023-11-07T05:31:56Z",
"createdAt": "2023-11-07T05:31:56Z",
"updatedAt": "2023-11-07T05:31:56Z"
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}{
"error": {
"message": "<string>",
"details": "<unknown>",
"requestId": "3c90c3cc-0d44-4b50-8888-8dd25736052a"
}
}