curl --request GET \
--url https://api.galtea.ai/evaluations \
--header 'Authorization: Bearer <token>'import requests
url = "https://api.galtea.ai/evaluations"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://api.galtea.ai/evaluations', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.galtea.ai/evaluations",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.galtea.ai/evaluations"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.galtea.ai/evaluations")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.galtea.ai/evaluations")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body[
{
"id": "eval_123",
"metricId": "metric_123",
"sessionId": "session_123",
"productId": "product_123",
"userId": "user_123",
"status": "SUCCESS",
"inferenceResultId": "ir_123",
"score": 0.95,
"reason": "High quality response",
"error": "<string>",
"canRetry": false,
"creditsUsed": 1,
"judgeInputTokens": 4100,
"judgeOutputTokens": 16800,
"judgeReasoningTokens": 12400,
"conversationSimulatorVersion": "1.0.0",
"humanEvaluatorId": "<string>",
"humanEvaluatorStartedAt": "2023-11-07T05:31:56Z",
"humanScore": 123,
"humanReason": "<string>",
"humanEvaluatorFinishedAt": "2023-11-07T05:31:56Z",
"failedTurns": [
"<string>"
],
"monitorId": "<string>",
"monitorSettledAtTurn": 123,
"monitorSamplingPercentage": 123,
"runId": "<string>",
"deletedAt": "2023-11-07T05:31:56Z",
"evaluatedAt": "2023-11-07T05:31:56Z",
"metricLegacyAt": "2023-11-07T05:31:56Z",
"metricDisabledAt": "2023-11-07T05:31:56Z",
"metricExcludedFromAnalyticsAt": "2023-11-07T05:31:56Z",
"metricExcludedByUserId": "<string>",
"testCaseLegacyAt": "2023-11-07T05:31:56Z",
"createdAt": "2023-11-07T05:31:56Z"
}
]{
"error": "Error type",
"message": "Error message description"
}{
"error": "Error type",
"message": "Error message description"
}Get evaluations
Get list of evaluations with pagination and filtering. See Evaluations.
curl --request GET \
--url https://api.galtea.ai/evaluations \
--header 'Authorization: Bearer <token>'import requests
url = "https://api.galtea.ai/evaluations"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://api.galtea.ai/evaluations', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.galtea.ai/evaluations",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.galtea.ai/evaluations"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.galtea.ai/evaluations")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.galtea.ai/evaluations")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body[
{
"id": "eval_123",
"metricId": "metric_123",
"sessionId": "session_123",
"productId": "product_123",
"userId": "user_123",
"status": "SUCCESS",
"inferenceResultId": "ir_123",
"score": 0.95,
"reason": "High quality response",
"error": "<string>",
"canRetry": false,
"creditsUsed": 1,
"judgeInputTokens": 4100,
"judgeOutputTokens": 16800,
"judgeReasoningTokens": 12400,
"conversationSimulatorVersion": "1.0.0",
"humanEvaluatorId": "<string>",
"humanEvaluatorStartedAt": "2023-11-07T05:31:56Z",
"humanScore": 123,
"humanReason": "<string>",
"humanEvaluatorFinishedAt": "2023-11-07T05:31:56Z",
"failedTurns": [
"<string>"
],
"monitorId": "<string>",
"monitorSettledAtTurn": 123,
"monitorSamplingPercentage": 123,
"runId": "<string>",
"deletedAt": "2023-11-07T05:31:56Z",
"evaluatedAt": "2023-11-07T05:31:56Z",
"metricLegacyAt": "2023-11-07T05:31:56Z",
"metricDisabledAt": "2023-11-07T05:31:56Z",
"metricExcludedFromAnalyticsAt": "2023-11-07T05:31:56Z",
"metricExcludedByUserId": "<string>",
"testCaseLegacyAt": "2023-11-07T05:31:56Z",
"createdAt": "2023-11-07T05:31:56Z"
}
]{
"error": "Error type",
"message": "Error message description"
}{
"error": "Error type",
"message": "Error message description"
}Authorizations
API key authorization. Pass your API key in the Authorization header as a Bearer token. Both new (gsk_*) and legacy (gsk-) API keys are accepted, e.g. Authorization: Bearer gsk_... or Authorization: Bearer gsk-....
Query Parameters
Filter by evaluation IDs
Filter by product IDs
Filter by session IDs
Filter by trace IDs (for single-turn evaluations)
Filter by metric IDs
Filter by metric group IDs (matches all revisions of a metric)
Filter by monitor IDs (returns only evaluations dispatched by the given monitors)
Filter by run IDs (returns only evaluations launched by the given runs)
Filter by monitor alert IDs (returns only the evaluations of the sessions these alerts linked as bad when they opened, made by the alert's monitor, on the alert's metric family). At most 10 IDs.
10Filter by test case IDs
Filter by test IDs (include only). Use an empty string ("") to select evaluations without a test
Omit evaluations linked to the specified test IDs. Use an empty string ("") to omit evaluations without a test
Filter by version IDs
Filter by specification IDs (returns evaluations whose metric is linked to any of the given specifications)
Filter by evaluation statuses
PENDING, PENDING_HUMAN, SUCCESS, FAILED, SKIPPED, CANCELLED, OUTDATED Filter by evaluation types
Filter by metric sources
Sort instructions (field and direction pairs)
Filter evaluations that can be retried
Filter evaluations by whether their test case is augmented
Filter evaluations by the joined Session.isProduction flag. Prefer this over the legacy testIds/excludeTestIds empty-string trick.
Include evaluations whose metric revision is excluded from analytics. Omit or pass false to hide them.
Filter by human evaluator user ID
Filter by the launching user's id
Filter evaluations created at or after this timestamp (ISO 8601 format)
Filter evaluations created at or before this timestamp (ISO 8601 format)
Keep evaluations whose judge score is greater than this value (0 to 1). Reads the judge's score, not the human review. An evaluation with no judge score never matches. A value outside 0 to 1 returns 400.
Keep evaluations whose judge score is greater than or equal to this value (0 to 1). Reads the judge's score, not the human review. An evaluation with no judge score never matches. A value outside 0 to 1 returns 400.
Keep evaluations whose judge score is less than this value (0 to 1). Reads the judge's score, not the human review. An evaluation with no judge score never matches. A value outside 0 to 1 returns 400.
Keep evaluations whose judge score is less than or equal to this value (0 to 1). Reads the judge's score, not the human review. An evaluation with no judge score never matches. A value outside 0 to 1 returns 400.
Keep evaluations whose human review score is greater than this value (0 to 1). An evaluation with no human score never matches. A value outside 0 to 1 returns 400.
Keep evaluations whose human review score is greater than or equal to this value (0 to 1). An evaluation with no human score never matches. A value outside 0 to 1 returns 400.
Keep evaluations whose human review score is less than this value (0 to 1). An evaluation with no human score never matches. A value outside 0 to 1 returns 400.
Keep evaluations whose human review score is less than or equal to this value (0 to 1). An evaluation with no human score never matches. A value outside 0 to 1 returns 400.
Maximum number of results
0 <= x <= 9007199254740991Number of results to skip
x >= 0Response
Evaluations retrieved successfully
"eval_123"
"metric_123"
"session_123"
Product the evaluated session belongs to
"product_123"
"user_123"
PENDING, PENDING_HUMAN, SUCCESS, FAILED, SKIPPED, CANCELLED, OUTDATED "SUCCESS"
"ir_123"
0.95
"High quality response"
false
1
Tokens the judge received, prompt plus session content
4100
Tokens the judge produced, reasoning included
16800
Reasoning tokens, a subset of judgeOutputTokens
12400
"1.0.0"
User ID of the human evaluator
Human-provided annotation score
Human-provided annotation reason
Timestamp when human evaluation was submitted
Conversation turns that failed
ID of the monitor that dispatched this evaluation
Session turn count when the monitor scored it
Sampling percentage in effect when the monitor scored this session
ID of the run that launched this evaluation