curl --request POST \
--url https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"golden_set_ids": [
"vevg_golden_set_001",
"vevg_golden_set_002"
],
"agent_id": "agt_8b21d0",
"trigger_source": "manual",
"stt_provider": "<string>",
"tts_provider": "<string>",
"llm_model": "<string>",
"p95_regression_threshold": 123,
"judge_score_regression_threshold": 123
}
'import requests
url = "https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run"
payload = {
"golden_set_ids": ["vevg_golden_set_001", "vevg_golden_set_002"],
"agent_id": "agt_8b21d0",
"trigger_source": "manual",
"stt_provider": "<string>",
"tts_provider": "<string>",
"llm_model": "<string>",
"p95_regression_threshold": 123,
"judge_score_regression_threshold": 123
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
golden_set_ids: ['vevg_golden_set_001', 'vevg_golden_set_002'],
agent_id: 'agt_8b21d0',
trigger_source: 'manual',
stt_provider: '<string>',
tts_provider: '<string>',
llm_model: '<string>',
p95_regression_threshold: 123,
judge_score_regression_threshold: 123
})
};
fetch('https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'golden_set_ids' => [
'vevg_golden_set_001',
'vevg_golden_set_002'
],
'agent_id' => 'agt_8b21d0',
'trigger_source' => 'manual',
'stt_provider' => '<string>',
'tts_provider' => '<string>',
'llm_model' => '<string>',
'p95_regression_threshold' => 123,
'judge_score_regression_threshold' => 123
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run"
payload := strings.NewReader("{\n \"golden_set_ids\": [\n \"vevg_golden_set_001\",\n \"vevg_golden_set_002\"\n ],\n \"agent_id\": \"agt_8b21d0\",\n \"trigger_source\": \"manual\",\n \"stt_provider\": \"<string>\",\n \"tts_provider\": \"<string>\",\n \"llm_model\": \"<string>\",\n \"p95_regression_threshold\": 123,\n \"judge_score_regression_threshold\": 123\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"golden_set_ids\": [\n \"vevg_golden_set_001\",\n \"vevg_golden_set_002\"\n ],\n \"agent_id\": \"agt_8b21d0\",\n \"trigger_source\": \"manual\",\n \"stt_provider\": \"<string>\",\n \"tts_provider\": \"<string>\",\n \"llm_model\": \"<string>\",\n \"p95_regression_threshold\": 123,\n \"judge_score_regression_threshold\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"golden_set_ids\": [\n \"vevg_golden_set_001\",\n \"vevg_golden_set_002\"\n ],\n \"agent_id\": \"agt_8b21d0\",\n \"trigger_source\": \"manual\",\n \"stt_provider\": \"<string>\",\n \"tts_provider\": \"<string>\",\n \"llm_model\": \"<string>\",\n \"p95_regression_threshold\": 123,\n \"judge_score_regression_threshold\": 123\n}"
response = http.request(request)
puts response.read_body{
"data": {
"suite_run_id": "vevs_a1b2c3",
"sets": [
{
"golden_set_id": "<string>",
"run_id": "<string>",
"cases_total": 123,
"cases_passed": 123,
"cases_failed": 123,
"pass_rate": 123,
"p95_latency_ms": 123,
"baseline_run_id": "<string>",
"p95_regression_pct": 123,
"judge_score_regression_pct": 123,
"has_regression": true,
"error": "<string>"
}
],
"aggregate": {
"sets_total": 123,
"sets_failed_to_run": 123,
"sets_with_regression": 123,
"cases_total": 123,
"cases_passed": 123,
"cases_failed": 123,
"pass_rate": 123,
"worst_p95_regression_pct": 123,
"worst_judge_score_regression_pct": 123,
"has_regression": true,
"is_simulated": true
},
"simulation_notice": "STT and TTS latencies are stubbed (synthetic). Do not use these numbers for STT/TTS provider procurement decisions."
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"status": 123,
"details": {}
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"status": 123,
"details": {}
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"status": 123,
"details": {}
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"status": 123,
"details": {}
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"status": 123,
"details": {}
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}{
"error": {
"code": "RATE_LIMITED",
"message": "<string>",
"status": 429,
"retry_after": 2
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"status": 123,
"details": {}
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}Run a golden-set eval suite and score regressions in one batch
Score an agent against an ordered suite of existing golden sets in a single batch run. Each golden set is replayed through the same LLM-judge + WER runner as a standalone voice evaluation run, then diffed against its own prior run to compute a per-set regression delta. Returns per-set results plus a suite-wide aggregate (pooled pass rate, worst-case P95 and judge-score regression, and count of regressed sets) — the batch view an operator needs before publishing a prompt, model, or knowledge-base change. Accepts 1 to 10 golden set ids (duplicates are deduped); a failure on one set does not abort the others.
curl --request POST \
--url https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"golden_set_ids": [
"vevg_golden_set_001",
"vevg_golden_set_002"
],
"agent_id": "agt_8b21d0",
"trigger_source": "manual",
"stt_provider": "<string>",
"tts_provider": "<string>",
"llm_model": "<string>",
"p95_regression_threshold": 123,
"judge_score_regression_threshold": 123
}
'import requests
url = "https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run"
payload = {
"golden_set_ids": ["vevg_golden_set_001", "vevg_golden_set_002"],
"agent_id": "agt_8b21d0",
"trigger_source": "manual",
"stt_provider": "<string>",
"tts_provider": "<string>",
"llm_model": "<string>",
"p95_regression_threshold": 123,
"judge_score_regression_threshold": 123
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
golden_set_ids: ['vevg_golden_set_001', 'vevg_golden_set_002'],
agent_id: 'agt_8b21d0',
trigger_source: 'manual',
stt_provider: '<string>',
tts_provider: '<string>',
llm_model: '<string>',
p95_regression_threshold: 123,
judge_score_regression_threshold: 123
})
};
fetch('https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'golden_set_ids' => [
'vevg_golden_set_001',
'vevg_golden_set_002'
],
'agent_id' => 'agt_8b21d0',
'trigger_source' => 'manual',
'stt_provider' => '<string>',
'tts_provider' => '<string>',
'llm_model' => '<string>',
'p95_regression_threshold' => 123,
'judge_score_regression_threshold' => 123
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run"
payload := strings.NewReader("{\n \"golden_set_ids\": [\n \"vevg_golden_set_001\",\n \"vevg_golden_set_002\"\n ],\n \"agent_id\": \"agt_8b21d0\",\n \"trigger_source\": \"manual\",\n \"stt_provider\": \"<string>\",\n \"tts_provider\": \"<string>\",\n \"llm_model\": \"<string>\",\n \"p95_regression_threshold\": 123,\n \"judge_score_regression_threshold\": 123\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"golden_set_ids\": [\n \"vevg_golden_set_001\",\n \"vevg_golden_set_002\"\n ],\n \"agent_id\": \"agt_8b21d0\",\n \"trigger_source\": \"manual\",\n \"stt_provider\": \"<string>\",\n \"tts_provider\": \"<string>\",\n \"llm_model\": \"<string>\",\n \"p95_regression_threshold\": 123,\n \"judge_score_regression_threshold\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.orbit.devotel.io/api/v1/agents/voice-eval/suites/run")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"golden_set_ids\": [\n \"vevg_golden_set_001\",\n \"vevg_golden_set_002\"\n ],\n \"agent_id\": \"agt_8b21d0\",\n \"trigger_source\": \"manual\",\n \"stt_provider\": \"<string>\",\n \"tts_provider\": \"<string>\",\n \"llm_model\": \"<string>\",\n \"p95_regression_threshold\": 123,\n \"judge_score_regression_threshold\": 123\n}"
response = http.request(request)
puts response.read_body{
"data": {
"suite_run_id": "vevs_a1b2c3",
"sets": [
{
"golden_set_id": "<string>",
"run_id": "<string>",
"cases_total": 123,
"cases_passed": 123,
"cases_failed": 123,
"pass_rate": 123,
"p95_latency_ms": 123,
"baseline_run_id": "<string>",
"p95_regression_pct": 123,
"judge_score_regression_pct": 123,
"has_regression": true,
"error": "<string>"
}
],
"aggregate": {
"sets_total": 123,
"sets_failed_to_run": 123,
"sets_with_regression": 123,
"cases_total": 123,
"cases_passed": 123,
"cases_failed": 123,
"pass_rate": 123,
"worst_p95_regression_pct": 123,
"worst_judge_score_regression_pct": 123,
"has_regression": true,
"is_simulated": true
},
"simulation_notice": "STT and TTS latencies are stubbed (synthetic). Do not use these numbers for STT/TTS provider procurement decisions."
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"status": 123,
"details": {}
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"status": 123,
"details": {}
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"status": 123,
"details": {}
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"status": 123,
"details": {}
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"status": 123,
"details": {}
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}{
"error": {
"code": "RATE_LIMITED",
"message": "<string>",
"status": 429,
"retry_after": 2
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"status": 123,
"details": {}
},
"meta": {
"request_id": "<string>",
"timestamp": "2023-11-07T05:31:56Z",
"docs_url": "<string>"
}
}Authorizations
Dashboard JWT token from Clerk
Headers
Stripe-style idempotency token. Pass a stable, client-generated value (1-255 chars) to dedupe retries on transient timeouts. The same key+credential+path replays the original response for 24h on 2xx (5min on 4xx, 30s on 5xx). Returns 409 if a concurrent request with the same key is already in flight; replayed responses include the Idempotency-Replay: true response header.
1 - 255Sandbox opt-in for Clerk-session-authenticated requests. Set to true to route the call through the test-mode pipeline: no real provider delivery, no credits deducted, response meta.test_mode: true. Ignored for live API keys (dv_live_sk_*) — server-to-server clients must use a test-prefixed key (dv_test_sk_*) to exercise sandbox. Test-prefixed keys unconditionally enable sandbox regardless of this header.
true, false Body
1 - 10 elements[
"vevg_golden_set_001",
"vevg_golden_set_002"
]
"agt_8b21d0"
manual, pre-deploy-ci, scheduled Fractional P95 latency regression threshold (e.g. 0.2 = 20% slower). Defaults to 0.2.
Fractional pass-rate (judge-score) regression threshold. Defaults to 0.1.