Compare test runs
curl --request GET \
--url https://guidinghand.ai/v1/test_runs/compare \
--header 'Authorization: Bearer <token>'const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://guidinghand.ai/v1/test_runs/compare', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));import requests
url = "https://guidinghand.ai/v1/test_runs/compare"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text){
"object": "test_run_comparison",
"baseline": "<string>",
"runs": [
{
"test_run_id": "<string>",
"name": "<string>",
"test_set_id": "<string>",
"status": "queued",
"agents": [
"<string>"
],
"created_at": "2023-11-07T05:31:56Z",
"totals": {
"tries": 123,
"attempts": 123,
"passed": 123,
"failed": 123,
"error": 123,
"pass_rate": 123,
"active_seconds": 123,
"price_cents": 123,
"model_cost_cents": 123,
"started_at": "2023-11-07T05:31:56Z",
"finished_at": "2023-11-07T05:31:56Z",
"wall_seconds": 123
},
"by_agent": {}
}
],
"changes": {},
"cases": [
{
"test_case_id": "<string>",
"key": "<string>",
"name": "<string>",
"category": "<string>",
"tags": [
"<string>"
],
"results": {},
"changes": {}
}
]
}{
"error": {
"type": "invalid_request",
"message": "<string>",
"code": "<string>",
"answered_by": "customer",
"test_case_ids": [
"<string>"
]
}
}{
"error": {
"type": "invalid_request",
"message": "<string>",
"code": "<string>",
"answered_by": "customer",
"test_case_ids": [
"<string>"
]
}
}{
"error": {
"type": "invalid_request",
"message": "<string>",
"code": "<string>",
"answered_by": "customer",
"test_case_ids": [
"<string>"
]
}
}Test runs
Compare test runs
2 to 10 runs side by side, the first as the baseline: each run’s totals and agents, each case’s results per run and agent, and which (case, agent) pairs improved or regressed. Cases line up by test case and agents by agent id, so runs of a set, reruns and runs with other agents all compare.
GET
/
v1
/
test_runs
/
compare
Compare test runs
curl --request GET \
--url https://guidinghand.ai/v1/test_runs/compare \
--header 'Authorization: Bearer <token>'const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://guidinghand.ai/v1/test_runs/compare', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));import requests
url = "https://guidinghand.ai/v1/test_runs/compare"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text){
"object": "test_run_comparison",
"baseline": "<string>",
"runs": [
{
"test_run_id": "<string>",
"name": "<string>",
"test_set_id": "<string>",
"status": "queued",
"agents": [
"<string>"
],
"created_at": "2023-11-07T05:31:56Z",
"totals": {
"tries": 123,
"attempts": 123,
"passed": 123,
"failed": 123,
"error": 123,
"pass_rate": 123,
"active_seconds": 123,
"price_cents": 123,
"model_cost_cents": 123,
"started_at": "2023-11-07T05:31:56Z",
"finished_at": "2023-11-07T05:31:56Z",
"wall_seconds": 123
},
"by_agent": {}
}
],
"changes": {},
"cases": [
{
"test_case_id": "<string>",
"key": "<string>",
"name": "<string>",
"category": "<string>",
"tags": [
"<string>"
],
"results": {},
"changes": {}
}
]
}{
"error": {
"type": "invalid_request",
"message": "<string>",
"code": "<string>",
"answered_by": "customer",
"test_case_ids": [
"<string>"
]
}
}{
"error": {
"type": "invalid_request",
"message": "<string>",
"code": "<string>",
"answered_by": "customer",
"test_case_ids": [
"<string>"
]
}
}{
"error": {
"type": "invalid_request",
"message": "<string>",
"code": "<string>",
"answered_by": "customer",
"test_case_ids": [
"<string>"
]
}
}Authorizations
An org API key (gh_live_…) from Settings → API keys in the console.
Query Parameters
2 to 10 test run ids, comma-separated. The first is the baseline.
Response
OK
Runs side by side. The first run is the baseline; cases line up by test case and agents by agent id.