Create an experiment
Runs a pinned dataset snapshot through a target and scores every item with pinned evaluator versions, in the background. Everything that decides what the run means is resolved and pinned here: the snapshot (@latest may cut one), each evaluator version and judge model, and the target (a deployment; a model version, resolved to the fleet adapter’s versioned internal name or the single-model deployment serving exactly those weights; or none, which scores recorded outputs). Poll GET /v1/experiments/, or subscribe to the experiment.completed webhook.
curl --request POST \
--url https://api.veri.studio/v1/experiments \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"name": "<string>",
"dataset": "<string>",
"target": {
"deployment_id": "<string>",
"kind": "deployment",
"model": "<string>"
},
"evaluators": [
"<string>"
],
"bindings": "<unknown>",
"generation": {
"temperature": 123,
"max_tokens": 123
},
"trials": 123,
"baseline_experiment_id": "<string>"
}
'import requests
url = "https://api.veri.studio/v1/experiments"
payload = {
"name": "<string>",
"dataset": "<string>",
"target": {
"deployment_id": "<string>",
"kind": "deployment",
"model": "<string>"
},
"evaluators": ["<string>"],
"bindings": "<unknown>",
"generation": {
"temperature": 123,
"max_tokens": 123
},
"trials": 123,
"baseline_experiment_id": "<string>"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
name: '<string>',
dataset: '<string>',
target: {deployment_id: '<string>', kind: 'deployment', model: '<string>'},
evaluators: ['<string>'],
bindings: '<unknown>',
generation: {temperature: 123, max_tokens: 123},
trials: 123,
baseline_experiment_id: '<string>'
})
};
fetch('https://api.veri.studio/v1/experiments', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));{
"object": "<string>",
"id": "<string>",
"name": "<string>",
"status": "<string>",
"dataset_id": "<string>",
"dataset_snapshot_id": "<string>",
"target": {},
"evaluators": [
{
"evaluator_id": "<string>",
"version": 123,
"name": "<string>",
"scope": "<string>",
"method": "<string>",
"judge_model": "<string>"
}
],
"bindings": {},
"generation": {
"temperature": 123,
"max_tokens": 123
},
"trials": 123,
"items_total": 123,
"items_done": 123,
"created_at": "2023-11-07T05:31:56Z",
"target_deployment_id": "<string>",
"target_model": "<string>",
"target_model_version_id": "<string>",
"target_served_model": "<string>",
"aggregates": {},
"passed": true,
"baseline_experiment_id": "<string>",
"diff": {},
"error": "<string>",
"created_by": "<string>",
"started_at": "2023-11-07T05:31:56Z",
"finished_at": "2023-11-07T05:31:56Z"
}Authorizations
API key with the vk_ prefix. Create one from the dashboard.
Body
1-128 characters.
Dataset id or name, optionally @snap-N or @latest. A stream with
no suffix pins @latest (reusing the newest snapshot when it is at
the head, else cutting one).
{kind: "deployment", deployment_id, model?} |
{kind: "model_version", model_version_id} | {kind: "none"} (score
each row's recorded output).
- Option 1
- Option 2
- Option 3
Show child attributes
Show child attributes
Evaluator ids or names, each optionally @N (default: the latest
version). llm_judge and code evaluators only.
Variable name -> JSONPath, layered over every evaluator's bindings.
{temperature?, max_tokens?} for the target's calls.
Show child attributes
Show child attributes
Runs per row, 1-10 (default 1).
A previous experiment to diff against at completion (comparable only on the same snapshot with identical evaluator pins).
Response
Accepted (status queued); items are materialized
One experiment.
Always "experiment".
exp_...
queued | running | awaiting_local | completed | failed | canceled
The snapshot the run is pinned to (<dataset_id>@snap-N).
The target as requested.
The pinned evaluator versions (+ each judge's served model).
Show child attributes
Show child attributes
Sampling params for the target's calls. Omitted = the server default.
Show child attributes
Show child attributes
Items = rows x trials; done counts finished items (ok or error).
Resolved at create: the deployment called, the request model sent
(an adapter's versioned internal name for a model version on a
fleet), the model version, and the served model's identity.
Per evaluator at completion: n, ok, errors, skipped, pass_rate (all items in the denominator), mean, label_counts, critical_failures.
Every evaluator with a pass rule passed every item with no errors; null until completed, or when no evaluator has a pass rule.
The diff against the baseline, computed at completion.
curl --request POST \
--url https://api.veri.studio/v1/experiments \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"name": "<string>",
"dataset": "<string>",
"target": {
"deployment_id": "<string>",
"kind": "deployment",
"model": "<string>"
},
"evaluators": [
"<string>"
],
"bindings": "<unknown>",
"generation": {
"temperature": 123,
"max_tokens": 123
},
"trials": 123,
"baseline_experiment_id": "<string>"
}
'import requests
url = "https://api.veri.studio/v1/experiments"
payload = {
"name": "<string>",
"dataset": "<string>",
"target": {
"deployment_id": "<string>",
"kind": "deployment",
"model": "<string>"
},
"evaluators": ["<string>"],
"bindings": "<unknown>",
"generation": {
"temperature": 123,
"max_tokens": 123
},
"trials": 123,
"baseline_experiment_id": "<string>"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
name: '<string>',
dataset: '<string>',
target: {deployment_id: '<string>', kind: 'deployment', model: '<string>'},
evaluators: ['<string>'],
bindings: '<unknown>',
generation: {temperature: 123, max_tokens: 123},
trials: 123,
baseline_experiment_id: '<string>'
})
};
fetch('https://api.veri.studio/v1/experiments', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));{
"object": "<string>",
"id": "<string>",
"name": "<string>",
"status": "<string>",
"dataset_id": "<string>",
"dataset_snapshot_id": "<string>",
"target": {},
"evaluators": [
{
"evaluator_id": "<string>",
"version": 123,
"name": "<string>",
"scope": "<string>",
"method": "<string>",
"judge_model": "<string>"
}
],
"bindings": {},
"generation": {
"temperature": 123,
"max_tokens": 123
},
"trials": 123,
"items_total": 123,
"items_done": 123,
"created_at": "2023-11-07T05:31:56Z",
"target_deployment_id": "<string>",
"target_model": "<string>",
"target_model_version_id": "<string>",
"target_served_model": "<string>",
"aggregates": {},
"passed": true,
"baseline_experiment_id": "<string>",
"diff": {},
"error": "<string>",
"created_by": "<string>",
"started_at": "2023-11-07T05:31:56Z",
"finished_at": "2023-11-07T05:31:56Z"
}
