Mettre à jour et réconcilier une évaluation
curl --request PUT \
--url https://distillation.training.wandb.ai/v1/evals/{evalId} \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"name": "<string>",
"spec": {
"type": "h2h_judge",
"judge_model_ref": "<string>",
"judge_prompt": "You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.",
"reference_conversation_count": 0
},
"participants": [
{
"kind": "original"
}
],
"reference": {
"kind": "original"
},
"sample_size": 123
}
'import requests
url = "https://distillation.training.wandb.ai/v1/evals/{evalId}"
payload = {
"name": "<string>",
"spec": {
"type": "h2h_judge",
"judge_model_ref": "<string>",
"judge_prompt": "You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.",
"reference_conversation_count": 0
},
"participants": [{ "kind": "original" }],
"reference": { "kind": "original" },
"sample_size": 123
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.put(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'PUT',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
name: '<string>',
spec: {
type: 'h2h_judge',
judge_model_ref: '<string>',
judge_prompt: 'You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.',
reference_conversation_count: 0
},
participants: [{kind: 'original'}],
reference: {kind: 'original'},
sample_size: 123
})
};
fetch('https://distillation.training.wandb.ai/v1/evals/{evalId}', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://distillation.training.wandb.ai/v1/evals/{evalId}",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "PUT",
CURLOPT_POSTFIELDS => json_encode([
'name' => '<string>',
'spec' => [
'type' => 'h2h_judge',
'judge_model_ref' => '<string>',
'judge_prompt' => 'You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.',
'reference_conversation_count' => 0
],
'participants' => [
[
'kind' => 'original'
]
],
'reference' => [
'kind' => 'original'
],
'sample_size' => 123
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://distillation.training.wandb.ai/v1/evals/{evalId}"
payload := strings.NewReader("{\n \"name\": \"<string>\",\n \"spec\": {\n \"type\": \"h2h_judge\",\n \"judge_model_ref\": \"<string>\",\n \"judge_prompt\": \"You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.\",\n \"reference_conversation_count\": 0\n },\n \"participants\": [\n {\n \"kind\": \"original\"\n }\n ],\n \"reference\": {\n \"kind\": \"original\"\n },\n \"sample_size\": 123\n}")
req, _ := http.NewRequest("PUT", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.put("https://distillation.training.wandb.ai/v1/evals/{evalId}")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"name\": \"<string>\",\n \"spec\": {\n \"type\": \"h2h_judge\",\n \"judge_model_ref\": \"<string>\",\n \"judge_prompt\": \"You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.\",\n \"reference_conversation_count\": 0\n },\n \"participants\": [\n {\n \"kind\": \"original\"\n }\n ],\n \"reference\": {\n \"kind\": \"original\"\n },\n \"sample_size\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://distillation.training.wandb.ai/v1/evals/{evalId}")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Put.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"name\": \"<string>\",\n \"spec\": {\n \"type\": \"h2h_judge\",\n \"judge_model_ref\": \"<string>\",\n \"judge_prompt\": \"You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.\",\n \"reference_conversation_count\": 0\n },\n \"participants\": [\n {\n \"kind\": \"original\"\n }\n ],\n \"reference\": {\n \"kind\": \"original\"\n },\n \"sample_size\": 123\n}"
response = http.request(request)
puts response.read_body{
"id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"dataset_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"name": "<string>",
"spec": {
"type": "h2h_judge",
"judge_model_ref": "openai/gpt-5.6-sol",
"judge_prompt": "<string>"
},
"participants": [
{
"kind": "original"
}
],
"sample_size": 123,
"status": "queued",
"progress": {
"total": 123,
"queued": 123,
"running": 123,
"completed": 123,
"failed": 123
},
"created_at": "2023-11-07T05:31:56Z",
"updated_at": "2023-11-07T05:31:56Z",
"reference": {
"kind": "original"
},
"results": {},
"failure_summary": {}
}{
"error": {
"message": "Task 'missing' not found in entity 'your-team'",
"type": "not_found"
}
}Mettre à jour et réconcilier une évaluation
Conserve les cas terminés qui n’ont pas changé et ne met en file d’attente que le travail ajouté ou invalidé.
PUT
/
evals
/
{evalId}
Mettre à jour et réconcilier une évaluation
curl --request PUT \
--url https://distillation.training.wandb.ai/v1/evals/{evalId} \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"name": "<string>",
"spec": {
"type": "h2h_judge",
"judge_model_ref": "<string>",
"judge_prompt": "You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.",
"reference_conversation_count": 0
},
"participants": [
{
"kind": "original"
}
],
"reference": {
"kind": "original"
},
"sample_size": 123
}
'import requests
url = "https://distillation.training.wandb.ai/v1/evals/{evalId}"
payload = {
"name": "<string>",
"spec": {
"type": "h2h_judge",
"judge_model_ref": "<string>",
"judge_prompt": "You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.",
"reference_conversation_count": 0
},
"participants": [{ "kind": "original" }],
"reference": { "kind": "original" },
"sample_size": 123
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.put(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'PUT',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
name: '<string>',
spec: {
type: 'h2h_judge',
judge_model_ref: '<string>',
judge_prompt: 'You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.',
reference_conversation_count: 0
},
participants: [{kind: 'original'}],
reference: {kind: 'original'},
sample_size: 123
})
};
fetch('https://distillation.training.wandb.ai/v1/evals/{evalId}', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://distillation.training.wandb.ai/v1/evals/{evalId}",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "PUT",
CURLOPT_POSTFIELDS => json_encode([
'name' => '<string>',
'spec' => [
'type' => 'h2h_judge',
'judge_model_ref' => '<string>',
'judge_prompt' => 'You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.',
'reference_conversation_count' => 0
],
'participants' => [
[
'kind' => 'original'
]
],
'reference' => [
'kind' => 'original'
],
'sample_size' => 123
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://distillation.training.wandb.ai/v1/evals/{evalId}"
payload := strings.NewReader("{\n \"name\": \"<string>\",\n \"spec\": {\n \"type\": \"h2h_judge\",\n \"judge_model_ref\": \"<string>\",\n \"judge_prompt\": \"You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.\",\n \"reference_conversation_count\": 0\n },\n \"participants\": [\n {\n \"kind\": \"original\"\n }\n ],\n \"reference\": {\n \"kind\": \"original\"\n },\n \"sample_size\": 123\n}")
req, _ := http.NewRequest("PUT", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.put("https://distillation.training.wandb.ai/v1/evals/{evalId}")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"name\": \"<string>\",\n \"spec\": {\n \"type\": \"h2h_judge\",\n \"judge_model_ref\": \"<string>\",\n \"judge_prompt\": \"You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.\",\n \"reference_conversation_count\": 0\n },\n \"participants\": [\n {\n \"kind\": \"original\"\n }\n ],\n \"reference\": {\n \"kind\": \"original\"\n },\n \"sample_size\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://distillation.training.wandb.ai/v1/evals/{evalId}")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Put.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"name\": \"<string>\",\n \"spec\": {\n \"type\": \"h2h_judge\",\n \"judge_model_ref\": \"<string>\",\n \"judge_prompt\": \"You are an intelligent and fair judge of chatbots. Evaluate the following two responses and choose which one is better. If the outputs are of similar quality, you can mark them as a tie.\",\n \"reference_conversation_count\": 0\n },\n \"participants\": [\n {\n \"kind\": \"original\"\n }\n ],\n \"reference\": {\n \"kind\": \"original\"\n },\n \"sample_size\": 123\n}"
response = http.request(request)
puts response.read_body{
"id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"dataset_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"name": "<string>",
"spec": {
"type": "h2h_judge",
"judge_model_ref": "openai/gpt-5.6-sol",
"judge_prompt": "<string>"
},
"participants": [
{
"kind": "original"
}
],
"sample_size": 123,
"status": "queued",
"progress": {
"total": 123,
"queued": 123,
"running": 123,
"completed": 123,
"failed": 123
},
"created_at": "2023-11-07T05:31:56Z",
"updated_at": "2023-11-07T05:31:56Z",
"reference": {
"kind": "original"
},
"results": {},
"failure_summary": {}
}{
"error": {
"message": "Task 'missing' not found in entity 'your-team'",
"type": "not_found"
}
}Autorisations
Une clé API W&B.
En-têtes
entity personnel ou d’équipe accessible. Omettez ce champ pour utiliser l’entity par défaut de l’utilisateur authentifié.
Paramètres de chemin
Corps
application/json
Un nom d’évaluation court et facile à repérer. Privilégiez 2 à 5 mots et évitez d’inclure le nom du jeu de données, la liste de tous les modèles comparés ou des détails de configuration.
Required string length:
1 - 128- Option 1
- Option 2
- Option 3
Show child attributes
Show child attributes
Minimum array length:
1- Option 1
- Option 2
- Option 3
- Option 4
- Option 5
- Option 6
Show child attributes
Show child attributes
- Option 1
- Option 2
- Option 3
Show child attributes
Show child attributes
Plage requise:
x <= 9007199254740991Réponse
Évaluation mise à jour avec un lot de cas réconcilié.
- Option 1
- Option 2
- Option 3
Show child attributes
Show child attributes
- Option 1
- Option 2
- Option 3
Show child attributes
Show child attributes
Options disponibles:
queued, running, completed, failed, stale Show child attributes
Show child attributes
- Option 1
- Option 2
- Option 3
Show child attributes
Show child attributes
Dernière modification le 30 septembre 2026
Cette page vous a-t-elle été utile ?