ListEvalSuiteLeaderboardRunsData
Example Usage
typescript
import { ListEvalSuiteLeaderboardRunsData } from "@meetkai/mka1/models/operations";
let value: ListEvalSuiteLeaderboardRunsData = {
id: "<id>",
object: "eval.run",
suiteId: "<id>",
suiteVersion: 666,
suiteVersionId: "<id>",
orgId: "<id>",
teamId: "<id>",
status: "completed",
models: [
"<value 1>",
],
taskIds: [
"<value 1>",
"<value 2>",
"<value 3>",
],
judgeModel: "<value>",
embeddingModel: "<value>",
generation: {},
requestCounts: {
total: 843051,
completed: 4960,
failed: 868344,
},
metrics: {
"key": "<value>",
"key1": "<value>",
},
error: null,
artifactFileIds: [
"<value 1>",
],
metadata: {
"key": "<value>",
},
agentName: "<value>",
agentVersion: "<value>",
agentEffort: "<value>",
costUsd: 2127.7,
createdAt: 73112,
startedAt: 456398,
completedAt: 38270,
cancelledAt: 40283,
failedAt: null,
};Fields
| Field | Type | Required | Description |
|---|---|---|---|
id | string | ✔️ | N/A |
object | "eval.run" | ✔️ | N/A |
suiteId | string | ✔️ | N/A |
suiteVersion | number | ✔️ | N/A |
suiteVersionId | string | ✔️ | N/A |
orgId | string | ✔️ | The org that owns this run. |
teamId | string | ✔️ | The team that owns this run. |
status | components.EvalRunStatus | ✔️ | N/A |
models | string[] | ✔️ | N/A |
taskIds | string[] | ✔️ | N/A |
judgeModel | string | ✔️ | N/A |
embeddingModel | string | ✔️ | N/A |
generation | operations.ListEvalSuiteLeaderboardRunsGeneration | ✔️ | N/A |
requestCounts | operations.ListEvalSuiteLeaderboardRunsRequestCounts | ✔️ | N/A |
metrics | Record<string, any> | ✔️ | N/A |
error | Record<string, any> | ✔️ | N/A |
artifactFileIds | string[] | ✔️ | N/A |
metadata | Record<string, string> | ✔️ | N/A |
agentName | string | ✔️ | Coding-agent harness that produced this run (e.g. omp, claude-code). Null for every non-harness eval kind — a null here is not a missing value, it means the run is not an agent run. |
agentVersion | string | ✔️ | Version of agent_name, as the harness reported it. Null when agent_name is. |
agentEffort | string | ✔️ | Reasoning-effort setting the agent ran at (e.g. medium, xhigh). Part of the leaderboard's row identity: the same agent and model at two efforts are two rows, not one. Null when agent_name is. |
costUsd | number | ✔️ | Total spend for the run in USD. Null when the harness did not report cost. |
createdAt | number | ✔️ | N/A |
startedAt | number | ✔️ | N/A |
completedAt | number | ✔️ | N/A |
cancelledAt | number | ✔️ | N/A |
failedAt | number | ✔️ | N/A |