Download OpenAPI specification:
API REST server for evaluation backend orchestration
Create and execute evaluation request using the simplified benchmark schema.
| name required | string The evaluation job name. |
| description | string The evaluation job description. |
| tags | Array of strings The evaluation job tags. |
required | object (ModelRef) The model to evaluate, or the model that was used to generate the pre-recorded data. |
required | Array of objects (EvaluationBenchmarkConfig) The evaluation benchmarks to run. |
object (PassCriteriaWithDefault) The overall pass criteria for the evaluation job. | |
object (ExperimentConfig) The MLFlow experiment configuration. When provided, the evaluation job will be tracked in MLFlow. | |
object (EvaluationExports) Optional exports configuration for the evaluation job. When provided, the evaluation job results will be exported to the specified location. | |
object (BenchmarkHardwareConfig) Optional evaluation-level hardware override for Kubernetes-backed jobs. Applied as a fallback for every benchmark that does not specify its own | |
object (QueueConfig) Deprecated Deprecated. Prefer | |
object Custom request data. This can be used for user specific job data. |
{- "name": "granite-3.1-8b-safety-eval",
- "description": "Safety and reasoning evaluation for Granite 3.1 8B Instruct",
- "tags": [
- "nightly",
- "granite"
], - "model": {
- "name": "granite-3.1-8b-instruct"
}, - "benchmarks": [
- {
- "id": "arc_easy",
- "provider_id": "lm_evaluation_harness",
- "weight": 0.6,
- "primary_score": {
- "metric": "acc_norm",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.25
}, - "parameters": {
- "num_fewshot": 0,
- "limit": 100
}
}, - {
- "id": "owasp_llm_top10",
- "provider_id": "garak",
- "weight": 0.4,
- "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}
], - "pass_criteria": {
- "threshold": 0.5
}
}{- "resource": {
- "id": "a1b2c3d4-5678-9abc-def0-1234567890ab",
- "tenant": "default",
- "created_at": "2026-01-15T09:30:00Z",
- "updated_at": "2026-01-15T09:30:00Z",
- "owner": "user@example.com"
}, - "status": {
- "state": "pending",
- "message": {
- "message": "Evaluation job created.",
- "message_code": "evaluation_job_created"
}
}, - "name": "granite-3.1-8b-safety-eval",
- "description": "Safety and reasoning evaluation for Granite 3.1 8B Instruct",
- "tags": [
- "nightly",
- "granite"
], - "model": {
- "name": "granite-3.1-8b-instruct"
}, - "benchmarks": [
- {
- "id": "arc_easy",
- "provider_id": "lm_evaluation_harness",
- "weight": 0.6,
- "primary_score": {
- "metric": "acc_norm",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.25
}, - "parameters": {
- "num_fewshot": 0,
- "limit": 100
}
}, - {
- "id": "owasp_llm_top10",
- "provider_id": "garak",
- "weight": 0.4,
- "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}
], - "pass_criteria": {
- "threshold": 0.5
}
}List all evaluation requests.
| limit | integer (Limit) [ 1 .. 100 ] Default: 50 Maximum number of evaluations to return |
| offset | integer (Offset) >= 0 Default: 0 Offset for pagination |
| status | string (Status Filter) Filter by status |
| name | string (Name) Name to search for |
| tags | string (Tags) Tags to search for |
| experiment_id | string (Experiment Id) Filter by MLflow experiment ID |
| collection_id | string (Collection Id) Filter by the ID of the collection the job was created from. Maps to the collection.id field on the job record. Returns all jobs for the collection. |
{- "first": {
- "href": "/api/v1/evaluations/jobs?limit=50&offset=0"
}, - "next": {
- "href": "/api/v1/evaluations/jobs?limit=50&offset=50"
}, - "limit": 50,
- "total_count": 73,
- "items": [
- {
- "resource": {
- "id": "a1b2c3d4-5678-9abc-def0-1234567890ab",
- "tenant": "default",
- "created_at": "2026-01-15T09:30:00Z",
- "updated_at": "2026-01-15T09:42:15Z",
- "owner": "user@example.com"
}, - "status": {
- "state": "completed",
- "message": {
- "message": "Evaluation job completed.",
- "message_code": "evaluation_job_updated"
}
}, - "results": {
- "benchmarks": [
- {
- "id": "arc_easy",
- "provider_id": "lm_evaluation_harness",
- "benchmark_index": 0,
- "metrics": {
- "acc": 0.82,
- "acc_norm": 0.85
}, - "test": {
- "primary_score": 0.85,
- "threshold": 0.25,
- "pass": true
}
}
], - "test": {
- "score": 0.85,
- "threshold": 0.5,
- "pass": true
}
}, - "name": "granite-3.1-8b-safety-eval",
- "model": {
- "name": "granite-3.1-8b-instruct"
}, - "benchmarks": [
- {
- "id": "arc_easy",
- "provider_id": "lm_evaluation_harness",
- "weight": 0.6
}
], - "pass_criteria": {
- "threshold": 0.5
}
}
]
}Returns the evaluation job resource with the current status and results.
| id required | string (Id) |
{- "resource": {
- "id": "a1b2c3d4-5678-9abc-def0-1234567890ab",
- "tenant": "default",
- "created_at": "2026-01-15T09:30:00Z",
- "updated_at": "2026-01-15T09:42:15Z",
- "owner": "user@example.com"
}, - "status": {
- "state": "completed",
- "message": {
- "message": "Evaluation job completed.",
- "message_code": "evaluation_job_updated"
}, - "benchmarks": [
- {
- "provider_id": "lm_evaluation_harness",
- "id": "arc_easy",
- "benchmark_index": 0,
- "status": "completed",
- "started_at": "2026-01-15T09:31:00Z",
- "completed_at": "2026-01-15T09:38:45Z"
}, - {
- "provider_id": "garak",
- "id": "owasp_llm_top10",
- "benchmark_index": 1,
- "status": "completed",
- "started_at": "2026-01-15T09:31:00Z",
- "completed_at": "2026-01-15T09:42:15Z"
}
]
}, - "results": {
- "benchmarks": [
- {
- "id": "arc_easy",
- "provider_id": "lm_evaluation_harness",
- "benchmark_index": 0,
- "metrics": {
- "acc": 0.82,
- "acc_norm": 0.85
}, - "mlflow_run_id": "run-7f3a1b2c",
- "logs_path": "/data/logs/a1b2c3d4.log",
- "test": {
- "primary_score": 0.85,
- "threshold": 0.25,
- "pass": true
}
}, - {
- "id": "owasp_llm_top10",
- "provider_id": "garak",
- "benchmark_index": 1,
- "metrics": {
- "attack_success_rate": 0.12
}, - "mlflow_run_id": "run-9e8d7c6b",
- "logs_path": "/data/logs/a1b2c3d4-garak.log",
- "test": {
- "primary_score": 0.12,
- "threshold": 0.3,
- "pass": true
}
}
], - "test": {
- "score": 0.85,
- "threshold": 0.5,
- "pass": true
}, - "post_processing_ref": {
- "id": "pp-1234567890ab"
}
}, - "name": "granite-3.1-8b-safety-eval",
- "description": "Safety and reasoning evaluation for Granite 3.1 8B Instruct",
- "tags": [
- "nightly",
- "granite"
], - "model": {
- "name": "granite-3.1-8b-instruct"
}, - "benchmarks": [
- {
- "id": "arc_easy",
- "provider_id": "lm_evaluation_harness",
- "weight": 0.6,
- "primary_score": {
- "metric": "acc_norm",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.25
}, - "parameters": {
- "num_fewshot": 0,
- "limit": 100
}
}, - {
- "id": "owasp_llm_top10",
- "provider_id": "garak",
- "weight": 0.4,
- "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}
], - "pass_criteria": {
- "threshold": 0.5
}
}Cancel a running evaluation.
| id required | string (Id) |
| hard_delete | boolean (Hard Delete) Default: false If |
{- "message": "The field 'state' is not valid.",
- "message_code": "invalid_value",
- "trace": "b12692e1-8582-4628-88ca-7a13fefb73e2"
}Returns plain-text workload logs for all benchmarks in an evaluation job.
Kubernetes runtime: adapter container stdout/stderr via the Kubernetes API.
Local runtime: contents of each benchmark's jobrun.log file under
/tmp/evalhub-jobs/{job_id}/{benchmark_index}/{provider_id}/{benchmark_id}/.
Logs are fetched on demand from the active runtime. Distinct from logs_path on
benchmark results, which refers to adapter-written artifact files.
| id required | string (Id) |
-1 (integer) or integer Default: 1000 Maximum number of log lines to return per benchmark. The response
concatenates one section per benchmark; each section is capped
independently, not as a total across the full response.
Use | |
| timestamps | boolean Default: false Include Kubernetes log timestamps |
| since_seconds | integer >= 1 Only return logs newer than this many seconds |
=== pod=a1b2c3d4-405ef22a-abc12 container=adapter benchmark_id=arc_easy === INFO starting evaluation INFO benchmark completed
Returns plain-text workload logs for a single benchmark within an evaluation job.
The benchmark is identified by benchmark_index in the request path.
Kubernetes runtime: adapter container stdout/stderr via the Kubernetes API.
Local runtime: contents of the benchmark's jobrun.log file. See
GET /api/v1/evaluations/jobs/{id}/logs for shared query parameters.
| id required | string (Id) |
| benchmark_index required | integer (Benchmark Index) >= 0 |
-1 (integer) or integer Default: 1000 Maximum number of log lines to return.
Use | |
| timestamps | boolean Default: false Include Kubernetes log timestamps |
| since_seconds | integer >= 1 Only return logs newer than this many seconds |
INFO starting evaluation INFO benchmark completed
Create an asynchronous post-processing computation.
| name | string (PostProcessingName) non-empty Name of the confidence interval post-processing computation. |
object (PostProcessingHardwareConfig) Optional hardware override for the post-processing job. | |
required | Array of objects (StandalonePostProcessingOperations) non-empty Post-processing operations executed by a standalone resource. |
{- "name": "inspect-accuracy-confidence-interval",
- "hardware_config": {
- "cpu": {
- "request": "500m",
- "limit": "2"
}, - "memory": {
- "request": "1Gi",
- "limit": "4Gi"
}
}, - "operations": [
- {
- "confidence_interval_config": {
- "results_data_ref": {
- "eval_job": {
- "id": "8e9d1d6b-5ca1-4a52-91b4-6f0f00c2f8d7",
- "num_parallel_threads": 2
}
}, - "calibration_data_ref": [
- {
- "s3": {
- "bucket": "evaluation-data",
- "key": "calibration/accuracy",
- "secret_ref": "s3-secret"
}, - "data_config": {
- "format": "jsonl",
- "columns": {
- "sample_id": "example_id",
- "label": "human_score",
- "prediction": "judge_score",
- "benchmark_id": "benchmark_id",
- "provider_id": "provider_id"
}, - "selection": {
- "metric": "accuracy"
}
}
}
], - "significance_level": 0.7
}
}
]
}{- "name": "string",
- "hardware_config": {
- "hardware_profile_name": "default-profile",
- "queue": {
- "kind": "kueue",
- "name": "string"
}, - "cpu": {
- "request": "1",
- "limit": "2"
}, - "memory": {
- "request": "1",
- "limit": "2"
}, - "gpu": {
- "name": "nvidia.com/gpu",
- "count": 1
}
}, - "resource": {
- "id": "string",
- "tenant": "string",
- "created_at": "2019-08-24T14:15:22Z",
- "updated_at": "2019-08-24T14:15:22Z",
- "owner": "string"
}, - "operations": [
- {
- "confidence_interval_config": {
- "calibration_data_ref": [
- {
- "s3": {
- "bucket": "my-eval-bucket",
- "key": "datasets/benchmark-a/v1",
- "secret_ref": "my-s3-connection-secret"
}, - "pvc": {
- "claim_name": "eval-datasets-pvc",
- "sub_path": "benchmark-a/v1"
}, - "git": {
- "ref": "main",
- "sub_path": "datasets/lm-eval",
- "secret_ref": "my-git-credentials"
}, - "hf": {
- "repo_id": "cais/mmlu",
- "revision": "main",
- "sub_path": "data/train",
- "secret_ref": "my-hf-credentials"
}, - "resolved_sha": "a3f9c12e8b4d6f1c2e9a7b3d5f0e8c4a2b6d9e1f",
- "type": "data_set",
- "data_config": {
- "format": "string",
- "columns": {
- "sample_id": "string",
- "label": "string",
- "prediction": "string",
- "benchmark_id": "string",
- "provider_id": "string"
}, - "selection": {
- "metric": "string"
}
}
}
], - "significance_level": 0.5,
- "results_data_ref": {
- "eval_job": {
- "id": "string",
- "num_parallel_threads": 1
}
}
}
}
], - "status": {
- "state": "pending",
- "error_message": {
- "message": "string",
- "message_code": "string",
- "message_origin": "server"
}, - "warning_message": {
- "message": "string",
- "message_code": "string",
- "message_origin": "server"
}, - "started_at": "2019-08-24T14:15:22Z",
- "completed_at": "2019-08-24T14:15:22Z",
- "benchmarks": [
- {
- "provider_id": "string",
- "id": "string",
- "benchmark_index": 0,
- "status": "pending",
- "phase": "initializing",
- "error_message": {
- "message": "string",
- "message_code": "string",
- "message_origin": "server"
}, - "warning_message": {
- "message": "string",
- "message_code": "string",
- "message_origin": "server"
}, - "started_at": "2019-08-24T14:15:22Z",
- "completed_at": "2019-08-24T14:15:22Z"
}
]
}, - "results": {
- "benchmarks": [
- {
- "id": "string",
- "provider_id": "string",
- "benchmark_index": 0,
- "confidence_interval": {
- "lower": 0.1,
- "upper": 0.1
}
}
]
}
}Get a standalone confidence interval post-processing resource.
| id required | string |
{- "resource": {
- "id": "4b6510bd-cee3-4a80-8028-4a6cb60ae1cc",
- "tenant": "sagar",
- "created_at": "2026-09-21T05:57:34.477263Z",
- "updated_at": "2026-09-21T05:58:29.768663Z",
- "owner": "nbs"
}, - "name": "inspect-accuracy-confidence-interval",
- "hardware_config": {
- "cpu": {
- "request": "500m",
- "limit": "2"
}, - "memory": {
- "request": "1Gi",
- "limit": "4Gi"
}
}, - "operations": [
- {
- "confidence_interval_config": {
- "results_data_ref": {
- "eval_job": {
- "id": "8e9d1d6b-5ca1-4a52-91b4-6f0f00c2f8d7",
- "num_parallel_threads": 2
}
}, - "calibration_data_ref": [
- {
- "s3": {
- "bucket": "evaluation-data",
- "key": "calibration/accuracy",
- "secret_ref": "s3-secret"
}, - "data_config": {
- "format": "jsonl",
- "columns": {
- "sample_id": "example_id",
- "label": "human_score",
- "prediction": "judge_score",
- "benchmark_id": "benchmark_id",
- "provider_id": "provider_id"
}, - "selection": {
- "metric": "accuracy"
}
}
}
], - "significance_level": 0.7
}
}
], - "status": {
- "state": "completed",
- "benchmarks": [
- {
- "id": "accuracy-benchmark",
- "provider_id": "inspect",
- "benchmark_index": 0,
- "status": "completed"
}
]
}, - "results": {
- "benchmarks": [
- {
- "id": "accuracy-benchmark",
- "provider_id": "inspect",
- "benchmark_index": 0,
- "confidence_interval": {
- "lower": 0.75,
- "upper": 0.85
}
}
]
}
}Delete a standalone confidence interval post-processing resource.
| id required | string |
{- "message": "The bearer token is not valid.",
- "message_code": "invalid_auth_token",
- "trace": "b12692e1-8582-4628-88ca-7a13fefb73e2"
}Create confidence interval computation for the benchmarks in an existing evaluation job. The operation is idempotent: if the same job already has a confidence interval computation, the existing post-processing resource is returned.
The computation runs independently for each benchmark. Each item in
operations describes one post-processing operation. hardware_config
applies to the post-processing job as a whole.
| job_id required | string (Job Id) |
| name | string (PostProcessingName) non-empty Name of the confidence interval post-processing computation. |
object (PostProcessingHardwareConfig) Optional hardware override for the post-processing job. | |
required | Array of objects (JobPostProcessingOperations) non-empty Post-processing operations associated with an evaluation job. |
{- "name": "inspect-accuracy-confidence-interval",
- "hardware_config": {
- "cpu": {
- "request": "500m",
- "limit": "2"
}, - "memory": {
- "request": "1Gi",
- "limit": "4Gi"
}
}, - "operations": [
- {
- "confidence_interval_config": {
- "num_parallel_threads": 2,
- "calibration_data_ref": [
- {
- "s3": {
- "bucket": "evaluation-data",
- "key": "calibration/accuracy",
- "secret_ref": "s3-secret"
}, - "data_config": {
- "format": "jsonl",
- "columns": {
- "sample_id": "example_id",
- "label": "human_score",
- "prediction": "judge_score",
- "benchmark_id": "benchmark_id",
- "provider_id": "provider_id"
}, - "selection": {
- "metric": "accuracy"
}
}
}
], - "significance_level": 0.7
}
}
]
}{- "resource": {
- "id": "4b6510bd-cee3-4a80-8028-4a6cb60ae1cc",
- "tenant": "sagar",
- "created_at": "2026-09-21T05:57:34.477263Z",
- "updated_at": "2026-09-21T05:58:29.768663Z",
- "owner": "nbs"
}, - "name": "inspect-accuracy-confidence-interval",
- "hardware_config": {
- "cpu": {
- "request": "500m",
- "limit": "2"
}, - "memory": {
- "request": "1Gi",
- "limit": "4Gi"
}
}, - "operations": [
- {
- "confidence_interval_config": {
- "num_parallel_threads": 2,
- "calibration_data_ref": [
- {
- "s3": {
- "bucket": "evaluation-data",
- "key": "calibration/accuracy",
- "secret_ref": "s3-secret"
}, - "data_config": {
- "format": "jsonl",
- "columns": {
- "sample_id": "example_id",
- "label": "human_score",
- "prediction": "judge_score",
- "benchmark_id": "benchmark_id",
- "provider_id": "provider_id"
}, - "selection": {
- "metric": "accuracy"
}
}
}
], - "significance_level": 0.7
}
}
], - "status": {
- "state": "completed",
- "benchmarks": [
- {
- "id": "accuracy-benchmark",
- "provider_id": "inspect",
- "benchmark_index": 0,
- "status": "completed"
}
]
}, - "results": {
- "benchmarks": [
- {
- "id": "accuracy-benchmark",
- "provider_id": "inspect",
- "benchmark_index": 0,
- "confidence_interval": {
- "lower": 0.75,
- "upper": 0.85
}
}
]
}
}Get a confidence interval computation associated with an evaluation job.
| job_id required | string (Job Id) |
| post_processing_id required | string (Post Processing Id) |
{- "resource": {
- "id": "pp-1234567890ab",
- "tenant": "sagar",
- "created_at": "2026-09-21T05:57:34.477263Z",
- "updated_at": "2026-09-21T05:58:29.768663Z",
- "owner": "nbs"
}, - "name": "inspect-accuracy-confidence-interval",
- "hardware_config": {
- "cpu": {
- "request": "500m",
- "limit": "2"
}, - "memory": {
- "request": "1Gi",
- "limit": "4Gi"
}
}, - "operations": [
- {
- "confidence_interval_config": {
- "num_parallel_threads": 2,
- "calibration_data_ref": [
- {
- "s3": {
- "bucket": "evaluation-data",
- "key": "calibration/accuracy",
- "secret_ref": "s3-secret"
}, - "data_config": {
- "format": "jsonl",
- "columns": {
- "sample_id": "example_id",
- "label": "human_score",
- "prediction": "judge_score",
- "benchmark_id": "benchmark_id",
- "provider_id": "provider_id"
}, - "selection": {
- "metric": "accuracy"
}
}
}
], - "significance_level": 0.7
}
}
], - "status": {
- "state": "completed",
- "benchmarks": [
- {
- "id": "accuracy-benchmark",
- "provider_id": "inspect",
- "benchmark_index": 0,
- "status": "completed"
}
]
}, - "results": {
- "benchmarks": [
- {
- "id": "accuracy-benchmark",
- "provider_id": "inspect",
- "benchmark_index": 0,
- "confidence_interval": {
- "lower": 0.75,
- "upper": 0.85
}
}
]
}
}Delete a confidence interval computation associated with an evaluation job.
| job_id required | string (Job Id) |
| post_processing_id required | string (Post Processing Id) |
{- "message": "The bearer token is not valid.",
- "message_code": "invalid_auth_token",
- "trace": "b12692e1-8582-4628-88ca-7a13fefb73e2"
}List all benchmark collections.
| limit | integer (Limit) [ 1 .. 100 ] Default: 50 Maximum number of collections to return |
| offset | integer (Offset) >= 0 Default: 0 Offset for pagination |
| name | string (Name) Name to search for |
| category | string (Category) Category to search for |
| tags | string (Tags) Tags to search for |
| scope | string (Scope of collections) Enum: "system" "tenant" Filters the result set by collection scope, within the collections visible to the requesting tenant. |
| modalities | string (Modality filter) Filter collections by modality (snake_case). Returns collections whose modalities array contains this value. May be repeated for multiple values. |
| tasks | string (Task filter) Filter collections by task type (snake_case). Returns collections whose tasks array contains this value. May be repeated for multiple values. |
| domains | string (Domain filter) Filter collections by evaluation domain (snake_case). Returns collections whose domains array contains this value. May be repeated for multiple values. |
| industries | string (Industry filter) Filter collections by target industry (snake_case). Returns collections whose industries array contains this value. May be repeated for multiple values. |
| evaluation_targets | string (Evaluation target filter) Filter collections by evaluation target type (snake_case). Example values: model, agent. Returns collections whose evaluation_targets array contains this value. May be repeated for multiple values. |
| sort_by | string (Sort order) Value: "curation_order" Sort field for the result set. Applied server-side across all matching collections before pagination. curation_order sorts curated collections ascending (0 or absent last). When omitted, the existing default collection ordering is preserved. |
{- "first": {
- "href": "/api/v1/evaluations/collections?limit=50&offset=0"
}, - "limit": 50,
- "total_count": 2,
- "items": [
- {
- "resource": {
- "id": "e5f6a7b8-9012-3456-cdef-0123456789ab",
- "tenant": "default",
- "created_at": "2025-12-01T10:00:00Z",
- "updated_at": "2025-12-01T10:00:00Z"
}, - "name": "llm-safety-suite",
- "category": "safety",
- "description": "Comprehensive safety evaluation combining reasoning accuracy and vulnerability scanning",
- "tags": [
- "safety",
- "nightly"
], - "pass_criteria": {
- "threshold": 0.5
}, - "benchmarks": [
- {
- "id": "arc_easy",
- "provider_id": "lm_evaluation_harness",
- "weight": 0.6,
- "primary_score": {
- "metric": "acc_norm",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.25
}
}, - {
- "id": "owasp_llm_top10",
- "provider_id": "garak",
- "weight": 0.4,
- "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}
]
}
]
}Create a new collection.
| name required | string Collection name. |
| category required | string [ 1 .. 128 ] characters Deprecated DEPRECATED — use |
| description | string Optional description. |
| tags | Array of strings Tags. |
object Custom key-value data. | |
object (PassCriteria) Pass criteria for the collection. | |
required | Array of objects (CollectionBenchmarkConfig) Benchmarks in the collection. |
| curation_order | integer Controls the listing priority of this collection within collection retrieval results. 0 (or absent) means not curated. Positive integers specify priority — lower values have higher priority. Set by system operators via YAML configuration only; the server rejects writes from tenant API consumers. |
| domains | Array of strings High-level evaluation domains this collection addresses (snake_case). If not explicitly set, the handler returns the union of domains from the collection's benchmarks (via BenchmarkResource.domains). Canonical values: knowledge_and_reasoning, grounded_document_understanding, instruction_and_output_reliability, tool_use_and_function_calling, software, trustworthiness, multilingual, multimodal. |
| tasks | Array of strings ML tasks this collection evaluates (snake_case). If not explicitly set, the handler returns the union of tasks from the collection's benchmarks (via BenchmarkResource.tasks). Known values: reasoning, data_analysis, extraction, summarization, full_document_qa, long_context_understanding, rag, citation_attribution, grounding_discipline, instruction_following, structured_output, constraint_following, call_generation, code_generation, code_understanding, code_repair, safety, calibration, abstention, adversarial_injection, translation, document_chart_vqa, general_visual_reasoning. |
| modalities | Array of strings Data modalities this collection covers (snake_case). If not explicitly set, the handler returns the union of modalities from the collection's benchmarks (via BenchmarkResource.modalities). Known values: text, vision, multimodal. |
| industries | Array of strings Business industries this collection is relevant for (snake_case). Collection-level field; the same benchmarks may serve different industries. Example values: health, telco, financial, government. |
| evaluation_targets | Array of strings AI entity types this collection evaluates (snake_case). If not explicitly set, the handler returns the union of evaluation_targets from the collection's benchmarks (via BenchmarkResource.evaluation_targets). Example values: model, agent. |
object (CollectionAgentMetadata) Structured metadata for AI agent discoverability at the collection level. |
{- "name": "release-gate-safety",
- "category": "safety",
- "description": "Release-gate collection combining reasoning and red-teaming benchmarks",
- "tags": [
- "release-gate",
- "safety"
], - "pass_criteria": {
- "threshold": 0.5
}, - "benchmarks": [
- {
- "id": "arc_easy",
- "provider_id": "lm_evaluation_harness",
- "weight": 0.6,
- "primary_score": {
- "metric": "acc_norm",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.25
}, - "parameters": {
- "num_fewshot": 0,
- "limit": 100
}
}, - {
- "id": "owasp_llm_top10",
- "provider_id": "garak",
- "weight": 0.4,
- "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}
]
}{- "resource": {
- "id": "f6a7b8c9-0123-4567-def0-123456789abc",
- "tenant": "default",
- "created_at": "2026-02-01T09:00:00Z",
- "updated_at": "2026-02-01T09:00:00Z",
- "owner": "user@example.com"
}, - "name": "release-gate-safety",
- "category": "safety",
- "description": "Release-gate collection combining reasoning and red-teaming benchmarks",
- "tags": [
- "release-gate",
- "safety"
], - "pass_criteria": {
- "threshold": 0.5
}, - "benchmarks": [
- {
- "id": "arc_easy",
- "provider_id": "lm_evaluation_harness",
- "weight": 0.6,
- "primary_score": {
- "metric": "acc_norm",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.25
}, - "parameters": {
- "num_fewshot": 0,
- "limit": 100
}
}, - {
- "id": "owasp_llm_top10",
- "provider_id": "garak",
- "weight": 0.4,
- "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}
]
}Get details of a specific collection.
| id required | string (Collection Id) |
{- "resource": {
- "id": "e5f6a7b8-9012-3456-cdef-0123456789ab",
- "tenant": "default",
- "created_at": "2025-12-01T10:00:00Z",
- "updated_at": "2025-12-01T10:00:00Z"
}, - "name": "llm-safety-suite",
- "category": "safety",
- "description": "Comprehensive safety evaluation combining reasoning accuracy and vulnerability scanning",
- "tags": [
- "safety",
- "nightly"
], - "pass_criteria": {
- "threshold": 0.5
}, - "benchmarks": [
- {
- "id": "arc_easy",
- "provider_id": "lm_evaluation_harness",
- "weight": 0.6,
- "primary_score": {
- "metric": "acc_norm",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.25
}
}, - {
- "id": "owasp_llm_top10",
- "provider_id": "garak",
- "weight": 0.4,
- "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}
]
}Update an existing collection.
| id required | string (Collection Id) |
| name required | string Collection name. |
| category required | string [ 1 .. 128 ] characters Deprecated DEPRECATED — use |
| description | string Optional description. |
| tags | Array of strings Tags. |
object Custom key-value data. | |
object (PassCriteria) Pass criteria for the collection. | |
required | Array of objects (CollectionBenchmarkConfig) Benchmarks in the collection. |
| curation_order | integer Controls the listing priority of this collection within collection retrieval results. 0 (or absent) means not curated. Positive integers specify priority — lower values have higher priority. Set by system operators via YAML configuration only; the server rejects writes from tenant API consumers. |
| domains | Array of strings High-level evaluation domains this collection addresses (snake_case). If not explicitly set, the handler returns the union of domains from the collection's benchmarks (via BenchmarkResource.domains). Canonical values: knowledge_and_reasoning, grounded_document_understanding, instruction_and_output_reliability, tool_use_and_function_calling, software, trustworthiness, multilingual, multimodal. |
| tasks | Array of strings ML tasks this collection evaluates (snake_case). If not explicitly set, the handler returns the union of tasks from the collection's benchmarks (via BenchmarkResource.tasks). Known values: reasoning, data_analysis, extraction, summarization, full_document_qa, long_context_understanding, rag, citation_attribution, grounding_discipline, instruction_following, structured_output, constraint_following, call_generation, code_generation, code_understanding, code_repair, safety, calibration, abstention, adversarial_injection, translation, document_chart_vqa, general_visual_reasoning. |
| modalities | Array of strings Data modalities this collection covers (snake_case). If not explicitly set, the handler returns the union of modalities from the collection's benchmarks (via BenchmarkResource.modalities). Known values: text, vision, multimodal. |
| industries | Array of strings Business industries this collection is relevant for (snake_case). Collection-level field; the same benchmarks may serve different industries. Example values: health, telco, financial, government. |
| evaluation_targets | Array of strings AI entity types this collection evaluates (snake_case). If not explicitly set, the handler returns the union of evaluation_targets from the collection's benchmarks (via BenchmarkResource.evaluation_targets). Example values: model, agent. |
object (CollectionAgentMetadata) Structured metadata for AI agent discoverability at the collection level. |
{- "name": "llm-safety-suite",
- "category": "safety",
- "description": "Safety evaluation with reasoning, OWASP risks, and content quality",
- "tags": [
- "safety",
- "nightly",
- "updated"
], - "pass_criteria": {
- "threshold": 0.5
}, - "benchmarks": [
- {
- "id": "arc_easy",
- "provider_id": "lm_evaluation_harness",
- "weight": 0.5,
- "primary_score": {
- "metric": "acc_norm",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.25
}
}, - {
- "id": "owasp_llm_top10",
- "provider_id": "garak",
- "weight": 0.3,
- "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}, - {
- "id": "quality",
- "provider_id": "garak",
- "weight": 0.2,
- "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}
]
}{- "resource": {
- "id": "e5f6a7b8-9012-3456-cdef-0123456789ab",
- "tenant": "default",
- "created_at": "2025-12-01T10:00:00Z",
- "updated_at": "2026-02-10T11:00:00Z"
}, - "name": "llm-safety-suite",
- "category": "safety",
- "description": "Safety evaluation with reasoning, OWASP risks, and content quality",
- "tags": [
- "safety",
- "nightly",
- "updated"
], - "pass_criteria": {
- "threshold": 0.5
}, - "benchmarks": [
- {
- "id": "arc_easy",
- "provider_id": "lm_evaluation_harness",
- "weight": 0.5,
- "primary_score": {
- "metric": "acc_norm",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.25
}
}, - {
- "id": "owasp_llm_top10",
- "provider_id": "garak",
- "weight": 0.3,
- "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}, - {
- "id": "quality",
- "provider_id": "garak",
- "weight": 0.2,
- "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}
]
}Partially update an existing collection.
| id required | string (Collection Id) |
| op required | string (PatchOp) Enum: "replace" "add" "remove" Patch operation type |
| path required | string JSON Pointer path |
| value | any Value for add/replace (omit for remove) |
[- {
- "op": "replace",
- "path": "/pass_criteria/threshold",
- "value": 0.6
}, - {
- "op": "replace",
- "path": "/description",
- "value": "Safety evaluation with stricter pass threshold"
}
]{- "resource": {
- "id": "e5f6a7b8-9012-3456-cdef-0123456789ab",
- "tenant": "default",
- "created_at": "2025-12-01T10:00:00Z",
- "updated_at": "2026-02-10T12:30:00Z"
}, - "name": "llm-safety-suite",
- "category": "safety",
- "description": "Safety evaluation with stricter pass threshold",
- "tags": [
- "safety",
- "nightly"
], - "pass_criteria": {
- "threshold": 0.6
}, - "benchmarks": [
- {
- "id": "arc_easy",
- "provider_id": "lm_evaluation_harness",
- "weight": 0.6,
- "primary_score": {
- "metric": "acc_norm",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.25
}
}, - {
- "id": "owasp_llm_top10",
- "provider_id": "garak",
- "weight": 0.4,
- "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}
]
}Creates a new tenant-scoped (custom) collection as a copy of the specified source collection. The server copies all CollectionConfig fields from the source, assigns a new ID with scope=tenant, and records the source ID in derived_from for provenance. The request body may include optional overrides (e.g. a new name).
| id required | string (Collection Id) ID of the source collection to copy. |
Optional field overrides for the cloned collection. Any field omitted here is inherited from the source collection, except curation_order. Clone requests never accept curation_order, and the server always resets it to 0.
| name | string New name for the copy. Defaults to source name. |
| description | string Override description. |
| category | string Deprecated — use domains. |
| tags | Array of strings Override tags. |
object (PassCriteria) Override pass criteria. | |
Array of objects (CollectionBenchmarkConfig) Override benchmarks. | |
| domains | Array of strings Override domains. |
| tasks | Array of strings Override tasks. |
| modalities | Array of strings Override modalities. |
| industries | Array of strings Override industries. |
| evaluation_targets | Array of strings Override evaluation targets. |
object Override custom key-value data. | |
object (CollectionAgentMetadata) Override agent metadata for AI agent consumption. |
{- "name": "my-custom-rag-eval",
- "domains": [
- "grounded_document_understanding"
]
}{- "resource": {
- "id": "b2c3d4e5-6789-0abc-def1-23456789abcd",
- "tenant": "my-tenant",
- "created_at": "2026-08-25T10:00:00Z",
- "updated_at": "2026-08-25T10:00:00Z",
- "owner": "user@example.com"
}, - "derived_from": "e5f6a7b8-9012-3456-cdef-0123456789ab",
- "pinned_order": 0,
- "status": {
- "run_count": 0
}, - "name": "my-rag-evaluation",
- "category": "document_understanding",
- "tasks": [
- "rag",
- "grounding_discipline"
], - "modalities": [
- "text"
], - "benchmarks": [
- {
- "id": "crag",
- "provider_id": "ragas",
- "weight": 1,
- "primary_score": {
- "metric": "answer_correctness",
- "lower_is_better": false
}
}
]
}List all registered evaluation providers.
| limit | integer (Limit) [ 1 .. 100 ] Default: 50 Maximum number of providers to return |
| offset | integer (Offset) >= 0 Default: 0 Offset for pagination |
| benchmarks | boolean (Benchmarks) Default: true Include or exclude benchmarks supported by this provider in the response |
| name | string (Name) Name to search for |
| tags | string (Tags) Tags to search for |
| scope | string (Scope of providers) Enum: "system" "tenant" Set to |
{- "first": {
- "href": "/api/v1/evaluations/providers?limit=50&offset=0"
}, - "limit": 50,
- "total_count": 3,
- "items": [
- {
- "resource": {
- "id": "b3f1a2c4-1234-5678-abcd-ef0123456789",
- "tenant": "default",
- "created_at": "2025-10-01T00:00:00Z",
- "updated_at": "2025-10-01T00:00:00Z"
}, - "name": "lm_evaluation_harness",
- "title": "LM Evaluation Harness",
- "description": "Comprehensive evaluation framework for language models with 180 benchmarks",
- "tags": [
- "reasoning",
- "science",
- "lm_eval"
], - "runtime": {
- "k8s": {
- "image": "quay.io/opendatahub/ta-lmes-job:odh-3.4-ea2",
- "entrypoint": [
- "/opt/app-root/bin/python",
- "/opt/app-root/src/main.py"
], - "cpu_request": "100m",
- "memory_request": "128Mi",
- "cpu_limit": "500m",
- "memory_limit": "4Gi"
}
}, - "benchmarks": [
- {
- "id": "arc_easy",
- "name": "Basic science Q&A",
- "description": "Grade-school science questions testing basic reasoning and scientific knowledge (AI2 Reasoning Challenge, easy split).",
- "category": "reasoning",
- "metrics": [
- "acc",
- "acc_norm"
], - "num_few_shot": 0,
- "dataset_size": 2376,
- "tags": [
- "reasoning",
- "science",
- "lm_eval"
], - "primary_score": {
- "metric": "acc_norm",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.25
}
}
]
}
]
}Create a new provider scoped to the current tenant (Bring Your Own Provider)
| name required | string Provider name |
| title | string Provider display title |
| description | string Provider description |
| tags | Array of strings Provider tags |
object (AgentMetadata) Agent discoverability metadata for this provider | |
required | object (Runtime) Provider runtime configuration |
required | Array of objects (BenchmarkResource) Benchmarks offered by this provider |
{- "name": "my-custom-evaluator",
- "title": "Custom Internal Evaluator",
- "description": "Internal evaluation adapter for domain-specific benchmarks",
- "tags": [
- "custom",
- "internal"
], - "runtime": {
- "k8s": {
- "image": "registry.internal.example.com/eval/custom-adapter:v1.2",
- "entrypoint": [
- "/opt/app-root/bin/python",
- "/opt/app-root/src/main.py"
], - "cpu_request": "250m",
- "memory_request": "512Mi",
- "cpu_limit": "1",
- "memory_limit": "2Gi"
}
}, - "benchmarks": [
- {
- "id": "domain-qa",
- "name": "Domain Q&A Accuracy",
- "description": "Measures accuracy on domain-specific question answering",
- "category": "reasoning",
- "metrics": [
- "acc",
- "f1"
], - "primary_score": {
- "metric": "acc",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.5
}
}
]
}{- "resource": {
- "id": "c4d5e6f7-8901-2345-bcde-f67890123456",
- "tenant": "default",
- "created_at": "2026-01-20T10:00:00Z",
- "updated_at": "2026-01-20T10:00:00Z",
- "owner": "user@example.com"
}, - "name": "my-custom-evaluator",
- "title": "Custom Internal Evaluator",
- "description": "Internal evaluation adapter for domain-specific benchmarks",
- "tags": [
- "custom",
- "internal"
], - "runtime": {
- "k8s": {
- "image": "registry.internal.example.com/eval/custom-adapter:v1.2",
- "entrypoint": [
- "/opt/app-root/bin/python",
- "/opt/app-root/src/main.py"
], - "cpu_request": "250m",
- "memory_request": "512Mi",
- "cpu_limit": "1",
- "memory_limit": "2Gi"
}
}, - "benchmarks": [
- {
- "id": "domain-qa",
- "name": "Domain Q&A Accuracy",
- "description": "Measures accuracy on domain-specific question answering",
- "category": "reasoning",
- "metrics": [
- "acc",
- "f1"
], - "primary_score": {
- "metric": "acc",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.5
}
}
]
}Get a provider by ID.
| id required | string (Provider Id) Provider ID |
{- "resource": {
- "id": "d8e9f0a1-2345-6789-cdef-012345678901",
- "tenant": "default",
- "created_at": "2025-10-01T00:00:00Z",
- "updated_at": "2025-10-01T00:00:00Z"
}, - "name": "garak",
- "title": "Garak",
- "description": "LLM vulnerability scanner and red-teaming framework",
- "tags": [
- "security",
- "red_team"
], - "runtime": {
- "k8s": {
- "image": "quay.io/trustyai/trustyai-garak-lls-provider-dsp:latest",
- "entrypoint": [
- "python",
- "-m",
- "llama_stack_provider_trustyai_garak.evalhub"
], - "cpu_request": "500m",
- "memory_request": "512Mi",
- "cpu_limit": "2000m",
- "memory_limit": "4Gi"
}
}, - "benchmarks": [
- {
- "id": "owasp_llm_top10",
- "name": "OWASP LLM top 10 risk scan",
- "description": "Tests against the top 10 security risks specific to LLM applications.",
- "category": "security",
- "metrics": [
- "attack_success_rate"
], - "tags": [
- "security",
- "owasp",
- "red_team"
], - "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}, - {
- "id": "quality",
- "name": "Toxic & harmful content scan",
- "description": "Scans for violence, profanity, toxicity, hate speech, and integrity issues.",
- "category": "safety",
- "metrics": [
- "attack_success_rate"
], - "tags": [
- "safety",
- "quality",
- "toxicity",
- "red_team"
], - "primary_score": {
- "metric": "attack_success_rate",
- "lower_is_better": true
}, - "pass_criteria": {
- "threshold": 0.3
}
}
]
}Update an existing provider.
| id required | string (Provider Id) Provider ID |
| name required | string Provider name |
| title | string Provider display title |
| description | string Provider description |
| tags | Array of strings Provider tags |
object (AgentMetadata) Agent discoverability metadata for this provider | |
required | object (Runtime) Provider runtime configuration |
required | Array of objects (BenchmarkResource) Benchmarks offered by this provider |
{- "name": "my-custom-evaluator",
- "title": "Custom Internal Evaluator",
- "description": "Updated evaluation adapter with improved tokenization",
- "tags": [
- "custom",
- "internal"
], - "runtime": {
- "k8s": {
- "image": "registry.internal.example.com/eval/custom-adapter:v2.0",
- "entrypoint": [
- "/opt/app-root/bin/python",
- "/opt/app-root/src/main.py"
], - "cpu_request": "500m",
- "memory_request": "1Gi",
- "cpu_limit": "2",
- "memory_limit": "4Gi"
}
}, - "benchmarks": [
- {
- "id": "domain-qa",
- "name": "Domain Q&A Accuracy",
- "description": "Measures accuracy on domain-specific question answering",
- "category": "reasoning",
- "metrics": [
- "acc",
- "f1"
], - "primary_score": {
- "metric": "acc",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.5
}
}
]
}{- "resource": {
- "id": "c4d5e6f7-8901-2345-bcde-f67890123456",
- "tenant": "default",
- "created_at": "2026-01-20T10:00:00Z",
- "updated_at": "2026-02-05T14:30:00Z"
}, - "name": "my-custom-evaluator",
- "title": "Custom Internal Evaluator",
- "description": "Updated evaluation adapter with improved tokenization",
- "tags": [
- "custom",
- "internal"
], - "runtime": {
- "k8s": {
- "image": "registry.internal.example.com/eval/custom-adapter:v2.0",
- "entrypoint": [
- "/opt/app-root/bin/python",
- "/opt/app-root/src/main.py"
], - "cpu_request": "500m",
- "memory_request": "1Gi",
- "cpu_limit": "2",
- "memory_limit": "4Gi"
}
}, - "benchmarks": [
- {
- "id": "domain-qa",
- "name": "Domain Q&A Accuracy",
- "description": "Measures accuracy on domain-specific question answering",
- "category": "reasoning",
- "metrics": [
- "acc",
- "f1"
], - "primary_score": {
- "metric": "acc",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.5
}
}
]
}Partially update an existing provider.
| id required | string (Provider Id) |
| op required | string (PatchOp) Enum: "replace" "add" "remove" Patch operation type |
| path required | string JSON Pointer path |
| value | any Value for add/replace (omit for remove) |
[- {
- "op": "replace",
- "path": "/runtime/k8s/image",
- "value": "registry.internal.example.com/eval/custom-adapter:v2.1"
}, - {
- "op": "replace",
- "path": "/description",
- "value": "Updated evaluation adapter with bug fixes"
}
]{- "resource": {
- "id": "c4d5e6f7-8901-2345-bcde-f67890123456",
- "tenant": "default",
- "created_at": "2026-01-20T10:00:00Z",
- "updated_at": "2026-02-06T09:15:00Z"
}, - "name": "my-custom-evaluator",
- "title": "Custom Internal Evaluator",
- "description": "Updated evaluation adapter with bug fixes",
- "tags": [
- "custom",
- "internal"
], - "runtime": {
- "k8s": {
- "image": "registry.internal.example.com/eval/custom-adapter:v2.1",
- "entrypoint": [
- "/opt/app-root/bin/python",
- "/opt/app-root/src/main.py"
], - "cpu_request": "500m",
- "memory_request": "1Gi",
- "cpu_limit": "2",
- "memory_limit": "4Gi"
}
}, - "benchmarks": [
- {
- "id": "domain-qa",
- "name": "Domain Q&A Accuracy",
- "description": "Measures accuracy on domain-specific question answering",
- "category": "reasoning",
- "metrics": [
- "acc",
- "f1"
], - "primary_score": {
- "metric": "acc",
- "lower_is_better": false
}, - "pass_criteria": {
- "threshold": 0.5
}
}
]
}