diff --git a/web/packages/studio/src/components/evaluation/Jobs/datasetEval/DatasetEvalRowResultsPanel.test.tsx b/web/packages/studio/src/components/evaluation/Jobs/datasetEval/DatasetEvalRowResultsPanel.test.tsx
new file mode 100644
index 0000000000..e449d10aab
--- /dev/null
+++ b/web/packages/studio/src/components/evaluation/Jobs/datasetEval/DatasetEvalRowResultsPanel.test.tsx
@@ -0,0 +1,68 @@
+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+// SPDX-License-Identifier: Apache-2.0
+
+import {
+ DatasetEvalRowResultsPanel,
+ type DatasetEvalRow,
+} from '@studio/components/evaluation/Jobs/datasetEval/DatasetEvalRowResultsPanel';
+import { fireEvent, render, screen } from '@studio/tests/util/render';
+
+const row = (requests: DatasetEvalRow['requests']): DatasetEvalRow => ({
+ row_index: 0,
+ item: { prompt: 'raw row', label: 'phishing' },
+ sample: { output_text: 'phishing' },
+ requests,
+});
+
+const openPanel = () => fireEvent.click(screen.getByRole('button', { name: /Row Results \(1\)/ }));
+
+describe('DatasetEvalRowResultsPanel', () => {
+ it('shows the empty state when there are no rows', () => {
+ render();
+ expect(
+ screen.getByText('No per-row results recorded for this evaluation.')
+ ).toBeInTheDocument();
+ });
+
+ it('renders the prompt from a chat-completions request body', async () => {
+ render(
+
+ );
+ openPanel();
+
+ expect(await screen.findByText('rendered prompt')).toBeInTheDocument();
+ });
+
+ it('renders the last message when the body carries a full transcript', async () => {
+ render(
+
+ );
+ openPanel();
+
+ expect(await screen.findByText('the task')).toBeInTheDocument();
+ });
+
+ it('falls back to input_message for jobs submitted before chat completions', async () => {
+ render(
+
+ );
+ openPanel();
+
+ expect(await screen.findByText('legacy body')).toBeInTheDocument();
+ });
+
+ it('falls back to the raw row when no request was recorded', async () => {
+ render();
+ openPanel();
+
+ expect(await screen.findByText(/raw row/)).toBeInTheDocument();
+ });
+});
diff --git a/web/packages/studio/src/components/evaluation/Jobs/datasetEval/DatasetEvalRowResultsPanel.tsx b/web/packages/studio/src/components/evaluation/Jobs/datasetEval/DatasetEvalRowResultsPanel.tsx
index 955c3fd02f..74fbfef767 100644
--- a/web/packages/studio/src/components/evaluation/Jobs/datasetEval/DatasetEvalRowResultsPanel.tsx
+++ b/web/packages/studio/src/components/evaluation/Jobs/datasetEval/DatasetEvalRowResultsPanel.tsx
@@ -18,7 +18,9 @@ export interface DatasetEvalRow {
item?: Record;
sample?: { output_text?: string };
metrics?: Record;
- requests?: { request?: { input_message?: string } }[];
+ requests?: {
+ request?: { messages?: { content?: string }[]; input_message?: string };
+ }[];
}
interface DatasetEvalRowResultsPanelProps {
@@ -38,7 +40,9 @@ const expectedValue = (item?: Record): string | null => {
};
const inputText = (row: DatasetEvalRow): string => {
- const rendered = row.requests?.[0]?.request?.input_message;
+ const request = row.requests?.[0]?.request;
+ // `input_message` is the pre-chat-completions body; jobs submitted then still render.
+ const rendered = request?.messages?.at(-1)?.content ?? request?.input_message;
if (typeof rendered === 'string' && rendered) return rendered;
return row.item ? JSON.stringify(row.item, null, 2) : '';
};
diff --git a/web/packages/studio/src/components/evaluation/submitEvaluationJob.test.ts b/web/packages/studio/src/components/evaluation/submitEvaluationJob.test.ts
index e9d639efef..1368c08844 100644
--- a/web/packages/studio/src/components/evaluation/submitEvaluationJob.test.ts
+++ b/web/packages/studio/src/components/evaluation/submitEvaluationJob.test.ts
@@ -5,6 +5,7 @@ import {
bareName,
buildAgentEvalRequestBody,
buildAgentTarget,
+ buildDatasetAgentTarget,
buildEvalJobName,
buildPersistedSpec,
injectJudgeModel,
@@ -48,15 +49,39 @@ describe('bareName', () => {
});
describe('buildAgentTarget', () => {
- it('targets the non-streaming /generate endpoint of the agent', () => {
+ it('targets the non-streaming chat-completions endpoint of the agent', () => {
const target = buildAgentTarget('ws-a', 'support-bot');
expect(target.kind).toBe('agent');
expect(target.agent.format).toBe('generic');
expect(target.agent.stream).toBe(false);
- expect(target.agent.response_path).toBe('$.value');
- expect(target.agent.body).toEqual({ input_message: '{{ instruction }}' });
- expect(target.agent.url).toContain('/agents/support-bot/-/generate');
- expect(target.agent.url).not.toContain('/generate/full');
+ expect(target.agent.response_path).toBe('$.choices[0].message.content');
+ expect(target.agent.body).toEqual({
+ model: 'support-bot',
+ messages: [{ role: 'user', content: '{{ instruction }}' }],
+ stream: false,
+ });
+ expect(target.agent.url).toContain('/agents/support-bot/-/v1/chat/completions');
+ });
+
+ it('strips a workspace prefix from the agent name', () => {
+ const target = buildAgentTarget('ws-a', 'ws-a/support-bot');
+ expect(target.agent.name).toBe('support-bot');
+ expect(target.agent.body.model).toBe('support-bot');
+ expect(target.agent.url).toContain('/agents/support-bot/-/');
+ });
+});
+
+describe('buildDatasetAgentTarget', () => {
+ it('renders the row prompt into the chat message', () => {
+ const target = buildDatasetAgentTarget('ws-a', 'support-bot');
+ expect(target.format).toBe('generic');
+ expect(target.response_path).toBe('$.choices[0].message.content');
+ expect(target.body).toEqual({
+ model: 'support-bot',
+ messages: [{ role: 'user', content: '{{ prompt }}' }],
+ stream: false,
+ });
+ expect(target.url).toContain('/agents/support-bot/-/v1/chat/completions');
});
});
diff --git a/web/packages/studio/src/components/evaluation/submitEvaluationJob.ts b/web/packages/studio/src/components/evaluation/submitEvaluationJob.ts
index 95aadfca1d..a3b13b7d4d 100644
--- a/web/packages/studio/src/components/evaluation/submitEvaluationJob.ts
+++ b/web/packages/studio/src/components/evaluation/submitEvaluationJob.ts
@@ -106,17 +106,23 @@ export interface SubmitSelections {
export const bareName = (value: string): string =>
value.includes('/') ? (value.split('/').pop() ?? value) : value;
-/** The generic agent target: the deployed agent's non-streaming ``/generate``. */
-export const buildAgentTarget = (workspace: string, agent: string) => ({
- kind: 'agent' as const,
- agent: {
- format: 'generic' as const,
- url: `${PLATFORM_BASE_URL}/apis/agents/v2/workspaces/${encodeURIComponent(workspace)}/agents/${encodeURIComponent(bareName(agent))}/-/generate`,
- name: bareName(agent),
- body: { input_message: '{{ instruction }}' },
- response_path: '$.value',
+/** Chat completions is the one endpoint both config formats serve, so this needs no branch. */
+const agentEndpoint = (workspace: string, agent: string, promptVar: string) => ({
+ format: 'generic' as const,
+ url: `${PLATFORM_BASE_URL}/apis/agents/v2/workspaces/${encodeURIComponent(workspace)}/agents/${encodeURIComponent(bareName(agent))}/-/v1/chat/completions`,
+ name: bareName(agent),
+ body: {
+ model: bareName(agent),
+ messages: [{ role: 'user', content: `{{ ${promptVar} }}` }],
stream: false,
},
+ response_path: '$.choices[0].message.content',
+ stream: false,
+});
+
+export const buildAgentTarget = (workspace: string, agent: string) => ({
+ kind: 'agent' as const,
+ agent: agentEndpoint(workspace, agent, 'instruction'),
params: AGENT_RUN_PARAMS,
});
@@ -178,14 +184,8 @@ export const buildAgentEvalRequestBody = (
* this is NOT wrapped in {kind, agent}: EvaluateInputSpec forbids extra keys and
* takes the agent object directly. The body renders the row-based ``prompt``
* rather than a task ``instruction``. */
-export const buildDatasetAgentTarget = (workspace: string, agent: string) => ({
- format: 'generic' as const,
- url: `${PLATFORM_BASE_URL}/apis/agents/v2/workspaces/${encodeURIComponent(workspace)}/agents/${encodeURIComponent(bareName(agent))}/-/generate`,
- name: bareName(agent),
- body: { input_message: '{{ prompt }}' },
- response_path: '$.value',
- stream: false,
-});
+export const buildDatasetAgentTarget = (workspace: string, agent: string) =>
+ agentEndpoint(workspace, agent, 'prompt');
/** Build the ``evaluate/jobs`` POST body from a dataset-driven config. ``params``
* must be exactly RunConfigOnline for an agent target, and ``prompt_template``
diff --git a/web/packages/studio/src/routes/agents/AgentEvaluationsRoute/AGENTS.md b/web/packages/studio/src/routes/agents/AgentEvaluationsRoute/AGENTS.md
index 8df57f69d6..cd324b787d 100644
--- a/web/packages/studio/src/routes/agents/AgentEvaluationsRoute/AGENTS.md
+++ b/web/packages/studio/src/routes/agents/AgentEvaluationsRoute/AGENTS.md
@@ -131,21 +131,32 @@ Submit body is wrapped: `{"spec": { ...AgentEvalInputSpec }}`.
"kind": "agent",
"agent": {
"format": "generic",
- "url": ".../agents//-/generate",
+ "url": ".../agents//-/v1/chat/completions",
"name": "",
- "body": { "input_message": "{{ instruction }}" },
- "response_path": "$.value",
+ "body": {
+ "model": "",
+ "messages": [{ "role": "user", "content": "{{ instruction }}" }],
+ "stream": false
+ },
+ "response_path": "$.choices[0].message.content",
"stream": false
}
}
```
-Use the non-streaming `/generate` endpoint. Do **not** use `/generate/full` — its per-token
-SSE stream leaves only the last token in the captured output and every score collapses to 0.
+Use the non-streaming chat-completions endpoint, which both agent config formats serve: a
+`nemo-agents-spec-v1` agent through the Platform-owned Fabric server, a `nat-workflow-v1`
+agent through NAT's FastAPI front end (`workflow.openai_api_v1_path`, on by default). NAT
+also serves the legacy `/generate`, but Fabric does not — it 404s — so the target must not
+branch on the agent's format.
-**`body` renders against the task inputs directly.** A generic agent's request is a passthrough
-of the task row, so `body` references task input fields by name — `{{ instruction }}` — not a
-chat wrapper. `instruction` is the single canonical task input.
+Do **not** use NAT's `/generate/full` — its per-token SSE stream leaves only the last token
+in the captured output and every score collapses to 0. Keep `stream: false` in the body for
+the same reason.
+
+**`body` renders against the task inputs.** `render_template` recurses into dicts and lists,
+so `{{ instruction }}` substitutes inside the nested `messages` entry. `instruction` is the
+single canonical task input; the dataset-driven target renders `{{ prompt }}` instead.
### Task
@@ -272,13 +283,22 @@ per-task bundle (trials, evidence, traces) lives in the fileset referenced by `b
dataset/fileset/taskset reference yet (planned). Large datasets must be inlined for now.
- **Agent must be deployed and running before submit** — a not-yet-ready agent connection
fails the job.
-- **Use `/generate`, not `/generate/full`** (per-token SSE zeroes the score).
+- **Use `/-/v1/chat/completions`, never `/generate`.** Only NAT serves `/generate`; a Fabric
+ (`nemo-agents-spec-v1`) agent 404s on it, and Fabric is what `nemo-build-agent` and Studio's
+ Create Example Agent produce. Chat completions is the one shape both formats serve.
+- **Never use `/generate/full`** (per-token SSE zeroes the score).
- **Run tasks serially (`max_concurrent_tasks: 1`) by default.** NAT currently reports workflow
failures such as output truncation as **422**; `422` is not retried, so one failure kills the
whole job. Serial execution is conservative but does not fix truncation. Configure an adequate
agent output budget, or set `target.params.ignore_request_failure: true` to accept `NaN` trials.
-- **`body` uses `{{ instruction }}`, not a `messages` wrapper** — a generic agent's request is a
- task-row passthrough with no `messages` key to index.
+- **`{{ instruction }}` is the task input, wherever it sits in `body`.** The template variable
+ names a task-row field, not a chat field — it happens to be rendered inside the `messages`
+ wrapper the agent's endpoint expects. `render_template` recurses through dicts and lists, so
+ nesting it is fine.
+- **Every eval request opens a new Fabric session.** Fabric starts a fresh runtime per
+ chat-completions call that carries no `X-Nemo-Session-Id`, and the evaluator sends none.
+ Sessions are reclaimed only by the 30-minute idle sweep, so a long task list leaves that many
+ runtimes alive and pays a cold start per task.
---
@@ -302,8 +322,13 @@ Shape (reference only):
],
"target": {
"format": "generic",
- "url": ".../-/generate",
- "response_path": "$.value",
+ "url": ".../-/v1/chat/completions",
+ "body": {
+ "model": "",
+ "messages": [{ "role": "user", "content": "{{ prompt }}" }],
+ "stream": false
+ },
+ "response_path": "$.choices[0].message.content",
"stream": false
},
"prompt_template": { "messages": [{ "role": "user", "content": "{{ item. }}" }] },
diff --git a/web/packages/studio/src/routes/agents/AgentEvaluationsRoute/components/submitEvaluationSpec.ts b/web/packages/studio/src/routes/agents/AgentEvaluationsRoute/components/submitEvaluationSpec.ts
deleted file mode 100644
index 51bac8b05f..0000000000
--- a/web/packages/studio/src/routes/agents/AgentEvaluationsRoute/components/submitEvaluationSpec.ts
+++ /dev/null
@@ -1,196 +0,0 @@
-// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
-// SPDX-License-Identifier: Apache-2.0
-
-import { FILESET_NAME_MAX_LENGTH, toValidFilesetName } from '@nemo/common/src/utils/filesetName';
-import { generateDefaultName } from '@nemo/common/src/utils/generateDefaultName';
-import { PLATFORM_BASE_URL } from '@studio/constants/environment';
-
-/** Sentinel ``evalConfig`` value that switches the form into create mode. */
-export const CREATE_NEW = '__create_new__';
-
-export const MODE_DEFAULT = 'default';
-export const MODE_FILESET = 'fileset';
-
-/** Suggested name for a new eval-config fileset (e.g. "wise-blue"). */
-export const generateEvalConfigName = (): string => generateDefaultName({ length: 2 });
-
-/** Default parallelism for a submitted eval (Studio default; the config value is a hint). */
-export const DEFAULT_MAX_CONCURRENT_TASKS = 1;
-
-export const buildEvalJobName = (filesetName: string): string => {
- const suffix = Math.random().toString(36).slice(2, 10).padEnd(8, '0');
- const base = toValidFilesetName(filesetName)
- .slice(0, FILESET_NAME_MAX_LENGTH - suffix.length - 1)
- .replace(/-+$/, '');
- return `${base}-${suffix}`;
-};
-
-// ---------------------------------------------------------------------------
-// eval-config.json shape (stored in a fileset, read at submit)
-// ---------------------------------------------------------------------------
-
-/** One inline metric bundle as stored in eval-config.json (no judge_model —
- * it is injected at submit). Kept loose: Studio does not re-validate the
- * built-in metric shape, it only injects the model and fans it onto tasks. */
-export interface InlineMetricBundle {
- bundle_kind: string;
- bundle_format_version: string;
- metric_type: string;
- metadata?: Record;
- outputs?: unknown[];
- secrets?: Record;
- payload: {
- kind: 'inline';
- metric: Record & { model?: unknown };
- };
-}
-
-export interface EvalConfigTask {
- id: string;
- intent: string;
- inputs?: { instruction?: string | null };
- reference?: Record;
-}
-
-/** The example template: inline tasks + one shared metric (metric not yet fanned). */
-export interface EvalConfig {
- tasks: EvalConfigTask[];
- metric: InlineMetricBundle;
- max_concurrent_tasks?: number;
-}
-
-/** A task with the shared metric fanned onto it (judge baked in). */
-export type EvalSpecTask = EvalConfigTask & { metrics: InlineMetricBundle[] };
-
-/** The persisted yardstick stored in a fileset: tasks-with-metrics, no target.
- * An `AgentEvalInputSpec` minus `target` — submit injects the per-run agent. */
-export interface PersistedEvalSpec {
- tasks: EvalSpecTask[];
- max_concurrent_tasks?: number;
-}
-
-// ---------------------------------------------------------------------------
-// Submit-time selections + request assembly
-// ---------------------------------------------------------------------------
-
-export interface SubmitSelections {
- workspace: string;
- /** Agent (bare name) to evaluate; used to build the generic target. */
- agent: string;
- /** Eval-config fileset name, stored under spec.labels.eval_config_fileset for display. */
- filesetName?: string;
-}
-
-/** Strip an optional ``workspace/`` prefix, returning the bare model/agent name. */
-export const bareName = (value: string): string =>
- value.includes('/') ? (value.split('/').pop() ?? value) : value;
-
-/** The generic agent target: the deployed agent's non-streaming ``/generate``. */
-export const buildAgentTarget = (workspace: string, agent: string) => ({
- kind: 'agent' as const,
- agent: {
- format: 'generic' as const,
- url: `${PLATFORM_BASE_URL}/apis/agents/v2/workspaces/${encodeURIComponent(workspace)}/agents/${encodeURIComponent(bareName(agent))}/-/generate`,
- name: bareName(agent),
- body: { input_message: '{{ instruction }}' },
- response_path: '$.value',
- stream: false,
- },
-});
-
-/** Set the metric's judge model to a ``workspace/name`` ModelRef (resolved to a
- * reachable Model server-side). Does not mutate input. */
-export const injectJudgeModel = (
- metric: InlineMetricBundle,
- judgeModel: string
-): InlineMetricBundle => ({
- ...metric,
- payload: {
- ...metric.payload,
- metric: { ...metric.payload.metric, model: judgeModel },
- },
-});
-
-/** Fan the shared metric onto every task. A judge model is injected only when
- * one is supplied; otherwise the template metric's own model is kept as-is. */
-export const fanMetricOntoTasks = (
- config: EvalConfig,
- judgeModel: string | null
-): EvalSpecTask[] => {
- const metric = judgeModel ? injectJudgeModel(config.metric, judgeModel) : config.metric;
- return config.tasks.map((task) => ({ ...task, metrics: [metric] }));
-};
-
-/** Build the persisted yardstick from an example template: fan the shared metric
- * (judge baked in) onto every task. This is what gets stored in the fileset. */
-export const buildPersistedSpec = (
- config: EvalConfig,
- judgeModel: string | null
-): PersistedEvalSpec => ({
- tasks: fanMetricOntoTasks(config, judgeModel),
- max_concurrent_tasks: config.max_concurrent_tasks ?? DEFAULT_MAX_CONCURRENT_TASKS,
-});
-
-/** Build the ``agent-evaluate/jobs`` POST body from a persisted spec + selections. */
-export const buildAgentEvalRequestBody = (
- spec: PersistedEvalSpec,
- selections: SubmitSelections
-) => ({
- ...(selections.filesetName ? { name: buildEvalJobName(selections.filesetName) } : {}),
- spec: {
- tasks: spec.tasks,
- target: buildAgentTarget(selections.workspace, selections.agent),
- max_concurrent_tasks: spec.max_concurrent_tasks ?? DEFAULT_MAX_CONCURRENT_TASKS,
- ...(selections.filesetName ? { labels: { eval_config_fileset: selections.filesetName } } : {}),
- },
-});
-
-/** Parse an example template blob, validating the minimal required shape. */
-export const parseEvalConfig = (text: string): EvalConfig => {
- const parsed = JSON.parse(text) as Partial;
- if (!Array.isArray(parsed.tasks) || parsed.tasks.length === 0) {
- throw new Error('eval-config.json must contain a non-empty "tasks" array');
- }
- if (!parsed.metric || typeof parsed.metric !== 'object') {
- throw new Error('eval-config.json must contain a "metric"');
- }
-
- const { payload } = parsed.metric;
- if (
- !payload ||
- typeof payload !== 'object' ||
- !payload.metric ||
- typeof payload.metric !== 'object'
- ) {
- throw new Error('eval-config.json "metric" must contain a "payload.metric" object');
- }
- return {
- tasks: parsed.tasks,
- metric: parsed.metric,
- max_concurrent_tasks: parsed.max_concurrent_tasks,
- };
-};
-
-/** Parse a persisted yardstick spec (the reuse path): tasks each carry their own
- * metrics, no top-level ``metric``. Submitted as-is with only a target injected. */
-export const parsePersistedSpec = (text: string): PersistedEvalSpec => {
- const parsed = JSON.parse(text) as Partial;
- if (!Array.isArray(parsed.tasks) || parsed.tasks.length === 0) {
- throw new Error('eval-config.json must contain a non-empty "tasks" array');
- }
- for (const task of parsed.tasks) {
- if (!Array.isArray(task.metrics) || task.metrics.length === 0) {
- throw new Error('eval-config.json every task must contain a non-empty "metrics" array');
- }
- const payload = task.metrics[0]?.payload;
- if (
- !payload ||
- typeof payload !== 'object' ||
- !payload.metric ||
- typeof payload.metric !== 'object'
- ) {
- throw new Error('eval-config.json task metric must contain a "payload.metric" object');
- }
- }
- return { tasks: parsed.tasks, max_concurrent_tasks: parsed.max_concurrent_tasks };
-};