diff --git a/AGENTS.md b/AGENTS.md index 50e9cd1..2b28045 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -144,6 +144,8 @@ Two silent-failure modes used to leak a fabricated composite (goal 0 + default c - **Empty-work signature**: a run that produced no transcript AND no result. `hasEmptyOutput()` in `types/output.ts` defines it; `isFailedRun()` treats it as failed even on a clean exit; the base adapter (`agent-adapter.ts`) stamps `"Agent produced no output"` so the error column is descriptive. - **Judge death / unparseable judge response**: `callJudge()` throws `ScoringError` when the judge invocation failed or returned nothing; `goal-achievement.ts` and `deep-eval.ts` throw `ScoringError` when a judge response can't be parsed at all (a parsed-but-incomplete response still gets per-item defaults). `scoreRunResult()` catches `ScoringError`, stamps `"Score withheld: …"` on the run, and returns a zero/failed result. Non-`ScoringError` throws propagate (real bugs should fail loud). +Before withholding, `callJudgeAndParse()` (`scoring/judge-retry.ts`) retries the judge call + parse up to `judging.retries` times (default 0) on `ScoringError` only; `judging.timeout_minutes` becomes `AgentInput.timeoutMs` for every judge call (default: the adapter's 10 min). It lives in its own module so tests that mock `judge.js`'s `callJudge` still drive it. + `buildScoredOutput()` recomputes `completed`/`failed` from the scored results, so a run that scoring flips to failed is reflected in the summary and exit code. The withheld path also sets `score.withheld = true`. Multi-run aggregation depends on that flag to keep a judge outage out of the agent's reliability denominator, so a new withhold site must set it rather than relying on the stamped error message. diff --git a/src/cli.ts b/src/cli.ts index 223193a..d9c24e7 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -10,6 +10,7 @@ import { run, buildJobEnv } from "./runner/runner.js"; import { runLifecyclePhase } from "./runner/lifecycle.js"; import { loadConfig } from "./config/loader.js"; import { scoreRunResult, buildScoredOutput } from "./scoring/index.js"; +import { judgeLimitsFromConfig } from "./scoring/judge.js"; import { initReport, finalizeReport } from "./reports/writer.js"; import { listReports, readReport, readScenarioResults } from "./reports/reader.js"; import { setBaseline, readBaseline, listBaselines, deleteBaseline, DEFAULT_BASELINE_NAME } from "./baselines/store.js"; @@ -410,6 +411,7 @@ async function runPipelineBody( // AgentConfig. The list is preserved so the scoring layer can pick the // best-matching judge per run (skip the run's own agent when possible). const judging = config.judging?.agents.filter((a): a is AgentConfig => typeof a === "object"); + const judgeLimits = judgeLimitsFromConfig(config.judging); let runOutput: RunOutput | undefined; let output: ScoredOutput | RunOutput | undefined; @@ -441,6 +443,7 @@ async function runPipelineBody( logger, reportDir, judging, + judgeLimits, onProgress: (event) => { if (event.phase === "start") onScoringStart?.(event); }, @@ -476,6 +479,7 @@ async function runPipelineBody( logger, reportDir, judging, + judgeLimits, }).catch((err) => { logger.error(`Scoring fallback failed for ${r.scenarioKey} (${r.agentName}): ${formatError(err)}`); return undefined; diff --git a/src/config/validator.ts b/src/config/validator.ts index 8e3da4c..7cac8e9 100644 --- a/src/config/validator.ts +++ b/src/config/validator.ts @@ -140,6 +140,16 @@ function validateJudging(data: unknown, filePath: string): void { } throw new Error(`Invalid config at ${filePath}: "judging.agents[${i}]" must be a string or an AgentConfig object`); } + if (obj.timeout_minutes !== undefined) { + if (typeof obj.timeout_minutes !== "number" || !(obj.timeout_minutes > 0)) { + throw new Error(`Invalid config at ${filePath}: "judging.timeout_minutes" must be a positive number`); + } + } + if (obj.retries !== undefined) { + if (typeof obj.retries !== "number" || !Number.isInteger(obj.retries) || obj.retries < 0 || obj.retries > 5) { + throw new Error(`Invalid config at ${filePath}: "judging.retries" must be a whole number from 0 to 5`); + } + } } function validateScenariosField(data: unknown, filePath: string): void { diff --git a/src/docs-site/src/pages/configuration.astro b/src/docs-site/src/pages/configuration.astro index fab5ff8..96242dc 100644 --- a/src/docs-site/src/pages/configuration.astro +++ b/src/docs-site/src/pages/configuration.astro @@ -264,7 +264,9 @@ export default async () => { No Precedence-ordered list of judge agents for scoring. See Judging - Agents. When omitted, each run is judged by its own agent. + Agents. When omitted, each run is judged by its own agent. Optional + timeout_minutes and retries (see + Judge timeout and retries). @@ -868,6 +870,47 @@ curl -X POST "$SLACK_WEBHOOK" \\ ] } }`} +

Judge timeout and retries

+

+ A judge that dies or replies with something AXIS can't parse + withholds the run's score. Two optional fields make that rarer: +

+ + + + + + + + + + + + + + + + + + + + +
FieldDefaultDescription
timeout_minutesadapter default (10) + Time limit for each judge invocation. Raise it when judges verify large workspaces and + time out. +
retries0 + Extra attempts (0–5) when a judge invocation fails (crash, timeout, rate limit) or its + reply can't be parsed. Each retry is a full judge call and costs as much as the first. + Retries are logged with --verbose; a score withheld after retries says how + many attempts were made. +
+
{`{
+  "judging": {
+    "agents": ["claude-code"],
+    "timeout_minutes": 15,
+    "retries": 1
+  }
+}`}

Skills

diff --git a/src/docs-site/src/pages/scoring.astro b/src/docs-site/src/pages/scoring.astro index fc8bee3..e134e43 100644 --- a/src/docs-site/src/pages/scoring.astro +++ b/src/docs-site/src/pages/scoring.astro @@ -518,6 +518,12 @@ import DocsLayout from "../layouts/DocsLayout.astro"; Withholding keeps a transient infrastructure failure from polluting your gate: the run fails loudly and can be retried, instead of quietly dragging an average down with a fabricated number.

+

+ To retry a dying or unparseable judge before withholding, and to give judges longer than the + adapter's 10-minute default, set judging.retries and + judging.timeout_minutes (see + Judge timeout and retries). +