diff --git a/AGENTS.md b/AGENTS.md index 50e9cd1..2b28045 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -144,6 +144,8 @@ Two silent-failure modes used to leak a fabricated composite (goal 0 + default c - **Empty-work signature**: a run that produced no transcript AND no result. `hasEmptyOutput()` in `types/output.ts` defines it; `isFailedRun()` treats it as failed even on a clean exit; the base adapter (`agent-adapter.ts`) stamps `"Agent produced no output"` so the error column is descriptive. - **Judge death / unparseable judge response**: `callJudge()` throws `ScoringError` when the judge invocation failed or returned nothing; `goal-achievement.ts` and `deep-eval.ts` throw `ScoringError` when a judge response can't be parsed at all (a parsed-but-incomplete response still gets per-item defaults). `scoreRunResult()` catches `ScoringError`, stamps `"Score withheld: …"` on the run, and returns a zero/failed result. Non-`ScoringError` throws propagate (real bugs should fail loud). +Before withholding, `callJudgeAndParse()` (`scoring/judge-retry.ts`) retries the judge call + parse up to `judging.retries` times (default 0) on `ScoringError` only; `judging.timeout_minutes` becomes `AgentInput.timeoutMs` for every judge call (default: the adapter's 10 min). It lives in its own module so tests that mock `judge.js`'s `callJudge` still drive it. + `buildScoredOutput()` recomputes `completed`/`failed` from the scored results, so a run that scoring flips to failed is reflected in the summary and exit code. The withheld path also sets `score.withheld = true`. Multi-run aggregation depends on that flag to keep a judge outage out of the agent's reliability denominator, so a new withhold site must set it rather than relying on the stamped error message. diff --git a/src/cli.ts b/src/cli.ts index 223193a..d9c24e7 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -10,6 +10,7 @@ import { run, buildJobEnv } from "./runner/runner.js"; import { runLifecyclePhase } from "./runner/lifecycle.js"; import { loadConfig } from "./config/loader.js"; import { scoreRunResult, buildScoredOutput } from "./scoring/index.js"; +import { judgeLimitsFromConfig } from "./scoring/judge.js"; import { initReport, finalizeReport } from "./reports/writer.js"; import { listReports, readReport, readScenarioResults } from "./reports/reader.js"; import { setBaseline, readBaseline, listBaselines, deleteBaseline, DEFAULT_BASELINE_NAME } from "./baselines/store.js"; @@ -410,6 +411,7 @@ async function runPipelineBody( // AgentConfig. The list is preserved so the scoring layer can pick the // best-matching judge per run (skip the run's own agent when possible). const judging = config.judging?.agents.filter((a): a is AgentConfig => typeof a === "object"); + const judgeLimits = judgeLimitsFromConfig(config.judging); let runOutput: RunOutput | undefined; let output: ScoredOutput | RunOutput | undefined; @@ -441,6 +443,7 @@ async function runPipelineBody( logger, reportDir, judging, + judgeLimits, onProgress: (event) => { if (event.phase === "start") onScoringStart?.(event); }, @@ -476,6 +479,7 @@ async function runPipelineBody( logger, reportDir, judging, + judgeLimits, }).catch((err) => { logger.error(`Scoring fallback failed for ${r.scenarioKey} (${r.agentName}): ${formatError(err)}`); return undefined; diff --git a/src/config/validator.ts b/src/config/validator.ts index 8e3da4c..7cac8e9 100644 --- a/src/config/validator.ts +++ b/src/config/validator.ts @@ -140,6 +140,16 @@ function validateJudging(data: unknown, filePath: string): void { } throw new Error(`Invalid config at ${filePath}: "judging.agents[${i}]" must be a string or an AgentConfig object`); } + if (obj.timeout_minutes !== undefined) { + if (typeof obj.timeout_minutes !== "number" || !(obj.timeout_minutes > 0)) { + throw new Error(`Invalid config at ${filePath}: "judging.timeout_minutes" must be a positive number`); + } + } + if (obj.retries !== undefined) { + if (typeof obj.retries !== "number" || !Number.isInteger(obj.retries) || obj.retries < 0 || obj.retries > 5) { + throw new Error(`Invalid config at ${filePath}: "judging.retries" must be a whole number from 0 to 5`); + } + } } function validateScenariosField(data: unknown, filePath: string): void { diff --git a/src/docs-site/src/pages/configuration.astro b/src/docs-site/src/pages/configuration.astro index fab5ff8..96242dc 100644 --- a/src/docs-site/src/pages/configuration.astro +++ b/src/docs-site/src/pages/configuration.astro @@ -264,7 +264,9 @@ export default async () => {
timeout_minutes and retries (see
+ Judge timeout and retries).
+ A judge that dies or replies with something AXIS can't parse + withholds the run's score. Two optional fields make that rarer: +
+| Field | +Default | +Description | +
|---|---|---|
timeout_minutes |
+ adapter default (10) | ++ Time limit for each judge invocation. Raise it when judges verify large workspaces and + time out. + | +
retries |
+ 0 |
+
+ Extra attempts (0–5) when a judge invocation fails (crash, timeout, rate limit) or its
+ reply can't be parsed. Each retry is a full judge call and costs as much as the first.
+ Retries are logged with --verbose; a score withheld after retries says how
+ many attempts were made.
+ |
+
{`{
+ "judging": {
+ "agents": ["claude-code"],
+ "timeout_minutes": 15,
+ "retries": 1
+ }
+}`}
diff --git a/src/docs-site/src/pages/scoring.astro b/src/docs-site/src/pages/scoring.astro index fc8bee3..e134e43 100644 --- a/src/docs-site/src/pages/scoring.astro +++ b/src/docs-site/src/pages/scoring.astro @@ -518,6 +518,12 @@ import DocsLayout from "../layouts/DocsLayout.astro"; Withholding keeps a transient infrastructure failure from polluting your gate: the run fails loudly and can be retried, instead of quietly dragging an average down with a fabricated number.
+
+ To retry a dying or unparseable judge before withholding, and to give judges longer than the
+ adapter's 10-minute default, set judging.retries and
+ judging.timeout_minutes (see
+ Judge timeout and retries).
+