diff --git a/docs/source/environments/tbench2.md b/docs/source/environments/tbench2.md index ee2132e5e4..8c642a6c29 100644 --- a/docs/source/environments/tbench2.md +++ b/docs/source/environments/tbench2.md @@ -65,6 +65,13 @@ docker build -t tbench2-env:latest -f envs/tbench2_env/server/Dockerfile . | `action_type` | str | The action type that produced this observation | | `info` | dict | Additional metadata | +**Execution budgets in the `reset` observation.** `reset` reports the time budgets that bound a single env op in `info`, so clients can derive per-message deadlines (budget plus margin) instead of guessing: + +| Key | Modes | Description | +|-----|-------|-------------| +| `verifier_timeout_sec` | local, Docker | Budget for one `evaluate` call: the task's `task.toml` `[verifier].timeout_sec`, else 900. | +| `command_timeout_s` | local only | Per-command budget for `exec` (`TB2_COMMAND_TIMEOUT_S`). Docker mode has no server-side per-command timeout, so it does not report one. | + ### State **Tbench2State**: Server-side state for the task session diff --git a/envs/tbench2_env/README.md b/envs/tbench2_env/README.md index d2d4a2a38b..03716b5db1 100644 --- a/envs/tbench2_env/README.md +++ b/envs/tbench2_env/README.md @@ -79,6 +79,13 @@ docker build -t tbench2-env:latest -f envs/tbench2_env/server/Dockerfile . | `action_type` | str | The action type that produced this observation | | `info` | dict | Additional metadata | +**Execution budgets in the `reset` observation.** `reset` reports the time budgets that bound a single env op in `info`, so clients can derive per-message deadlines (budget plus margin) instead of guessing: + +| Key | Modes | Description | +|-----|-------|-------------| +| `verifier_timeout_sec` | local, Docker | Budget for one `evaluate` call: the task's `task.toml` `[verifier].timeout_sec`, else 900. | +| `command_timeout_s` | local only | Per-command budget for `exec` (`TB2_COMMAND_TIMEOUT_S`). Docker mode has no server-side per-command timeout, so it does not report one. | + ### State **Tbench2State**: Server-side state for the task session diff --git a/envs/tbench2_env/server/tbench2_env_environment.py b/envs/tbench2_env/server/tbench2_env_environment.py index f731b1117f..b7dffefeb3 100644 --- a/envs/tbench2_env/server/tbench2_env_environment.py +++ b/envs/tbench2_env/server/tbench2_env_environment.py @@ -10,6 +10,7 @@ import io import logging +import math import os import re import shlex @@ -153,6 +154,11 @@ def _workdir_is_server_tree(workdir: str) -> bool: return True # unresolvable path: fail toward the task-dir fallback +# Verifier budget when task.toml declares no [verifier].timeout_sec. Shared by +# reset() (which reports it) and evaluation (which enforces it). +_DEFAULT_VERIFIER_TIMEOUT_S = 900.0 + + def _read_timeout(task_dir: Path, fallback: float) -> float: task_toml = task_dir / "task.toml" if not task_toml.exists(): @@ -162,7 +168,13 @@ def _read_timeout(task_dir: Path, fallback: float) -> float: except Exception: return fallback verifier = data.get("verifier", {}) - return float(verifier.get("timeout_sec", fallback)) + if not isinstance(verifier, dict): + return fallback + try: + timeout_s = float(verifier.get("timeout_sec", fallback)) + except (TypeError, ValueError, OverflowError): + return fallback + return timeout_s if math.isfinite(timeout_s) and timeout_s > 0 else fallback # The scoring exec echoes its verdict on a marker line so the caller can parse @@ -185,14 +197,14 @@ def _canonical_eval_cmd(workdir: str, timeout_s: float | None = None) -> str: then, and callers need the pytest diagnostics on failure; the reward marker line stays last for parsing. - ``timeout_s`` bounds test.sh with coreutils ``timeout`` when present in - the image. The local mode passes None: its terminal toolkit enforces the + ``timeout_s`` bounds test.sh with coreutils ``timeout`` (required in the + image). The local mode passes None: its terminal toolkit enforces the budget itself. Docker exec has no server-side timeout, so the task's own verifier budget is enforced in-shell. """ run = f"bash {_VERIFY_TESTS_DIR}/test.sh" if timeout_s is not None: - run = f"if command -v timeout >/dev/null 2>&1; then timeout {int(timeout_s)} {run}; else {run}; fi" + run = f"timeout {timeout_s:g} {run}" return ( f"cd {shlex.quote(workdir)} && " f"{run} > {_VERIFIER_LOG_DIR}/testsh.log 2>&1; " @@ -233,20 +245,21 @@ def _require_canonical_verdict(reward: float | None, output: str) -> float: return reward -def _fallback_eval_cmd(workdir: str) -> str: +def _fallback_eval_cmd(workdir: str, timeout_s: float | None = None) -> str: """pytest against the staged tests copy, for task dirs without the canonical harness (none of the 89 official TB2 tasks — they all ship test.sh — but custom task dirs may only have bare pytest tests). Prefer uvx so pytest comes with its own toolchain like the canonical harness does. Verify from the same directory the agent worked in. """ - return ( - f"cd {shlex.quote(workdir)} && " + run = ( "if command -v uvx >/dev/null 2>&1; " f"then uvx --with pytest==8.4.1 pytest -q {_VERIFY_TESTS_DIR} -rA; " - f"else python -m pytest -q {_VERIFY_TESTS_DIR} -rA; fi; " - f"echo {_EXIT_CODE_MARKER}$?" + f"else python -m pytest -q {_VERIFY_TESTS_DIR} -rA; fi" ) + if timeout_s is not None: + run = f"timeout {timeout_s:g} bash -c {shlex.quote(run)}" + return f"cd {shlex.quote(workdir)} && {run}; echo {_EXIT_CODE_MARKER}$?" def _parse_exit_code_marker(output: str) -> int: @@ -298,6 +311,7 @@ def __init__( self._state = Tbench2State() self._task_dir: Path | None = None + self._verifier_timeout_s = _DEFAULT_VERIFIER_TIMEOUT_S self._terminal_toolkit = None self._instruction = "" self._workdir = "" @@ -356,6 +370,9 @@ def reset( session_logs_dir=session_logs_dir, safe_mode=self.safe_mode, ) + self._verifier_timeout_s = _read_timeout( + task_dir, fallback=_DEFAULT_VERIFIER_TIMEOUT_S + ) self._state = Tbench2State( episode_id=episode_id or str(uuid4()), @@ -374,7 +391,10 @@ def reset( task_path=str(task_dir), session_id=None, action_type="reset", - info={}, + info={ + "verifier_timeout_sec": self._verifier_timeout_s, + "command_timeout_s": self.command_timeout_s, + }, reward=0.0, done=False, ) @@ -479,6 +499,7 @@ def state(self) -> Tbench2State: def close(self) -> None: self._terminal_toolkit = None self._task_dir = None + self._verifier_timeout_s = _DEFAULT_VERIFIER_TIMEOUT_S self._instruction = "" def _resolve_task_path(self, task_id: str | None, task_path: str | None) -> Path: @@ -583,9 +604,8 @@ def _evaluate_task(self) -> tuple[str, float, dict[str, Any]]: if self._terminal_toolkit is None: raise RuntimeError("Terminal toolkit not initialized.") - # The task's own verifier budget (task.toml [verifier].timeout_sec) — - # heavy tests legitimately run minutes (circuit-fibsqrt declares 3600s). - verifier_timeout_s = _read_timeout(self._task_dir, fallback=900.0) + # Honor the reset-time budget even if the agent edits task.toml. + verifier_timeout_s = self._verifier_timeout_s with self._CANONICAL_EVAL_LOCK: try: @@ -708,6 +728,7 @@ def __init__( self._state = Tbench2State() self._task_dir: Path | None = None + self._verifier_timeout_s = _DEFAULT_VERIFIER_TIMEOUT_S self._docker_client = None self._container = None self._instruction = "" @@ -789,6 +810,9 @@ def reset( except Exception: self.close() raise + self._verifier_timeout_s = _read_timeout( + task_dir, fallback=_DEFAULT_VERIFIER_TIMEOUT_S + ) return Tbench2Observation( instruction=self._instruction, @@ -799,7 +823,10 @@ def reset( task_path=str(task_dir), session_id=None, action_type="reset", - info={"docker_image": self._task_image}, + info={ + "docker_image": self._task_image, + "verifier_timeout_sec": self._verifier_timeout_s, + }, reward=0.0, done=False, ) @@ -1027,9 +1054,7 @@ def _evaluate_docker(self) -> tuple[str, float, dict[str, Any]]: {"tests_passed": False, "error": "missing tests"}, ) - # The task's own verifier budget (task.toml [verifier].timeout_sec) — - # heavy tests legitimately run minutes (circuit-fibsqrt declares 3600s). - verifier_timeout_s = _read_timeout(self._task_dir, fallback=900.0) + verifier_timeout_s = self._verifier_timeout_s workdir = self._workdir or "/task" wipe_ec, wipe_out = self._exec_in_container( @@ -1054,7 +1079,9 @@ def _evaluate_docker(self) -> tuple[str, float, dict[str, Any]]: ) info = {"tests_passed": reward == 1.0, "harness": "tests/test.sh"} else: - _, output = self._exec_in_container(_fallback_eval_cmd(workdir)) + _, output = self._exec_in_container( + _fallback_eval_cmd(workdir, timeout_s=verifier_timeout_s) + ) exit_code = _parse_exit_code_marker(output) reward = 1.0 if exit_code == 0 else 0.0 info = {"tests_passed": exit_code == 0, "exit_code": exit_code} @@ -1097,6 +1124,7 @@ def close(self) -> None: pass self._container = None self._task_dir = None + self._verifier_timeout_s = _DEFAULT_VERIFIER_TIMEOUT_S self._instruction = "" self._workdir = "" diff --git a/tests/envs/test_tbench2_env.py b/tests/envs/test_tbench2_env.py index d9ce71dd22..17956b8668 100644 --- a/tests/envs/test_tbench2_env.py +++ b/tests/envs/test_tbench2_env.py @@ -1,6 +1,8 @@ import io +import json import logging import os +import subprocess import sys import tarfile from pathlib import Path @@ -62,6 +64,82 @@ def test_tbench2_reset_uses_default_task_id(monkeypatch, tmp_path: Path): assert "terminal task" in observation.instruction +def test_tbench2_reset_reports_execution_budgets(monkeypatch, tmp_path: Path): + """reset() surfaces the budgets that bound a single env op, so clients can + derive per-message deadlines from the server instead of guessing.""" + task_dir = tmp_path / "budget-task" + task_dir.mkdir() + (task_dir / "instruction.md").write_text("do it\n") + (task_dir / "task.toml").write_text("[verifier]\ntimeout_sec = 3600\n") + + monkeypatch.setattr( + tbench2_env_environment, + "_require_terminal_toolkit", + lambda: _FakeTerminalToolkit, + ) + env = Tbench2Environment( + tasks_dir=str(tmp_path), + output_dir=str(tmp_path / "runs"), + command_timeout_s=42.0, + default_task_id="budget-task", + ) + + observation = env.reset() + + assert observation.info["verifier_timeout_sec"] == 3600.0 + assert observation.info["command_timeout_s"] == 42.0 + + +def test_tbench2_reset_budget_falls_back_without_task_toml(monkeypatch, tmp_path: Path): + """A task with no [verifier] budget reports the same fallback that + evaluate enforces, not a missing key.""" + task_dir = tmp_path / "plain-task" + task_dir.mkdir() + (task_dir / "instruction.md").write_text("do it\n") + + env = _make_env(monkeypatch, tmp_path, "plain-task") + + observation = env.reset() + + assert ( + observation.info["verifier_timeout_sec"] + == tbench2_env_environment._DEFAULT_VERIFIER_TIMEOUT_S + ) + + +@pytest.mark.parametrize( + "verifier_config", + [ + f"[verifier]\ntimeout_sec = {value}\n" + for value in ['"slow"', "[1, 2, 3]", "nan", "inf", "-inf", "0", "-1"] + ] + + ['verifier = "slow"\n', "verifier = [1, 2, 3]\n"], +) +@pytest.mark.parametrize("docker_mode", [False, True]) +def test_reset_invalid_verifier_budget_uses_finite_default( + monkeypatch, tmp_path: Path, verifier_config: str, docker_mode: bool +): + task = _make_task_dir(tmp_path) + (task / "task.toml").write_text( + verifier_config + '\n[environment]\ndocker_image = "debian:12"\n' + ) + if docker_mode: + monkeypatch.setattr( + Tbench2DockerEnvironment, "_start_container", lambda self, *args: None + ) + env = Tbench2DockerEnvironment(tasks_dir=str(tmp_path)) + else: + env = _make_env(monkeypatch, tmp_path, task.name) + + observation = env.reset(task_id=task.name) + + assert observation.info["verifier_timeout_sec"] == 900.0 + assert ( + json.loads(observation.model_dump_json())["info"]["verifier_timeout_sec"] + == 900.0 + ) + + def _make_env(monkeypatch, tmp_path: Path, task_id: str) -> Tbench2Environment: monkeypatch.setattr( tbench2_env_environment, @@ -216,6 +294,67 @@ def shell_exec(self, **kwargs): return self.output +@pytest.mark.parametrize("docker_mode", [False, True]) +@pytest.mark.parametrize("canonical", [False, True]) +@pytest.mark.parametrize( + "timeout_value, expected_timeout", [("37.5", 37.5), ("nan", 900.0), (None, 900.0)] +) +def test_verifier_budget_frozen_until_next_reset( + monkeypatch, + tmp_path: Path, + staged_paths, + docker_mode, + canonical, + timeout_value, + expected_timeout, +): + task = _make_task_dir(tmp_path) + if not canonical: + (task / "tests" / "test.sh").unlink() + original_config = (task / "task.toml").read_text() + if timeout_value is not None: + (task / "task.toml").write_text( + original_config + f"\n[verifier]\ntimeout_sec = {timeout_value}\n" + ) + output = "__TB2_REWARD__:1" if canonical else "__TB2_EXIT_CODE__:0" + if docker_mode: + monkeypatch.setattr( + Tbench2DockerEnvironment, + "_start_container", + lambda self, *args: setattr( + self, "_container", _FakeContainer(output.encode()) + ), + ) + env = Tbench2DockerEnvironment(tasks_dir=str(tmp_path)) + else: + toolkit = _RecordingToolkit(output) + monkeypatch.setattr( + tbench2_env_environment, + "_require_terminal_toolkit", + lambda: lambda **kwargs: toolkit, + ) + env = Tbench2Environment(tasks_dir=str(tmp_path), withhold_tests=False) + + observation = env.reset(task_id=task.name) + assert observation.info["verifier_timeout_sec"] == expected_timeout + (task / "task.toml").write_text( + original_config + "\n[verifier]\ntimeout_sec = 7200\n" + ) + + for reset_again, budget in enumerate((expected_timeout, 7200.0)): + if reset_again: + observation = env.reset(task_id=task.name) + assert observation.info["verifier_timeout_sec"] == budget + if docker_mode: + _, reward, _ = env._evaluate_docker() + command = env._container.events[-2][1] + assert f"timeout {budget:g} " in command + else: + _, reward, _ = env._evaluate_task() + assert toolkit.calls[-1]["timeout"] == budget + assert reward == 1.0 + + def test_evaluate_canonical_from_withheld_copy(tmp_path: Path, staged_paths): """Full evaluate flow: staging from memory + canonical test.sh scoring.""" stage_tests, stage_logs = staged_paths @@ -418,12 +557,16 @@ def test_evaluate_docker_missing_verdict_raises_not_zero(tmp_path: Path): assert "rm -rf /tests /logs/verifier" in container.events[-1][1] -def test_evaluate_docker_stages_tests_at_verify(tmp_path: Path): +@pytest.mark.parametrize("timeout_s", [900.0, 0.5]) +def test_evaluate_docker_stages_tests_at_verify(tmp_path: Path, timeout_s: float): task = _make_task_dir(tmp_path) + with (task / "task.toml").open("a") as handle: + handle.write(f"\n[verifier]\ntimeout_sec = {timeout_s}\n") env = Tbench2DockerEnvironment() container = _FakeContainer() env._container = container env._task_dir = task + env._verifier_timeout_s = timeout_s output, reward, info = env._evaluate_docker() @@ -441,7 +584,7 @@ def test_evaluate_docker_stages_tests_at_verify(tmp_path: Path): # budget, verdict read from reward.txt — not bare pytest in /task. assert "bash /tests/test.sh" in eval_cmd assert eval_cmd.startswith("cd /task && ") # no resolved workdir → /task - assert "timeout 900" in eval_cmd + assert f"timeout {timeout_s:g}" in eval_cmd assert "/logs/verifier/reward.txt" in eval_cmd assert "rm -rf /tests /logs/verifier" in container.events[3][1] @@ -465,10 +608,13 @@ def test_evaluate_docker_fallback_without_testsh(tmp_path: Path): still against the staged /tests copy.""" task = _make_task_dir(tmp_path) (task / "tests" / "test.sh").unlink() + with (task / "task.toml").open("a") as handle: + handle.write("\n[verifier]\ntimeout_sec = 0.5\n") env = Tbench2DockerEnvironment() container = _FakeContainer(exec_output=b"__TB2_EXIT_CODE__:0\n") env._container = container env._task_dir = task + env._verifier_timeout_s = 0.5 output, reward, info = env._evaluate_docker() @@ -477,6 +623,48 @@ def test_evaluate_docker_fallback_without_testsh(tmp_path: Path): eval_cmd = container.events[2][1] assert "pytest -q /tests -rA" in eval_cmd assert "test.sh" not in eval_cmd + assert "timeout 0.5" in eval_cmd + + +@pytest.mark.parametrize("has_timeout", [False, True]) +def test_fallback_timeout_preserves_exit_marker(tmp_path: Path, has_timeout: bool): + """Timeout failure or absence must never run an unbounded verifier.""" + timeout_args = tmp_path / "timeout-args" + if has_timeout: + timeout = tmp_path / "timeout" + timeout.write_text( + '#!/bin/sh\nprintf "%s\\n" "$@" > "$TIMEOUT_ARGS"\nexit 124\n' + ) + timeout.chmod(0o755) + verifier_ran = tmp_path / "verifier-ran" + uvx = tmp_path / "uvx" + uvx.write_text('#!/bin/sh\nprintf ran > "$VERIFIER_RAN"\n') + uvx.chmod(0o755) + result = subprocess.run( + [ + "/bin/bash", + "-c", + tbench2_env_environment._fallback_eval_cmd(str(tmp_path), timeout_s=0.5), + ], + env={ + **os.environ, + "PATH": str(tmp_path), + "TIMEOUT_ARGS": str(timeout_args), + "VERIFIER_RAN": str(verifier_ran), + }, + capture_output=True, + text=True, + timeout=5, + check=True, + ) + + assert not verifier_ran.exists() + if has_timeout: + assert timeout_args.read_text().splitlines()[:3] == ["0.5", "bash", "-c"] + expected_exit = 124 if has_timeout else 127 + assert ( + tbench2_env_environment._parse_exit_code_marker(result.stdout) == expected_exit + ) def test_evaluate_docker_cleans_up_when_scoring_raises(tmp_path: Path): @@ -623,6 +811,46 @@ def remove(self, force): assert env._workdir == "" +def _docker_reset_info(monkeypatch, tmp_path: Path, task_toml: str) -> dict: + task = tmp_path / "docker-task" + task.mkdir() + (task / "task.toml").write_text(task_toml) + (task / "instruction.md").write_text("do it\n") + monkeypatch.setattr( + Tbench2DockerEnvironment, "_start_container", lambda self, *args: None + ) + env = Tbench2DockerEnvironment( + tasks_dir=str(tmp_path), output_dir=str(tmp_path / "runs") + ) + return env.reset(task_id="docker-task").info + + +def test_docker_reset_reports_verifier_budget(monkeypatch, tmp_path: Path): + """Docker mode enforces the task's verifier budget in-shell, so reset() + reports it alongside the image. It must not advertise command_timeout_s: + exec_run has no server-side timeout, so that value is never enforced.""" + info = _docker_reset_info( + monkeypatch, + tmp_path, + '[environment]\ndocker_image = "tb2/demo:1"\n\n[verifier]\ntimeout_sec = 1800\n', + ) + + assert info == {"docker_image": "tb2/demo:1", "verifier_timeout_sec": 1800.0} + + +def test_docker_reset_budget_falls_back_without_verifier_section( + monkeypatch, tmp_path: Path +): + info = _docker_reset_info( + monkeypatch, tmp_path, '[environment]\ndocker_image = "tb2/demo:1"\n' + ) + + assert ( + info["verifier_timeout_sec"] + == tbench2_env_environment._DEFAULT_VERIFIER_TIMEOUT_S + ) + + class _FakeImage: def __init__(self, working_dir: str): self.attrs = {"Config": {"WorkingDir": working_dir}}