Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
22 changes: 22 additions & 0 deletions docs/source/environments/jupyter.md
Original file line number Diff line number Diff line change
Expand Up @@ -93,6 +93,28 @@ negative rewards and values above 1 are accepted too. A value that is not a
finite number is ignored, the pass rate is used, and the reason is recorded in
`state.reward_override_ignored`.

Verify commands, the clearing of that file and the read of it run as their own
sandbox processes through E2B's process API, as `root` in `/home/user`, rather
than inside the notebook kernel. A cell can rebind `subprocess.run`, change the
working directory or edit `os.environ`, and none of that reaches verification.
This removes the coupling to the kernel; it does not isolate verification from
the agent. The notebook kernel runs as root in E2B's default code-interpreter
template, so agent code can still change anything in the sandbox, including the
files a verify command reads and the startup files of the shell it runs in.
E2B starts each command as a login shell, so a profile the agent writes decides
what a verify command reports: an `exit 0` or an `EXIT` trap in
`/root/.bash_profile` makes a failing command look like a passing one. A test
pins that behaviour rather than leaving it to be found in a reward curve, and
#1232 tracks verification outside the agent's sandbox, which is what closes it.

Only a command that ran and exited decides a verify result. If the sandbox
itself fails, because it cannot be reached, the command cannot be started, or
the wait for it ends without an exit status, the error is raised instead:
verification stops, no reward is produced, and the episode does not finish. A sandbox that
is unreachable, expired or unauthenticated says nothing about the agent's work,
and a command with no exit status may still be running, since E2B's kill
signals the command's own process and not the children it started.

## Notes

This first version intentionally keeps sandbox provider selection local to the
Expand Down
22 changes: 22 additions & 0 deletions envs/jupyter_env/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -108,6 +108,28 @@ negative rewards and values above 1 are accepted too. A value that is not a
finite number is ignored, the pass rate is used, and the reason is recorded in
`state.reward_override_ignored`.

Verify commands, the clearing of that file and the read of it run as their own
sandbox processes through E2B's process API, as `root` in `/home/user`, rather
than inside the notebook kernel. A cell can rebind `subprocess.run`, change the
working directory or edit `os.environ`, and none of that reaches verification.
This removes the coupling to the kernel; it does not isolate verification from
the agent. The notebook kernel runs as root in E2B's default code-interpreter
template, so agent code can still change anything in the sandbox, including the
files a verify command reads and the startup files of the shell it runs in.
E2B starts each command as a login shell, so a profile the agent writes decides
what a verify command reports: an `exit 0` or an `EXIT` trap in
`/root/.bash_profile` makes a failing command look like a passing one. A test
pins that behaviour rather than leaving it to be found in a reward curve, and
#1232 tracks verification outside the agent's sandbox, which is what closes it.

Only a command that ran and exited decides a verify result. If the sandbox
itself fails, because it cannot be reached, the command cannot be started, or
the wait for it ends without an exit status, the error is raised instead:
verification stops, no reward is produced, and the episode does not finish. A sandbox that
is unreachable, expired or unauthenticated says nothing about the agent's work,
and a command with no exit status may still be running, since E2B's kill
signals the command's own process and not the children it started.

## Notes

This first version intentionally keeps sandbox provider selection local to the
Expand Down
92 changes: 92 additions & 0 deletions envs/jupyter_env/server/e2b_sandbox.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@
that decouples the rest of the environment from the E2B API surface.
"""

import contextlib
from dataclasses import dataclass
from typing import Any, List, Optional

Expand All @@ -15,6 +16,13 @@
_E2B_IMPORT_ERROR = _e2b_import_error
Sandbox = None # type: ignore[assignment]

# E2B's default code-interpreter template runs its Jupyter server, and so every
# notebook kernel, as root with the notebook in /home/user. Verification has
# always run from that kernel, so ``run_command`` keeps the same user and
# directory and verify commands see the same permissions they did before.
_COMMAND_USER = "root"
_COMMAND_CWD = "/home/user"


@dataclass
class CellResult:
Expand All @@ -30,6 +38,26 @@ class CellResult:
success: bool


def _failed_command(exc: Exception, exit_code: int) -> CellResult:
"""Describe a command that exited non-zero as a failed `CellResult`.

A non-zero exit raises the SDK's ``CommandExitException``, which carries the
exit code and output. It is matched by attribute so this module still
imports without the SDK installed.
"""
error = f"exit code {exit_code}"
return CellResult(
stdout=getattr(exc, "stdout", "") or "",
stderr=getattr(exc, "stderr", "") or "",
error=error,
error_name=type(exc).__name__,
text_results=[],
images=[],
execution_count=0,
success=False,
)


class E2BSandbox:
"""
Manages a single E2B Code Interpreter sandbox session.
Expand Down Expand Up @@ -105,6 +133,70 @@ def run_shell(self, command: str, timeout_s: float = 120) -> CellResult:
)
return self.run_code(shell_code)

def run_command(self, command: str, timeout_s: float = 120) -> CellResult:
"""
Execute a shell command as its own sandbox process, outside the kernel.

``run_shell`` runs inside the notebook kernel, so it inherits whatever
the notebook has done to it: a rebound ``subprocess.run``, a changed
working directory, edited ``os.environ``. That is right for the agent's
own shell tool and wrong for verification. This goes through E2B's
process API instead and reports the command's real exit status.

It removes the coupling to the notebook kernel. It is not an isolation
boundary: the kernel runs as root, so agent code can still change what
any process in the sandbox sees, including this command's login shell.

Args:
command (`str`):
Shell command to run.
timeout_s (`float`, *optional*, defaults to `120`):
Seconds to wait for the command before giving up on it.

Returns:
[`CellResult`]: `success` is `True` only for a zero exit status.

Raises:
`Exception`: whatever the SDK raises when the command cannot be
started, or when the wait for it ends without an exit status.
Only a command that ran and exited is reported as a result:
a sandbox that is unreachable, expired or unauthenticated says
nothing about the work being verified.
"""
# Started in the background so there is a handle to kill if the wait
# fails: E2B keeps a command running when its connection drops.
handle = self._sbx.commands.run(
command,
background=True,
user=_COMMAND_USER,
cwd=_COMMAND_CWD,
timeout=timeout_s,
)
try:
result = handle.wait()
except Exception as exc:
exit_code = getattr(exc, "exit_code", None)
if exit_code is not None:
return _failed_command(exc, exit_code)
# No exit status: the command timed out, or the connection to it
# broke. Ask E2B to stop it, but a kill signals the command's own
# process, so a child of it can stay alive and still write to the
# sandbox. Nothing here can call the command finished, so the
# failure is raised rather than counted as a verify result.
with contextlib.suppress(Exception):
handle.kill()
raise
return CellResult(
stdout=result.stdout or "",
stderr=result.stderr or "",
error=None,
error_name=None,
text_results=[],
images=[],
execution_count=0,
success=True,
)

def write_file(self, filename: str, content: bytes) -> None:
"""Upload a file into the sandbox filesystem."""
self._sbx.files.write(filename, content)
Expand Down
16 changes: 12 additions & 4 deletions envs/jupyter_env/server/jupyter_environment.py
Original file line number Diff line number Diff line change
Expand Up @@ -388,11 +388,19 @@ def _run_verify_commands(self) -> dict[str, Any]:
if not self._sandbox:
return {"passed": 0, "total": 0, "reward": None}

self._sandbox.run_shell("mkdir -p /home/user/logs/verifier")
# Verification runs as separate sandbox processes, not in the notebook
# kernel, so nothing a cell does to the kernel changes how verify
# commands run or what they report.
self._sandbox.run_command("mkdir -p /home/user/logs/verifier")
# The override is documented as verify-written, so drop anything the
# policy wrote there during the episode before verification starts.
cleared = self._sandbox.run_shell(_CLEAR_REWARD_FILE).success
verify_results = self._run_shell_commands(self._state.verify_commands)
cleared = self._sandbox.run_command(_CLEAR_REWARD_FILE).success
verify_results = [
_command_result_from_cell_result(
command, self._sandbox.run_command(command)
)
for command in self._state.verify_commands
]
self._state.verify_results = verify_results

passed = sum(1 for result in verify_results if result.success)
Expand Down Expand Up @@ -453,7 +461,7 @@ def _read_reward_override(
) -> tuple[Optional[float], Optional[str]]:
if not cleared:
return None, "reward file could not be removed before verification"
result = sandbox.run_shell(f"cat {REWARD_FILE} 2>/dev/null || true")
result = sandbox.run_command(f"cat {REWARD_FILE} 2>/dev/null || true")
return _parse_reward_override(result.stdout or "")


Expand Down
Loading