Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion benchmark/LHTB/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -179,7 +179,7 @@ Defaults leave cleanup room between nested layers:
| Harbor agent limit | 5400 s |
| LoopX scheduler process | 5080 s |
| One wake command | 4800 s |
| One Codex exec | 4700 s |
| One Codex exec | Remaining total phase budget, minus cleanup reserve |

The scheduler can perform multiple shorter wakes within 5080 seconds. A single
long Codex wake can consume most of that budget, which is expected; Harbor's
Expand Down
8 changes: 6 additions & 2 deletions benchmark/LHTB/run.sh
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@ REASONING_EFFORT="${REASONING_EFFORT:-max}"
CONCURRENCY="${CONCURRENCY:-4}"
AGENT_TIMEOUT_SEC="${AGENT_TIMEOUT_SEC:-5400}"
LOOPX_SCHEDULER_TIMEOUT_SEC="${LOOPX_SCHEDULER_TIMEOUT_SEC:-5080}"
LOOPX_CODEX_TURN_TIMEOUT_SEC="${LOOPX_CODEX_TURN_TIMEOUT_SEC:-4700}"
LOOPX_CODEX_TURN_TIMEOUT_SEC="${LOOPX_CODEX_TURN_TIMEOUT_SEC:-}"
LHTB_MAX_RETRIES="${LHTB_MAX_RETRIES:-2}"
RUNNER_RESTARTS="${RUNNER_RESTARTS:-2}"
LHTB_MODELONLY_NETWORK="${LHTB_MODELONLY_NETWORK:-lhtb-modelonly}"
Expand Down Expand Up @@ -121,6 +121,10 @@ job_name="lhtb-${LOOPX_EXECUTION_MODE}-${LOOPX_TASK_ENTRY}-${LOOPX_ITERATION_CON
generated_config="$CODE_DIR/.generated/${job_name}.yaml"
jobs_dir="$CODE_DIR/runs"

turn_timeout_args=()
if [[ -n "$LOOPX_CODEX_TURN_TIMEOUT_SEC" ]]; then
turn_timeout_args=(--turn-timeout "$LOOPX_CODEX_TURN_TIMEOUT_SEC")
fi
"$VENV/bin/python" "$CODE_DIR/scripts/render_config.py" \
--template "$CODE_DIR/configs/heartbeat-generic-cli.yaml" \
--output "$generated_config" \
Expand All @@ -135,7 +139,7 @@ jobs_dir="$CODE_DIR/runs"
--planning-timeout "$LOOPX_PLANNING_TIMEOUT_SEC" \
--iteration-context "$LOOPX_ITERATION_CONTEXT" \
--validation-command-json "$LOOPX_VALIDATION_COMMAND_JSON" \
--turn-timeout "$LOOPX_CODEX_TURN_TIMEOUT_SEC" \
"${turn_timeout_args[@]}" \
--scheduler-timeout "$LOOPX_SCHEDULER_TIMEOUT_SEC" \
"${task_args[@]}"

Expand Down
2 changes: 1 addition & 1 deletion benchmark/LHTB/scripts/render_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,7 +28,7 @@ def main() -> int:
parser.add_argument("--task-entry", choices=TASK_ENTRIES, default="seeded-todo")
parser.add_argument("--planning-timeout", type=float, default=300)
parser.add_argument("--validation-command-json", default="[]")
parser.add_argument("--turn-timeout", type=float, default=4700)
parser.add_argument("--turn-timeout", type=float, default=None)
parser.add_argument("--scheduler-timeout", type=int, default=5080)
args = parser.parse_args()

Expand Down
6 changes: 4 additions & 2 deletions benchmark/runtime/RUNTIME.md
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,7 @@ agents:
iteration_context: fresh
reasoning_effort: max
codex_sandbox: danger-full-access
turn_timeout_sec: 4700
turn_timeout_sec: null
scheduler_timeout_sec: 5080
replan_after_todos: 3
```
Expand Down Expand Up @@ -111,7 +111,7 @@ kwargs:
task_entry: loopx-planned
planning_timeout_sec: 300
iteration_context: fresh
turn_timeout_sec: 4700
turn_timeout_sec: null
scheduler_timeout_sec: 5080
```

Expand Down Expand Up @@ -201,3 +201,5 @@ Install the intended Harbor version for the adapter tests. Real qualification
also needs installed Codex, the native Harbor backend and independently checked
task output. Unit tests establish no score or model-uplift claim. Validate small
jobs through each benchmark's native configuration before launching a study.

By default, worker calls have no independent turn deadline. Harbor derives their available time from the remaining total phase budget, reserving cleanup and settlement time. An explicit `turn_timeout_sec` remains supported as an operator override.
6 changes: 4 additions & 2 deletions benchmark/runtime/codex.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,7 @@ class Execution:
mode: str = "heartbeat"
context: str = "fresh"
sandbox: str = "danger-full-access"
timeout_seconds: float = 4700
timeout_seconds: float | None = None
validation_command: tuple[str, ...] = ()
task_entry: str = "seeded-todo"

Expand All @@ -35,7 +35,9 @@ def __post_init__(self) -> None:
raise ValueError("resume requires mode=turn or mode=heartbeat")
if self.sandbox not in SANDBOXES:
raise ValueError("unsupported Codex sandbox")
if not math.isfinite(self.timeout_seconds) or self.timeout_seconds <= 0:
if self.timeout_seconds is not None and (
not math.isfinite(self.timeout_seconds) or self.timeout_seconds <= 0
):
raise ValueError("execution timeout must be finite and positive")
if not isinstance(self.validation_command, (list, tuple)):
raise ValueError("validation_command must be an argv list")
Expand Down
4 changes: 2 additions & 2 deletions benchmark/runtime/harbor.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,7 +53,7 @@ def __init__(
iteration_context="fresh",
codex_sandbox="danger-full-access",
validation_command=None,
turn_timeout_sec=4700,
turn_timeout_sec=None,
scheduler_timeout_sec=5080,
replan_after_todos=3,
task_entry="seeded-todo",
Expand All @@ -66,7 +66,7 @@ def __init__(
execution_mode,
iteration_context,
codex_sandbox,
float(turn_timeout_sec),
float(scheduler_timeout_sec) - 160 if turn_timeout_sec is None else float(turn_timeout_sec),
validation_command if validation_command is not None else (),
task_entry,
)
Expand Down
11 changes: 7 additions & 4 deletions benchmark/runtime/worker.py
Original file line number Diff line number Diff line change
Expand Up @@ -144,8 +144,8 @@ def turn_command(
env["MODEL_NAME"],
"--codex-sandbox",
execution.sandbox,
"--timeout-seconds",
str(execution.timeout_seconds),
*(["--timeout-seconds", str(execution.timeout_seconds)]
if execution.timeout_seconds is not None else []),
"--validation-command-json",
json.dumps(execution.validation_command),
"--available-capability",
Expand Down Expand Up @@ -235,7 +235,8 @@ def run_once(env: dict[str, str]) -> dict:
mode=env.get("LOOPX_EXECUTION_MODE", "heartbeat"),
context=env.get("LOOPX_ITERATION_CONTEXT", "fresh"),
sandbox=env.get("LOOPX_CODEX_SANDBOX", "danger-full-access"),
timeout_seconds=float(env.get("LOOPX_CODEX_TURN_TIMEOUT_SEC", "4700")),
timeout_seconds=(float(env["LOOPX_CODEX_TURN_TIMEOUT_SEC"])
if env.get("LOOPX_CODEX_TURN_TIMEOUT_SEC") else None),
validation_command=json.loads(env.get("LOOPX_VALIDATION_COMMAND_JSON", "[]")),
task_entry=env.get("LOOPX_TASK_ENTRY", "seeded-todo"),
)
Expand Down Expand Up @@ -272,7 +273,8 @@ def run_once(env: dict[str, str]) -> dict:
receipt.update(ok=True, budget_exhausted=True, host_invoked=False)
return receipt
execution = replace(
execution, timeout_seconds=min(execution.timeout_seconds, remaining - 160)
execution, timeout_seconds=(remaining - 160 if execution.timeout_seconds is None
else min(execution.timeout_seconds, remaining - 160))
)
prepare_codex_home(
home,
Expand Down Expand Up @@ -330,6 +332,7 @@ def run_once(env: dict[str, str]) -> dict:
) as process:
allowance = 150 if execution.mode == "turn" and stage == "execute" else 0
timeout = (float(env["LOOPX_PLANNING_TIMEOUT_SEC"]) if stage == "plan"
else None if execution.timeout_seconds is None
else execution.timeout_seconds + allowance)
process.communicate(
input=body, timeout=timeout
Expand Down
2 changes: 1 addition & 1 deletion benchmark/swe-marathon/configs/shared-heartbeat.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9,5 +9,5 @@ agents:
task_entry: seeded-todo # Use loopx-planned for the product planning checkpoint.
iteration_context: fresh
reasoning_effort: high
turn_timeout_sec: 4700
turn_timeout_sec: null
scheduler_timeout_sec: 5080
10 changes: 6 additions & 4 deletions benchmark/tests/test_native_codex_goal.py
Original file line number Diff line number Diff line change
Expand Up @@ -293,10 +293,11 @@ def test_terminal_event_preserves_failed_turn_status() -> None:
assert turn.turn_status == "failed"


def test_goal_runtime_waits_for_automatic_continuation_until_terminal() -> None:
@pytest.mark.parametrize("timeout", [None, 1])
def test_goal_runtime_waits_for_automatic_continuation_until_terminal(timeout) -> None:
transport = ContinuationTransport()

turn = run_native_goal_until_terminal(transport, _config(), timeout_sec=1)
turn = run_native_goal_until_terminal(transport, _config(), timeout_sec=timeout)

methods = [method for method, _ in transport.calls]
assert methods.count("turn/start") == 1
Expand Down Expand Up @@ -492,8 +493,9 @@ def test_process_cwd_can_differ_from_goal_thread_cwd(tmp_path: Path) -> None:
assert turn.terminal_event_observed is True


@pytest.mark.parametrize("timeout", [None, 2])
def test_real_stdio_process_waits_until_native_goal_is_terminal(
tmp_path: Path,
tmp_path: Path, timeout,
) -> None:
fake_server = tmp_path / "fake-continuing-codex"
_write_fake_app_server(fake_server)
Expand All @@ -512,7 +514,7 @@ def test_real_stdio_process_waits_until_native_goal_is_terminal(
process_command=[sys.executable, str(fake_server)],
process_env=process_env,
response_timeout_sec=2,
goal_timeout_sec=2,
goal_timeout_sec=timeout,
)

assert turn.post_goal_status == "complete"
Expand Down
17 changes: 17 additions & 0 deletions benchmark/tests/test_shared_codex_runtime.py
Original file line number Diff line number Diff line change
Expand Up @@ -438,3 +438,20 @@ async def unpack(*args, **kwargs):
assert staged == original
assert marker.read_text() == "successor"
assert uploaded == [b"original"]


@pytest.mark.parametrize("total", [1800, 64800])
def test_harbor_default_execution_budget_tracks_total_trial(tmp_path, total):
pytest.importorskip("harbor")
from benchmark.runtime.harbor import BenchmarkCodex
agent = BenchmarkCodex(logs_dir=tmp_path, model_name="fixture",
scheduler_timeout_sec=total)
assert agent.execution.timeout_seconds == total - 160
assert agent.scheduler_timeout == total


def test_default_execution_has_no_independent_turn_deadline(tmp_path):
execution = Execution(mode="turn", validation_command=("true",))
env = worker_env(tmp_path) | {"LOOPX_CLI": "loopx", "LOOPX_GOAL_ID": "fixture", "LOOPX_AGENT_ID": "worker", "LOOPX_REGISTRY": "registry", "LOOPX_RUNTIME_ROOT": "runtime"}
assert execution.timeout_seconds is None
assert "--timeout-seconds" not in turn_command(env, execution, "wake-default")
24 changes: 24 additions & 0 deletions docs/reference/protocols/loopx-turn-v0.md
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,30 @@ The protocol is host-neutral. A Codex CLI adapter is the first target, but the
driver lifecycle must not depend on Codex-specific session files, transcript
formats, or benchmark task schemas.

### Execution deadline defaults

`turn run-once` now has no default single-turn wall-time limit (previously
120 seconds). The built-in Codex hosts and generic managed host transport wait
for natural completion or cancellation. `--timeout-seconds` remains an explicit
operator-selected deadline for bounded execution or recovery tests.

The external heartbeat scheduler likewise has no default wake-command deadline
(previously 600 seconds); `--wake-timeout-seconds` opts into one. Quota probes,
provider request/idle liveness checks, validation commands, output budgets,
lease fencing and process-group cleanup retain their own boundaries. Cancellation
still terminates the owned process tree; disabling an execution timer does not
authorize continued effects after cancellation or lease loss.

Benchmark adapters retain their declared total trial deadline. The shared Harbor
adapter defaults a call to the remaining trial allowance, including its existing
startup/cleanup reserve, instead of imposing an independent 4700-second wake cap.
Explicit per-call settings in existing experiment configurations remain explicit
protocol choices; omit them for natural continuation.

单轮执行默认不再计时终止:Turn 的原 120 秒上限和外部 heartbeat 的原 600 秒
wake 上限均改为显式选择。取消、租约失效、输出限制、请求存活检查和进程清理
仍生效。benchmark 仍遵守整场总预算,默认不另设 4700 秒的单轮上限。

## Mental Model

LoopX Turn is a four-stage control loop, not another agent runtime:
Expand Down
28 changes: 14 additions & 14 deletions loopx/capabilities/benchmark_toolkit/native_codex_goal.py
Original file line number Diff line number Diff line change
Expand Up @@ -349,7 +349,7 @@ def wait_native_goal_turn(
transport: NativeGoalEventTransport,
turn: NativeGoalTurn,
*,
timeout_sec: float,
timeout_sec: float | None,
completed_before: int | None = None,
) -> NativeGoalTurn:
"""Drain events until one more correlated turn reaches a terminal event.
Expand All @@ -359,18 +359,18 @@ def wait_native_goal_turn(
preserves the single-turn behavior for existing callers.
"""

if timeout_sec <= 0:
if timeout_sec is not None and timeout_sec <= 0:
raise ValueError("timeout_sec must be positive")
deadline = time.monotonic() + timeout_sec
deadline = None if timeout_sec is None else time.monotonic() + timeout_sec
if completed_before is None:
if turn.terminal_event_observed:
return turn
completed_before = turn.turn_completed_count
while turn.turn_completed_count <= completed_before:
remaining = deadline - time.monotonic()
if remaining <= 0:
remaining = None if deadline is None else deadline - time.monotonic()
if remaining is not None and remaining <= 0:
raise NativeGoalProtocolError("goal_turn_timeout")
event = transport.next_event(timeout_sec=min(0.25, remaining))
event = transport.next_event(timeout_sec=0.25 if remaining is None else min(0.25, remaining))
if event is not None:
observe_native_goal_event(turn, event)
return turn
Expand Down Expand Up @@ -399,7 +399,7 @@ def run_native_goal_turn(
transport: NativeGoalEventTransport,
config: NativeGoalConfig,
*,
timeout_sec: float,
timeout_sec: float | None,
) -> NativeGoalTurn:
"""Execute the complete native Goal transaction over an admitted transport."""

Expand All @@ -413,7 +413,7 @@ def run_native_goal_until_terminal(
transport: NativeGoalEventTransport,
config: NativeGoalConfig,
*,
timeout_sec: float,
timeout_sec: float | None,
on_turn_started: Callable[[NativeGoalTurn], None] | None = None,
) -> NativeGoalTurn:
"""Run one native Goal until its status leaves ``active``.
Expand All @@ -423,16 +423,16 @@ def run_native_goal_until_terminal(
continuation events and reading the Goal status under one total timeout.
"""

if timeout_sec <= 0:
if timeout_sec is not None and timeout_sec <= 0:
raise ValueError("timeout_sec must be positive")
turn = start_native_goal_turn(transport, config)
if on_turn_started is not None:
on_turn_started(turn)
deadline = time.monotonic() + timeout_sec
deadline = None if timeout_sec is None else time.monotonic() + timeout_sec
completed_before = turn.turn_completed_count
while True:
remaining = deadline - time.monotonic()
if remaining <= 0:
remaining = None if deadline is None else deadline - time.monotonic()
if remaining is not None and remaining <= 0:
raise NativeGoalDeadlineExceeded("goal_timeout_before_terminal")
try:
wait_native_goal_turn(
Expand Down Expand Up @@ -639,7 +639,7 @@ def run_native_goal_process(
process_env: Mapping[str, str] | None = None,
process_cwd: str | None = None,
response_timeout_sec: float = 30,
goal_timeout_sec: float = 21_600,
goal_timeout_sec: float | None = 21_600,
) -> NativeGoalTurn:
"""Spawn a real app-server and execute one complete native Goal turn."""

Expand All @@ -661,7 +661,7 @@ def run_native_goal_process_until_terminal(
process_env: Mapping[str, str] | None = None,
process_cwd: str | None = None,
response_timeout_sec: float = 30,
goal_timeout_sec: float = 21_600,
goal_timeout_sec: float | None = 21_600,
) -> NativeGoalTurn:
"""Spawn app-server and keep the native Goal alive through continuations."""

Expand Down
15 changes: 7 additions & 8 deletions loopx/chat_agent.py
Original file line number Diff line number Diff line change
Expand Up @@ -432,7 +432,7 @@ class CodexChatAgentSession:
reasoning_effort: str | None = None
response_timeout_sec: float = 30.0
idle_timeout_sec: float = 180.0
hard_timeout_sec: float = 900.0
hard_timeout_sec: float | None = 900.0
next_request_id: int = 5
current_turn_id: str = ""
model_catalog_compatibility_applied: bool = False
Expand Down Expand Up @@ -468,7 +468,7 @@ def start(
objective: str,
response_timeout_sec: float = 30.0,
idle_timeout_sec: float = 180.0,
hard_timeout_sec: float = 900.0,
hard_timeout_sec: float | None = 900.0,
resume_thread_id: str | None = None,
execution_mode: bool = False,
isolate_process_tree: bool = False,
Expand Down Expand Up @@ -988,23 +988,22 @@ def send(
last_activity_at = started_at
while True:
now = time.monotonic()
if now - started_at >= self.hard_timeout_sec:
if self.hard_timeout_sec is not None and now - started_at >= self.hard_timeout_sec:
raise self._timeout_error(
"hard_timeout", "Codex Chat turn reached its hard time limit."
)
if now - last_activity_at >= self.idle_timeout_sec:
raise self._timeout_error(
"idle_timeout", "Codex Chat turn stopped producing activity."
)
deadline = min(
started_at + self.hard_timeout_sec,
last_activity_at + self.idle_timeout_sec,
)
deadline = last_activity_at + self.idle_timeout_sec
if self.hard_timeout_sec is not None:
deadline = min(deadline, started_at + self.hard_timeout_sec)
try:
message = self._next_event(deadline=deadline)
except CodexChatAgentError:
now = time.monotonic()
if now - started_at >= self.hard_timeout_sec:
if self.hard_timeout_sec is not None and now - started_at >= self.hard_timeout_sec:
raise self._timeout_error(
"hard_timeout",
"Codex Chat turn reached its hard time limit.",
Expand Down
Loading
Loading