Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 5 additions & 3 deletions benchmark/LHTB/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -107,7 +107,8 @@ document. Registries are inside their own task containers, so no Goal, Todo,
or scheduler state is shared between the 46 trials.

The adapter registers the project through the current public bootstrap CLI and
adds the benchmark phase as an explicit P0 Todo. Retired onboarding flags are
plans phase Todos from the task by default (or adds a generic P0 Todo when
seeded entry is explicitly selected). Retired onboarding flags are
not replayed; bootstrap and Todo lifecycle follow the installed product version.

## Shared execution configuration
Expand All @@ -118,8 +119,9 @@ accept `resume`, sharing the same Goal/Agent session across planning, wakes and
an argv array for the independently protected task validator. No generic
benchmark scoring or hidden-verifier feedback is introduced.

`LOOPX_TASK_ENTRY=seeded-todo` preserves the generic phase Todo default.
`LOOPX_TASK_ENTRY=loopx-planned` invokes the product planning checkpoint before
`LOOPX_TASK_ENTRY=seeded-todo` retains the former generic phase Todo entry.
Omitting this setting now selects `loopx-planned` for LoopX modes and invokes
the product planning checkpoint before
heartbeat, Turn or LoopX Goal execution. `LOOPX_PLANNING_TIMEOUT_SEC` defaults
to 300 and consumes the existing phase budget. Both entry policies preserve
existing waits when new phases arrive. See the shared runtime for session and
Expand Down
10 changes: 7 additions & 3 deletions benchmark/LHTB/run.sh
Original file line number Diff line number Diff line change
Expand Up @@ -47,7 +47,7 @@ LHTB_MODELONLY_GATEWAY="${LHTB_MODELONLY_GATEWAY:-192.0.2.1}"
LOOPX_SRC_DIR="${LOOPX_SRC_DIR:-$LOOPX_ROOT}"
export PYTHONPATH="$LOOPX_SRC_DIR${PYTHONPATH:+:$PYTHONPATH}"
LOOPX_EXECUTION_MODE="${LOOPX_EXECUTION_MODE:-heartbeat}"
LOOPX_TASK_ENTRY="${LOOPX_TASK_ENTRY:-seeded-todo}"
LOOPX_TASK_ENTRY="${LOOPX_TASK_ENTRY:-}"
LOOPX_PLANNING_TIMEOUT_SEC="${LOOPX_PLANNING_TIMEOUT_SEC:-300}"
LOOPX_ITERATION_CONTEXT="${LOOPX_ITERATION_CONTEXT:-fresh}"
LOOPX_VALIDATION_COMMAND_JSON="${LOOPX_VALIDATION_COMMAND_JSON:-[]}"
Expand Down Expand Up @@ -117,10 +117,14 @@ if [[ "$MODE" == smoke ]]; then
expected_task_count=1
job_suffix="smoke-${SMOKE_TASK}"
fi
job_name="lhtb-${LOOPX_EXECUTION_MODE}-${LOOPX_TASK_ENTRY}-${LOOPX_ITERATION_CONTEXT}-${job_suffix}-${run_stamp}"
job_name="lhtb-${LOOPX_EXECUTION_MODE}-${LOOPX_TASK_ENTRY:-default}-${LOOPX_ITERATION_CONTEXT}-${job_suffix}-${run_stamp}"
generated_config="$CODE_DIR/.generated/${job_name}.yaml"
jobs_dir="$CODE_DIR/runs"

task_entry_args=()
if [[ -n "$LOOPX_TASK_ENTRY" ]]; then
task_entry_args=(--task-entry "$LOOPX_TASK_ENTRY")
fi
turn_timeout_args=()
if [[ -n "$LOOPX_CODEX_TURN_TIMEOUT_SEC" ]]; then
turn_timeout_args=(--turn-timeout "$LOOPX_CODEX_TURN_TIMEOUT_SEC")
Expand All @@ -135,7 +139,7 @@ fi
--effort "$REASONING_EFFORT" \
--timeout "$AGENT_TIMEOUT_SEC" \
--execution-mode "$LOOPX_EXECUTION_MODE" \
--task-entry "$LOOPX_TASK_ENTRY" \
"${task_entry_args[@]}" \
--planning-timeout "$LOOPX_PLANNING_TIMEOUT_SEC" \
--iteration-context "$LOOPX_ITERATION_CONTEXT" \
--validation-command-json "$LOOPX_VALIDATION_COMMAND_JSON" \
Expand Down
2 changes: 1 addition & 1 deletion benchmark/LHTB/scripts/preflight.py
Original file line number Diff line number Diff line change
Expand Up @@ -128,7 +128,7 @@ def check(label: str, passed: bool, detail: str) -> None:
)
execution = Execution(
mode=kwargs.get("execution_mode", "heartbeat"),
task_entry=kwargs.get("task_entry", "seeded-todo"),
task_entry=kwargs.get("task_entry"),
context=kwargs.get("iteration_context", "fresh"),
validation_command=kwargs.get("validation_command", []),
)
Expand Down
3 changes: 2 additions & 1 deletion benchmark/LHTB/scripts/render_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,8 @@ def main() -> int:
parser.add_argument("--task", action="append", default=[])
parser.add_argument("--execution-mode", choices=MODES, default="heartbeat")
parser.add_argument("--iteration-context", choices=CONTEXTS, default="fresh")
parser.add_argument("--task-entry", choices=TASK_ENTRIES, default="seeded-todo")
parser.add_argument("--task-entry", choices=TASK_ENTRIES,
help="LoopX modes default to loopx-planned; explicit seeded-todo retains legacy entry")
parser.add_argument("--planning-timeout", type=float, default=300)
parser.add_argument("--validation-command-json", default="[]")
parser.add_argument("--turn-timeout", type=float, default=None)
Expand Down
7 changes: 4 additions & 3 deletions benchmark/edgebench/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -93,7 +93,8 @@ def main(argv=None):
parser.add_argument("--effort", choices=("low", "medium", "high", "xhigh"), required=True)
parser.add_argument("--timeout", type=int,
help="Total trial seconds; task-defaults.json overrides the 18h fallback")
parser.add_argument("--task-entry", choices=TASK_ENTRIES, default="seeded-todo")
parser.add_argument("--task-entry", choices=TASK_ENTRIES,
help="Heartbeat default: loopx-planned; seeded-todo is an explicit ablation")
parser.add_argument("--replan-after-turns", type=int, choices=range(1, 6),
help="Opt in to settled work Turn cadence for heartbeat profiles")
parser.add_argument("--eval-interval", type=int,
Expand All @@ -104,7 +105,7 @@ def main(argv=None):
args = parser.parse_args(argv)
args.timeout = _task_default(args.task, "timeout_seconds", args.timeout, DEFAULT_TIMEOUT_SECONDS)
args.eval_interval = _task_default(args.task, "eval_interval_seconds", args.eval_interval, 300)
if args.task_entry != "seeded-todo" and not args.worker.startswith("heartbeat-"):
if args.task_entry == "loopx-planned" and not args.worker.startswith("heartbeat-"):
parser.error("--task-entry loopx-planned requires a heartbeat worker")
if args.turn_envelope and args.worker not in {"heartbeat-resume", "heartbeat-explore"}:
parser.error("--turn-envelope requires a heartbeat worker")
Expand Down Expand Up @@ -159,7 +160,7 @@ def main(argv=None):
raise RuntimeError(f"Missing native image: {image}")
receipt = {
"run_id": args.run_id, "task": args.task, "worker": args.worker,
"task_entry": args.task_entry,
"task_entry": agent.task_entry,
"model": args.model, "effort": args.effort, "timeout_seconds": args.timeout,
"loopx_commit": pins[0], "runner_commit": pins[1],
**({"turn_envelope": True} if args.turn_envelope else {}),
Expand Down
11 changes: 7 additions & 4 deletions benchmark/runtime/RUNTIME.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@ agents:
override_timeout_sec: 5400
kwargs:
execution_mode: heartbeat
task_entry: seeded-todo
task_entry: loopx-planned
iteration_context: fresh
reasoning_effort: max
codex_sandbox: danger-full-access
Expand Down Expand Up @@ -76,14 +76,14 @@ validator protection remains the environment owner's responsibility.

`task_entry` is independent of the execution mode:

- `seeded-todo` (the compatibility default) writes a generic execution Todo.
- `seeded-todo` (explicit compatibility/ablation choice) writes a generic execution Todo.
Follow-up phases update that Todo while it remains live and owned by this
agent; completed or deferred work gets a new Todo. Updates preserve blocked
state. The agent can still plan and replan during execution. The seed asks the
worker to read, implement and validate the task. It leaves task decomposition
to the worker and the existing task protocol. This removes the previous
unconditional successor instruction for newly seeded or updated phases.
- `loopx-planned` runs the installed `$loopx` skill against the public
- `loopx-planned` (the default for LoopX modes) runs the installed `$loopx` skill against the public
`loopx todo plan` checkpoint before execution. The checkpoint shares the
product's planner and continuation-aware Todo delta; it creates no planning
Todo and starts no host loop. Select it only for heartbeat, Turn or LoopX Goal.
Expand Down Expand Up @@ -233,7 +233,7 @@ By default, worker calls have no independent turn deadline. Harbor derives their
### Native SForge task entry

The EdgeBench runner accepts `--task-entry seeded-todo|loopx-planned` for
`heartbeat-resume` and `heartbeat-explore`; the default remains `seeded-todo`.
`heartbeat-resume` and `heartbeat-explore`; the heartbeat default is `loopx-planned`.
Official, single and native-Goal profiles reject planned entry before creating
an attempt. Select only this flag for a task-entry ablation and keep all other
inputs fixed. Runtime/profile receipts record the selected entry.
Expand Down Expand Up @@ -271,3 +271,6 @@ The two options are independent: `--task-entry` selects where the initial Todo
comes from, while `--replan-after-turns` selects which cadence the shared control
plane uses afterwards. A trial may set either, both, or neither; receipts record
both selections so a comparison keeps every other input fixed.

See [default settings and recommended ablations](SETTINGS.md) for explicit launch flags,
matched controls and rollback.
98 changes: 98 additions & 0 deletions benchmark/runtime/SETTINGS.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,98 @@
# Runner defaults and controlled ablations

New LoopX executions default to **loopx-planned** task entry. This applies to
Harbor heartbeat, Turn and LoopX Goal modes, and EdgeBench heartbeat-resume and
heartbeat-explore profiles. It replaces the seeded-todo default; pass
`task_entry: seeded-todo` / `--task-entry seeded-todo` to retain the previous
entry. Explicit settings, archived study configs and existing attempts retain
their original meaning. Official, single and native-goal profiles do not start
LoopX planning. The task-entry value recorded for these non-LoopX profiles is
`seeded-todo`, an inert compatibility value.

Planning decomposes the authorized task before execution; it is not proof that
planned entry improves scores. Treat this default as an operational choice and
use matched repetitions to test its effect.

## Defaults versus study settings

| Setting | New-run behavior | Explicit study choice |
| --- | --- | --- |
| Task entry | LoopX modes: loopx-planned | Record planned or seeded for every arm |
| Explore | Off in heartbeat-resume; on in heartbeat-explore | Keep resume as the reference |
| Turn envelope | Off | Enable only in its ablation |
| Replan cadence | 3 completed Todos | For long Todos, study 3 settled effective work turns |
| Iteration context | Harbor: fresh; EdgeBench heartbeat: resume | Freeze the provider and context within a comparison |
| Model and effort | Caller-selected | Pin both; never infer them from a profile name |
| Time and sampling | EdgeBench task defaults in [task settings](../edgebench/README.md#trial-timeouts); explicit flags override | Pin resolved seconds in the study manifest |

`replan_after_turns` counts settled effective work turns through the shared
control-plane contract, not tool calls or idle heartbeat wakes. It remains an
explicit opt-in; changing task-entry defaults does not silently change cadence.
Task timeouts bound attempts, not a requirement to consume every second.

## Recommended small study

Start with a planned Resume reference and vary one factor per comparison.
These are recommendations, not automatically launched experiments.

| Arm | EdgeBench flags relative to the reference | Question |
| --- | --- | --- |
| Reference | `--worker heartbeat-resume --task-entry loopx-planned --replan-after-turns 3` | Planned entry with effective-turn replanning |
| Seeded | Replace only `--task-entry` with `seeded-todo` | Does initial task decomposition help? |
| Explore | Replace only `--worker` with `heartbeat-explore` | Are recorded evidence and subsequent route choices useful? |
| Short envelope | Add `--turn-envelope` | Does progressive context loading reduce overhead without losing decisions? |
| Todo cadence | Omit `--replan-after-turns` (3 completed Todos) | Does effective-turn cadence avoid postponing replans on long Todos? |
| No LoopX | `--worker official`; omit LoopX-specific flags | What is the net effect of the whole LoopX treatment? |

The no-LoopX comparison changes several mechanisms; do not attribute its delta
to one component. Native versus blind feedback is a separate factor: repeat
selected matched arms within each feedback setting rather than mixing them.
Prioritize entry and cadence first; enable the other arms after verifying that
planning, settlement and evaluation work. For Explore, inspect whether evidence
was written, retrieved and adopted; an enabled hook with an empty graph is not
full mechanism use. For the envelope, check deferred context retrieval as well
as initial prompt size.

A Portfolio reference command after source, image and credential preflight:

```sh
python -m benchmark.edgebench.run \
--task portfolio_risk_calibration --tasks-dir "$TASKS_DIR" \
--log-dir "$RUNS_DIR" --run-id "$NEW_ATTEMPT_ID" \
--worker heartbeat-resume --task-entry loopx-planned \
--replan-after-turns 3 --feedback blind \
--model "$MODEL" --effort xhigh --timeout 43200 --eval-interval 300 \
--judge-url "$JUDGE_URL"
```

Use a fresh attempt id for each arm; set the runtime's documented source pins,
credentials and API proxy before launch. This example fixes the comparison
budget and sampling explicitly rather than relying on changing defaults.
For Lean, use `--task lean_analysis_proofs --eval-interval 1800` and the same
explicit total budget for all arms. Evaluator concurrency is an independently
recorded service setting, not inherited from another task.

Harbor uses the same task-entry owner. In the agent kwargs, set
`execution_mode: heartbeat`, `task_entry: loopx-planned`,
`replan_after_turns: 3`, `iteration_context: resume` and `turn_envelope: false`
for an equivalent mechanism reference. For Todo cadence, remove
`replan_after_turns` and set `replan_after_todos: 3`; the two cadence fields are
mutually exclusive. The named Explore profile in this matrix is EdgeBench-specific; do not assume
an equivalent Harbor kwargs switch. Keep the dataset, provider and validation
configuration unchanged.

## Readback and interpretation

Inspect resolved runtime/profile receipts, installed source revisions and the
first planning/seed checkpoint before scoring a treatment. Pin task, evaluator,
images, model, effort, budget and sampling schedule. Retain missing points as
missing; compare common elapsed-time windows, and keep each terminal snapshot.
Record early closure, planning cost, useful work turns and evidence adoption
alongside score. A single stochastic run or a cross-version historical delta
cannot isolate an individual fix.

Old attempts are immutable references. A rerun after multiple fixes measures
the combined revision change unless each fix has a matched control. Roll back
this default through explicit seeded entry in a new attempt, not by rewriting
old receipts or changing an active worker. None of these settings grants model,
submission, credential or launch authority.
4 changes: 3 additions & 1 deletion benchmark/runtime/codex.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,14 +22,16 @@ class Execution:
sandbox: str = "danger-full-access"
timeout_seconds: float | None = None
validation_command: tuple[str, ...] = ()
task_entry: str = "seeded-todo"
task_entry: str | None = None
turn_envelope: bool = False

def __post_init__(self) -> None:
if not isinstance(self.turn_envelope, bool) or (self.turn_envelope and self.mode != "heartbeat"):
raise ValueError("turn_envelope requires heartbeat execution and a boolean opt-in")
if self.mode not in MODES or self.context not in CONTEXTS:
raise ValueError("unsupported execution mode or iteration context")
if self.task_entry is None:
object.__setattr__(self, "task_entry", "loopx-planned" if self.uses_loopx else "seeded-todo")
if self.task_entry not in TASK_ENTRIES:
raise ValueError("unsupported task entry")
if self.task_entry == "loopx-planned" and not self.uses_loopx:
Expand Down
2 changes: 1 addition & 1 deletion benchmark/runtime/harbor.py
Original file line number Diff line number Diff line change
Expand Up @@ -56,7 +56,7 @@ def __init__(
scheduler_timeout_sec=5080,
replan_after_todos=None,
replan_after_turns=None,
task_entry="seeded-todo",
task_entry=None,
planning_timeout_sec=300,
turn_envelope=False,
**kwargs,
Expand Down
13 changes: 7 additions & 6 deletions benchmark/runtime/sforge.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,7 +15,7 @@

from sforge.harness.agent.codex import CodexAgent

from .codex import Execution, TASK_ENTRIES, prepare_codex_home
from .codex import Execution, prepare_codex_home
from .codex_offline import CodexOffline
from .harbor import (
BenchmarkCodex, _GOAL_ID, _PYTHON, _SCHEDULER_STATE, _SRC,
Expand Down Expand Up @@ -81,7 +81,7 @@ class SForgeWorker(CodexAgent):
def __init__(self, config, *, profile: str, cwd: str,
timeout_seconds: int = DEFAULT_TIMEOUT_SECONDS,
blind_prompt: str | None = None,
task_entry: str = "seeded-todo",
task_entry: str | None = None,
turn_envelope: bool = False,
replan_after_turns: int | None = None):
super().__init__(config)
Expand All @@ -98,11 +98,12 @@ def __init__(self, config, *, profile: str, cwd: str,
raise ValueError("Explicit model and reasoning effort are required")
if not os.environ.get("CODEX_AUTH_JSON_PATH"):
raise ValueError("Set CODEX_AUTH_JSON_PATH to the trial credential source")
if task_entry not in TASK_ENTRIES:
raise ValueError("unsupported task entry")
if task_entry != "seeded-todo" and not profile.startswith("heartbeat-"):
if task_entry == "loopx-planned" and not profile.startswith("heartbeat-"):
raise ValueError("loopx-planned requires a heartbeat profile")
self.task_entry = task_entry
self.task_entry = Execution(
mode="heartbeat" if profile.startswith("heartbeat-") else "plain",
task_entry=task_entry,
).task_entry
if replan_after_turns is not None:
if (type(replan_after_turns) is not int or
not 1 <= replan_after_turns <= 5):
Expand Down
2 changes: 1 addition & 1 deletion benchmark/runtime/worker.py
Original file line number Diff line number Diff line change
Expand Up @@ -249,7 +249,7 @@ def run_once(env: dict[str, str]) -> dict:
timeout_seconds=(float(env["LOOPX_CODEX_TURN_TIMEOUT_SEC"])
if env.get("LOOPX_CODEX_TURN_TIMEOUT_SEC") else None),
validation_command=json.loads(env.get("LOOPX_VALIDATION_COMMAND_JSON", "[]")),
task_entry=env.get("LOOPX_TASK_ENTRY", "seeded-todo"),
task_entry=env.get("LOOPX_TASK_ENTRY"),
turn_envelope=env.get("LOOPX_TURN_ENVELOPE", "0") == "1",
)
stage = env.get("LOOPX_TASK_STAGE", "execute")
Expand Down
2 changes: 1 addition & 1 deletion benchmark/swe-marathon/configs/shared-heartbeat.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ agents:
override_timeout_sec: 5400
kwargs:
execution_mode: heartbeat
task_entry: seeded-todo # Use loopx-planned for the product planning checkpoint.
task_entry: loopx-planned # Set seeded-todo for the entry-policy ablation.
iteration_context: fresh
reasoning_effort: high
turn_timeout_sec: null
Expand Down
Loading
Loading