diff --git a/benchmark/LHTB/README.md b/benchmark/LHTB/README.md index 765a59035a..0e885debb1 100644 --- a/benchmark/LHTB/README.md +++ b/benchmark/LHTB/README.md @@ -107,7 +107,8 @@ document. Registries are inside their own task containers, so no Goal, Todo, or scheduler state is shared between the 46 trials. The adapter registers the project through the current public bootstrap CLI and -adds the benchmark phase as an explicit P0 Todo. Retired onboarding flags are +plans phase Todos from the task by default (or adds a generic P0 Todo when +seeded entry is explicitly selected). Retired onboarding flags are not replayed; bootstrap and Todo lifecycle follow the installed product version. ## Shared execution configuration @@ -118,8 +119,9 @@ accept `resume`, sharing the same Goal/Agent session across planning, wakes and an argv array for the independently protected task validator. No generic benchmark scoring or hidden-verifier feedback is introduced. -`LOOPX_TASK_ENTRY=seeded-todo` preserves the generic phase Todo default. -`LOOPX_TASK_ENTRY=loopx-planned` invokes the product planning checkpoint before +`LOOPX_TASK_ENTRY=seeded-todo` retains the former generic phase Todo entry. +Omitting this setting now selects `loopx-planned` for LoopX modes and invokes +the product planning checkpoint before heartbeat, Turn or LoopX Goal execution. `LOOPX_PLANNING_TIMEOUT_SEC` defaults to 300 and consumes the existing phase budget. Both entry policies preserve existing waits when new phases arrive. See the shared runtime for session and diff --git a/benchmark/LHTB/run.sh b/benchmark/LHTB/run.sh index 7596ee06ef..95a0471353 100755 --- a/benchmark/LHTB/run.sh +++ b/benchmark/LHTB/run.sh @@ -47,7 +47,7 @@ LHTB_MODELONLY_GATEWAY="${LHTB_MODELONLY_GATEWAY:-192.0.2.1}" LOOPX_SRC_DIR="${LOOPX_SRC_DIR:-$LOOPX_ROOT}" export PYTHONPATH="$LOOPX_SRC_DIR${PYTHONPATH:+:$PYTHONPATH}" LOOPX_EXECUTION_MODE="${LOOPX_EXECUTION_MODE:-heartbeat}" -LOOPX_TASK_ENTRY="${LOOPX_TASK_ENTRY:-seeded-todo}" +LOOPX_TASK_ENTRY="${LOOPX_TASK_ENTRY:-}" LOOPX_PLANNING_TIMEOUT_SEC="${LOOPX_PLANNING_TIMEOUT_SEC:-300}" LOOPX_ITERATION_CONTEXT="${LOOPX_ITERATION_CONTEXT:-fresh}" LOOPX_VALIDATION_COMMAND_JSON="${LOOPX_VALIDATION_COMMAND_JSON:-[]}" @@ -117,10 +117,14 @@ if [[ "$MODE" == smoke ]]; then expected_task_count=1 job_suffix="smoke-${SMOKE_TASK}" fi -job_name="lhtb-${LOOPX_EXECUTION_MODE}-${LOOPX_TASK_ENTRY}-${LOOPX_ITERATION_CONTEXT}-${job_suffix}-${run_stamp}" +job_name="lhtb-${LOOPX_EXECUTION_MODE}-${LOOPX_TASK_ENTRY:-default}-${LOOPX_ITERATION_CONTEXT}-${job_suffix}-${run_stamp}" generated_config="$CODE_DIR/.generated/${job_name}.yaml" jobs_dir="$CODE_DIR/runs" +task_entry_args=() +if [[ -n "$LOOPX_TASK_ENTRY" ]]; then + task_entry_args=(--task-entry "$LOOPX_TASK_ENTRY") +fi turn_timeout_args=() if [[ -n "$LOOPX_CODEX_TURN_TIMEOUT_SEC" ]]; then turn_timeout_args=(--turn-timeout "$LOOPX_CODEX_TURN_TIMEOUT_SEC") @@ -135,7 +139,7 @@ fi --effort "$REASONING_EFFORT" \ --timeout "$AGENT_TIMEOUT_SEC" \ --execution-mode "$LOOPX_EXECUTION_MODE" \ - --task-entry "$LOOPX_TASK_ENTRY" \ + "${task_entry_args[@]}" \ --planning-timeout "$LOOPX_PLANNING_TIMEOUT_SEC" \ --iteration-context "$LOOPX_ITERATION_CONTEXT" \ --validation-command-json "$LOOPX_VALIDATION_COMMAND_JSON" \ diff --git a/benchmark/LHTB/scripts/preflight.py b/benchmark/LHTB/scripts/preflight.py index cc3fd909d7..5cc78fdca5 100755 --- a/benchmark/LHTB/scripts/preflight.py +++ b/benchmark/LHTB/scripts/preflight.py @@ -128,7 +128,7 @@ def check(label: str, passed: bool, detail: str) -> None: ) execution = Execution( mode=kwargs.get("execution_mode", "heartbeat"), - task_entry=kwargs.get("task_entry", "seeded-todo"), + task_entry=kwargs.get("task_entry"), context=kwargs.get("iteration_context", "fresh"), validation_command=kwargs.get("validation_command", []), ) diff --git a/benchmark/LHTB/scripts/render_config.py b/benchmark/LHTB/scripts/render_config.py index 668fb12445..942552e931 100755 --- a/benchmark/LHTB/scripts/render_config.py +++ b/benchmark/LHTB/scripts/render_config.py @@ -25,7 +25,8 @@ def main() -> int: parser.add_argument("--task", action="append", default=[]) parser.add_argument("--execution-mode", choices=MODES, default="heartbeat") parser.add_argument("--iteration-context", choices=CONTEXTS, default="fresh") - parser.add_argument("--task-entry", choices=TASK_ENTRIES, default="seeded-todo") + parser.add_argument("--task-entry", choices=TASK_ENTRIES, + help="LoopX modes default to loopx-planned; explicit seeded-todo retains legacy entry") parser.add_argument("--planning-timeout", type=float, default=300) parser.add_argument("--validation-command-json", default="[]") parser.add_argument("--turn-timeout", type=float, default=None) diff --git a/benchmark/edgebench/run.py b/benchmark/edgebench/run.py index 164263daf4..be65e7fab9 100644 --- a/benchmark/edgebench/run.py +++ b/benchmark/edgebench/run.py @@ -93,7 +93,8 @@ def main(argv=None): parser.add_argument("--effort", choices=("low", "medium", "high", "xhigh"), required=True) parser.add_argument("--timeout", type=int, help="Total trial seconds; task-defaults.json overrides the 18h fallback") - parser.add_argument("--task-entry", choices=TASK_ENTRIES, default="seeded-todo") + parser.add_argument("--task-entry", choices=TASK_ENTRIES, + help="Heartbeat default: loopx-planned; seeded-todo is an explicit ablation") parser.add_argument("--replan-after-turns", type=int, choices=range(1, 6), help="Opt in to settled work Turn cadence for heartbeat profiles") parser.add_argument("--eval-interval", type=int, @@ -104,7 +105,7 @@ def main(argv=None): args = parser.parse_args(argv) args.timeout = _task_default(args.task, "timeout_seconds", args.timeout, DEFAULT_TIMEOUT_SECONDS) args.eval_interval = _task_default(args.task, "eval_interval_seconds", args.eval_interval, 300) - if args.task_entry != "seeded-todo" and not args.worker.startswith("heartbeat-"): + if args.task_entry == "loopx-planned" and not args.worker.startswith("heartbeat-"): parser.error("--task-entry loopx-planned requires a heartbeat worker") if args.turn_envelope and args.worker not in {"heartbeat-resume", "heartbeat-explore"}: parser.error("--turn-envelope requires a heartbeat worker") @@ -159,7 +160,7 @@ def main(argv=None): raise RuntimeError(f"Missing native image: {image}") receipt = { "run_id": args.run_id, "task": args.task, "worker": args.worker, - "task_entry": args.task_entry, + "task_entry": agent.task_entry, "model": args.model, "effort": args.effort, "timeout_seconds": args.timeout, "loopx_commit": pins[0], "runner_commit": pins[1], **({"turn_envelope": True} if args.turn_envelope else {}), diff --git a/benchmark/runtime/RUNTIME.md b/benchmark/runtime/RUNTIME.md index d20d2c9b3b..4b7135cd6c 100644 --- a/benchmark/runtime/RUNTIME.md +++ b/benchmark/runtime/RUNTIME.md @@ -17,7 +17,7 @@ agents: override_timeout_sec: 5400 kwargs: execution_mode: heartbeat - task_entry: seeded-todo + task_entry: loopx-planned iteration_context: fresh reasoning_effort: max codex_sandbox: danger-full-access @@ -76,14 +76,14 @@ validator protection remains the environment owner's responsibility. `task_entry` is independent of the execution mode: -- `seeded-todo` (the compatibility default) writes a generic execution Todo. +- `seeded-todo` (explicit compatibility/ablation choice) writes a generic execution Todo. Follow-up phases update that Todo while it remains live and owned by this agent; completed or deferred work gets a new Todo. Updates preserve blocked state. The agent can still plan and replan during execution. The seed asks the worker to read, implement and validate the task. It leaves task decomposition to the worker and the existing task protocol. This removes the previous unconditional successor instruction for newly seeded or updated phases. -- `loopx-planned` runs the installed `$loopx` skill against the public +- `loopx-planned` (the default for LoopX modes) runs the installed `$loopx` skill against the public `loopx todo plan` checkpoint before execution. The checkpoint shares the product's planner and continuation-aware Todo delta; it creates no planning Todo and starts no host loop. Select it only for heartbeat, Turn or LoopX Goal. @@ -233,7 +233,7 @@ By default, worker calls have no independent turn deadline. Harbor derives their ### Native SForge task entry The EdgeBench runner accepts `--task-entry seeded-todo|loopx-planned` for -`heartbeat-resume` and `heartbeat-explore`; the default remains `seeded-todo`. +`heartbeat-resume` and `heartbeat-explore`; the heartbeat default is `loopx-planned`. Official, single and native-Goal profiles reject planned entry before creating an attempt. Select only this flag for a task-entry ablation and keep all other inputs fixed. Runtime/profile receipts record the selected entry. @@ -271,3 +271,6 @@ The two options are independent: `--task-entry` selects where the initial Todo comes from, while `--replan-after-turns` selects which cadence the shared control plane uses afterwards. A trial may set either, both, or neither; receipts record both selections so a comparison keeps every other input fixed. + +See [default settings and recommended ablations](SETTINGS.md) for explicit launch flags, +matched controls and rollback. diff --git a/benchmark/runtime/SETTINGS.md b/benchmark/runtime/SETTINGS.md new file mode 100644 index 0000000000..e616117740 --- /dev/null +++ b/benchmark/runtime/SETTINGS.md @@ -0,0 +1,98 @@ +# Runner defaults and controlled ablations + +New LoopX executions default to **loopx-planned** task entry. This applies to +Harbor heartbeat, Turn and LoopX Goal modes, and EdgeBench heartbeat-resume and +heartbeat-explore profiles. It replaces the seeded-todo default; pass +`task_entry: seeded-todo` / `--task-entry seeded-todo` to retain the previous +entry. Explicit settings, archived study configs and existing attempts retain +their original meaning. Official, single and native-goal profiles do not start +LoopX planning. The task-entry value recorded for these non-LoopX profiles is +`seeded-todo`, an inert compatibility value. + +Planning decomposes the authorized task before execution; it is not proof that +planned entry improves scores. Treat this default as an operational choice and +use matched repetitions to test its effect. + +## Defaults versus study settings + +| Setting | New-run behavior | Explicit study choice | +| --- | --- | --- | +| Task entry | LoopX modes: loopx-planned | Record planned or seeded for every arm | +| Explore | Off in heartbeat-resume; on in heartbeat-explore | Keep resume as the reference | +| Turn envelope | Off | Enable only in its ablation | +| Replan cadence | 3 completed Todos | For long Todos, study 3 settled effective work turns | +| Iteration context | Harbor: fresh; EdgeBench heartbeat: resume | Freeze the provider and context within a comparison | +| Model and effort | Caller-selected | Pin both; never infer them from a profile name | +| Time and sampling | EdgeBench task defaults in [task settings](../edgebench/README.md#trial-timeouts); explicit flags override | Pin resolved seconds in the study manifest | + +`replan_after_turns` counts settled effective work turns through the shared +control-plane contract, not tool calls or idle heartbeat wakes. It remains an +explicit opt-in; changing task-entry defaults does not silently change cadence. +Task timeouts bound attempts, not a requirement to consume every second. + +## Recommended small study + +Start with a planned Resume reference and vary one factor per comparison. +These are recommendations, not automatically launched experiments. + +| Arm | EdgeBench flags relative to the reference | Question | +| --- | --- | --- | +| Reference | `--worker heartbeat-resume --task-entry loopx-planned --replan-after-turns 3` | Planned entry with effective-turn replanning | +| Seeded | Replace only `--task-entry` with `seeded-todo` | Does initial task decomposition help? | +| Explore | Replace only `--worker` with `heartbeat-explore` | Are recorded evidence and subsequent route choices useful? | +| Short envelope | Add `--turn-envelope` | Does progressive context loading reduce overhead without losing decisions? | +| Todo cadence | Omit `--replan-after-turns` (3 completed Todos) | Does effective-turn cadence avoid postponing replans on long Todos? | +| No LoopX | `--worker official`; omit LoopX-specific flags | What is the net effect of the whole LoopX treatment? | + +The no-LoopX comparison changes several mechanisms; do not attribute its delta +to one component. Native versus blind feedback is a separate factor: repeat +selected matched arms within each feedback setting rather than mixing them. +Prioritize entry and cadence first; enable the other arms after verifying that +planning, settlement and evaluation work. For Explore, inspect whether evidence +was written, retrieved and adopted; an enabled hook with an empty graph is not +full mechanism use. For the envelope, check deferred context retrieval as well +as initial prompt size. + +A Portfolio reference command after source, image and credential preflight: + +```sh +python -m benchmark.edgebench.run \ + --task portfolio_risk_calibration --tasks-dir "$TASKS_DIR" \ + --log-dir "$RUNS_DIR" --run-id "$NEW_ATTEMPT_ID" \ + --worker heartbeat-resume --task-entry loopx-planned \ + --replan-after-turns 3 --feedback blind \ + --model "$MODEL" --effort xhigh --timeout 43200 --eval-interval 300 \ + --judge-url "$JUDGE_URL" +``` + +Use a fresh attempt id for each arm; set the runtime's documented source pins, +credentials and API proxy before launch. This example fixes the comparison +budget and sampling explicitly rather than relying on changing defaults. +For Lean, use `--task lean_analysis_proofs --eval-interval 1800` and the same +explicit total budget for all arms. Evaluator concurrency is an independently +recorded service setting, not inherited from another task. + +Harbor uses the same task-entry owner. In the agent kwargs, set +`execution_mode: heartbeat`, `task_entry: loopx-planned`, +`replan_after_turns: 3`, `iteration_context: resume` and `turn_envelope: false` +for an equivalent mechanism reference. For Todo cadence, remove +`replan_after_turns` and set `replan_after_todos: 3`; the two cadence fields are +mutually exclusive. The named Explore profile in this matrix is EdgeBench-specific; do not assume +an equivalent Harbor kwargs switch. Keep the dataset, provider and validation +configuration unchanged. + +## Readback and interpretation + +Inspect resolved runtime/profile receipts, installed source revisions and the +first planning/seed checkpoint before scoring a treatment. Pin task, evaluator, +images, model, effort, budget and sampling schedule. Retain missing points as +missing; compare common elapsed-time windows, and keep each terminal snapshot. +Record early closure, planning cost, useful work turns and evidence adoption +alongside score. A single stochastic run or a cross-version historical delta +cannot isolate an individual fix. + +Old attempts are immutable references. A rerun after multiple fixes measures +the combined revision change unless each fix has a matched control. Roll back +this default through explicit seeded entry in a new attempt, not by rewriting +old receipts or changing an active worker. None of these settings grants model, +submission, credential or launch authority. diff --git a/benchmark/runtime/codex.py b/benchmark/runtime/codex.py index 4bc6b329c0..3684895ada 100644 --- a/benchmark/runtime/codex.py +++ b/benchmark/runtime/codex.py @@ -22,7 +22,7 @@ class Execution: sandbox: str = "danger-full-access" timeout_seconds: float | None = None validation_command: tuple[str, ...] = () - task_entry: str = "seeded-todo" + task_entry: str | None = None turn_envelope: bool = False def __post_init__(self) -> None: @@ -30,6 +30,8 @@ def __post_init__(self) -> None: raise ValueError("turn_envelope requires heartbeat execution and a boolean opt-in") if self.mode not in MODES or self.context not in CONTEXTS: raise ValueError("unsupported execution mode or iteration context") + if self.task_entry is None: + object.__setattr__(self, "task_entry", "loopx-planned" if self.uses_loopx else "seeded-todo") if self.task_entry not in TASK_ENTRIES: raise ValueError("unsupported task entry") if self.task_entry == "loopx-planned" and not self.uses_loopx: diff --git a/benchmark/runtime/harbor.py b/benchmark/runtime/harbor.py index ab0f8f2f15..dd6cdc5eb9 100644 --- a/benchmark/runtime/harbor.py +++ b/benchmark/runtime/harbor.py @@ -56,7 +56,7 @@ def __init__( scheduler_timeout_sec=5080, replan_after_todos=None, replan_after_turns=None, - task_entry="seeded-todo", + task_entry=None, planning_timeout_sec=300, turn_envelope=False, **kwargs, diff --git a/benchmark/runtime/sforge.py b/benchmark/runtime/sforge.py index fcb1bdcc81..25e62bd003 100644 --- a/benchmark/runtime/sforge.py +++ b/benchmark/runtime/sforge.py @@ -15,7 +15,7 @@ from sforge.harness.agent.codex import CodexAgent -from .codex import Execution, TASK_ENTRIES, prepare_codex_home +from .codex import Execution, prepare_codex_home from .codex_offline import CodexOffline from .harbor import ( BenchmarkCodex, _GOAL_ID, _PYTHON, _SCHEDULER_STATE, _SRC, @@ -81,7 +81,7 @@ class SForgeWorker(CodexAgent): def __init__(self, config, *, profile: str, cwd: str, timeout_seconds: int = DEFAULT_TIMEOUT_SECONDS, blind_prompt: str | None = None, - task_entry: str = "seeded-todo", + task_entry: str | None = None, turn_envelope: bool = False, replan_after_turns: int | None = None): super().__init__(config) @@ -98,11 +98,12 @@ def __init__(self, config, *, profile: str, cwd: str, raise ValueError("Explicit model and reasoning effort are required") if not os.environ.get("CODEX_AUTH_JSON_PATH"): raise ValueError("Set CODEX_AUTH_JSON_PATH to the trial credential source") - if task_entry not in TASK_ENTRIES: - raise ValueError("unsupported task entry") - if task_entry != "seeded-todo" and not profile.startswith("heartbeat-"): + if task_entry == "loopx-planned" and not profile.startswith("heartbeat-"): raise ValueError("loopx-planned requires a heartbeat profile") - self.task_entry = task_entry + self.task_entry = Execution( + mode="heartbeat" if profile.startswith("heartbeat-") else "plain", + task_entry=task_entry, + ).task_entry if replan_after_turns is not None: if (type(replan_after_turns) is not int or not 1 <= replan_after_turns <= 5): diff --git a/benchmark/runtime/worker.py b/benchmark/runtime/worker.py index 9df8d22761..bc58d07e9e 100644 --- a/benchmark/runtime/worker.py +++ b/benchmark/runtime/worker.py @@ -249,7 +249,7 @@ def run_once(env: dict[str, str]) -> dict: timeout_seconds=(float(env["LOOPX_CODEX_TURN_TIMEOUT_SEC"]) if env.get("LOOPX_CODEX_TURN_TIMEOUT_SEC") else None), validation_command=json.loads(env.get("LOOPX_VALIDATION_COMMAND_JSON", "[]")), - task_entry=env.get("LOOPX_TASK_ENTRY", "seeded-todo"), + task_entry=env.get("LOOPX_TASK_ENTRY"), turn_envelope=env.get("LOOPX_TURN_ENVELOPE", "0") == "1", ) stage = env.get("LOOPX_TASK_STAGE", "execute") diff --git a/benchmark/swe-marathon/configs/shared-heartbeat.yaml b/benchmark/swe-marathon/configs/shared-heartbeat.yaml index 1409399d62..a3b726ae45 100644 --- a/benchmark/swe-marathon/configs/shared-heartbeat.yaml +++ b/benchmark/swe-marathon/configs/shared-heartbeat.yaml @@ -6,7 +6,7 @@ agents: override_timeout_sec: 5400 kwargs: execution_mode: heartbeat - task_entry: seeded-todo # Use loopx-planned for the product planning checkpoint. + task_entry: loopx-planned # Set seeded-todo for the entry-policy ablation. iteration_context: fresh reasoning_effort: high turn_timeout_sec: null diff --git a/benchmark/tests/test_sforge_runtime.py b/benchmark/tests/test_sforge_runtime.py index 79a36b86b7..1665f9ed1d 100644 --- a/benchmark/tests/test_sforge_runtime.py +++ b/benchmark/tests/test_sforge_runtime.py @@ -87,6 +87,7 @@ def test_worker_completion_authority(profile, resumes, monkeypatch): profile=profile, cwd="/task") assert (worker.resume_cmd is not None) is resumes assert worker.timeout_seconds == 64800 + assert worker.task_entry == ("loopx-planned" if profile.startswith("heartbeat-") else "seeded-todo") if profile in {"official", "single"}: command = worker.format_run_cmd("/task.md", internet=False) assert 'model_reasoning_effort="xhigh"' in command @@ -396,13 +397,14 @@ def test_edgebench_rejects_envelope_before_creating_trial(tmp_path): @pytest.mark.parametrize("enabled", [False, True]) +@pytest.mark.parametrize("entry", [None, "seeded-todo"]) @pytest.mark.parametrize("task,timeout_args,expected,interval", [ ("fixture", [], 64800, 300), ("portfolio_risk_calibration", [], 43200, 300), ("lean_analysis_proofs", [], 64800, 1800), ("portfolio_risk_calibration", ["--timeout", "1800", "--eval-interval", "60"], 1800, 60), ("lean_analysis_proofs", ["--eval-interval", "0"], 64800, 0)]) -def test_edgebench_receipt_records_only_enabled_treatment(tmp_path, monkeypatch, enabled, task, timeout_args, expected, interval): +def test_edgebench_receipt_records_resolved_entry_and_enabled_treatment(tmp_path, monkeypatch, enabled, task, timeout_args, expected, interval, entry): pytest.importorskip("sforge") pytest.importorskip("harbor") from types import SimpleNamespace @@ -415,7 +417,7 @@ def test_edgebench_receipt_records_only_enabled_treatment(tmp_path, monkeypatch, monkeypatch.setattr(run, "load_benchmark", lambda *a: None) monkeypatch.setattr(run, "make_task_spec", lambda *a: SimpleNamespace( cwd="/task", work_image_key="work", judge_image_key="judge", internet=False)) - monkeypatch.setattr(run, "SForgeWorker", lambda *a, **k: SimpleNamespace(resume_cmd="resume")) + monkeypatch.setenv("CODEX_AUTH_JSON_PATH", "/synthetic-credential") monkeypatch.setattr(run, "RecordingDockerBackend", lambda **k: SimpleNamespace(image_exists=lambda image: True)) def stop_before_solver(**kwargs): assert kwargs["timeout"] == kwargs["config"].agent_timeout == expected @@ -425,9 +427,12 @@ def stop_before_solver(**kwargs): args = ["--task", task, "--tasks-dir", str(tmp_path), "--log-dir", str(tmp_path), "--run-id", "receipt", "--worker", "heartbeat-resume", "--model", "fixture", "--effort", "xhigh", "--judge-url", "http://127.0.0.1:9999"] + if entry: + args += ["--task-entry", entry] with pytest.raises(RuntimeError, match="synthetic launch failure"): run.main(args + timeout_args + (["--turn-envelope"] if enabled else [])) receipt = json.loads((tmp_path / f"runs/receipt/{task}/runtime-receipt.json").read_text()) + assert receipt["task_entry"] == (entry or "loopx-planned") assert receipt["timeout_seconds"] == expected assert receipt["eval_interval"] == interval assert ("turn_envelope" in receipt) is enabled diff --git a/benchmark/tests/test_shared_codex_runtime.py b/benchmark/tests/test_shared_codex_runtime.py index 0c4b0207c3..518820b85e 100644 --- a/benchmark/tests/test_shared_codex_runtime.py +++ b/benchmark/tests/test_shared_codex_runtime.py @@ -353,13 +353,13 @@ def test_baseline_and_treatment_use_same_harbor_entry(tmp_path): @pytest.mark.parametrize("existing", [False, True]) @pytest.mark.parametrize("turns", [None, 2]) -def test_phase_bootstrap_uses_current_public_cli(tmp_path, monkeypatch, existing, turns): +def test_explicit_seeded_phase_bootstrap_uses_current_public_cli(tmp_path, monkeypatch, existing, turns): pytest.importorskip("harbor") from benchmark.runtime.harbor import BenchmarkCodex from loopx.cli import build_parser agent = BenchmarkCodex(logs_dir=tmp_path, model_name="openai/fixture", - replan_after_turns=turns) + replan_after_turns=turns, task_entry="seeded-todo") field = "replan_after_effective_turns" if turns else "replan_after_completed_todos" calls = [] diff --git a/benchmark/tests/test_task_entry.py b/benchmark/tests/test_task_entry.py index 5c17cfa1dd..295d2c5c83 100644 --- a/benchmark/tests/test_task_entry.py +++ b/benchmark/tests/test_task_entry.py @@ -29,6 +29,16 @@ def test_invalid_entry_rejected_before_model_call(kwargs): Execution(**kwargs) +@pytest.mark.parametrize("mode", ["heartbeat", "turn", "loopx-goal", "plain", "native-goal"]) +def test_task_entry_defaults_to_planning_only_for_loopx_modes(mode): + kwargs = {"mode": mode} + if mode == "turn": + kwargs["validation_command"] = ("python", "validate.py") + expected = "loopx-planned" if mode in {"heartbeat", "turn", "loopx-goal"} else "seeded-todo" + assert Execution(**kwargs).task_entry == expected + assert Execution(**kwargs, task_entry="seeded-todo").task_entry == "seeded-todo" + + @pytest.fixture def planning_env(tmp_path): project = tmp_path / "project" @@ -114,6 +124,8 @@ def test_planning_writes_real_todo_then_reuses_it_without_executing_task( planning_env, context ): planning_env["LOOPX_ITERATION_CONTEXT"] = context + if context == "fresh": + planning_env.pop("LOOPX_TASK_ENTRY") # Omitted worker setting uses the same planner. ids = [] for invocation in range(2): receipt = run_once(planning_env) @@ -240,7 +252,6 @@ def test_planning_budget_and_blocked_handoff_use_the_real_adapter_run( agent = harbor.BenchmarkCodex( logs_dir=tmp_path, model_name="openai/fixture", - task_entry="loopx-planned", turn_timeout_sec=250, scheduler_timeout_sec=500, )