Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
49 changes: 45 additions & 4 deletions devtools/verify.py
Original file line number Diff line number Diff line change
Expand Up @@ -1834,6 +1834,21 @@ def _stop_after_failed_step(label: str) -> bool:
return label.startswith("pytest") or label in {"lab smoke", "bench slo"}


def _seed_shard_failure_requires_stop(step: Mapping[str, Any], *, shard_complete: bool) -> bool:
"""Stop shard admission after harness failure while retaining red-test evidence.

A normal pytest exit 1 with a structured ``pytest_failed`` diagnosis is
useful seed evidence: later shards can still populate the resumable
dependency graph. Timeouts, resource refusals, worker/internal errors,
usage errors, and unclassified failures mean the harness is no longer
healthy enough to admit another expensive shard.
"""
exit_code = step.get("exit")
if exit_code == 0:
return False
return not (exit_code == 1 and step.get("diagnosis") == "pytest_failed" and shard_complete)


# ── step builder ────────────────────────────────────────────────────


Expand Down Expand Up @@ -3440,8 +3455,26 @@ def main(argv: list[str] | None = None) -> int:
shard_index=shard_index,
step=shard_result,
)
if shard_rc != 0 and exit_code == 0:
exit_code = shard_rc
checkpointed_shards = prepared_seed_attempt.get("shards")
if (
not isinstance(checkpointed_shards, list)
or shard_index > len(checkpointed_shards)
or not isinstance(checkpointed_shards[shard_index - 1], Mapping)
):
raise RuntimeError("testmon seed shard checkpoint is malformed")
shard_complete = checkpointed_shards[shard_index - 1].get("status") == SeedShardStatus.COMPLETE.value
if shard_rc != 0:
stop_seed = _seed_shard_failure_requires_stop(
shard_result,
shard_complete=shard_complete,
)
if exit_code == 0 or stop_seed:
# A later infrastructure failure is the terminal
# condition even when an earlier shard recorded
# ordinary red-test evidence.
exit_code = shard_rc
Comment thread
coderabbitai[bot] marked this conversation as resolved.
if stop_seed:
break
continue
if label in {"pytest testmon", "pytest testmon (broad)"} and not args.seed_testmon and not full_pytest:
_refresh_testmon_selection_attempt(step=step_result, run=verify_run, exit_code=rc)
Expand Down Expand Up @@ -3477,14 +3510,22 @@ def main(argv: list[str] | None = None) -> int:
"total_duration_s": total_duration,
"exit_code": exit_code,
}
pytest_diagnosis = next(
fallback_pytest_diagnosis = next(
(
str(step["diagnosis"])
for step in step_results
for step in reversed(step_results)
if str(step.get("name", "")).startswith("pytest") and "diagnosis" in step
),
None,
)
pytest_diagnosis = next(
(
str(step["diagnosis"])
for step in reversed(step_results)
if str(step.get("name", "")).startswith("pytest") and step.get("exit") == exit_code and "diagnosis" in step
),
fallback_pytest_diagnosis,
)
if pytest_diagnosis is not None:
history_entry["diagnosis"] = pytest_diagnosis
if seed_receipt is not None:
Expand Down
131 changes: 131 additions & 0 deletions tests/unit/devtools/test_verify.py
Original file line number Diff line number Diff line change
Expand Up @@ -3161,6 +3161,137 @@ def fake_run(label: str, command: list[str], **kwargs: object) -> tuple[int, flo
assert '"exit_code": 1' in payload


@pytest.mark.parametrize(
("shard_results", "expected_exit", "expected_diagnosis", "expected_statuses"),
[
(
[(124, "pytest_timeout"), (0, "pytest_passed")],
124,
"pytest_timeout",
["incomplete", "pending"],
),
(
[(1, "pytest_failed"), (0, "pytest_passed")],
1,
"pytest_failed",
["complete", "complete"],
),
(
[(1, "pytest_failed"), (124, "pytest_timeout")],
124,
"pytest_timeout",
["complete", "incomplete"],
),
(
[(1, "pytest_failed"), (0, "pytest_passed")],
1,
"pytest_failed",
["incomplete", "pending"],
),
],
)
def test_seed_testmon_stops_only_after_infrastructure_failed_shard(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
capsys: pytest.CaptureFixture[str],
shard_results: list[tuple[int, str]],
expected_exit: int,
expected_diagnosis: str,
expected_statuses: list[str],
) -> None:
nodeids = ["tests/test_seed.py::test_one", "tests/test_seed.py::test_two"]
collection_dir = tmp_path / "collection"
collection_dir.mkdir()
(collection_dir / "selection.json").write_text(
json.dumps(
{
"selected_count": len(nodeids),
"selected_nodeids": nodeids,
"selected_nodeids_omitted": 0,
}
)
)
calls: list[str] = []
checkpointed: list[int] = []
finalized_shard_statuses: list[str] = []

def fake_run(label: str, command: list[str], **kwargs: object) -> tuple[int, float, dict[str, object]]:
del command, kwargs
calls.append(label)
if label == "pytest seed-testmon collect":
return 0, 0.01, {"artifact_dir": str(collection_dir)}
if label.startswith("pytest seed-testmon shard "):
shard_index = int(label.rsplit(" ", 1)[1].split("/", 1)[0])
shard_exit, diagnosis = shard_results[shard_index - 1]
return shard_exit, 0.01, {"diagnosis": diagnosis}
pytest.fail(f"unexpected seed step: {label}")

def fake_checkpoint(*, prepared: dict[str, object], shard_index: int, step: dict[str, object]) -> dict[str, object]:
del step
checkpointed.append(shard_index)
raw_shards = prepared["shards"]
assert isinstance(raw_shards, list)
assert all(isinstance(shard, dict) for shard in raw_shards)
shards = [dict(shard) for shard in raw_shards]
shards[shard_index - 1]["status"] = expected_statuses[shard_index - 1]
return {**prepared, "shards": shards}

def fake_finalize(
*, prepared: dict[str, object], step_results: list[dict[str, object]], exit_code: int
) -> dict[str, object]:
del step_results
assert exit_code == expected_exit
raw_shards = prepared["shards"]
assert isinstance(raw_shards, list)
assert all(isinstance(shard, dict) for shard in raw_shards)
finalized_shard_statuses.extend(str(shard["status"]) for shard in raw_shards)
return {
"status": "incomplete" if "incomplete" in expected_statuses else "complete",
"outcome": "resource_timeout" if expected_exit == 124 else "red-baseline",
"resume": False,
"expected_count": len(nodeids),
"release_baseline_allowed": False,
}

monkeypatch.setattr(verify, "TESTMON_SEED_SHARD_SIZE", 1)
with (
patch("devtools.verify._anchor_verification_paths"),
patch("devtools.verify.maybe_bootstrap_testmon_seed", return_value=None),
patch("devtools.verify._run", side_effect=fake_run),
patch(
"devtools.verify.build_verify_steps",
return_value=[("pytest seed-testmon collect", ["pytest", "--collect-only"])],
),
patch("devtools.verify._git_head", return_value="head"),
patch("devtools.verify._git_committed_tree", return_value="tree"),
patch(
"devtools.verify._testmon_seed_identity",
return_value={"git_head": "head", "git_tree": "tree", "skip_slow": False, "lab": False},
),
patch("devtools.verify._testmon_seed_can_resume", return_value=False),
patch("devtools.verify._checkpoint_testmon_seed_shard", side_effect=fake_checkpoint),
patch("devtools.verify._finalize_testmon_seed_attempt", side_effect=fake_finalize),
patch("devtools.verify._testmon_release_baseline_permission", return_value=False),
patch("devtools.verify._warn_low_memory"),
patch("devtools.verify._save_history"),
patch("devtools.verify._stamp_head"),
patch("devtools.verify._notify"),
):
rc = main(["--seed-testmon", "--json"])

assert rc == expected_exit
executed_shards = sum(status != "pending" for status in expected_statuses)
assert calls == [
"pytest seed-testmon collect",
*(f"pytest seed-testmon shard {index}/2" for index in range(1, executed_shards + 1)),
]
assert checkpointed == list(range(1, executed_shards + 1))
assert finalized_shard_statuses == expected_statuses
output = json.loads(capsys.readouterr().out)
assert output["exit_code"] == expected_exit
assert output["diagnosis"] == expected_diagnosis


@pytest.mark.parametrize(
("argv", "expected_scope", "expected_permission"),
[
Expand Down