From 26f7b8a774652c480821c027ca7350d18ae35a41 Mon Sep 17 00:00:00 2001 From: Aryan Date: Mon, 28 Sep 2026 21:54:20 +0530 Subject: [PATCH 1/3] feat: add PPO learning-curve benchmark (#245) Train and evaluate fresh PPO models across configured timestep budgets, recording fixed-seed metrics and exporting JSON/CSV results with optional plots. Add CLI, validation, truncation handling, regression coverage, and usage documentation. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- README.md | 10 +- docs/EXPERIMENT.md | 59 +- src/adaptive_rl/__init__.py | 14 + src/adaptive_rl/benchmarking/__init__.py | 17 + .../benchmarking/learning_curve.py | 514 +++++++++++++++++ src/adaptive_rl/cli.py | 98 ++++ src/adaptive_rl/config.py | 57 +- src/adaptive_rl/environments/drone.py | 4 + src/adaptive_rl/evaluation/evaluator.py | 40 +- tests/test_cli.py | 1 + tests/test_drone.py | 56 ++ tests/test_learning_curve_benchmark.py | 527 ++++++++++++++++++ 12 files changed, 1382 insertions(+), 15 deletions(-) create mode 100644 src/adaptive_rl/benchmarking/__init__.py create mode 100644 src/adaptive_rl/benchmarking/learning_curve.py create mode 100644 tests/test_learning_curve_benchmark.py diff --git a/README.md b/README.md index 3662fb6..1cf94db 100644 --- a/README.md +++ b/README.md @@ -76,10 +76,11 @@ AdaptiveRL is an educational reinforcement-learning project in which a PPO agent - **Deterministic Evaluation**: Reusable evaluation pipeline with reproducible seed control. - **Untrained Random Policy Baseline**: Built-in non-learning baseline to scientifically validate policy improvement. - **Obstacle-Density Experiment**: Controlled testing across 4, 6, and 8 obstacles to demonstrate environmental difficulty scaling. -- **Command-Line Interface (CLI)**: Typer-based CLI for training, evaluation, environment inspection, and trajectory demonstration. +- **PPO Learning-Curve Benchmark**: Train fresh PPO models across configurable timestep budgets and export evaluation metrics as JSON/CSV with optional plots. +- **Command-Line Interface (CLI)**: Typer-based CLI for training, evaluation, learning-curve benchmarking, environment inspection, and trajectory demonstration. - **Streamlit + Plotly 3D GUI**: Interactive browser-based presentation flight deck with a trajectory playback scrubber and live sensor visualization. - **Training Checkpoints**: Automatic model weight checkpointing (`.zip`) and JSON metadata export. -- **Automated Test Suite**: 49 unit and integration tests verifying kinematics, environment spaces, training lifecycle, and GUI charts. +- **Automated Test Suite**: Unit and integration tests verify kinematics, environment spaces, training lifecycle, benchmark outputs, and GUI charts. --- @@ -677,6 +678,8 @@ ARL/ │ ├── __init__.py # Package version declaration │ ├── cli.py # Typer CLI implementation │ ├── config.py # Pydantic configuration schemas and YAML loader +│ ├── benchmarking/ +│ │ └── learning_curve.py # PPO budget sweep, evaluation, and JSON/CSV/plot exports │ ├── algorithms/ │ │ ├── base.py # BaseAlgorithm abstract interface │ │ ├── ppo.py # Stable-Baselines3 PPO wrapper @@ -694,12 +697,13 @@ ARL/ │ └── training/ │ ├── callbacks.py # Episode metric logging and checkpoint callbacks │ └── trainer.py # PPOTrainer training orchestrator -└── tests/ # 49 automated unit and integration tests +└── tests/ # Automated unit and integration tests ├── test_cli.py ├── test_configuration.py ├── test_drone.py ├── test_evaluation.py ├── test_gui.py + ├── test_learning_curve_benchmark.py └── test_training.py ``` diff --git a/docs/EXPERIMENT.md b/docs/EXPERIMENT.md index 580ccd7..a3692e5 100644 --- a/docs/EXPERIMENT.md +++ b/docs/EXPERIMENT.md @@ -33,7 +33,60 @@ This document details the experimental methodology, hypotheses, benchmark variab --- -## 3. Expected Results (Hypothesized Prior to Testing) +## 3. PPO Learning-Curve Benchmark + +The budget benchmark trains a fresh PPO model from the same base configuration at each requested training budget. Every model is evaluated with the same ordered evaluation seeds, episode count per seed, and deterministic-action setting; evaluation uses the saved model and a separate fresh environment. + +```bash +adaptive-rl benchmark budgets \ + --config configs/drone_ppo.yaml \ + --budgets 5000,10000,25000,50000 \ + --training-seed 42 \ + --eval-seeds 42,43,44,45,46 \ + --episodes 20 \ + --deterministic +``` + +The command reports the budget list and output locations when complete. By default, machine-readable artifacts are written beneath `artifacts/benchmarks/`: + +```text +Learning Curve Benchmark +PPO learning-curve benchmark complete +Budgets: 5,000, 10,000, 25,000, 50,000 +Training seed: 42 +Evaluation seeds: [42, 43, 44, 45, 46] +JSON: artifacts/benchmarks/learning_curve_budget.json +CSV: artifacts/benchmarks/learning_curve_budget.csv +Plot: not generated + +artifacts/benchmarks/ +├── learning_curve_budget.json +├── learning_curve_budget.csv +└── learning_curve/ + ├── budget_5000/models/ppo_budget_5000_final.zip + ├── budget_10000/models/ppo_budget_10000_final.zip + └── ... +``` + +JSON contains benchmark settings, one result object per requested budget, and plot-ready series. CSV contains the same per-budget performance values. Pass `--plot` to additionally render `learning_curve_budget.png`; Matplotlib must be installed for that optional output. + +`budget_timesteps` records the requested budget, while `trained_timesteps` records the actual environment interactions reported by Stable-Baselines3. PPO collects complete rollouts, so a requested budget that is not a multiple of its configured `n_steps` can be exceeded up to the next rollout boundary. Compare results using `trained_timesteps` when budgets are not aligned to rollout sizes. Training duration is informational and should not be interpreted as a hardware-independent performance metric. + +Interpret the curves jointly: rising success rate and mean reward with a falling collision or timeout rate suggest improvement; flat metrics may indicate a plateau. A timeout is counted only when Gymnasium returns `truncated=True`, not merely because an episode has a particular length. The same seeds make evaluation conditions comparable, but do not remove variation from training or guarantee bit-for-bit results across hardware, PyTorch versions, or CUDA kernels. + +For a CI-sized run, copy the experiment YAML and set PPO `n_steps: 64` and `batch_size: 32` in that copy. Then run a short evaluation: + +```bash +cp configs/drone_ppo_demo.yaml /tmp/drone_ppo_ci.yaml +# Edit /tmp/drone_ppo_ci.yaml: set n_steps to 64 and batch_size to 32. +adaptive-rl benchmark budgets --config /tmp/drone_ppo_ci.yaml --budgets 64,128 --episodes 1 +``` + +The committed demo config uses `n_steps: 1024`, so those tiny budgets would be rounded up to its rollout boundary; keep the shipped training hyperparameters unchanged and use the copied config only for this CI-sized run. + +--- + +## 4. Expected Results (Hypothesized Prior to Testing) 1. **Random Action Baseline**: - Success Rate: $0.0\%$ (probability of randomly stumbling into a $1.5\text{ m}$ sphere across a $13,500\text{ m}^3$ arena without striking walls is practically zero). @@ -50,7 +103,7 @@ This document details the experimental methodology, hypotheses, benchmark variab --- -## 4. Actual Measured Results (Empirical Verification) +## 5. Actual Measured Results (Empirical Verification) All results below were generated through genuine Python 3.12 CPU execution using the canonical project commands: ```bash @@ -87,7 +140,7 @@ adaptive-rl experiment-density --model artifacts/models/drone_ppo_demo_final.zip --- -## 5. Reproducibility Guarantee +## 6. Reproducibility Guarantee To independently reproduce the identical metrics on any student laptop: ```bash diff --git a/src/adaptive_rl/__init__.py b/src/adaptive_rl/__init__.py index b5627ef..f86dade 100644 --- a/src/adaptive_rl/__init__.py +++ b/src/adaptive_rl/__init__.py @@ -4,8 +4,16 @@ a simulated 3D drone through obstacles toward a target waypoint. """ +from adaptive_rl.benchmarking import ( + LearningCurveBenchmarkResult, + LearningCurvePoint, + plot_learning_curve, + run_learning_curve_benchmark, + validate_budgets, +) from adaptive_rl.config import ( AlgorithmConfig, + BenchmarkConfig, ConfigError, EnvironmentConfig, EvaluationConfig, @@ -28,6 +36,7 @@ __all__ = [ "__version__", "AlgorithmConfig", + "BenchmarkConfig", "ConfigError", "DefaultOutcomePolicy", "EnvironmentConfig", @@ -35,10 +44,15 @@ "EpisodeMetricsAccumulator", "EvaluationConfig", "ExperimentConfig", + "LearningCurveBenchmarkResult", + "LearningCurvePoint", "OutcomePolicy", "TrainingConfig", "compute_rate", "extract_episode_metrics", "load_config", + "plot_learning_curve", + "run_learning_curve_benchmark", "save_config", + "validate_budgets", ] diff --git a/src/adaptive_rl/benchmarking/__init__.py b/src/adaptive_rl/benchmarking/__init__.py new file mode 100644 index 0000000..4ce7a2b --- /dev/null +++ b/src/adaptive_rl/benchmarking/__init__.py @@ -0,0 +1,17 @@ +"""Benchmarking utilities for AdaptiveRL.""" + +from adaptive_rl.benchmarking.learning_curve import ( + LearningCurveBenchmarkResult, + LearningCurvePoint, + plot_learning_curve, + run_learning_curve_benchmark, + validate_budgets, +) + +__all__ = [ + "LearningCurveBenchmarkResult", + "LearningCurvePoint", + "plot_learning_curve", + "run_learning_curve_benchmark", + "validate_budgets", +] diff --git a/src/adaptive_rl/benchmarking/learning_curve.py b/src/adaptive_rl/benchmarking/learning_curve.py new file mode 100644 index 0000000..0639611 --- /dev/null +++ b/src/adaptive_rl/benchmarking/learning_curve.py @@ -0,0 +1,514 @@ +"""Benchmark orchestration for PPO learning curves across training budgets.""" + +from __future__ import annotations + +import csv +import json +import math +import time +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Sequence + +from adaptive_rl.algorithms.ppo import PPOAlgorithm +from adaptive_rl.config import BenchmarkConfig, ExperimentConfig +from adaptive_rl.environments.registry import make_env +from adaptive_rl.evaluation.evaluator import Evaluator +from adaptive_rl.training.trainer import PPOTrainer + + +def validate_budgets( + raw_budgets: Sequence[int] | str | None, *, allow_empty: bool = False +) -> list[int]: + """Validate and normalize benchmark budgets. + + Rules: + - reject empty, zero, negative, non-integer, malformed comma-separated values + - reject duplicates after normalization + - sort ascending for reproducibility + """ + if raw_budgets is None: + if allow_empty: + return [] + raise ValueError("Benchmark budgets cannot be empty.") + + if isinstance(raw_budgets, str): + values = [part.strip() for part in raw_budgets.split(",")] + if not values or any(part == "" for part in values): + raise ValueError("Malformed budget list: expected comma-separated integers.") + normalized: list[int] = [] + for token in values: + try: + normalized.append(int(token)) + except ValueError as exc: # pragma: no cover - explicit validation path + raise ValueError(f"Malformed budget value: {token!r}") from exc + else: + normalized = list(raw_budgets) + + if not normalized: + if allow_empty: + return [] + raise ValueError("Benchmark budgets cannot be empty.") + + cleaned: list[int] = [] + seen: set[int] = set() + for value in normalized: + if isinstance(value, bool): + raise ValueError(f"Budget values must be integers, got boolean {value!r}.") + if not isinstance(value, int): + raise ValueError( + f"Budget values must be integers, got {type(value).__name__}: {value!r}" + ) + if value <= 0: + raise ValueError(f"Budget values must be positive, got {value!r}.") + if value in seen: + raise ValueError(f"Duplicate budget value detected: {value}.") + seen.add(value) + cleaned.append(value) + + cleaned.sort() + return cleaned + + +@dataclass(frozen=True) +class LearningCurvePoint: + """Performance record for a single training-budget evaluation point.""" + + budget_timesteps: int + trained_timesteps: int + success_rate: float | None + collision_rate: float | None + timeout_rate: float | None + mean_reward: float + std_reward: float + mean_episode_length: float + training_time_seconds: float + model_path: str + training_seed: int + evaluation_seeds: list[int] + evaluation_episodes: int + deterministic: bool + algorithm: str + environment: str + + +@dataclass +class LearningCurveBenchmarkResult: + """Top-level container for the ordered learning-curve benchmark output.""" + + benchmark_name: str + algorithm: str + environment: str + budgets: list[int] + training_seed: int + evaluation_seeds: list[int] + evaluation_episodes: int + deterministic: bool + points: list[LearningCurvePoint] = field(default_factory=list) + plot_data: dict[str, list[float | int | None]] = field(default_factory=dict) + output_dir: Path | None = None + json_path: Path | None = None + csv_path: Path | None = None + plot_path: Path | None = None + + def to_dict(self) -> dict[str, Any]: + """Serialize the benchmark result to a JSON-friendly dictionary.""" + payload: dict[str, Any] = { + "benchmark": { + "name": self.benchmark_name, + "algorithm": self.algorithm, + "environment": self.environment, + "training_seed": self.training_seed, + "evaluation_seeds": self.evaluation_seeds, + "evaluation_episodes": self.evaluation_episodes, + "deterministic": self.deterministic, + "budgets": self.budgets, + }, + "results": [ + { + "budget_timesteps": point.budget_timesteps, + "trained_timesteps": point.trained_timesteps, + "success_rate": point.success_rate, + "collision_rate": point.collision_rate, + "timeout_rate": point.timeout_rate, + "mean_reward": point.mean_reward, + "std_reward": point.std_reward, + "mean_episode_length": point.mean_episode_length, + "training_time_seconds": point.training_time_seconds, + "model_path": point.model_path, + "training_seed": point.training_seed, + "evaluation_seeds": point.evaluation_seeds, + "evaluation_episodes": point.evaluation_episodes, + "deterministic": point.deterministic, + "algorithm": point.algorithm, + "environment": point.environment, + } + for point in self.points + ], + "plot_data": self.plot_data, + } + return payload + + +def _resolve_benchmark_config( + base_config: ExperimentConfig, overrides: dict[str, Any] | None +) -> BenchmarkConfig: + """Merge benchmark settings into a config object without mutating the original.""" + if base_config.benchmark is not None: + benchmark_cfg = BenchmarkConfig.model_validate( + base_config.benchmark.model_dump(mode="python") + ) + else: + benchmark_cfg = BenchmarkConfig() + + if overrides: + benchmark_cfg = benchmark_cfg.model_copy(update=overrides) + return benchmark_cfg + + +def _budget_dir(base_output_dir: Path, budget: int) -> Path: + return base_output_dir / "learning_curve" / f"budget_{budget}" + + +def _evaluate_model( + model_path: Path, + *, + env_name: str, + env_kwargs: dict[str, Any], + evaluation_seeds: Sequence[int], + evaluation_episodes: int, + deterministic: bool, +) -> tuple[float | None, float | None, float | None, float, float, float]: + env = make_env(env_name, **env_kwargs) + try: + algo = PPOAlgorithm.from_pretrained(model_path, env=env) + evaluator = Evaluator(algorithm=algo, env=env) + + all_rewards: list[float] = [] + all_lengths: list[int] = [] + successes: list[bool] = [] + collisions: list[bool] = [] + timeout_count = 0 + total_episodes = 0 + + for seed in evaluation_seeds: + metrics = evaluator.evaluate( + num_episodes=evaluation_episodes, + deterministic=deterministic, + base_seed=seed, + ) + all_rewards.extend(metrics.additional_metrics.get("all_rewards", [])) + all_lengths.extend(metrics.additional_metrics.get("all_lengths", [])) + records = evaluator.last_episode_records + total_episodes += len(records) + successes.extend(record.success for record in records if record.success is not None) + collisions.extend( + record.collision for record in records if record.collision is not None + ) + timeout_count += sum(record.truncated for record in records) + + mean_reward = float(sum(all_rewards) / len(all_rewards)) if all_rewards else 0.0 + std_reward = ( + float( + ( + sum((reward - mean_reward) ** 2 for reward in all_rewards) + / max(1, len(all_rewards) - 1) + ) + ** 0.5 + ) + if len(all_rewards) > 1 + else 0.0 + ) + mean_episode_length = float(sum(all_lengths) / len(all_lengths)) if all_lengths else 0.0 + + success_rate = float(sum(successes) / len(successes)) if successes else None + collision_rate = float(sum(collisions) / len(collisions)) if collisions else None + timeout_rate = float(timeout_count / total_episodes) if total_episodes else None + + return ( + success_rate, + collision_rate, + timeout_rate, + mean_reward, + std_reward, + mean_episode_length, + ) + finally: + env.close() + + +def _run_single_budget( + config: ExperimentConfig, + budget: int, + *, + training_seed: int, + evaluation_seeds: Sequence[int], + evaluation_episodes: int, + deterministic: bool, + output_base_dir: Path, +) -> LearningCurvePoint: + """Run a single budget as a fresh training process from the same base configuration.""" + benchmark_dir = _budget_dir(output_base_dir, budget) + benchmark_dir.mkdir(parents=True, exist_ok=True) + + config_copy = config.model_copy(deep=True) + config_copy.name = f"ppo_budget_{budget}" + config_copy.seed = training_seed + config_copy.algorithm.parameters.pop("seed", None) + if config_copy.training is None: + raise ValueError("A training section is required to run the learning-curve benchmark.") + config_copy.training.total_timesteps = budget + config_copy.output_dir = benchmark_dir + config_copy.log_dir = benchmark_dir / "logs" + + env = make_env(config_copy.environment.name, **config_copy.environment.parameters) + trainer: PPOTrainer | None = None + start = time.perf_counter() + try: + trainer = PPOTrainer(config=config_copy, env=env) + result = trainer.fit() + training_time_seconds = time.perf_counter() - start + finally: + if trainer is not None: + trainer.close() + else: + env.close() + + model_path = result.final_model_path + if not model_path.exists(): + raise FileNotFoundError( + f"Training budget {budget} did not create model artifact: {model_path}" + ) + if trainer is None: + raise RuntimeError("Training did not create a PPO trainer.") + trained_timesteps = trainer.algorithm.num_timesteps + if not math.isfinite(training_time_seconds) or training_time_seconds < 0: + raise RuntimeError( + f"Invalid training duration for budget {budget}: {training_time_seconds}" + ) + + success_rate, collision_rate, timeout_rate, mean_reward, std_reward, mean_episode_length = ( + _evaluate_model( + model_path, + env_name=config_copy.environment.name, + env_kwargs=config_copy.environment.parameters, + evaluation_seeds=evaluation_seeds, + evaluation_episodes=evaluation_episodes, + deterministic=deterministic, + ) + ) + + return LearningCurvePoint( + budget_timesteps=budget, + trained_timesteps=trained_timesteps, + success_rate=success_rate, + collision_rate=collision_rate, + timeout_rate=timeout_rate, + mean_reward=mean_reward, + std_reward=std_reward, + mean_episode_length=mean_episode_length, + training_time_seconds=float(training_time_seconds), + model_path=str(model_path), + training_seed=training_seed, + evaluation_seeds=list(evaluation_seeds), + evaluation_episodes=evaluation_episodes, + deterministic=deterministic, + algorithm=config_copy.algorithm.name, + environment=config_copy.environment.name, + ) + + +def run_learning_curve_benchmark( + config: ExperimentConfig, + budgets: Sequence[int] | str | None = None, + *, + training_seed: int | None = None, + evaluation_seeds: Sequence[int] | None = None, + evaluation_episodes: int | None = None, + deterministic: bool | None = None, + output_dir: str | Path | None = None, + plot: bool = False, +) -> LearningCurveBenchmarkResult: + """Train a model for each budget, evaluate it, and serialize benchmark artifacts.""" + if config.algorithm.name.lower() != "ppo": + raise ValueError("The learning-curve benchmark currently supports only PPO.") + if config.training is None: + raise ValueError("A training section is required to run the learning-curve benchmark.") + + if budgets is None: + benchmark_cfg = _resolve_benchmark_config(config, None) + normalized = validate_budgets(benchmark_cfg.budgets) + else: + normalized = validate_budgets(budgets) + + if training_seed is not None: + final_training_seed = training_seed + elif config.benchmark is not None: + final_training_seed = config.benchmark.training_seed + else: + final_training_seed = config.seed + + if evaluation_seeds is None: + if config.benchmark is not None: + final_eval_seeds = list(config.benchmark.evaluation_seeds) + else: + final_eval_seeds = [config.seed + i for i in range(config.evaluation.eval_episodes)] + else: + final_eval_seeds = list(evaluation_seeds) + + if isinstance(final_training_seed, bool) or not isinstance(final_training_seed, int): + raise ValueError("Training seed must be an integer.") + if final_training_seed < 0: + raise ValueError("Training seed must be non-negative.") + if not final_eval_seeds: + raise ValueError("Evaluation seeds must not be empty.") + if any(isinstance(seed, bool) or not isinstance(seed, int) for seed in final_eval_seeds): + raise ValueError("Evaluation seeds must be integers.") + if any(seed < 0 for seed in final_eval_seeds): + raise ValueError("Evaluation seeds must be non-negative.") + if len(set(final_eval_seeds)) != len(final_eval_seeds): + raise ValueError("Evaluation seeds must not contain duplicates.") + + if evaluation_episodes is None: + if config.benchmark is not None: + final_eval_episodes = config.benchmark.evaluation_episodes + else: + final_eval_episodes = config.evaluation.eval_episodes + else: + final_eval_episodes = evaluation_episodes + if isinstance(final_eval_episodes, bool) or not isinstance(final_eval_episodes, int): + raise ValueError("Evaluation episodes per seed must be an integer.") + if final_eval_episodes <= 0: + raise ValueError("Evaluation episodes per seed must be positive.") + + if deterministic is None: + if config.benchmark is not None: + final_deterministic = config.benchmark.deterministic + else: + final_deterministic = config.evaluation.deterministic + else: + if not isinstance(deterministic, bool): + raise ValueError("Deterministic evaluation setting must be a boolean.") + final_deterministic = deterministic + + if output_dir is None: + target_dir = Path(config.output_dir) / "benchmarks" + else: + target_dir = Path(output_dir) + + target_dir.mkdir(parents=True, exist_ok=True) + json_path = target_dir / "learning_curve_budget.json" + csv_path = target_dir / "learning_curve_budget.csv" + plot_path = target_dir / "learning_curve_budget.png" if plot else None + + bench_points: list[LearningCurvePoint] = [] + for budget in normalized: + point = _run_single_budget( + config, + budget, + training_seed=final_training_seed, + evaluation_seeds=final_eval_seeds, + evaluation_episodes=final_eval_episodes, + deterministic=final_deterministic, + output_base_dir=target_dir, + ) + bench_points.append(point) + + result = LearningCurveBenchmarkResult( + benchmark_name="ppo_learning_curve", + algorithm=config.algorithm.name, + environment=config.environment.name, + budgets=normalized, + training_seed=final_training_seed, + evaluation_seeds=list(final_eval_seeds), + evaluation_episodes=final_eval_episodes, + deterministic=final_deterministic, + points=bench_points, + plot_data={ + "budgets": [int(point.budget_timesteps) for point in bench_points], + "success_rate": [point.success_rate for point in bench_points], + "mean_reward": [point.mean_reward for point in bench_points], + }, + output_dir=target_dir, + json_path=json_path, + csv_path=csv_path, + plot_path=plot_path, + ) + + with json_path.open("w", encoding="utf-8") as handle: + json.dump(result.to_dict(), handle, indent=2, allow_nan=False) + + fieldnames = [ + "budget_timesteps", + "trained_timesteps", + "success_rate", + "collision_rate", + "timeout_rate", + "mean_reward", + "std_reward", + "mean_episode_length", + "training_time_seconds", + "model_path", + ] + with csv_path.open("w", newline="", encoding="utf-8") as handle: + writer = csv.DictWriter(handle, fieldnames=fieldnames) + writer.writeheader() + for point in bench_points: + row = { + "budget_timesteps": point.budget_timesteps, + "trained_timesteps": point.trained_timesteps, + "success_rate": point.success_rate, + "collision_rate": point.collision_rate, + "timeout_rate": point.timeout_rate, + "mean_reward": point.mean_reward, + "std_reward": point.std_reward, + "mean_episode_length": point.mean_episode_length, + "training_time_seconds": point.training_time_seconds, + "model_path": point.model_path, + } + writer.writerow(row) + + if plot: + plot_learning_curve(result, plot_path) + + return result + + +def plot_learning_curve( + result: LearningCurveBenchmarkResult, output_path: str | Path | None = None +) -> Path: + """Render a lightweight learning curve plot for success rate and mean reward.""" + try: + import matplotlib + + matplotlib.use("Agg") + import matplotlib.pyplot as plt + except ImportError as exc: # pragma: no cover - optional dependency + raise RuntimeError( + "Plotting requires matplotlib. Install optional visualization dependencies to enable --plot." + ) from exc + + plot_target = Path(output_path) if output_path is not None else result.plot_path + if plot_target is None: + raise ValueError("An output path is required for plotting.") + plot_target.parent.mkdir(parents=True, exist_ok=True) + + budgets = [int(point.budget_timesteps) for point in result.points] + success_rates = [point.success_rate for point in result.points] + rewards = [point.mean_reward for point in result.points] + + fig, axes = plt.subplots(1, 2, figsize=(12, 4), constrained_layout=True) + axes[0].plot(budgets, success_rates, marker="o", linewidth=2) + axes[0].set_title("Success rate vs training budget") + axes[0].set_xlabel("Training budget (timesteps)") + axes[0].set_ylabel("Success rate") + axes[0].set_ylim(-0.05, 1.05) + + axes[1].plot(budgets, rewards, marker="s", linewidth=2, color="tab:orange") + axes[1].set_title("Mean reward vs training budget") + axes[1].set_xlabel("Training budget (timesteps)") + axes[1].set_ylabel("Mean reward") + + fig.savefig(plot_target, dpi=160) + plt.close(fig) + return plot_target diff --git a/src/adaptive_rl/cli.py b/src/adaptive_rl/cli.py index 2af48f1..9e0fa8a 100644 --- a/src/adaptive_rl/cli.py +++ b/src/adaptive_rl/cli.py @@ -26,6 +26,13 @@ no_args_is_help=True, ) +benchmark_app = typer.Typer( + name="benchmark", + help="Benchmarking commands for training-budget learning curves.", + no_args_is_help=True, +) +app.add_typer(benchmark_app, name="benchmark") + config_app = typer.Typer( name="config", help="Configuration inspection and validation commands.", @@ -207,6 +214,97 @@ def train( raise typer.Exit(code=1) +@benchmark_app.command(name="budgets") +def benchmark_budgets( + config: Optional[Path] = typer.Option( + None, "--config", "-c", help="Path to experiment configuration YAML" + ), + budgets: Optional[str] = typer.Option( + None, "--budgets", help="Comma-separated training budgets (for example: 5000,10000,25000)" + ), + training_seed: Optional[int] = typer.Option( + None, "--training-seed", help="Override the training seed used across budgets" + ), + eval_seeds: Optional[str] = typer.Option( + None, "--eval-seeds", help="Comma-separated evaluation seeds (for example: 42,43,44)" + ), + episodes: Optional[int] = typer.Option( + None, "--episodes", help="Override evaluation episodes per seed" + ), + deterministic: Optional[bool] = typer.Option( + None, "--deterministic/--stochastic", help="Use deterministic actions during evaluation" + ), + output_dir: Optional[Path] = typer.Option( + None, "--output-dir", help="Directory for benchmark JSON/CSV/plot artifacts" + ), + plot: bool = typer.Option(False, "--plot/--no-plot", help="Render a learning-curve plot"), +) -> None: + """Train and evaluate PPO across a set of training budgets.""" + if config is None: + for candidate in [Path("configs/drone_ppo.yaml"), Path("configs/drone_ppo_demo.yaml")]: + if candidate.exists(): + config = candidate + break + if config is None: + console.print( + "[bold red]No configuration file provided.[/bold red] Specify --config " + ) + raise typer.Exit(code=1) + + try: + exp_config = load_config(config) + except ConfigError as err: + console.print(f"[bold red]Configuration error:[/bold red] {err}") + raise typer.Exit(code=1) + + try: + from adaptive_rl.benchmarking import run_learning_curve_benchmark, validate_budgets + + budget_values = validate_budgets(budgets) if budgets is not None else None + if eval_seeds is not None: + seed_tokens = [part.strip() for part in eval_seeds.split(",")] + if not seed_tokens or any(not token for token in seed_tokens): + raise ValueError( + "Malformed evaluation seed list: expected comma-separated integers." + ) + parsed_eval_seeds = [] + for token in seed_tokens: + try: + parsed_eval_seeds.append(int(token)) + except ValueError as err: + raise ValueError(f"Malformed evaluation seed value: {token!r}") from err + else: + parsed_eval_seeds = None + + result = run_learning_curve_benchmark( + exp_config, + budgets=budget_values, + training_seed=training_seed, + evaluation_seeds=parsed_eval_seeds, + evaluation_episodes=episodes, + deterministic=deterministic, + output_dir=output_dir, + plot=plot, + ) + except Exception as err: + console.print(f"[bold red]Benchmark failed with error:[/bold red] {err}") + raise typer.Exit(code=1) + + console.print( + Panel.fit( + f"[bold green]PPO learning-curve benchmark complete[/bold green]\n\n" + f"• [bold]Budgets:[/bold] {', '.join(str(b) for b in result.budgets)}\n" + f"• [bold]Training seed:[/bold] {result.training_seed}\n" + f"• [bold]Evaluation seeds:[/bold] {result.evaluation_seeds}\n" + f"• [bold]JSON:[/bold] {result.json_path}\n" + f"• [bold]CSV:[/bold] {result.csv_path}\n" + f"• [bold]Plot:[/bold] {result.plot_path if result.plot_path else 'not generated'}", + title="Learning Curve Benchmark", + border_style="green", + ) + ) + + @app.command() def evaluate( config: Optional[Path] = typer.Option( diff --git a/src/adaptive_rl/config.py b/src/adaptive_rl/config.py index 3f3e530..7eccc6a 100644 --- a/src/adaptive_rl/config.py +++ b/src/adaptive_rl/config.py @@ -12,7 +12,15 @@ from typing import Any, Dict, Optional import yaml -from pydantic import BaseModel, ConfigDict, Field, ValidationError, model_validator +from pydantic import ( + BaseModel, + ConfigDict, + Field, + StrictInt, + ValidationError, + field_validator, + model_validator, +) class ConfigError(Exception): @@ -92,6 +100,49 @@ def _alias_episodes(cls, data: Any) -> Any: return data +class BenchmarkConfig(BaseModel): + """Configuration for PPO learning-curve benchmarking across training budgets.""" + + model_config = ConfigDict(extra="forbid", populate_by_name=True) + + budgets: list[StrictInt] = Field( + default_factory=lambda: [5000, 10000, 25000, 50000], + min_length=1, + description="Training budgets used for the learning-curve benchmark.", + ) + training_seed: int = Field(42, ge=0, description="Seed used for all benchmark training runs") + evaluation_seeds: list[StrictInt] = Field( + default_factory=lambda: [42, 43, 44, 45, 46], + min_length=1, + description="Fixed seed sequence used for evaluation across all budgets.", + ) + evaluation_episodes: int = Field( + 20, gt=0, description="Episodes per seed for benchmark evaluation" + ) + deterministic: bool = Field( + True, + description="Whether to evaluate using deterministic action selection for all budgets.", + ) + + @field_validator("budgets") + @classmethod + def _validate_budgets(cls, values: list[int]) -> list[int]: + if any(value <= 0 for value in values): + raise ValueError("Benchmark budgets must all be positive integers.") + if len(values) != len(set(values)): + raise ValueError("Benchmark budgets must not contain duplicates.") + return sorted(values) + + @field_validator("evaluation_seeds") + @classmethod + def _validate_evaluation_seeds(cls, values: list[int]) -> list[int]: + if any(value < 0 for value in values): + raise ValueError("Evaluation seeds must be non-negative integers.") + if len(values) != len(set(values)): + raise ValueError("Evaluation seeds must not contain duplicates.") + return values + + class ExperimentConfig(BaseModel): """Top-level configuration schema for an AdaptiveRL experiment.""" @@ -111,6 +162,10 @@ class ExperimentConfig(BaseModel): default_factory=lambda: Path("artifacts/logs"), description="Directory for logging and metrics", ) + benchmark: Optional[BenchmarkConfig] = Field( + default=None, + description="Optional benchmark settings for training-budget learning curves.", + ) @model_validator(mode="before") @classmethod diff --git a/src/adaptive_rl/environments/drone.py b/src/adaptive_rl/environments/drone.py index 17e1667..855dc01 100644 --- a/src/adaptive_rl/environments/drone.py +++ b/src/adaptive_rl/environments/drone.py @@ -630,6 +630,10 @@ def step( if self._current_step >= self.max_steps and not terminated: truncated = True + info["terminated"] = terminated + info["truncated"] = truncated + info["TimeLimit.truncated"] = truncated + if self.render_mode == "human": print(self.render()) diff --git a/src/adaptive_rl/evaluation/evaluator.py b/src/adaptive_rl/evaluation/evaluator.py index 6863427..d4871db 100644 --- a/src/adaptive_rl/evaluation/evaluator.py +++ b/src/adaptive_rl/evaluation/evaluator.py @@ -25,8 +25,9 @@ class EpisodeEvaluationRecord: seed: Optional[int] return_value: float episode_length: int - success: bool - collision: bool + success: Optional[bool] + collision: Optional[bool] + truncated: bool def to_dict(self) -> Dict[str, Any]: return { @@ -36,6 +37,7 @@ def to_dict(self) -> Dict[str, Any]: "episode_length": self.episode_length, "success": self.success, "collision": self.collision, + "truncated": self.truncated, } @@ -73,8 +75,9 @@ def evaluate( self.last_episode_records = [] rewards: List[float] = [] lengths: List[int] = [] - successes: List[bool] = [] - collisions: List[bool] = [] + successes: List[Optional[bool]] = [] + collisions: List[Optional[bool]] = [] + truncations: List[bool] = [] for ep in range(num_episodes): seed = (base_seed + ep) if base_seed is not None else None @@ -83,6 +86,7 @@ def evaluate( ep_length = 0 done = False last_info = dict(info or {}) + was_truncated = False while not done: action, _ = self.algorithm.predict(obs, deterministic=deterministic) @@ -90,15 +94,19 @@ def evaluate( ep_reward += float(reward) ep_length += 1 last_info = step_info + was_truncated = bool(truncated) done = terminated or truncated - is_success = bool(last_info.get("success", False)) - is_collision = bool(last_info.get("collision", False)) + success_value = last_info.get("success", last_info.get("is_success")) + collision_value = last_info.get("collision") + is_success = bool(success_value) if success_value is not None else None + is_collision = bool(collision_value) if collision_value is not None else None rewards.append(ep_reward) lengths.append(ep_length) successes.append(is_success) collisions.append(is_collision) + truncations.append(was_truncated) self.last_episode_records.append( EpisodeEvaluationRecord( @@ -108,6 +116,7 @@ def evaluate( episode_length=ep_length, success=is_success, collision=is_collision, + truncated=was_truncated, ) ) @@ -118,8 +127,17 @@ def evaluate( mean_len = float(np.mean(lengths)) std_len = float(np.std(lengths)) - succ_rate = float(sum(successes) / num_episodes) - coll_rate = float(sum(collisions) / num_episodes) + observed_successes = [value for value in successes if value is not None] + observed_collisions = [value for value in collisions if value is not None] + succ_rate = ( + float(sum(observed_successes) / len(observed_successes)) if observed_successes else None + ) + coll_rate = ( + float(sum(observed_collisions) / len(observed_collisions)) + if observed_collisions + else None + ) + trunc_rate = float(sum(truncations) / num_episodes) return EvaluationMetrics( episodes=num_episodes, @@ -129,6 +147,7 @@ def evaluate( max_reward=max_rew, success_rate=succ_rate, collision_rate=coll_rate, + truncation_rate=trunc_rate, mean_episode_length=mean_len, std_episode_length=std_len, additional_metrics={ @@ -136,6 +155,8 @@ def evaluate( "all_lengths": lengths, "deterministic": deterministic, "base_seed": base_seed, + "truncation_count": int(sum(truncations)), + "timeout_rate": trunc_rate, }, ) @@ -160,6 +181,9 @@ def save_report( "collision_rate": round(metrics.collision_rate, 4) if metrics.collision_rate is not None else None, + "truncation_rate": round(metrics.truncation_rate, 4) + if metrics.truncation_rate is not None + else None, "mean_episode_length": round(metrics.mean_episode_length, 2), "std_episode_length": round(metrics.std_episode_length, 2), } diff --git a/tests/test_cli.py b/tests/test_cli.py index 3cf8b84..8d4936e 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -17,6 +17,7 @@ def test_cli_help() -> None: assert "AdaptiveRL" in result.output assert "train" in result.output assert "evaluate" in result.output + assert "benchmark" in result.output assert "demo-drone" in result.output assert "gui" in result.output assert "experiment-density" in result.output diff --git a/tests/test_drone.py b/tests/test_drone.py index c5ca714..3cdc291 100644 --- a/tests/test_drone.py +++ b/tests/test_drone.py @@ -235,6 +235,9 @@ def test_drone_boundary_collision_termination() -> None: assert terminated is True assert info["collision"] is True + assert info["terminated"] is True + assert info["truncated"] is False + assert info["TimeLimit.truncated"] is False assert reward == env.collision_reward env.close() @@ -258,9 +261,62 @@ def test_drone_truncation_at_max_steps() -> None: assert trunc is True assert term is False assert info["step"] == max_steps + assert info["terminated"] is False + assert info["truncated"] is True + assert info["TimeLimit.truncated"] is True env.close() +def test_drone_step_outcome_metadata_matches_gymnasium_signals() -> None: + """Step metadata reflects final Gymnasium termination and truncation signals.""" + successful_env = DroneNavigation3DEnv( + bounds=(20.0, 20.0, 10.0), + start_pos=np.array([10.0, 10.0, 5.0]), + goal_pos=np.array([10.5, 10.0, 5.0]), + target_radius=1.5, + max_steps=1, + num_obstacles=0, + ) + successful_env.reset(seed=5) + transition = successful_env.step(np.zeros(3, dtype=np.float32)) + assert len(transition) == 5 + observation, reward, terminated, truncated, info = transition + assert successful_env.observation_space.contains(observation) + assert isinstance(reward, float) + assert terminated is True + assert truncated is False + assert info["terminated"] is terminated + assert info["truncated"] is truncated + assert info["TimeLimit.truncated"] is truncated + assert info["success"] is True + assert info["is_success"] is True + assert "position" in info + assert "distance_to_goal" in info + successful_env.close() + + timeout_env = DroneNavigation3DEnv( + bounds=(30.0, 30.0, 15.0), + max_steps=1, + num_obstacles=0, + ) + timeout_env.reset(seed=5) + observation, reward, terminated, truncated, info = timeout_env.step( + np.zeros(3, dtype=np.float32) + ) + assert timeout_env.observation_space.contains(observation) + assert isinstance(reward, float) + assert terminated is False + assert truncated is True + assert info["terminated"] is terminated + assert info["truncated"] is truncated + assert info["TimeLimit.truncated"] is truncated + assert info["success"] is False + assert info["collision"] is False + assert "position" in info + assert "distance_to_goal" in info + timeout_env.close() + + def test_drone_render_modes() -> None: """Verify textual 3D flight dashboard rendering.""" env = DroneNavigation3DEnv(render_mode="ansi") diff --git a/tests/test_learning_curve_benchmark.py b/tests/test_learning_curve_benchmark.py new file mode 100644 index 0000000..3b33a43 --- /dev/null +++ b/tests/test_learning_curve_benchmark.py @@ -0,0 +1,527 @@ +"""Focused tests for PPO learning-curve benchmarking.""" + +import csv +import json +import math +from pathlib import Path +from typing import Any + +import gymnasium as gym +import numpy as np +import pytest +from pydantic import ValidationError +from typer.testing import CliRunner + +from adaptive_rl.benchmarking import ( + LearningCurveBenchmarkResult, + run_learning_curve_benchmark, + validate_budgets, +) +from adaptive_rl.config import ( + AlgorithmConfig, + BenchmarkConfig, + EnvironmentConfig, + EvaluationConfig, + ExperimentConfig, + TrainingConfig, + load_config, + save_config, +) +from adaptive_rl.evaluation.evaluator import Evaluator + + +def _make_config(tmp_path: Path) -> ExperimentConfig: + return ExperimentConfig( + name="benchmark_test", + seed=31, + algorithm=AlgorithmConfig( + name="ppo", + batch_size=32, + parameters={"n_steps": 64, "n_epochs": 1, "seed": 999}, + ), + environment=EnvironmentConfig( + name="drone", + max_steps=8, + parameters={"bounds": [20.0, 20.0, 10.0], "num_obstacles": 1}, + ), + training=TrainingConfig(total_timesteps=256, checkpoint_freq=0, log_interval=1), + evaluation=EvaluationConfig(eval_episodes=1, deterministic=True), + output_dir=tmp_path / "artifacts", + log_dir=tmp_path / "logs", + benchmark=BenchmarkConfig( + budgets=[128, 64], + training_seed=17, + evaluation_seeds=[11, 12], + evaluation_episodes=1, + deterministic=True, + ), + ) + + +@pytest.mark.parametrize( + ("raw_budgets", "message"), + [ + ([], "empty"), + ([0], "positive"), + ([-1], "positive"), + ("foo,128", "Malformed budget value"), + ("64,,128", "Malformed budget list"), + ([64, 64], "Duplicate"), + ], +) +def test_budget_validation_rejects_invalid_values(raw_budgets: Any, message: str) -> None: + with pytest.raises(ValueError, match=message): + validate_budgets(raw_budgets) + + +def test_budget_validation_normalizes_valid_values() -> None: + assert validate_budgets("128,64") == [64, 128] + assert validate_budgets([128, 64]) == [64, 128] + + +@pytest.mark.parametrize("budgets", [[], [0], [-1], [64, 64], ["64"]]) +def test_benchmark_config_rejects_invalid_budgets(budgets: Any) -> None: + with pytest.raises(ValidationError): + BenchmarkConfig(budgets=budgets) + + +def test_benchmark_config_defaults_and_old_config_compatibility(tmp_path: Path) -> None: + benchmark = BenchmarkConfig() + assert benchmark.budgets == [5000, 10000, 25000, 50000] + assert benchmark.training_seed == 42 + assert benchmark.evaluation_seeds + assert benchmark.evaluation_episodes > 0 + + old_config_path = tmp_path / "old_config.yaml" + old_config_path.write_text( + """ +name: old_config +seed: 4 +algorithm: + name: ppo +environment: + name: drone +training: + total_timesteps: 128 +evaluation: + eval_episodes: 2 +""", + encoding="utf-8", + ) + loaded = load_config(old_config_path) + assert loaded.name == "old_config" + assert loaded.benchmark is None + + benchmark_config_path = tmp_path / "benchmark_config.yaml" + benchmark_config_path.write_text( + """ +name: configured_benchmark +algorithm: + name: ppo +environment: + name: drone +training: + total_timesteps: 128 +evaluation: + eval_episodes: 2 +benchmark: + budgets: [128, 64] + training_seed: 7 + evaluation_seeds: [20, 21] + evaluation_episodes: 3 + deterministic: false +""", + encoding="utf-8", + ) + configured = load_config(benchmark_config_path) + assert configured.benchmark is not None + assert configured.benchmark.budgets == [64, 128] + assert configured.benchmark.training_seed == 7 + assert configured.benchmark.evaluation_seeds == [20, 21] + assert configured.benchmark.evaluation_episodes == 3 + assert configured.benchmark.deterministic is False + + +def test_learning_curve_benchmark_execution_and_exports( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + from adaptive_rl.benchmarking import learning_curve + + config = _make_config(tmp_path) + original = config.model_dump(mode="python") + real_trainer_type = learning_curve.PPOTrainer + real_make_env = learning_curve.make_env + real_evaluate_model = learning_curve._evaluate_model + trainers: list[Any] = [] + envs: list[gym.Env] = [] + evaluation_calls: list[dict[str, Any]] = [] + + class TrackingTrainer(real_trainer_type): + def __init__(self, *args: Any, **kwargs: Any) -> None: + super().__init__(*args, **kwargs) + self.fit_calls = 0 + trainers.append(self) + + def fit(self) -> Any: + self.fit_calls += 1 + return super().fit() + + def track_env(*args: Any, **kwargs: Any) -> gym.Env: + env = real_make_env(*args, **kwargs) + envs.append(env) + return env + + def track_evaluation(*args: Any, **kwargs: Any) -> Any: + evaluation_calls.append(kwargs.copy()) + return real_evaluate_model(*args, **kwargs) + + monkeypatch.setattr(learning_curve, "PPOTrainer", TrackingTrainer) + monkeypatch.setattr(learning_curve, "make_env", track_env) + monkeypatch.setattr(learning_curve, "_evaluate_model", track_evaluation) + + result = run_learning_curve_benchmark(config, budgets=[128, 64]) + + assert result.budgets == [64, 128] + assert [point.budget_timesteps for point in result.points] == [64, 128] + assert len(result.points) == len(trainers) == 2 + assert [trainer.fit_calls for trainer in trainers] == [1, 1] + assert [trainer.config.training.total_timesteps for trainer in trainers] == [64, 128] + assert [trainer.config.seed for trainer in trainers] == [17, 17] + assert [trainer.config.name for trainer in trainers] == ["ppo_budget_64", "ppo_budget_128"] + assert trainers[0].config.output_dir != trainers[1].config.output_dir + assert trainers[0].config.log_dir != trainers[1].config.log_dir + assert [ + (trainer.config.training.checkpoint_freq, trainer.config.training.log_interval) + for trainer in trainers + ] == [(0, 1), (0, 1)] + assert [trainer.algorithm.hyperparameters["seed"] for trainer in trainers] == [17, 17] + expected_algorithm_config = config.algorithm.model_dump() + expected_algorithm_config["parameters"].pop("seed") + assert [trainer.config.algorithm.model_dump() for trainer in trainers] == [ + expected_algorithm_config, + expected_algorithm_config, + ] + assert [trainer.config.environment.model_dump() for trainer in trainers] == [ + config.environment.model_dump(), + config.environment.model_dump(), + ] + assert [trainer.config.evaluation.model_dump() for trainer in trainers] == [ + config.evaluation.model_dump(), + config.evaluation.model_dump(), + ] + assert len({id(trainer.env) for trainer in trainers}) == 2 + assert len({id(trainer.algorithm.model) for trainer in trainers}) == 2 + assert [trainer.algorithm.num_timesteps for trainer in trainers] == [64, 128] + assert all(env is trainer.env for env, trainer in zip((envs[0], envs[2]), trainers)) + assert len({id(env) for env in envs}) == 4 + + for call in evaluation_calls: + assert call["evaluation_seeds"] == [11, 12] + assert call["evaluation_episodes"] == 1 + assert call["deterministic"] is True + assert call["env_name"] == config.environment.name + assert call["env_kwargs"] == config.environment.parameters + assert len(evaluation_calls) == 2 + + model_paths = [Path(point.model_path) for point in result.points] + assert len(set(model_paths)) == 2 + assert all(path.is_file() for path in model_paths) + assert all(path.parent.parent.parent.name == "learning_curve" for path in model_paths) + assert result.json_path == config.output_dir / "benchmarks" / "learning_curve_budget.json" + assert result.csv_path == config.output_dir / "benchmarks" / "learning_curve_budget.csv" + + for point in result.points: + assert point.trained_timesteps == point.budget_timesteps + assert point.success_rate is None or 0.0 <= point.success_rate <= 1.0 + assert point.collision_rate is None or 0.0 <= point.collision_rate <= 1.0 + assert point.timeout_rate is None or 0.0 <= point.timeout_rate <= 1.0 + assert math.isfinite(point.mean_reward) + assert point.std_reward >= 0.0 + assert point.mean_episode_length >= 0.0 + assert math.isfinite(point.training_time_seconds) + assert point.training_time_seconds >= 0.0 + + assert config.name == original["name"] + assert config.training.total_timesteps == original["training"]["total_timesteps"] + assert config.output_dir == original["output_dir"] + assert config.log_dir == original["log_dir"] + assert config.evaluation.model_dump() == original["evaluation"] + assert config.algorithm.parameters == original["algorithm"]["parameters"] + assert config.model_dump(mode="python") == original + + assert result.json_path is not None and result.json_path.is_file() + data = json.loads(result.json_path.read_text(encoding="utf-8")) + json.dumps(data, allow_nan=False) + assert data["benchmark"]["training_seed"] == 17 + assert data["benchmark"]["evaluation_seeds"] == [11, 12] + assert data["benchmark"]["evaluation_episodes"] == 1 + assert data["benchmark"]["deterministic"] is True + assert data["benchmark"]["budgets"] == [64, 128] + assert len(data["results"]) == 2 + required_metrics = { + "budget_timesteps", + "trained_timesteps", + "success_rate", + "collision_rate", + "timeout_rate", + "mean_reward", + "std_reward", + "mean_episode_length", + "training_time_seconds", + "model_path", + } + for row in data["results"]: + assert required_metrics <= row.keys() + assert Path(row["model_path"]).is_file() + assert set(data["plot_data"]) == {"budgets", "success_rate", "mean_reward"} + + assert result.csv_path is not None and result.csv_path.is_file() + with result.csv_path.open(newline="", encoding="utf-8") as handle: + reader = csv.DictReader(handle) + rows = list(reader) + assert reader.fieldnames == [ + "budget_timesteps", + "trained_timesteps", + "success_rate", + "collision_rate", + "timeout_rate", + "mean_reward", + "std_reward", + "mean_episode_length", + "training_time_seconds", + "model_path", + ] + assert len(rows) == len(result.points) + assert not any(field.startswith("Unnamed:") for field in reader.fieldnames or []) + assert [int(row["budget_timesteps"]) for row in rows] == [64, 128] + assert all(math.isfinite(float(row["mean_reward"])) for row in rows) + assert all(Path(row["model_path"]).is_file() for row in rows) + + +def test_learning_curve_benchmark_repeats_deterministically(tmp_path: Path) -> None: + config = _make_config(tmp_path) + first = run_learning_curve_benchmark( + config, + budgets=[65], + output_dir=tmp_path / "first", + ) + second = run_learning_curve_benchmark( + config, + budgets=[65], + output_dir=tmp_path / "second", + ) + + first_point = first.points[0] + second_point = second.points[0] + assert first.budgets == second.budgets == [65] + assert first_point.trained_timesteps == second_point.trained_timesteps == 128 + assert first.training_seed == second.training_seed == 17 + assert first.evaluation_seeds == second.evaluation_seeds == [11, 12] + assert first.evaluation_episodes == second.evaluation_episodes == 1 + assert first.deterministic is second.deterministic is True + for field in ( + "budget_timesteps", + "trained_timesteps", + "success_rate", + "collision_rate", + "timeout_rate", + "mean_reward", + "std_reward", + "mean_episode_length", + "training_seed", + "evaluation_seeds", + "evaluation_episodes", + "deterministic", + ): + assert getattr(first_point, field) == getattr(second_point, field) + + +class _OutcomeEnv(gym.Env): + observation_space = gym.spaces.Box(low=-1.0, high=1.0, shape=(1,), dtype=np.float32) + action_space = gym.spaces.Box(low=-1.0, high=1.0, shape=(1,), dtype=np.float32) + + def __init__(self, outcome: str) -> None: + super().__init__() + self.outcome = outcome + + def reset( + self, *, seed: int | None = None, options: dict[str, Any] | None = None + ) -> tuple[np.ndarray, dict[str, Any]]: + super().reset(seed=seed) + return np.zeros(1, dtype=np.float32), {} + + def step(self, action: np.ndarray) -> tuple[np.ndarray, float, bool, bool, dict[str, Any]]: + if self.outcome == "truncated": + return ( + np.zeros(1, dtype=np.float32), + 1.0, + False, + True, + { + "success": False, + "collision": False, + }, + ) + if self.outcome == "success": + return ( + np.zeros(1, dtype=np.float32), + 1.0, + True, + False, + { + "success": True, + "collision": False, + }, + ) + if self.outcome == "termination": + return ( + np.zeros(1, dtype=np.float32), + 1.0, + True, + False, + { + "success": False, + "collision": False, + }, + ) + return np.zeros(1, dtype=np.float32), 1.0, True, False, {} + + +class _ConstantPolicy: + def predict(self, observation: Any, deterministic: bool = True) -> tuple[np.ndarray, None]: + return np.zeros(1, dtype=np.float32), None + + +@pytest.mark.parametrize( + ("outcome", "expected_success", "expected_timeout"), + [ + ("termination", 0.0, 0.0), + ("truncated", 0.0, 1.0), + ("success", 1.0, 0.0), + ], +) +def test_evaluator_uses_gymnasium_truncation_signal( + outcome: str, expected_success: float, expected_timeout: float +) -> None: + env = _OutcomeEnv(outcome) + evaluator = Evaluator(algorithm=_ConstantPolicy(), env=env) # type: ignore[arg-type] + metrics = evaluator.evaluate(num_episodes=1, deterministic=True, base_seed=2) + + assert metrics.success_rate == expected_success + assert metrics.truncation_rate == expected_timeout + assert evaluator.last_episode_records[0].truncated is (expected_timeout == 1.0) + evaluator.close() + + +def test_evaluator_preserves_unavailable_outcome_metrics() -> None: + env = _OutcomeEnv("unavailable") + evaluator = Evaluator(algorithm=_ConstantPolicy(), env=env) # type: ignore[arg-type] + metrics = evaluator.evaluate(num_episodes=1) + assert metrics.success_rate is None + assert metrics.collision_rate is None + evaluator.close() + + +def test_benchmark_cli_dispatch_and_budget_validation( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + from adaptive_rl import benchmarking + from adaptive_rl.cli import app + + config_path = tmp_path / "benchmark.yaml" + save_config(_make_config(tmp_path), config_path) + runner = CliRunner() + dispatch: list[dict[str, Any]] = [] + fake_result = LearningCurveBenchmarkResult( + benchmark_name="ppo_learning_curve", + algorithm="ppo", + environment="drone", + budgets=[64, 128], + training_seed=17, + evaluation_seeds=[11, 12], + evaluation_episodes=1, + deterministic=True, + json_path=tmp_path / "learning_curve_budget.json", + csv_path=tmp_path / "learning_curve_budget.csv", + ) + + def fake_benchmark(config: ExperimentConfig, **kwargs: Any) -> LearningCurveBenchmarkResult: + dispatch.append({"config": config, **kwargs}) + return fake_result + + monkeypatch.setattr(benchmarking, "run_learning_curve_benchmark", fake_benchmark) + success = runner.invoke( + app, + [ + "benchmark", + "budgets", + "--config", + str(config_path), + "--budgets", + "64,128", + "--training-seed", + "17", + "--eval-seeds", + "11,12", + "--episodes", + "1", + ], + ) + assert success.exit_code == 0, success.output + assert "learning-curve benchmark complete" in success.output + assert "64, 128" in success.output + assert len(dispatch) == 1 + assert dispatch[0]["budgets"] == [64, 128] + assert dispatch[0]["training_seed"] == 17 + assert dispatch[0]["evaluation_seeds"] == [11, 12] + assert dispatch[0]["evaluation_episodes"] == 1 + + for invalid in ("64,-1", "foo,128"): + result = runner.invoke( + app, + [ + "benchmark", + "budgets", + "--config", + str(config_path), + "--budgets", + invalid, + ], + ) + assert result.exit_code == 1 + assert "Benchmark failed with error" in result.output + assert len(dispatch) == 1 + + malformed_seeds = runner.invoke( + app, + [ + "benchmark", + "budgets", + "--config", + str(config_path), + "--budgets", + "64,128", + "--eval-seeds", + "11,,12", + ], + ) + assert malformed_seeds.exit_code == 1 + assert "Malformed evaluation seed list" in malformed_seeds.output + assert len(dispatch) == 1 + + invalid_seed_value = runner.invoke( + app, + [ + "benchmark", + "budgets", + "--config", + str(config_path), + "--budgets", + "64,128", + "--eval-seeds", + "foo,12", + ], + ) + assert invalid_seed_value.exit_code == 1 + assert "Malformed evaluation seed value: 'foo'" in invalid_seed_value.output + assert len(dispatch) == 1 From f87949ecdcafa8e54c0659f0328163416315b8fe Mon Sep 17 00:00:00 2001 From: Aryan Date: Mon, 28 Sep 2026 22:21:30 +0530 Subject: [PATCH 2/3] feat: add reproducible multi-seed evaluation (#244) Aggregate evaluation metrics across explicit seeds with Student's t confidence intervals, raw episode records, and stable JSON/CSV reports. Preserve single-seed CLI behavior while adding multi-seed options and documentation. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- docs/EXPERIMENT.md | 34 +++- src/adaptive_rl/cli.py | 167 ++++++++++++---- src/adaptive_rl/evaluation/__init__.py | 7 + src/adaptive_rl/evaluation/evaluator.py | 236 ++++++++++++++++++++++- src/adaptive_rl/evaluation/statistics.py | 150 ++++++++++++++ tests/test_cli.py | 126 ++++++++++++ tests/test_evaluation.py | 202 +++++++++++++++++++ 7 files changed, 879 insertions(+), 43 deletions(-) create mode 100644 src/adaptive_rl/evaluation/statistics.py diff --git a/docs/EXPERIMENT.md b/docs/EXPERIMENT.md index a3692e5..7e95712 100644 --- a/docs/EXPERIMENT.md +++ b/docs/EXPERIMENT.md @@ -86,7 +86,35 @@ The committed demo config uses `n_steps: 1024`, so those tiny budgets would be r --- -## 4. Expected Results (Hypothesized Prior to Testing) +## 4. Multi-Seed Evaluation and Confidence Intervals + +Evaluation over several independent environment seeds helps show how policy performance varies with randomized starts and obstacles, instead of depending on one seed sequence. `--episodes` is the number of episodes run for each listed seed. Each requested seed owns a disjoint block of actual environment reset seeds (`seed * episodes_per_seed + episode_index`), avoiding overlap between adjacent requested seed groups; the requested seed and actual per-episode reset seed are both recorded. Duplicate requested seeds are rejected to avoid overweighting a repeated condition. + +```bash +adaptive-rl evaluate \ + --config configs/drone_ppo.yaml \ + --model artifacts/models/drone_ppo_final.zip \ + --seeds 0 1 2 3 4 \ + --episodes 10 \ + --deterministic +``` + +The existing invocation remains single-seed and uses the configuration seed unless overridden with `--seed`: + +```bash +adaptive-rl evaluate --config configs/drone_ppo.yaml --episodes 10 +adaptive-rl evaluate --config configs/drone_ppo.yaml --seed 7 --episodes 10 +``` + +`--seed` and `--seeds` are mutually exclusive. In multi-seed mode, `--episodes` is per seed, and `--compare-random` is not supported. The command writes `artifacts/evaluation_multiseed.json` and `artifacts/evaluation_multiseed.csv` by default; `--output-report` and `--output-csv` can select alternate destinations. + +The JSON retains raw episode records (requested seed, episode index, actual reset seed, return, episode length, outcomes, truncation, and path length when the environment reports positions), per-seed summaries, aggregate metrics, and evaluation metadata. The CSV is a stable, aggregate-only table with one row per metric and columns `metric`, `mean`, `std`, `ci95_lower`, `ci95_upper`, `sample_count`, `seed_count`, `episodes_per_seed`, and `total_episodes`. + +Cross-seed means and confidence intervals are calculated from the per-seed summaries, not pooled episodes. The standard deviation is the sample standard deviation (`ddof=1`); two-sided 95% confidence intervals use Student's t critical values and `mean ± t * s / sqrt(n)`. Missing values are excluded per metric. With fewer than two valid seeds, sample standard deviation and CI bounds are `null`/unavailable; they are not replaced with zero. The interval describes uncertainty in the estimated mean across the evaluated seeds under the independent, representative-seed and approximate t-model assumptions. It is not proof that one policy is superior. Identical seeds and deterministic actions reproduce equivalent episode results when the policy and environment implementation are unchanged. + +--- + +## 5. Expected Results (Hypothesized Prior to Testing) 1. **Random Action Baseline**: - Success Rate: $0.0\%$ (probability of randomly stumbling into a $1.5\text{ m}$ sphere across a $13,500\text{ m}^3$ arena without striking walls is practically zero). @@ -103,7 +131,7 @@ The committed demo config uses `n_steps: 1024`, so those tiny budgets would be r --- -## 5. Actual Measured Results (Empirical Verification) +## 6. Actual Measured Results (Empirical Verification) All results below were generated through genuine Python 3.12 CPU execution using the canonical project commands: ```bash @@ -140,7 +168,7 @@ adaptive-rl experiment-density --model artifacts/models/drone_ppo_demo_final.zip --- -## 6. Reproducibility Guarantee +## 7. Reproducibility Guarantee To independently reproduce the identical metrics on any student laptop: ```bash diff --git a/src/adaptive_rl/cli.py b/src/adaptive_rl/cli.py index 9e0fa8a..56c6f14 100644 --- a/src/adaptive_rl/cli.py +++ b/src/adaptive_rl/cli.py @@ -8,7 +8,7 @@ from __future__ import annotations from pathlib import Path -from typing import Optional +from typing import List, Optional import typer from rich.console import Console @@ -50,6 +50,10 @@ console = Console() +def _format_metric(value: float | None) -> str: + return f"{value:.3f}" if value is not None else "N/A" + + @app.command() def version() -> None: """Show the installed AdaptiveRL version and project story.""" @@ -305,8 +309,9 @@ def benchmark_budgets( ) -@app.command() +@app.command(context_settings={"allow_extra_args": True}) def evaluate( + ctx: typer.Context, config: Optional[Path] = typer.Option( None, "--config", "-c", help="Path to experiment configuration YAML" ), @@ -316,12 +321,21 @@ def evaluate( episodes: Optional[int] = typer.Option( 20, "--episodes", "-e", help="Number of evaluation episodes" ), + seed: Optional[int] = typer.Option( + None, "--seed", "-s", help="Base seed for single-seed evaluation (defaults to config seed)" + ), + seeds: Optional[List[int]] = typer.Option( + None, "--seeds", help="Explicit evaluation seeds; each seed runs --episodes episodes" + ), deterministic: bool = typer.Option( True, "--deterministic/--stochastic", help="Use deterministic action selection" ), output_report: Optional[Path] = typer.Option( None, "--output-report", "-o", help="Optional path to export JSON metrics report" ), + output_csv: Optional[Path] = typer.Option( + None, "--output-csv", help="Optional path to export aggregated multi-seed CSV" + ), compare_random: bool = typer.Option( False, "--compare-random", @@ -329,6 +343,20 @@ def evaluate( ), ) -> None: """Evaluate a trained agent over multiple benchmark episodes.""" + if ctx.args: + if seeds is None: + console.print( + "[bold red]Unexpected positional values; provide seeds with --seeds.[/bold red]" + ) + raise typer.Exit(code=1) + try: + seeds.extend(int(value) for value in ctx.args) + except ValueError: + console.print( + "[bold red]Seeds must be integers; use --seeds followed by space-separated values.[/bold red]" + ) + raise typer.Exit(code=1) + if config is None: for candidate in [Path("configs/drone_ppo.yaml"), Path("configs/drone_ppo_demo.yaml")]: if candidate.exists(): @@ -346,7 +374,22 @@ def evaluate( console.print(f"[bold red]Configuration error:[/bold red] {err}") raise typer.Exit(code=1) - num_episodes = episodes or exp_config.evaluation.eval_episodes + num_episodes = episodes if episodes is not None else exp_config.evaluation.eval_episodes + if num_episodes <= 0: + console.print("[bold red]Evaluation episodes must be positive.[/bold red]") + raise typer.Exit(code=1) + if seed is not None and seed < 0: + console.print("[bold red]Evaluation seed must be non-negative.[/bold red]") + raise typer.Exit(code=1) + if seeds is not None and seed is not None: + console.print("[bold red]Use either --seed or --seeds, not both.[/bold red]") + raise typer.Exit(code=1) + if output_csv is not None and seeds is None: + console.print("[bold red]--output-csv requires --seeds.[/bold red]") + raise typer.Exit(code=1) + if seeds is not None and compare_random: + console.print("[bold red]--compare-random cannot be combined with --seeds.[/bold red]") + raise typer.Exit(code=1) # Resolve model path if model is None: @@ -365,7 +408,8 @@ def evaluate( f"• [bold]Model:[/bold] {model}\n" f"• [bold]Environment:[/bold] {exp_config.environment.name}\n" f"• [bold]Episodes:[/bold] {num_episodes}\n" - f"• [bold]Deterministic:[/bold] {deterministic}", + f"• [bold]Deterministic:[/bold] {deterministic}\n" + f"• [bold]Seeds:[/bold] {seeds if seeds is not None else seed if seed is not None else exp_config.seed}", title="Evaluation Engine", border_style="cyan", ) @@ -382,41 +426,85 @@ def evaluate( algo = PPOAlgorithm.from_pretrained(model, env=env) evaluator = Evaluator(algorithm=algo, env=env) - metrics = evaluator.evaluate( - num_episodes=num_episodes, - deterministic=deterministic, - base_seed=exp_config.seed, - ) + metrics = None + if seeds is not None: + multi_seed_result = evaluator.evaluate_seeds( + seeds=seeds, + episodes_per_seed=num_episodes, + deterministic=deterministic, + ) + console.print( + f"\nSeeds: {len(multi_seed_result.seeds)} | " + f"Episodes per seed: {multi_seed_result.episodes_per_seed} | " + f"Total episodes: {multi_seed_result.total_episodes}" + ) + table = Table(title="Multi-Seed Evaluation (95% Student's t CI)") + table.add_column("Metric", style="cyan") + table.add_column("Mean", justify="right") + table.add_column("Std", justify="right") + table.add_column("95% CI lower", justify="right") + table.add_column("95% CI upper", justify="right") + for metric_name, label in ( + ("mean_reward", "Mean reward"), + ("success_rate", "Success rate"), + ("collision_rate", "Collision rate"), + ("mean_episode_length", "Episode length"), + ): + summary = multi_seed_result.aggregate[metric_name] + table.add_row( + label, + _format_metric(summary.mean), + _format_metric(summary.std), + _format_metric(summary.ci95_lower), + _format_metric(summary.ci95_upper), + ) + console.print(table) + report_target = output_report or (exp_config.output_dir / "evaluation_multiseed.json") + csv_target = output_csv or (exp_config.output_dir / "evaluation_multiseed.csv") + json_saved, csv_saved = evaluator.save_multiseed_report( + multi_seed_result, report_target, csv_target + ) + console.print(f"\n[bold green]JSON report saved to:[/bold green] {json_saved}") + console.print(f"[bold green]CSV report saved to:[/bold green] {csv_saved}") + else: + metrics = evaluator.evaluate( + num_episodes=num_episodes, + deterministic=deterministic, + base_seed=exp_config.seed if seed is None else seed, + ) - success_pct = ( - f"{metrics.success_rate * 100:.1f}%" if metrics.success_rate is not None else "N/A" - ) - collision_pct = ( - f"{metrics.collision_rate * 100:.1f}%" if metrics.collision_rate is not None else "N/A" - ) + success_pct = ( + f"{metrics.success_rate * 100:.1f}%" if metrics.success_rate is not None else "N/A" + ) + collision_pct = ( + f"{metrics.collision_rate * 100:.1f}%" + if metrics.collision_rate is not None + else "N/A" + ) - console.print("\n[bold]## Evaluation[/bold]") - console.print(f"Episodes: {metrics.episodes}") - console.print(f"Success rate: {success_pct}") - console.print(f"Collision rate: {collision_pct}") - console.print(f"Mean reward: {metrics.mean_reward:.2f}") - console.print(f"Mean episode length: {metrics.mean_episode_length:.1f}\n") - - table = Table(title=f"Benchmark Results ({num_episodes} episodes)") - table.add_column("Metric", style="cyan") - table.add_column("Value", style="green", justify="right") - - table.add_row("Mean Reward", f"{metrics.mean_reward:.2f} ± {metrics.std_reward:.2f}") - table.add_row("Min / Max Reward", f"{metrics.min_reward:.2f} / {metrics.max_reward:.2f}") - table.add_row("Success Rate", success_pct) - table.add_row("Collision Rate", collision_pct) - table.add_row( - "Mean Episode Length", - f"{metrics.mean_episode_length:.1f} ± {metrics.std_episode_length:.1f}", - ) - console.print(table) + console.print("\n[bold]## Evaluation[/bold]") + console.print(f"Episodes: {metrics.episodes}") + console.print(f"Success rate: {success_pct}") + console.print(f"Collision rate: {collision_pct}") + console.print(f"Mean reward: {metrics.mean_reward:.2f}") + console.print(f"Mean episode length: {metrics.mean_episode_length:.1f}\n") + + table = Table(title=f"Benchmark Results ({num_episodes} episodes)") + table.add_column("Metric", style="cyan") + table.add_column("Value", style="green", justify="right") + table.add_row("Mean Reward", f"{metrics.mean_reward:.2f} ± {metrics.std_reward:.2f}") + table.add_row( + "Min / Max Reward", f"{metrics.min_reward:.2f} / {metrics.max_reward:.2f}" + ) + table.add_row("Success Rate", success_pct) + table.add_row("Collision Rate", collision_pct) + table.add_row( + "Mean Episode Length", + f"{metrics.mean_episode_length:.1f} ± {metrics.std_episode_length:.1f}", + ) + console.print(table) - if compare_random: + if compare_random and metrics is not None: from adaptive_rl.evaluation.evaluator import compare_policies comp_results = compare_policies( @@ -445,9 +533,10 @@ def evaluate( console.print("\n") console.print(comp_table) - report_target = output_report or (exp_config.output_dir / "evaluation.json") - saved_path = evaluator.save_report(metrics, report_target) - console.print(f"\n[bold green]Report saved to:[/bold green] {saved_path}") + if metrics is not None: + report_target = output_report or (exp_config.output_dir / "evaluation.json") + saved_path = evaluator.save_report(metrics, report_target) + console.print(f"\n[bold green]Report saved to:[/bold green] {saved_path}") env.close() except Exception as err: diff --git a/src/adaptive_rl/evaluation/__init__.py b/src/adaptive_rl/evaluation/__init__.py index a6e23a3..d20821b 100644 --- a/src/adaptive_rl/evaluation/__init__.py +++ b/src/adaptive_rl/evaluation/__init__.py @@ -3,19 +3,26 @@ from adaptive_rl.evaluation.evaluator import ( EpisodeEvaluationRecord, Evaluator, + MultiSeedEvaluationResult, + SeedEvaluationSummary, compare_policies, evaluate_ppo_policy, evaluate_random_policy, run_obstacle_density_experiment, ) from adaptive_rl.evaluation.metrics import EvaluationMetrics +from adaptive_rl.evaluation.statistics import MetricStatistics, student_t_critical_value __all__ = [ "EpisodeEvaluationRecord", "EvaluationMetrics", "Evaluator", + "MetricStatistics", + "MultiSeedEvaluationResult", + "SeedEvaluationSummary", "compare_policies", "evaluate_ppo_policy", "evaluate_random_policy", "run_obstacle_density_experiment", + "student_t_critical_value", ] diff --git a/src/adaptive_rl/evaluation/evaluator.py b/src/adaptive_rl/evaluation/evaluator.py index d4871db..7024e03 100644 --- a/src/adaptive_rl/evaluation/evaluator.py +++ b/src/adaptive_rl/evaluation/evaluator.py @@ -2,8 +2,10 @@ from __future__ import annotations +import csv import json -from dataclasses import dataclass +import math +from dataclasses import dataclass, replace from pathlib import Path from typing import Any, Dict, List, Optional, Sequence, Tuple @@ -15,6 +17,7 @@ from adaptive_rl.environments.drone import DroneNavigation3DEnv from adaptive_rl.environments.registry import make_env from adaptive_rl.evaluation.metrics import EvaluationMetrics +from adaptive_rl.evaluation.statistics import MetricStatistics, summarize_seed_values @dataclass(frozen=True) @@ -28,6 +31,8 @@ class EpisodeEvaluationRecord: success: Optional[bool] collision: Optional[bool] truncated: bool + path_length: Optional[float] = None + episode_seed: Optional[int] = None def to_dict(self) -> Dict[str, Any]: return { @@ -38,6 +43,75 @@ def to_dict(self) -> Dict[str, Any]: "success": self.success, "collision": self.collision, "truncated": self.truncated, + "path_length": self.path_length, + "episode_seed": self.episode_seed, + } + + +@dataclass(frozen=True) +class SeedEvaluationSummary: + """Episode-level summary for a single evaluation seed.""" + + seed: int + episodes: int + success_rate: float | None + collision_rate: float | None + truncation_rate: float | None + mean_reward: float + std_reward: float + mean_episode_length: float + std_episode_length: float + path_length: float | None + + def to_dict(self) -> dict[str, int | float | None]: + return { + "seed": self.seed, + "episodes": self.episodes, + "success_rate": self.success_rate, + "collision_rate": self.collision_rate, + "truncation_rate": self.truncation_rate, + "mean_reward": self.mean_reward, + "std_reward": self.std_reward, + "mean_episode_length": self.mean_episode_length, + "std_episode_length": self.std_episode_length, + "path_length": self.path_length, + } + + +@dataclass(frozen=True) +class MultiSeedEvaluationResult: + """Complete multi-seed result retaining episodes and seed-level identity.""" + + seeds: list[int] + episodes_per_seed: int + deterministic: bool + per_seed: list[SeedEvaluationSummary] + episodes: list[EpisodeEvaluationRecord] + aggregate: dict[str, MetricStatistics] + environment: str + + @property + def total_episodes(self) -> int: + return len(self.episodes) + + def to_dict(self) -> dict[str, Any]: + return { + "metadata": { + "seeds": list(self.seeds), + "seed_count": len(self.seeds), + "episodes_per_seed": self.episodes_per_seed, + "total_episodes": self.total_episodes, + "deterministic": self.deterministic, + "environment": self.environment, + "confidence_interval": "two-sided 95% Student's t interval across seed summaries; " + "sample standard deviation; unavailable when fewer than two values exist", + "duplicate_seed_policy": "rejected", + }, + "episodes": [record.to_dict() for record in self.episodes], + "per_seed": [summary.to_dict() for summary in self.per_seed], + "aggregate": { + name: statistics.to_dict() for name, statistics in self.aggregate.items() + }, } @@ -87,6 +161,8 @@ def evaluate( done = False last_info = dict(info or {}) was_truncated = False + previous_position = self._position_from_info(last_info) + path_length = 0.0 if previous_position is not None else None while not done: action, _ = self.algorithm.predict(obs, deterministic=deterministic) @@ -95,6 +171,17 @@ def evaluate( ep_length += 1 last_info = step_info was_truncated = bool(truncated) + current_position = self._position_from_info(step_info) + if ( + path_length is not None + and previous_position is not None + and current_position is not None + and current_position.shape == previous_position.shape + ): + path_length += float(np.linalg.norm(current_position - previous_position)) + else: + path_length = None + previous_position = current_position done = terminated or truncated success_value = last_info.get("success", last_info.get("is_success")) @@ -107,6 +194,8 @@ def evaluate( successes.append(is_success) collisions.append(is_collision) truncations.append(was_truncated) + if not math.isfinite(ep_reward): + raise ValueError(f"Episode {ep} produced a non-finite cumulative reward.") self.last_episode_records.append( EpisodeEvaluationRecord( @@ -117,6 +206,8 @@ def evaluate( success=is_success, collision=is_collision, truncated=was_truncated, + path_length=path_length, + episode_seed=seed, ) ) @@ -153,6 +244,7 @@ def evaluate( additional_metrics={ "all_rewards": rewards, "all_lengths": lengths, + "all_path_lengths": [record.path_length for record in self.last_episode_records], "deterministic": deterministic, "base_seed": base_seed, "truncation_count": int(sum(truncations)), @@ -160,6 +252,148 @@ def evaluate( }, ) + @staticmethod + def _position_from_info(info: dict[str, Any]) -> np.ndarray | None: + position = info.get("position") + if position is None: + return None + try: + coordinates = np.asarray(position, dtype=np.float64) + except (TypeError, ValueError): + return None + if coordinates.ndim != 1 or coordinates.size == 0: + return None + if not np.all(np.isfinite(coordinates)): + return None + return coordinates + + def evaluate_seeds( + self, + seeds: Sequence[int], + episodes_per_seed: int, + deterministic: bool = True, + ) -> MultiSeedEvaluationResult: + """Evaluate a policy independently for each explicit seed. + + Duplicate seeds are rejected because repeated entries do not represent + independent test conditions and would over-weight that environment. + """ + seed_values = list(seeds) + if not seed_values: + raise ValueError("Evaluation seeds must not be empty.") + if any(isinstance(seed, bool) or not isinstance(seed, int) for seed in seed_values): + raise ValueError("Evaluation seeds must be integers.") + if any(seed < 0 for seed in seed_values): + raise ValueError("Evaluation seeds must be non-negative.") + if len(set(seed_values)) != len(seed_values): + raise ValueError("Evaluation seeds must be unique; duplicate seeds are not allowed.") + if isinstance(episodes_per_seed, bool) or not isinstance(episodes_per_seed, int): + raise ValueError("episodes_per_seed must be an integer.") + if episodes_per_seed <= 0: + raise ValueError(f"episodes_per_seed must be positive, got {episodes_per_seed}.") + if not isinstance(deterministic, bool): + raise ValueError("deterministic must be a boolean.") + + all_records: list[EpisodeEvaluationRecord] = [] + seed_summaries: list[SeedEvaluationSummary] = [] + for seed in seed_values: + episode_seed_base = seed * episodes_per_seed + metrics = self.evaluate( + num_episodes=episodes_per_seed, + deterministic=deterministic, + base_seed=episode_seed_base, + ) + records = [ + replace(record, seed=seed, episode_index=index) + for index, record in enumerate(self.last_episode_records) + ] + all_records.extend(records) + path_lengths = [ + record.path_length for record in records if record.path_length is not None + ] + seed_summaries.append( + SeedEvaluationSummary( + seed=seed, + episodes=metrics.episodes, + success_rate=metrics.success_rate, + collision_rate=metrics.collision_rate, + truncation_rate=metrics.truncation_rate, + mean_reward=metrics.mean_reward, + std_reward=metrics.std_reward, + mean_episode_length=metrics.mean_episode_length, + std_episode_length=metrics.std_episode_length, + path_length=( + float(math.fsum(path_lengths) / len(path_lengths)) if path_lengths else None + ), + ) + ) + + metric_names = ( + "mean_reward", + "success_rate", + "collision_rate", + "truncation_rate", + "mean_episode_length", + "std_reward", + "std_episode_length", + "path_length", + ) + aggregate = { + name: summarize_seed_values([getattr(summary, name) for summary in seed_summaries]) + for name in metric_names + } + self.last_episode_records = all_records + return MultiSeedEvaluationResult( + seeds=seed_values, + episodes_per_seed=episodes_per_seed, + deterministic=deterministic, + per_seed=seed_summaries, + episodes=all_records, + aggregate=aggregate, + environment=self.env_name, + ) + + @staticmethod + def save_multiseed_report( + result: MultiSeedEvaluationResult, + json_path: str | Path, + csv_path: str | Path, + ) -> tuple[Path, Path]: + """Write the complete multi-seed result to stable JSON and aggregate CSV.""" + json_target = Path(json_path) + csv_target = Path(csv_path) + json_target.parent.mkdir(parents=True, exist_ok=True) + csv_target.parent.mkdir(parents=True, exist_ok=True) + + with json_target.open("w", encoding="utf-8") as handle: + json.dump(result.to_dict(), handle, indent=2, allow_nan=False) + + columns = [ + "metric", + "mean", + "std", + "ci95_lower", + "ci95_upper", + "sample_count", + "seed_count", + "episodes_per_seed", + "total_episodes", + ] + with csv_target.open("w", newline="", encoding="utf-8") as handle: + writer = csv.DictWriter(handle, fieldnames=columns) + writer.writeheader() + for metric, statistics in result.aggregate.items(): + writer.writerow( + { + "metric": metric, + **statistics.to_dict(), + "seed_count": len(result.seeds), + "episodes_per_seed": result.episodes_per_seed, + "total_episodes": result.total_episodes, + } + ) + return json_target, csv_target + @staticmethod def save_report( metrics: EvaluationMetrics, diff --git a/src/adaptive_rl/evaluation/statistics.py b/src/adaptive_rl/evaluation/statistics.py new file mode 100644 index 0000000..8e044b1 --- /dev/null +++ b/src/adaptive_rl/evaluation/statistics.py @@ -0,0 +1,150 @@ +"""Statistical summaries for independent-seed evaluation results.""" + +from __future__ import annotations + +import math +from dataclasses import asdict, dataclass +from typing import Sequence + + +@dataclass(frozen=True) +class MetricStatistics: + """Across-seed statistics for one scalar metric.""" + + mean: float | None + std: float | None + ci95_lower: float | None + ci95_upper: float | None + sample_count: int + + def to_dict(self) -> dict[str, float | int | None]: + return asdict(self) + + +def _beta_continued_fraction(a: float, b: float, x: float) -> float: + """Evaluate the continued fraction used by the regularized beta function.""" + max_iterations = 300 + epsilon = 3e-14 + tiny = 1e-300 + qab = a + b + qap = a + 1.0 + qam = a - 1.0 + c = 1.0 + d = 1.0 - qab * x / qap + if abs(d) < tiny: + d = tiny + d = 1.0 / d + result = d + + for iteration in range(1, max_iterations + 1): + m2 = 2 * iteration + aa = iteration * (b - iteration) * x / ((qam + m2) * (a + m2)) + d = 1.0 + aa * d + if abs(d) < tiny: + d = tiny + c = 1.0 + aa / c + if abs(c) < tiny: + c = tiny + d = 1.0 / d + result *= d * c + + aa = -(a + iteration) * (qab + iteration) * x / ((a + m2) * (qap + m2)) + d = 1.0 + aa * d + if abs(d) < tiny: + d = tiny + c = 1.0 + aa / c + if abs(c) < tiny: + c = tiny + d = 1.0 / d + delta = d * c + result *= delta + if abs(delta - 1.0) < epsilon: + return result + + raise ArithmeticError("Incomplete beta continued fraction did not converge.") + + +def _regularized_incomplete_beta(a: float, b: float, x: float) -> float: + if not 0.0 <= x <= 1.0: + raise ValueError(f"Incomplete beta x must be in [0, 1], got {x}.") + if x == 0.0: + return 0.0 + if x == 1.0: + return 1.0 + + log_front = ( + math.lgamma(a + b) - math.lgamma(a) - math.lgamma(b) + a * math.log(x) + b * math.log1p(-x) + ) + front = math.exp(log_front) + if x < (a + 1.0) / (a + b + 2.0): + return front * _beta_continued_fraction(a, b, x) / a + return 1.0 - front * _beta_continued_fraction(b, a, 1.0 - x) / b + + +def _student_t_cdf(value: float, degrees_of_freedom: int) -> float: + if degrees_of_freedom <= 0: + raise ValueError("Student-t degrees of freedom must be positive.") + if value == 0.0: + return 0.5 + x = degrees_of_freedom / (degrees_of_freedom + value * value) + tail = 0.5 * _regularized_incomplete_beta(degrees_of_freedom / 2.0, 0.5, x) + return 1.0 - tail if value > 0 else tail + + +def student_t_critical_value(confidence: float, degrees_of_freedom: int) -> float: + """Return the positive two-sided Student-t critical value. + + The inverse is computed by bisection over the Student-t CDF, implemented + using the regularized incomplete beta function; no statistical dependency + or normal approximation is used. + """ + if not 0.0 < confidence < 1.0: + raise ValueError(f"Confidence must be between 0 and 1, got {confidence}.") + if degrees_of_freedom <= 0: + raise ValueError("Student-t degrees of freedom must be positive.") + + target = (1.0 + confidence) / 2.0 + lower = 0.0 + upper = 1.0 + while _student_t_cdf(upper, degrees_of_freedom) < target: + upper *= 2.0 + + for _ in range(100): + midpoint = (lower + upper) / 2.0 + if _student_t_cdf(midpoint, degrees_of_freedom) < target: + lower = midpoint + else: + upper = midpoint + return (lower + upper) / 2.0 + + +def summarize_seed_values(values: Sequence[float | None]) -> MetricStatistics: + """Summarize independent seed-level values using sample standard deviation. + + Missing values are excluded metric-by-metric. CI bounds are unavailable + unless at least two finite seed-level observations are present. + """ + observed: list[float] = [] + for value in values: + if value is None: + continue + if not math.isfinite(value): + raise ValueError(f"Metric values must be finite, got {value!r}.") + observed.append(float(value)) + + count = len(observed) + if count == 0: + return MetricStatistics(None, None, None, None, 0) + + mean = math.fsum(observed) / count + if count < 2: + return MetricStatistics(mean, None, None, None, count) + + variance = math.fsum((value - mean) ** 2 for value in observed) / (count - 1) + std = math.sqrt(variance) + critical = student_t_critical_value(0.95, count - 1) + margin = critical * std / math.sqrt(count) + return MetricStatistics(mean, std, mean - margin, mean + margin, count) + + +__all__ = ["MetricStatistics", "student_t_critical_value", "summarize_seed_values"] diff --git a/tests/test_cli.py b/tests/test_cli.py index 8d4936e..9a44900 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -1,10 +1,12 @@ """Tests verifying Typer CLI commands and execution.""" +import json import re from pathlib import Path from typer.testing import CliRunner +from adaptive_rl.algorithms.ppo import PPOAlgorithm from adaptive_rl.cli import app runner = CliRunner() @@ -85,6 +87,130 @@ def test_cli_env_inspect_failure() -> None: assert "Environment inspection failed" in result.output +def test_cli_multi_seed_evaluation_and_single_seed_compatibility( + tmp_path: Path, monkeypatch +) -> None: + class ZeroPolicy: + def predict(self, observation, deterministic=True): + import numpy as np + + return np.zeros(3, dtype=np.float32), None + + monkeypatch.setattr( + PPOAlgorithm, + "from_pretrained", + classmethod(lambda cls, path, env=None: ZeroPolicy()), + ) + config_path = tmp_path / "evaluation.yaml" + config_path.write_text( + f""" +name: cli_evaluation +seed: 42 +algorithm: + name: ppo +environment: + name: drone + max_steps: 2 + parameters: + bounds: [20.0, 20.0, 10.0] + num_obstacles: 0 +training: + total_timesteps: 64 +evaluation: + eval_episodes: 2 +output_dir: "{tmp_path / "artifacts"}" +log_dir: "{tmp_path / "logs"}" +""", + encoding="utf-8", + ) + model_path = tmp_path / "policy.zip" + model_path.touch() + + multi_json = tmp_path / "multi.json" + multi_csv = tmp_path / "multi.csv" + multi = runner.invoke( + app, + [ + "evaluate", + "--config", + str(config_path), + "--model", + str(model_path), + "--seeds", + "0", + "1", + "--episodes", + "1", + "--output-report", + str(multi_json), + "--output-csv", + str(multi_csv), + ], + ) + assert multi.exit_code == 0, multi.output + assert "Multi-Seed Evaluation" in multi.output + assert "95% CI lower" in multi.output + assert "Seeds: 2" in multi.output + assert "Total episodes: 2" in multi.output + assert multi_json.is_file() + assert multi_csv.is_file() + assert json.loads(multi_json.read_text(encoding="utf-8"))["metadata"]["seeds"] == [0, 1] + + single = runner.invoke( + app, + [ + "evaluate", + "--config", + str(config_path), + "--model", + str(model_path), + "--seed", + "9", + "--episodes", + "1", + "--output-report", + str(tmp_path / "single.json"), + ], + ) + assert single.exit_code == 0, single.output + assert "## Evaluation" in single.output + assert "Report saved to" in single.output + + conflict = runner.invoke( + app, + [ + "evaluate", + "--config", + str(config_path), + "--model", + str(model_path), + "--seed", + "9", + "--seeds", + "9", + "10", + ], + ) + assert conflict.exit_code == 1 + assert "Use either --seed or --seeds" in conflict.output + + duplicate_seeds = runner.invoke( + app, + [ + "evaluate", + "--config", + str(config_path), + "--model", + str(model_path), + "--seeds", + "2", + "2", + ], + ) + assert duplicate_seeds.exit_code == 1 + assert "seeds must be unique" in duplicate_seeds.output + + def test_cli_train_and_evaluate_and_demo(tmp_path: Path) -> None: """End-to-end CLI test: train -> evaluate -> demo-drone.""" test_config = tmp_path / "test_drone_cli.yaml" diff --git a/tests/test_evaluation.py b/tests/test_evaluation.py index ff6422b..876ffb7 100644 --- a/tests/test_evaluation.py +++ b/tests/test_evaluation.py @@ -1,8 +1,14 @@ """Tests for agent evaluation and JSON report generation.""" +import csv import json +import math from pathlib import Path +import gymnasium as gym +import numpy as np +import pytest + from adaptive_rl.algorithms.ppo import PPOAlgorithm from adaptive_rl.environments.drone import DroneNavigation3DEnv from adaptive_rl.evaluation.evaluator import ( @@ -11,6 +17,56 @@ evaluate_random_policy, run_obstacle_density_experiment, ) +from adaptive_rl.evaluation.statistics import ( + student_t_critical_value, + summarize_seed_values, +) + + +class _SeedOutcomeEnv(gym.Env): + observation_space = gym.spaces.Box(-1000.0, 1000.0, shape=(1,), dtype=np.float32) + action_space = gym.spaces.Box(-1.0, 1.0, shape=(1,), dtype=np.float32) + + def __init__(self, *, include_outcomes: bool = True) -> None: + super().__init__() + self.include_outcomes = include_outcomes + self.current_seed = 0 + self.position = np.zeros(2, dtype=np.float64) + + def reset( + self, *, seed: int | None = None, options: dict | None = None + ) -> tuple[np.ndarray, dict]: + super().reset(seed=seed) + self.current_seed = 0 if seed is None else seed + self.position = np.array([float(self.current_seed), 0.0]) + return np.zeros(1, dtype=np.float32), {"position": self.position.copy()} + + def step(self, action: np.ndarray) -> tuple[np.ndarray, float, bool, bool, dict]: + self.position = self.position + np.array([1.0, 0.0]) + info = {"position": self.position.copy()} + if self.include_outcomes: + outcome = self.current_seed % 3 + info.update( + { + "success": outcome == 0, + "collision": outcome == 1, + } + ) + return ( + np.zeros(1, dtype=np.float32), + float(self.current_seed), + outcome != 2, + outcome == 2, + info, + ) + return np.zeros(1, dtype=np.float32), float(self.current_seed), True, False, info + + +class _ZeroPolicy: + def predict( + self, observation: np.ndarray, deterministic: bool = True + ) -> tuple[np.ndarray, None]: + return np.zeros(1, dtype=np.float32), None def test_evaluator_deterministic_evaluation(tmp_path: Path) -> None: @@ -61,6 +117,152 @@ def test_evaluator_episode_records() -> None: env.close() +def test_student_t_statistics_match_analytical_values() -> None: + assert student_t_critical_value(0.95, 1) == pytest.approx(12.7062047364, rel=1e-9) + assert student_t_critical_value(0.95, 2) == pytest.approx(4.3026527297, rel=1e-9) + assert student_t_critical_value(0.95, 9) == pytest.approx(2.2621571627, rel=1e-9) + + stats = summarize_seed_values([1.0, 2.0, 3.0]) + margin = 4.3026527297 / math.sqrt(3.0) + assert stats.mean == pytest.approx(2.0) + assert stats.std == pytest.approx(1.0) + assert stats.ci95_lower == pytest.approx(2.0 - margin) + assert stats.ci95_upper == pytest.approx(2.0 + margin) + assert stats.sample_count == 3 + + +def test_student_t_statistics_handle_small_and_constant_samples() -> None: + one = summarize_seed_values([7.0]) + assert one.mean == 7.0 + assert one.std is None + assert one.ci95_lower is None + assert one.ci95_upper is None + + two = summarize_seed_values([0.0, 2.0]) + assert two.mean == 1.0 + assert two.std == pytest.approx(math.sqrt(2.0)) + assert two.ci95_lower is not None and math.isfinite(two.ci95_lower) + assert two.ci95_upper is not None and math.isfinite(two.ci95_upper) + + constant = summarize_seed_values([3.0, 3.0, 3.0]) + assert constant.mean == 3.0 + assert constant.std == 0.0 + assert constant.ci95_lower == 3.0 + assert constant.ci95_upper == 3.0 + + assert summarize_seed_values([None, None]).mean is None + with pytest.raises(ValueError, match="finite"): + summarize_seed_values([1.0, float("nan")]) + + +def test_evaluate_seeds_preserves_per_seed_records_and_metrics(tmp_path: Path) -> None: + env = _SeedOutcomeEnv() + evaluator = Evaluator(algorithm=_ZeroPolicy(), env=env) # type: ignore[arg-type] + requested_seeds = [10, 20] + result = evaluator.evaluate_seeds(requested_seeds, episodes_per_seed=2) + + assert result.seeds == [10, 20] + assert requested_seeds == [10, 20] + assert result.total_episodes == 4 + assert [(record.seed, record.episode_index) for record in result.episodes] == [ + (10, 0), + (10, 1), + (20, 0), + (20, 1), + ] + assert [record.episode_seed for record in result.episodes] == [20, 21, 40, 41] + assert all(record.path_length == 1.0 for record in result.episodes) + assert [summary.seed for summary in result.per_seed] == requested_seeds + assert [summary.episodes for summary in result.per_seed] == [2, 2] + assert [summary.mean_reward for summary in result.per_seed] == [20.5, 40.5] + assert result.per_seed[0].success_rate == pytest.approx(0.5) + assert result.per_seed[0].collision_rate == pytest.approx(0.0) + assert result.per_seed[0].truncation_rate == pytest.approx(0.5) + assert result.per_seed[0].mean_episode_length == 1.0 + assert result.per_seed[0].path_length == 1.0 + assert result.aggregate["mean_reward"].mean == pytest.approx(30.5) + assert result.aggregate["mean_reward"].std == pytest.approx(math.sqrt(200.0)) + + json_path = tmp_path / "evaluation_multiseed.json" + csv_path = tmp_path / "evaluation_multiseed.csv" + saved_json, saved_csv = evaluator.save_multiseed_report(result, json_path, csv_path) + assert saved_json.is_file() + assert saved_csv.is_file() + document = json.loads(saved_json.read_text(encoding="utf-8")) + json.dumps(document, allow_nan=False) + assert document["metadata"] == { + "seeds": [10, 20], + "seed_count": 2, + "episodes_per_seed": 2, + "total_episodes": 4, + "deterministic": True, + "environment": "drone", + "confidence_interval": ( + "two-sided 95% Student's t interval across seed summaries; " + "sample standard deviation; unavailable when fewer than two values exist" + ), + "duplicate_seed_policy": "rejected", + } + assert len(document["episodes"]) == 4 + assert len(document["per_seed"]) == 2 + assert document["aggregate"]["mean_reward"]["sample_count"] == 2 + assert document["episodes"][0]["episode_seed"] == 20 + assert document["episodes"][0]["path_length"] == 1.0 + + expected_headers = [ + "metric", + "mean", + "std", + "ci95_lower", + "ci95_upper", + "sample_count", + "seed_count", + "episodes_per_seed", + "total_episodes", + ] + with saved_csv.open(newline="", encoding="utf-8") as handle: + reader = csv.DictReader(handle) + rows = list(reader) + assert reader.fieldnames == expected_headers + assert len(rows) == len(result.aggregate) + assert rows[0]["metric"] == "mean_reward" + assert float(rows[0]["mean"]) == pytest.approx(30.5) + assert rows[0]["seed_count"] == "2" + assert rows[0]["episodes_per_seed"] == "2" + assert rows[0]["total_episodes"] == "4" + evaluator.close() + + +def test_evaluate_seeds_is_deterministic_and_rejects_invalid_inputs() -> None: + evaluator = Evaluator(algorithm=_ZeroPolicy(), env=_SeedOutcomeEnv()) # type: ignore[arg-type] + first = evaluator.evaluate_seeds([7, 13], episodes_per_seed=2, deterministic=True) + first_data = first.to_dict() + second = evaluator.evaluate_seeds([7, 13], episodes_per_seed=2, deterministic=True) + assert second.to_dict() == first_data + + with pytest.raises(ValueError, match="must not be empty"): + evaluator.evaluate_seeds([], episodes_per_seed=1) + with pytest.raises(ValueError, match="unique"): + evaluator.evaluate_seeds([7, 7], episodes_per_seed=1) + with pytest.raises(ValueError, match="positive"): + evaluator.evaluate_seeds([7], episodes_per_seed=0) + with pytest.raises(ValueError, match="non-negative"): + evaluator.evaluate_seeds([-1], episodes_per_seed=1) + evaluator.close() + + +def test_evaluate_seeds_preserves_unavailable_optional_metrics() -> None: + evaluator = Evaluator( + algorithm=_ZeroPolicy(), + env=_SeedOutcomeEnv(include_outcomes=False), # type: ignore[arg-type] + ) + result = evaluator.evaluate_seeds([2, 5], episodes_per_seed=1) + assert all(summary.success_rate is None for summary in result.per_seed) + assert all(summary.collision_rate is None for summary in result.per_seed) + assert result.aggregate["success_rate"].mean is None + evaluator.close() + + def test_evaluate_random_policy() -> None: """Verify uniform-random policy evaluation baseline executes and returns valid metrics.""" env = DroneNavigation3DEnv(bounds=(20.0, 20.0, 10.0), max_steps=15, num_obstacles=2) From 72e1ac4ad2234fe1072571479d4f3b9a71feef86 Mon Sep 17 00:00:00 2001 From: Aryan Date: Mon, 28 Sep 2026 22:46:14 +0530 Subject: [PATCH 3/3] fix: harden PPO benchmark evaluation semantics (#244/#245) Load RL dependencies only when benchmark execution begins, measure PPO optimization separately from artifact serialization, and preserve pooled plus per-seed evaluation semantics. Use fixed benchmark defaults and add focused regression coverage. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- README.md | 4 +- docs/EXPERIMENT.md | 14 +- src/adaptive_rl/__init__.py | 26 +- .../benchmarking/learning_curve.py | 201 +++++++++------ src/adaptive_rl/cli.py | 6 +- src/adaptive_rl/training/trainer.py | 8 + tests/test_evaluation.py | 3 + tests/test_learning_curve_benchmark.py | 239 +++++++++++++++++- tests/test_training.py | 49 ++++ 9 files changed, 439 insertions(+), 111 deletions(-) diff --git a/README.md b/README.md index 1cf94db..a87e94c 100644 --- a/README.md +++ b/README.md @@ -354,7 +354,7 @@ For detailed per-test execution traces: python -m pytest -v ``` -The repository includes **49 automated unit and integration tests** verifying: +The repository includes automated unit and integration tests verifying: - 3D kinematics equations and aerodynamic drag - 29-dimensional observation space bounds - Analytical 16-ray LiDAR raycasts and obstacle clearance @@ -397,7 +397,7 @@ When a training run completes, artifacts are automatically written to disk: *(e.g., `artifacts/models/drone_ppo_demo_final.zip`)* - **Training Metadata & Loss/Reward Log**: `artifacts/metadata/{experiment_name}_training.json` - *(contains total timesteps, duration in seconds, mean reward, and per-episode return lists)* + *(contains total timesteps, `training_time_seconds` for PPO optimization only, broader `duration_seconds` through model serialization, mean reward, and per-episode return lists)* - **Periodic Checkpoints** (if configured): `artifacts/checkpoints/{experiment_name}/` diff --git a/docs/EXPERIMENT.md b/docs/EXPERIMENT.md index 7e95712..379b292 100644 --- a/docs/EXPERIMENT.md +++ b/docs/EXPERIMENT.md @@ -35,7 +35,7 @@ This document details the experimental methodology, hypotheses, benchmark variab ## 3. PPO Learning-Curve Benchmark -The budget benchmark trains a fresh PPO model from the same base configuration at each requested training budget. Every model is evaluated with the same ordered evaluation seeds, episode count per seed, and deterministic-action setting; evaluation uses the saved model and a separate fresh environment. +The budget benchmark trains a fresh PPO model from the same base configuration at each requested training budget. Every model is evaluated with the same ordered evaluation seed groups, episode count per seed, environment parameters, algorithm settings, and deterministic-action setting; evaluation uses the saved model and a separate fresh environment. `--eval-seeds` selects the seed groups; `--episodes` is the number of episodes run within each group and does not determine how many seeds are evaluated. ```bash adaptive-rl benchmark budgets \ @@ -68,11 +68,17 @@ artifacts/benchmarks/ └── ... ``` -JSON contains benchmark settings, one result object per requested budget, and plot-ready series. CSV contains the same per-budget performance values. Pass `--plot` to additionally render `learning_curve_budget.png`; Matplotlib must be installed for that optional output. +JSON contains benchmark settings, one result object per requested budget, pooled metrics, per-seed summaries, cross-seed Student's t statistics, and plot-ready series. CSV contains the pooled per-budget performance values. Pass `--plot` to additionally render `learning_curve_budget.png`; Matplotlib is imported only when plotting is requested and must be installed for that optional output. Generated plot figures are closed after saving. -`budget_timesteps` records the requested budget, while `trained_timesteps` records the actual environment interactions reported by Stable-Baselines3. PPO collects complete rollouts, so a requested budget that is not a multiple of its configured `n_steps` can be exceeded up to the next rollout boundary. Compare results using `trained_timesteps` when budgets are not aligned to rollout sizes. Training duration is informational and should not be interpreted as a hardware-independent performance metric. +The built-in benchmark defaults are budgets `[5000, 10000, 25000, 50000]`, training seed `42`, evaluation seed groups `[42, 43, 44, 45, 46]`, and `20` episodes per seed. A `benchmark` section in the YAML supplies these values instead; explicit CLI options override the corresponding config values. Legacy `evaluation.eval_episodes` does not control the number of seed groups or the benchmark episode count. Thus, without overrides, the default evaluation runs five seed groups with twenty episodes each, not twenty seed groups with twenty episodes each. -Interpret the curves jointly: rising success rate and mean reward with a falling collision or timeout rate suggest improvement; flat metrics may indicate a plateau. A timeout is counted only when Gymnasium returns `truncated=True`, not merely because an episode has a particular length. The same seeds make evaluation conditions comparable, but do not remove variation from training or guarantee bit-for-bit results across hardware, PyTorch versions, or CUDA kernels. +`budget_timesteps` records the requested budget, while `trained_timesteps` records the actual environment interactions reported by Stable-Baselines3. For example, budget `65` with PPO `n_steps: 64` trains to `128` steps because PPO collects complete rollouts. Compare results using `trained_timesteps` when budgets are not aligned to rollout sizes. + +`training_time_seconds` measures only the call to `PPOAlgorithm.train()` using a monotonic clock. It excludes environment/model setup, final model serialization, metadata writing, evaluation, JSON/CSV export, and plotting. Training metadata also retains the broader legacy `duration_seconds` lifecycle measure, which is not the benchmark training-time metric. Neither duration is hardware-independent. + +The named benchmark metrics (`success_rate`, `collision_rate`, `timeout_rate`, `mean_reward`, `std_reward`, and `mean_episode_length`) are pooled descriptive summaries over all evaluated episodes for a budget. Reward standard deviation is the sample standard deviation across pooled episode returns and is unavailable (`null` in JSON, blank in CSV) with fewer than two episodes. Success and collision rates use episodes that reported the corresponding outcome field; timeout rate is based only on Gymnasium's actual `truncated` signal. The JSON additionally retains per-seed summaries and cross-seed Student's t statistics from the reusable evaluator; these are distinct from the pooled metrics and are not estimates based on the pooled episode sample. Within-seed reward and episode-length standard deviations follow the evaluator's existing population-standard-deviation convention; cross-seed uncertainty is then calculated over those seed summaries using sample-standard-deviation and Student's t conventions. + +Interpret the curves jointly: rising success rate and mean reward with a falling collision or timeout rate suggest improvement; flat metrics may indicate a plateau. A timeout is counted only when Gymnasium returns `truncated=True`, not merely because an episode has a particular length. The same seed groups and settings make evaluation conditions comparable, but do not remove variation from training or guarantee bit-for-bit results across hardware, PyTorch versions, or CUDA kernels. For a CI-sized run, copy the experiment YAML and set PPO `n_steps: 64` and `batch_size: 32` in that copy. Then run a short evaluation: diff --git a/src/adaptive_rl/__init__.py b/src/adaptive_rl/__init__.py index f86dade..aeebd14 100644 --- a/src/adaptive_rl/__init__.py +++ b/src/adaptive_rl/__init__.py @@ -4,13 +4,8 @@ a simulated 3D drone through obstacles toward a target waypoint. """ -from adaptive_rl.benchmarking import ( - LearningCurveBenchmarkResult, - LearningCurvePoint, - plot_learning_curve, - run_learning_curve_benchmark, - validate_budgets, -) +from typing import Any + from adaptive_rl.config import ( AlgorithmConfig, BenchmarkConfig, @@ -33,6 +28,23 @@ __version__ = "0.1.0" +_BENCHMARK_EXPORTS = { + "LearningCurveBenchmarkResult", + "LearningCurvePoint", + "plot_learning_curve", + "run_learning_curve_benchmark", + "validate_budgets", +} + + +def __getattr__(name: str) -> Any: + if name in _BENCHMARK_EXPORTS: + from adaptive_rl import benchmarking + + return getattr(benchmarking, name) + raise AttributeError(f"module {__name__!r} has no attribute {name!r}") + + __all__ = [ "__version__", "AlgorithmConfig", diff --git a/src/adaptive_rl/benchmarking/learning_curve.py b/src/adaptive_rl/benchmarking/learning_curve.py index 0639611..34ac1a7 100644 --- a/src/adaptive_rl/benchmarking/learning_curve.py +++ b/src/adaptive_rl/benchmarking/learning_curve.py @@ -5,16 +5,11 @@ import csv import json import math -import time from dataclasses import dataclass, field from pathlib import Path from typing import Any, Sequence -from adaptive_rl.algorithms.ppo import PPOAlgorithm from adaptive_rl.config import BenchmarkConfig, ExperimentConfig -from adaptive_rl.environments.registry import make_env -from adaptive_rl.evaluation.evaluator import Evaluator -from adaptive_rl.training.trainer import PPOTrainer def validate_budgets( @@ -80,7 +75,7 @@ class LearningCurvePoint: collision_rate: float | None timeout_rate: float | None mean_reward: float - std_reward: float + std_reward: float | None mean_episode_length: float training_time_seconds: float model_path: str @@ -90,6 +85,8 @@ class LearningCurvePoint: deterministic: bool algorithm: str environment: str + per_seed_summaries: list[dict[str, int | float | None]] = field(default_factory=list) + cross_seed_statistics: dict[str, dict[str, float | int | None]] = field(default_factory=dict) @dataclass @@ -123,6 +120,22 @@ def to_dict(self) -> dict[str, Any]: "evaluation_episodes": self.evaluation_episodes, "deterministic": self.deterministic, "budgets": self.budgets, + "metric_semantics": { + "aggregation": "pooled episode-level descriptive statistics", + "success_rate": "pooled over episodes with available success metadata", + "collision_rate": "pooled over episodes with available collision metadata", + "timeout_rate": "fraction of all evaluated episodes with truncated=True", + "mean_reward": "mean of pooled episode returns", + "std_reward": "sample standard deviation across pooled episode returns", + "mean_episode_length": "mean of pooled episode lengths", + "cross_seed_statistics": ( + "Student's t summaries across evaluator per-seed summaries; " + "within-seed reward and length standard deviations use evaluator semantics" + ), + }, + "training_time_semantics": ( + "monotonic elapsed time inside PPOAlgorithm.train() only" + ), }, "results": [ { @@ -142,6 +155,8 @@ def to_dict(self) -> dict[str, Any]: "deterministic": point.deterministic, "algorithm": point.algorithm, "environment": point.environment, + "per_seed_summaries": point.per_seed_summaries, + "cross_seed_statistics": point.cross_seed_statistics, } for point in self.points ], @@ -170,6 +185,26 @@ def _budget_dir(base_output_dir: Path, budget: int) -> Path: return base_output_dir / "learning_curve" / f"budget_{budget}" +def _make_env(env_name: str, **env_kwargs: Any) -> Any: + from adaptive_rl.environments.registry import make_env + + return make_env(env_name, **env_kwargs) + + +def _make_trainer(config: ExperimentConfig, env: Any) -> Any: + from adaptive_rl.training.trainer import PPOTrainer + + return PPOTrainer(config=config, env=env) + + +def _load_evaluator(model_path: Path, env: Any) -> tuple[Any, Any]: + from adaptive_rl.algorithms.ppo import PPOAlgorithm + from adaptive_rl.evaluation.evaluator import Evaluator + + algorithm = PPOAlgorithm.from_pretrained(model_path, env=env) + return algorithm, Evaluator(algorithm=algorithm, env=env) + + def _evaluate_model( model_path: Path, *, @@ -178,36 +213,35 @@ def _evaluate_model( evaluation_seeds: Sequence[int], evaluation_episodes: int, deterministic: bool, -) -> tuple[float | None, float | None, float | None, float, float, float]: - env = make_env(env_name, **env_kwargs) +) -> tuple[ + float | None, + float | None, + float | None, + float, + float | None, + float, + list[dict[str, Any]], + dict[str, dict[str, float | int | None]], +]: + env = _make_env(env_name, **env_kwargs) try: - algo = PPOAlgorithm.from_pretrained(model_path, env=env) - evaluator = Evaluator(algorithm=algo, env=env) - - all_rewards: list[float] = [] - all_lengths: list[int] = [] - successes: list[bool] = [] - collisions: list[bool] = [] - timeout_count = 0 - total_episodes = 0 - - for seed in evaluation_seeds: - metrics = evaluator.evaluate( - num_episodes=evaluation_episodes, - deterministic=deterministic, - base_seed=seed, - ) - all_rewards.extend(metrics.additional_metrics.get("all_rewards", [])) - all_lengths.extend(metrics.additional_metrics.get("all_lengths", [])) - records = evaluator.last_episode_records - total_episodes += len(records) - successes.extend(record.success for record in records if record.success is not None) - collisions.extend( - record.collision for record in records if record.collision is not None - ) - timeout_count += sum(record.truncated for record in records) - - mean_reward = float(sum(all_rewards) / len(all_rewards)) if all_rewards else 0.0 + _, evaluator = _load_evaluator(model_path, env) + evaluation = evaluator.evaluate_seeds( + seeds=evaluation_seeds, + episodes_per_seed=evaluation_episodes, + deterministic=deterministic, + ) + records = evaluation.episodes + if not records: + raise RuntimeError("Evaluation produced no episodes.") + all_rewards = [record.return_value for record in records] + all_lengths = [record.episode_length for record in records] + successes = [record.success for record in records if record.success is not None] + collisions = [record.collision for record in records if record.collision is not None] + timeout_count = sum(record.truncated for record in records) + total_episodes = len(records) + + mean_reward = float(sum(all_rewards) / len(all_rewards)) std_reward = ( float( ( @@ -217,9 +251,9 @@ def _evaluate_model( ** 0.5 ) if len(all_rewards) > 1 - else 0.0 + else None ) - mean_episode_length = float(sum(all_lengths) / len(all_lengths)) if all_lengths else 0.0 + mean_episode_length = float(sum(all_lengths) / len(all_lengths)) success_rate = float(sum(successes) / len(successes)) if successes else None collision_rate = float(sum(collisions) / len(collisions)) if collisions else None @@ -232,6 +266,8 @@ def _evaluate_model( mean_reward, std_reward, mean_episode_length, + [summary.to_dict() for summary in evaluation.per_seed], + {name: stats.to_dict() for name, stats in evaluation.aggregate.items()}, ) finally: env.close() @@ -261,13 +297,12 @@ def _run_single_budget( config_copy.output_dir = benchmark_dir config_copy.log_dir = benchmark_dir / "logs" - env = make_env(config_copy.environment.name, **config_copy.environment.parameters) - trainer: PPOTrainer | None = None - start = time.perf_counter() + env = _make_env(config_copy.environment.name, **config_copy.environment.parameters) + trainer: Any = None try: - trainer = PPOTrainer(config=config_copy, env=env) + trainer = _make_trainer(config_copy, env) result = trainer.fit() - training_time_seconds = time.perf_counter() - start + training_time_seconds = result.training_time_seconds finally: if trainer is not None: trainer.close() @@ -287,15 +322,22 @@ def _run_single_budget( f"Invalid training duration for budget {budget}: {training_time_seconds}" ) - success_rate, collision_rate, timeout_rate, mean_reward, std_reward, mean_episode_length = ( - _evaluate_model( - model_path, - env_name=config_copy.environment.name, - env_kwargs=config_copy.environment.parameters, - evaluation_seeds=evaluation_seeds, - evaluation_episodes=evaluation_episodes, - deterministic=deterministic, - ) + ( + success_rate, + collision_rate, + timeout_rate, + mean_reward, + std_reward, + mean_episode_length, + per_seed_summaries, + cross_seed_statistics, + ) = _evaluate_model( + model_path, + env_name=config_copy.environment.name, + env_kwargs=config_copy.environment.parameters, + evaluation_seeds=evaluation_seeds, + evaluation_episodes=evaluation_episodes, + deterministic=deterministic, ) return LearningCurvePoint( @@ -315,6 +357,8 @@ def _run_single_budget( deterministic=deterministic, algorithm=config_copy.algorithm.name, environment=config_copy.environment.name, + per_seed_summaries=per_seed_summaries, + cross_seed_statistics=cross_seed_statistics, ) @@ -335,11 +379,8 @@ def run_learning_curve_benchmark( if config.training is None: raise ValueError("A training section is required to run the learning-curve benchmark.") - if budgets is None: - benchmark_cfg = _resolve_benchmark_config(config, None) - normalized = validate_budgets(benchmark_cfg.budgets) - else: - normalized = validate_budgets(budgets) + benchmark_cfg = _resolve_benchmark_config(config, None) + normalized = validate_budgets(benchmark_cfg.budgets if budgets is None else budgets) if training_seed is not None: final_training_seed = training_seed @@ -348,13 +389,9 @@ def run_learning_curve_benchmark( else: final_training_seed = config.seed - if evaluation_seeds is None: - if config.benchmark is not None: - final_eval_seeds = list(config.benchmark.evaluation_seeds) - else: - final_eval_seeds = [config.seed + i for i in range(config.evaluation.eval_episodes)] - else: - final_eval_seeds = list(evaluation_seeds) + final_eval_seeds = ( + list(benchmark_cfg.evaluation_seeds) if evaluation_seeds is None else list(evaluation_seeds) + ) if isinstance(final_training_seed, bool) or not isinstance(final_training_seed, int): raise ValueError("Training seed must be an integer.") @@ -369,13 +406,9 @@ def run_learning_curve_benchmark( if len(set(final_eval_seeds)) != len(final_eval_seeds): raise ValueError("Evaluation seeds must not contain duplicates.") - if evaluation_episodes is None: - if config.benchmark is not None: - final_eval_episodes = config.benchmark.evaluation_episodes - else: - final_eval_episodes = config.evaluation.eval_episodes - else: - final_eval_episodes = evaluation_episodes + final_eval_episodes = ( + benchmark_cfg.evaluation_episodes if evaluation_episodes is None else evaluation_episodes + ) if isinstance(final_eval_episodes, bool) or not isinstance(final_eval_episodes, int): raise ValueError("Evaluation episodes per seed must be an integer.") if final_eval_episodes <= 0: @@ -498,17 +531,19 @@ def plot_learning_curve( rewards = [point.mean_reward for point in result.points] fig, axes = plt.subplots(1, 2, figsize=(12, 4), constrained_layout=True) - axes[0].plot(budgets, success_rates, marker="o", linewidth=2) - axes[0].set_title("Success rate vs training budget") - axes[0].set_xlabel("Training budget (timesteps)") - axes[0].set_ylabel("Success rate") - axes[0].set_ylim(-0.05, 1.05) - - axes[1].plot(budgets, rewards, marker="s", linewidth=2, color="tab:orange") - axes[1].set_title("Mean reward vs training budget") - axes[1].set_xlabel("Training budget (timesteps)") - axes[1].set_ylabel("Mean reward") - - fig.savefig(plot_target, dpi=160) - plt.close(fig) + try: + axes[0].plot(budgets, success_rates, marker="o", linewidth=2) + axes[0].set_title("Success rate vs training budget") + axes[0].set_xlabel("Training budget (timesteps)") + axes[0].set_ylabel("Success rate") + axes[0].set_ylim(-0.05, 1.05) + + axes[1].plot(budgets, rewards, marker="s", linewidth=2, color="tab:orange") + axes[1].set_title("Mean reward vs training budget") + axes[1].set_xlabel("Training budget (timesteps)") + axes[1].set_ylabel("Mean reward") + + fig.savefig(plot_target, dpi=160, metadata={"Software": "AdaptiveRL"}) + finally: + plt.close(fig) return plot_target diff --git a/src/adaptive_rl/cli.py b/src/adaptive_rl/cli.py index 56c6f14..6f4089a 100644 --- a/src/adaptive_rl/cli.py +++ b/src/adaptive_rl/cli.py @@ -230,10 +230,12 @@ def benchmark_budgets( None, "--training-seed", help="Override the training seed used across budgets" ), eval_seeds: Optional[str] = typer.Option( - None, "--eval-seeds", help="Comma-separated evaluation seeds (for example: 42,43,44)" + None, + "--eval-seeds", + help="Comma-separated evaluation seed groups; defaults to configured or benchmark seeds", ), episodes: Optional[int] = typer.Option( - None, "--episodes", help="Override evaluation episodes per seed" + None, "--episodes", help="Evaluation episodes per seed (separate from the seed count)" ), deterministic: Optional[bool] = typer.Option( None, "--deterministic/--stochastic", help="Use deterministic actions during evaluation" diff --git a/src/adaptive_rl/training/trainer.py b/src/adaptive_rl/training/trainer.py index aa39da3..a80bc0d 100644 --- a/src/adaptive_rl/training/trainer.py +++ b/src/adaptive_rl/training/trainer.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +import math import random import time from dataclasses import dataclass, field @@ -40,6 +41,7 @@ class TrainingResult: success_rate: Optional[float] = None collision_rate: Optional[float] = None metadata_path: Optional[Path] = None + training_time_seconds: float = 0.0 class PPOTrainer: @@ -109,10 +111,14 @@ def fit(self) -> TrainingResult: ) assert self.config.training is not None + training_started_at = time.perf_counter() self.algorithm.train( total_timesteps=self.config.training.total_timesteps, callback=adapter, ) + training_time_seconds = time.perf_counter() - training_started_at + if not math.isfinite(training_time_seconds) or training_time_seconds < 0: + raise RuntimeError(f"Invalid training duration: {training_time_seconds}") # Save model artifact models_dir = self.config.output_dir / "models" @@ -140,6 +146,7 @@ def fit(self) -> TrainingResult: "collision_rate": self.metric_logger.collision_rate, "final_model_path": str(final_model_path), "duration_seconds": round(duration, 2), + "training_time_seconds": training_time_seconds, "episode_rewards": [round(float(r), 2) for r in self.metric_logger.episode_rewards], "episode_lengths": [int(length) for length in self.metric_logger.episode_lengths], "version": adaptive_rl.__version__, @@ -159,6 +166,7 @@ def fit(self) -> TrainingResult: success_rate=self.metric_logger.success_rate, collision_rate=self.metric_logger.collision_rate, metadata_path=metadata_path, + training_time_seconds=training_time_seconds, ) def close(self) -> None: diff --git a/tests/test_evaluation.py b/tests/test_evaluation.py index 876ffb7..dcd4994 100644 --- a/tests/test_evaluation.py +++ b/tests/test_evaluation.py @@ -120,7 +120,10 @@ def test_evaluator_episode_records() -> None: def test_student_t_statistics_match_analytical_values() -> None: assert student_t_critical_value(0.95, 1) == pytest.approx(12.7062047364, rel=1e-9) assert student_t_critical_value(0.95, 2) == pytest.approx(4.3026527297, rel=1e-9) + assert student_t_critical_value(0.95, 5) == pytest.approx(2.5705818356, rel=1e-9) assert student_t_critical_value(0.95, 9) == pytest.approx(2.2621571627, rel=1e-9) + assert student_t_critical_value(0.95, 10) == pytest.approx(2.2281388520, rel=1e-9) + assert student_t_critical_value(0.95, 30) == pytest.approx(2.0422724563, rel=1e-9) stats = summarize_seed_values([1.0, 2.0, 3.0]) margin = 4.3026527297 / math.sqrt(3.0) diff --git a/tests/test_learning_curve_benchmark.py b/tests/test_learning_curve_benchmark.py index 3b33a43..2259740 100644 --- a/tests/test_learning_curve_benchmark.py +++ b/tests/test_learning_curve_benchmark.py @@ -3,8 +3,14 @@ import csv import json import math +import os +import subprocess +import sys +import textwrap from pathlib import Path +from types import ModuleType from typing import Any +from unittest.mock import MagicMock import gymnasium as gym import numpy as np @@ -142,21 +148,58 @@ def test_benchmark_config_defaults_and_old_config_compatibility(tmp_path: Path) assert configured.benchmark.deterministic is False +def test_core_and_benchmark_imports_do_not_load_optional_rl_stack() -> None: + repository_root = Path(__file__).resolve().parent.parent + script = textwrap.dedent( + """ + import sys + from importlib.abc import MetaPathFinder + + class BlockOptionalRLImports(MetaPathFinder): + def find_spec(self, fullname, path=None, target=None): + if fullname == "gymnasium" or fullname.startswith( + ("stable_baselines3", "torch") + ): + raise AssertionError(f"Optional RL dependency imported eagerly: {fullname}") + return None + + sys.meta_path.insert(0, BlockOptionalRLImports()) + import adaptive_rl + from adaptive_rl.benchmarking import run_learning_curve_benchmark + + assert callable(run_learning_curve_benchmark) + assert callable(adaptive_rl.run_learning_curve_benchmark) + """ + ) + environment = os.environ.copy() + environment["PYTHONPATH"] = os.pathsep.join( + [str(repository_root / "src"), environment.get("PYTHONPATH", "")] + ) + subprocess.run( + [sys.executable, "-c", script], + cwd=repository_root, + env=environment, + check=True, + capture_output=True, + text=True, + ) + + def test_learning_curve_benchmark_execution_and_exports( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: from adaptive_rl.benchmarking import learning_curve + from adaptive_rl.training.trainer import PPOTrainer config = _make_config(tmp_path) original = config.model_dump(mode="python") - real_trainer_type = learning_curve.PPOTrainer - real_make_env = learning_curve.make_env + real_make_env = learning_curve._make_env real_evaluate_model = learning_curve._evaluate_model trainers: list[Any] = [] envs: list[gym.Env] = [] evaluation_calls: list[dict[str, Any]] = [] - class TrackingTrainer(real_trainer_type): + class TrackingTrainer(PPOTrainer): def __init__(self, *args: Any, **kwargs: Any) -> None: super().__init__(*args, **kwargs) self.fit_calls = 0 @@ -166,17 +209,20 @@ def fit(self) -> Any: self.fit_calls += 1 return super().fit() - def track_env(*args: Any, **kwargs: Any) -> gym.Env: - env = real_make_env(*args, **kwargs) + def track_env(env_name: str, **kwargs: Any) -> gym.Env: + env = real_make_env(env_name, **kwargs) envs.append(env) return env + def track_trainer(config: ExperimentConfig, env: gym.Env) -> PPOTrainer: + return TrackingTrainer(config=config, env=env) + def track_evaluation(*args: Any, **kwargs: Any) -> Any: evaluation_calls.append(kwargs.copy()) return real_evaluate_model(*args, **kwargs) - monkeypatch.setattr(learning_curve, "PPOTrainer", TrackingTrainer) - monkeypatch.setattr(learning_curve, "make_env", track_env) + monkeypatch.setattr(learning_curve, "_make_trainer", track_trainer) + monkeypatch.setattr(learning_curve, "_make_env", track_env) monkeypatch.setattr(learning_curve, "_evaluate_model", track_evaluation) result = run_learning_curve_benchmark(config, budgets=[128, 64]) @@ -236,7 +282,7 @@ def track_evaluation(*args: Any, **kwargs: Any) -> Any: assert point.collision_rate is None or 0.0 <= point.collision_rate <= 1.0 assert point.timeout_rate is None or 0.0 <= point.timeout_rate <= 1.0 assert math.isfinite(point.mean_reward) - assert point.std_reward >= 0.0 + assert point.std_reward is None or point.std_reward >= 0.0 assert point.mean_episode_length >= 0.0 assert math.isfinite(point.training_time_seconds) assert point.training_time_seconds >= 0.0 @@ -274,6 +320,12 @@ def track_evaluation(*args: Any, **kwargs: Any) -> Any: assert required_metrics <= row.keys() assert Path(row["model_path"]).is_file() assert set(data["plot_data"]) == {"budgets", "success_rate", "mean_reward"} + assert data["benchmark"]["metric_semantics"]["aggregation"] == ( + "pooled episode-level descriptive statistics" + ) + assert data["benchmark"]["training_time_semantics"].startswith("monotonic elapsed time") + assert all(len(row["per_seed_summaries"]) == 2 for row in data["results"]) + assert all("success_rate" in row["cross_seed_statistics"] for row in data["results"]) assert result.csv_path is not None and result.csv_path.is_file() with result.csv_path.open(newline="", encoding="utf-8") as handle: @@ -336,6 +388,134 @@ def test_learning_curve_benchmark_repeats_deterministically(tmp_path: Path) -> N assert getattr(first_point, field) == getattr(second_point, field) +def test_benchmark_keeps_single_episode_standard_deviation_unavailable( + tmp_path: Path, +) -> None: + config = _make_config(tmp_path) + assert config.benchmark is not None + config.benchmark.evaluation_seeds = [11] + config.benchmark.evaluation_episodes = 1 + + result = run_learning_curve_benchmark( + config, + budgets=[64], + output_dir=tmp_path / "single_episode", + ) + + point = result.points[0] + assert point.std_reward is None + assert result.to_dict()["results"][0]["std_reward"] is None + assert result.csv_path is not None + with result.csv_path.open(newline="", encoding="utf-8") as handle: + row = next(csv.DictReader(handle)) + assert row["std_reward"] == "" + + +def test_default_benchmark_evaluation_does_not_derive_seed_count_from_episode_count( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + from adaptive_rl.benchmarking import learning_curve + + config = _make_config(tmp_path) + config.benchmark = None + config.evaluation.eval_episodes = 2 + calls: list[dict[str, Any]] = [] + + def fake_run( + config: ExperimentConfig, + budget: int, + **kwargs: Any, + ) -> learning_curve.LearningCurvePoint: + calls.append({"budget": budget, **kwargs}) + return learning_curve.LearningCurvePoint( + budget_timesteps=budget, + trained_timesteps=budget, + success_rate=None, + collision_rate=None, + timeout_rate=0.0, + mean_reward=1.0, + std_reward=0.0, + mean_episode_length=1.0, + training_time_seconds=0.0, + model_path=str(tmp_path / "model.zip"), + training_seed=kwargs["training_seed"], + evaluation_seeds=list(kwargs["evaluation_seeds"]), + evaluation_episodes=kwargs["evaluation_episodes"], + deterministic=kwargs["deterministic"], + algorithm="ppo", + environment="drone", + ) + + monkeypatch.setattr(learning_curve, "_run_single_budget", fake_run) + result = run_learning_curve_benchmark(config, budgets=[64], output_dir=tmp_path / "out") + + assert result.evaluation_seeds == [42, 43, 44, 45, 46] + assert result.evaluation_episodes == 20 + assert calls[0]["evaluation_seeds"] == [42, 43, 44, 45, 46] + assert calls[0]["evaluation_episodes"] == 20 + assert len(calls[0]["evaluation_seeds"]) * calls[0]["evaluation_episodes"] == 100 + + +def test_plotting_is_lazy_and_closes_figure( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + from adaptive_rl.benchmarking.learning_curve import ( + LearningCurveBenchmarkResult, + plot_learning_curve, + ) + + figure = MagicMock() + figure.savefig.side_effect = lambda path, dpi, metadata: Path(path).touch() + axes = [MagicMock(), MagicMock()] + pyplot = ModuleType("matplotlib.pyplot") + pyplot.subplots = MagicMock(return_value=(figure, axes)) # type: ignore[attr-defined] + pyplot.close = MagicMock() # type: ignore[attr-defined] + matplotlib = ModuleType("matplotlib") + matplotlib.use = MagicMock() # type: ignore[attr-defined] + monkeypatch.setitem(sys.modules, "matplotlib", matplotlib) + monkeypatch.setitem(sys.modules, "matplotlib.pyplot", pyplot) + + output_path = tmp_path / "curve.png" + result = LearningCurveBenchmarkResult( + benchmark_name="ppo_learning_curve", + algorithm="ppo", + environment="drone", + budgets=[], + training_seed=0, + evaluation_seeds=[], + evaluation_episodes=1, + deterministic=True, + ) + path = plot_learning_curve(result, output_path) + assert path == output_path + assert output_path.is_file() + assert figure.savefig.call_args.kwargs["metadata"] == {"Software": "AdaptiveRL"} + pyplot.close.assert_called_once_with(figure) # type: ignore[attr-defined] + + +def test_plotting_reports_missing_matplotlib( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + from adaptive_rl.benchmarking.learning_curve import ( + LearningCurveBenchmarkResult, + plot_learning_curve, + ) + + monkeypatch.setitem(sys.modules, "matplotlib", None) + result = LearningCurveBenchmarkResult( + benchmark_name="ppo_learning_curve", + algorithm="ppo", + environment="drone", + budgets=[], + training_seed=0, + evaluation_seeds=[], + evaluation_episodes=1, + deterministic=True, + ) + with pytest.raises(RuntimeError, match="requires matplotlib"): + plot_learning_curve(result, tmp_path / "curve.png") + + class _OutcomeEnv(gym.Env): observation_space = gym.spaces.Box(low=-1.0, high=1.0, shape=(1,), dtype=np.float32) action_space = gym.spaces.Box(low=-1.0, high=1.0, shape=(1,), dtype=np.float32) @@ -351,6 +531,14 @@ def reset( return np.zeros(1, dtype=np.float32), {} def step(self, action: np.ndarray) -> tuple[np.ndarray, float, bool, bool, dict[str, Any]]: + if self.outcome == "collision": + return ( + np.zeros(1, dtype=np.float32), + 1.0, + True, + False, + {"success": False, "collision": True}, + ) if self.outcome == "truncated": return ( np.zeros(1, dtype=np.float32), @@ -393,21 +581,26 @@ def predict(self, observation: Any, deterministic: bool = True) -> tuple[np.ndar @pytest.mark.parametrize( - ("outcome", "expected_success", "expected_timeout"), + ("outcome", "expected_success", "expected_collision", "expected_timeout"), [ - ("termination", 0.0, 0.0), - ("truncated", 0.0, 1.0), - ("success", 1.0, 0.0), + ("termination", 0.0, 0.0, 0.0), + ("collision", 0.0, 1.0, 0.0), + ("truncated", 0.0, 0.0, 1.0), + ("success", 1.0, 0.0, 0.0), ], ) def test_evaluator_uses_gymnasium_truncation_signal( - outcome: str, expected_success: float, expected_timeout: float + outcome: str, + expected_success: float, + expected_collision: float, + expected_timeout: float, ) -> None: env = _OutcomeEnv(outcome) evaluator = Evaluator(algorithm=_ConstantPolicy(), env=env) # type: ignore[arg-type] metrics = evaluator.evaluate(num_episodes=1, deterministic=True, base_seed=2) assert metrics.success_rate == expected_success + assert metrics.collision_rate == expected_collision assert metrics.truncation_rate == expected_timeout assert evaluator.last_episode_records[0].truncated is (expected_timeout == 1.0) evaluator.close() @@ -431,6 +624,24 @@ def test_benchmark_cli_dispatch_and_budget_validation( config_path = tmp_path / "benchmark.yaml" save_config(_make_config(tmp_path), config_path) runner = CliRunner() + benchmark_help = runner.invoke(app, ["benchmark", "--help"]) + assert benchmark_help.exit_code == 0 + assert "budgets" in benchmark_help.output + budget_help = runner.invoke(app, ["benchmark", "budgets", "--help"]) + assert budget_help.exit_code == 0 + for option in ( + "--config", + "--budgets", + "--training-seed", + "--eval-seeds", + "--episodes", + "--deterministic", + "--stochastic", + "--output-dir", + "--plot", + "--no-plot", + ): + assert option in budget_help.output dispatch: list[dict[str, Any]] = [] fake_result = LearningCurveBenchmarkResult( benchmark_name="ppo_learning_curve", @@ -490,6 +701,7 @@ def fake_benchmark(config: ExperimentConfig, **kwargs: Any) -> LearningCurveBenc ) assert result.exit_code == 1 assert "Benchmark failed with error" in result.output + assert "Traceback" not in result.output assert len(dispatch) == 1 malformed_seeds = runner.invoke( @@ -507,6 +719,7 @@ def fake_benchmark(config: ExperimentConfig, **kwargs: Any) -> LearningCurveBenc ) assert malformed_seeds.exit_code == 1 assert "Malformed evaluation seed list" in malformed_seeds.output + assert "Traceback" not in malformed_seeds.output assert len(dispatch) == 1 invalid_seed_value = runner.invoke( diff --git a/tests/test_training.py b/tests/test_training.py index 227d30e..ec03365 100644 --- a/tests/test_training.py +++ b/tests/test_training.py @@ -1,5 +1,6 @@ """Tests for PPO training pipeline and model persistence.""" +import json from pathlib import Path import numpy as np @@ -84,4 +85,52 @@ def test_ppo_trainer_full_lifecycle(tmp_path: Path) -> None: assert result.final_model_path.exists() assert result.metadata_path is not None and result.metadata_path.exists() assert isinstance(result.mean_reward, float) + assert result.training_time_seconds >= 0.0 + trainer.close() + + +def test_training_duration_excludes_model_serialization(tmp_path: Path, monkeypatch) -> None: + from adaptive_rl.training import trainer as trainer_module + + clock = [0.0] + + class FakeEnv: + def close(self) -> None: + pass + + class FakeAlgorithm: + def __init__(self, **kwargs) -> None: + self.num_timesteps = 64 + + def train(self, total_timesteps: int, callback=None) -> None: + clock[0] += 2.5 + + def save(self, path: str | Path) -> None: + clock[0] += 100.0 + Path(path).write_bytes(b"model") + + monkeypatch.setattr(trainer_module, "PPOAlgorithm", FakeAlgorithm) + monkeypatch.setattr(trainer_module, "SB3CallbackAdapter", lambda **kwargs: object()) + monkeypatch.setattr(trainer_module.time, "perf_counter", lambda: clock[0]) + config = ExperimentConfig( + name="timing_test", + output_dir=tmp_path / "artifacts", + log_dir=tmp_path / "logs", + algorithm=AlgorithmConfig( + name="ppo", + parameters={"n_steps": 64, "batch_size": 32}, + ), + environment=EnvironmentConfig(name="drone"), + training=TrainingConfig(total_timesteps=64, checkpoint_freq=0), + evaluation=EvaluationConfig(eval_episodes=1), + ) + + trainer = PPOTrainer(config=config, env=FakeEnv()) # type: ignore[arg-type] + result = trainer.fit() + metadata = json.loads(result.metadata_path.read_text(encoding="utf-8")) + + assert result.training_time_seconds == 2.5 + assert clock[0] == 102.5 + assert metadata["training_time_seconds"] == 2.5 + assert result.final_model_path.is_file() trainer.close()