diff --git a/.claude/docs/REPO_WALKTHROUGH.md b/.claude/docs/REPO_WALKTHROUGH.md index d08684da0e..cc87eb77c7 100644 --- a/.claude/docs/REPO_WALKTHROUGH.md +++ b/.claude/docs/REPO_WALKTHROUGH.md @@ -78,6 +78,7 @@ src/ │ │ ├── report.py # Versioned validation report models │ │ ├── policy.py # Severity-policy loading and application │ │ ├── runner.py # Local validation orchestration +│ │ ├── runtime/ # Bounded plans, launch requests and raw evidence │ │ ├── parsers/ # Manifest parser registry │ │ ├── providers/ # Validation provider contracts │ │ ├── graders/ # Grader registry and protocols diff --git a/docs/plans/rfc008-level2-implementation.md b/docs/plans/rfc008-level2-implementation.md new file mode 100644 index 0000000000..282a34815b --- /dev/null +++ b/docs/plans/rfc008-level2-implementation.md @@ -0,0 +1,431 @@ +# RFC 008 Level 2 implementation plan + +Planning baseline: 2026-09-16, `huggingface/OpenEnv` main at +`e3eb3fa5bc11ff0a019c30c7a7b1b7039eb8b615`. + +Delivery tracker: [Level 2 umbrella #1177](https://github.com/huggingface/OpenEnv/issues/1177). +This document describes the full roadmap. The first delivery implements PRs 1–3; +subsequent slices remain planned. See the umbrella and the shared case catalog for +current implementation coverage and validation evidence. + +## Outcome and scope + +Deliver a reproducible Docker-local implementation of all **15 existing local +`runtime.*` checks** in RFC 008, plus an explicit startup outcome. A reviewer should +be able to build the exact revision, run a known-good subject, introduce one known +defect, observe the expected finding, and inspect the same evidence bundle that CI +produces. + +The reference environment is Linux, x86-64, Docker and a qualified cgroup/storage +configuration. Fast tests work without Docker. Docker Desktop and other platforms +can run the same harness, but their capability reports determine which isolation +and resource claims they can establish. They do not replace the reference evidence. + +First support the served OpenEnv execution binding and CPU fixtures. Preserve a +format-neutral grader interface so subsequent Harbor/PostTrain execution bindings +and an HF Sandbox provider reuse the checks. Those implementations, GPU provider +qualification, Level 3 oracle scoring/floor checks, remaining Level 1 graders, +operator certification and publish gating are separate work. + +The implementation must handle LLM-judged declarations honestly: implement the +specified bounded variance path using a controlled test judge; if a real provider +or judge cannot run, report its missing capability or prerequisite. Live model +inference is not a prerequisite for the deterministic reference suite. + +## What exists and what must change + +- `validation/runner.py` currently ignores `skip_build`, registers only + `StaticManifestGrader`, and always reports level 1. +- `Subject` carries a manifest and running subject, but no replay plan or collected + runtime evidence. Registry selection orders by name, not dependency graph. +- `ValidationProvider.start()` cannot receive resource limits; `RunningSubject` + only exposes `base_url`, `exec` and `stop`. Neither interface describes effective + policy, agent identity, metrics, fresh launches or timeout cleanup. +- The core `LocalDockerProvider` does not enforce the needed launch settings. + Passing additional keyword arguments to it does not configure Docker security. +- No normalized runtime action fixture, session-bound rubric/attribution telemetry, + or environment-emitted trajectory contract exists. +- The existing static fixtures contain declarations, not runnable subjects. The + existing echo image uses a mutable base tag. There is no committed root + `uv.lock`; the ordinary test workflow excludes Docker/network/integration tests. + +Use the validation-specific provider protocol rather than widening the core +provider ABC. Reuse core transport and serialization contracts. Keep protocol and +Docker details out of graders. + +## Architecture + +```text +pure parser + data-only probe plan + | + validated RuntimePlan + | + runner: dependency plan -> build -> provider preflight -> start/readiness + | + session collector: reset / actions / state / discovery / telemetry + | + immutable RuntimeEvidence + independently inspected provider evidence + | + graders: manifest + evidence -> CheckResult + | + policy -> existing report + provenance/artifact bundle -> teardown verification +``` + +One collector owns the normal episode sequence. Graders consume immutable evidence +instead of independently stepping a shared environment in an order-dependent way. +Fresh-session and fresh-container replay are explicit experiments. Invasive +containment and resource probes get separate, disposable subjects. + +`RuntimePlan` is not a second capability manifest. Rewards, resources, network +policy, execution binding, observable capabilities, applicability and type selection +remain authoritative in the normalized manifest. The plan supplies bounded actions, +reset inputs and replay schedules only; it cannot suppress a grader or override +policy. Missing actions produce an unmet prerequisite, never silent deselection. +Only collector dispatch uses the execution binding; graders receive manifest +declarations and measured evidence. Include the plan hash in the run bundle so a +replay is tied to those exact inputs. + +Package code runs only inside the subject. Pure parsers do not import it; neither +does the host validator. A trusted probe worker may use the existing OpenEnv +protocol from an isolated helper when a host URL is unavailable. This is an +internal validation transport, not a new public environment API. + +## PR 1: settle the missing contracts + +Amend RFC 008 narrowly before implementing new abstractions or core wire fields. +Do not create a parallel validation architecture. The following are recommended +decisions for that review, with schemas and small fixtures making them concrete. + +| Decision | Proposed contract | +|---|---| +| Runtime input | A versioned, data-only `validation/runtime.json` plan: bounded reset arguments, action sequence and seed/replay schedule. Normalize execution binding, agent boundary and observable capabilities into the manifest, not this sidecar. A trusted parser normalizes inputs into `RuntimePlan`; graders never inspect package signature. No arbitrary host callbacks, guessed legal actions, or reuse of privileged L3 oracle actions. | +| Schema compatibility | Add the small execution/probe declaration in manifest schema v2 and version the report embedding it as v2. Preserve v1 schemas/models/fixtures and static-report compatibility; provide explicit parsing/migration tests. Attach `RuntimePlan` and collected evidence internally to `Subject`; keep expanded provenance and coverage in a separate versioned sidecar. Never silently change schema 1 or put capability selection in the sidecar to avoid versioning. | +| Startup outcome | Add `runtime.startup` in severity policy v2. Failed subject build/start/readiness is FAIL with the failed phase; a validator/provider defect is ERROR. Missing prerequisites are named SKIPs. One successful build does not pass `static.reproducible_build`. | +| Policy versions | Static validation retains v1 support. Runtime requires the policy version defining its new lifecycle outcome; reject an incompatible explicit policy before execution. Make v2 the documented runtime default and pin it in all fixtures. Never emit an unknown ID into v1. | +| Launch and evidence | Validation-only typed launch specification: immutable image identity, manifest resource/network settings, explicit environment variables, effective execution identity, writable roots, independent deadlines. Provider returns inspected settings, verified capabilities, bounded metrics/logs and cleanup evidence. | +| Seed control | Report framework-observed acceptance of the seed/reset contract separately from empirical replay determinism. A successful reset alone is insufficient because current server filtering can silently drop kwargs. A deterministic environment is allowed to produce the same output for different seeds. | +| Reward and comparison | Check raw types before coercion. Reject bool/non-finite/out-of-range rewards; explicitly specify when reset reward may be null. Compare full episode content with only policy-owned volatile metadata exclusions. Authors cannot exclude reward, done, tool output or state changes. | +| Session telemetry | Orchestrator-only, typed telemetry for seed handling, named rubric/configuration, child reward attribution and an emitted trajectory reference. Use the existing replay connection with an opt-in, per-run/session scoped authorization capability; do not open a second environment connection. Reject unauthorized and cross-session reads, and expose none of it as agent MCP tools. A validator transcript is not evidence that the subject emitted a record. | +| Applicability | Explicit predicates for capabilities/collections; an empty declared tool set must still be checked. Missing subject features, missing provider support and implementation-unavailable checks are distinct inventory reasons. | +| Agent boundary | Declare whether agent access is API-only or includes a process UID/filesystem. Oracle and host containment must test that boundary. Privileged exec identity is not an agent-access test. Missing measurable identity produces a named incomplete outcome. | +| Resources | CPU means an allocation ceiling; memory includes explicit swap behavior. Define `disk_mb` as the aggregate permitted writable subject storage, excluding immutable image layers, and name writable roots. Episode timeout includes descendants and is externally enforced. PID/log/build caps are additional supervisor budgets. | +| Network | Keep `public` egress as the existing default. Define any host/private-address restrictions explicitly. For allowlists, settle DNS, IPv4/IPv6, exact/wildcard hostnames, CIDRs and protocol/port semantics; do not silently weaken a hostname rule into an IP-only guarantee. | + +The allowlist recommendation is to specify a DNS-derived destination-address +policy, with controlled resolution and recorded address sets, explicitly disclosing +shared-IP limitations. If the RFC instead requires application-host identity, +PR 9 must use protocol-aware enforcement and reject unsupported protocols. This +decision must be made in PR 1; a permissive fallback is not an implementation. + +For judged rewards, specify a policy-owned repeated-sample procedure, its units, +sample count and total budget. A proposed default is 20 identical-input replays +and empirical reward variance, compared with the declared bound. This is a bounded +runtime check, not a confidence claim or Level 4 statistical evaluation. Fewer +completed samples cannot silently count as a passing full check. + +## Stacked delivery: ten reviewable PRs + +Use `ben/rfc008-l2-01-contracts` through `ben/rfc008-l2-10-episode-oracle` +as proposed branch names. Each PR targets its immediate predecessor. Merge +bottom-up, refresh the next dependency edge, and re-run checks on the resulting +exact head. Keep at most two or three unmerged implementation slices in active +review rather than opening the entire stack immediately. + +| PR | Incremental change | Independent acceptance gate | +|---|---|---| +| 1. Contracts and shared assets | RFC amendments above; `RuntimePlan`/launch/evidence types; policy v2; scenario catalog; small fake provider and fixture skeleton; schema compatibility. | Pure contract tests and the acceptance inventory run without Docker. Expected findings are written from the RFC, not generated by graders. | +| 2. Docker supervisor and reproducible build | Runnable shared fixture; pinned test project/wheelhouse; build snapshot; build/start/inspect/exec/stop; safe launch defaults, deadlines and idempotent cleanup; initial Linux CI lane. | Actual exact-head wheel and fixture image start, answer health and protocol probes, execute bounded commands and leave no run-owned resources on success, failure or cancellation. | +| 3. Runner and basic checks | Dependency scheduler, strict session collector, runtime CLI path, startup outcome; reward, observation and state graders. | Public CLI returns levels 1 and 2; good fixture passes these checks; malformed-wire, invalid-reward, wrong-state and failed-start cases produce expected IDs. No Docker calls under `--skip-build`. | +| 4. Session telemetry | Small reviewed core protocol additions for seed acceptance, rubric/configuration, attribution and subject-emitted record metadata. | Real session tests preserve the replayed instance; unauthorized and cross-session reads fail; records are independent from the collector transcript; production MCP cannot invoke telemetry/reset controls. Existing clients remain compatible. | +| 5. Repeatability and trajectories | `seed_control`, `episode_determinism`, `trajectory_record`; fixed-input fresh-session/fresh-container experiments; bounded variance mode. | Deterministic and seed-invariant cases pass; ignored-seed stochastic fixture, trace mismatch and nondeterminism fail; first divergent operation/path is reported. | +| 6. Discovery and rubric checks | `tool_declaration_accuracy`, `task_declaration_accuracy`, `rubric_introspectable`, `reward_attribution`. | Empty/extra/missing tools, discovery failure, wrong task counts, bounded task previews, missing rubric and inconsistent attribution are distinguishable. | +| 7. Host containment and resources | `host_containment`, `resource_bounds`; externally inspected settings, tested agent identity, bounded canaries and quota/deadline evidence. | Benign host sentinel inaccessible; missing/misconfigured limits detected; bounded stress contained; timeout kills descendants; unsupported quota backend cannot pass disk enforcement. | +| 8. Public/no-network modes | `network_policy` for these modes; trusted control transport; controlled reachable/blocked sinks. | Runtime communication works with network isolation; public positive connectivity succeeds; no-network has no egress; a control helper cannot bridge subject egress. | +| 9. Allowlist mode | Dedicated enforcement backend and qualification matrix for reviewed hostname/CIDR/DNS/IPv6 semantics. | Allowed destinations succeed; prohibited names, direct-IP and alternate-DNS/bypass cases fail as specified. Unsupported policy is refused before launch. Separate network/security review. | +| 10. Episode and oracle boundaries | `episode_isolation`, `oracle_containment`; agent-access probes, same-server A/reset/B experiments; final documentation and reference evidence. | Sticky memory/file/rubric/background-state cases and exposed harmless oracle canary are detected. Full reference inventory has all 15 original IDs plus startup, with no missing expected checks. | + +Security settings belong in PR 2 even though their dedicated graders arrive later. +The provider initially advertises only modes it can actually enforce. No-network +and allowlist requests must not be launched under public networking while their +implementation is pending. + +Every PR adds its fault cases, useful tests and reproduction instructions with the +behavior it implements. The required CI inventory expands immediately. During the +stack, expose unimplemented requested checks as explicit incomplete inventory and +named SKIPs rather than presenting a partial runtime run as complete. Preserve RFC +exit semantics: WARN remains exit 0. The test harness independently enforces the +expected check inventory. + +Review routing: Sergio as proposed architecture/core-contract lead; a named Docker +and networking reviewer for PRs 2 and 7–9; an environment/rubric maintainer for PRs +4–6 and 10. HF Sandbox-specific review belongs to the subsequent provider stack. +These are suggested review roles, not assigned or requested reviews. + +## One shared runtime fixture and scenario catalog + +Use a tiny CPU-only served fixture with a seeded counter/task selector, two tools, +two fixed task splits, a non-LLM rubric, bounded rewards, finite termination and a +harmless withheld oracle canary. Add a controlled stochastic judge mode to exercise +variance logic. All runtime actions are public, bounded JSON data. + +Fault modes live only in test assets. Each mode introduces one defect into a fresh +subject; no production bypass flags. Reuse the same image and collector across +unit, protocol and Docker tests. Preserve the existing manifest-only static +fixtures, and use `echo_env` as an additional compatibility smoke rather than the +reproducibility anchor. + +Proposed shared layout: + +```text +tests/fixtures/validation/runtime/ + README.md + served_probe/ # runnable package, Dockerfile, public runtime plan + cases.json # check -> case -> expected status/evidence predicate + expected/ # small semantic expectations, never full noisy logs +tests/test_validation/ + support/ # fake provider, evidence builders, case loader + runtime/ # grader and orchestration tests + integration/ # real protocol, Docker, installed-wheel tests +tests/validation_runtime/ + pyproject.toml + uv.lock + .python-version + toolchain.json # uv/Python/image/platform/build dependency pins +scripts/validation/ + reproduce.py # shared entry point for developers and CI + verify_artifacts.py +.github/workflows/validation-runtime.yml +``` + +The case catalog also declares applicability, required provider features, the PR +that implements the case and whether its test needs Docker. A missing case or an +unexpected skip fails the relevant CI suite; expected subject validation failures +are successful tests only when their exact findings and evidence match. + +## Check-by-check acceptance matrix + +| Existing Level 2 ID | Measurement and essential negative case | +|---|---| +| `runtime.reward_well_formed` | Raw reward type/finite/range checks under the explicit null rule; bool, string, NaN/Inf and out-of-range faults. | +| `runtime.observation_schema` | Strict envelope checks, then reconstruct full observation including reward/done for advertised-schema validation; missing/wrong-typed fields fail. | +| `runtime.state_contract` | Minimum state contract: requested episode identity, reset count and coherent step increments within one session; wrong episode or sticky counter fails. | +| `runtime.trajectory_record` | Fetch the subject-emitted record and compare with independently captured actions/results; missing, truncated or mismatched record fails. | +| `runtime.reward_attribution` | Match child scores to declared rubric structure and aggregation semantics; absent/misidentified/inconsistent attribution fails. Do not assume every rubric is an unweighted sum. | +| `runtime.rubric_introspectable` | Session-correct named tree and serializable configuration when claimed; declared-but-uninspectable rubric fails its warning-level check. | +| `runtime.tool_declaration_accuracy` | Compare normalized declared/discovered sets, including empty declaration; extra/missing tools fail, discovery error cannot become an empty success. | +| `runtime.task_declaration_accuracy` | Compare declared split counts with `num_tasks`; use bounded item sampling, never `len(list_tasks)` as the count; missing split/wrong count fails. | +| `runtime.seed_control` | Framework-observed seed contract plus scheduled resets; rejected/silently discarded seed in a seed-dependent fixture is detected. Different seeds need not change a deterministic subject. | +| `runtime.episode_determinism` | Same seed/actions across fresh sessions and containers; compare complete traces with tightly defined volatile fields; injected observation/reward drift fails. Judged path uses its separately pinned variance procedure. | +| `runtime.network_policy` | Effective inspected rules plus positive/negative connectivity from the subject namespace; controlled endpoints establish reachability before denial is interpreted. | +| `runtime.host_containment` | Effective mounts/namespaces/security settings plus a harmless agent-identity sentinel probe; use synthetic unsafe inspection records instead of giving a test real host privilege. | +| `runtime.resource_bounds` | Effective CPU/memory/swap/PID/writable-storage configuration and external time budget; bounded qualification probes detect missing enforcement and surviving children. | +| `runtime.episode_isolation` | In the same server, episode A writes a declared marker, reset starts B, B cannot observe A's state/file/rubric/task effects; fresh containers alone do not establish this property. | +| `runtime.oracle_containment` | Oracle artifact is absent/inaccessible through the declared agent boundary at serve time; a harmless readable canary fails. General solution leakage in observations stays in Level 3. | + +For each applicable check test: good input, one subject defect, missing prerequisite, +and malformed evidence/provider error where meaningful. A declared supported +capability that malfunctions is a failure/error, not an opportunistic SKIP. + +## Protocol and isolation details that tests must preserve + +Use the existing WebSocket session for reset/step/state. Do not compose an episode +out of HTTP reset/step calls, which can instantiate separate environments. Keep raw +responses before generic-client/Pydantic defaults can hide missing fields. The +wire envelope separates reward/done from observation; validate both representations +correctly. The current schema endpoint exposes base State, so check the minimum +state behavior directly. + +Do not combine MCP tool-return values with Gym rewards. Discovery failures must +remain failures, even if the ordinary convenience client returns an empty list. +Task discovery may legally expose a bounded preview. + +Each current WebSocket connection creates its own environment. Telemetry therefore +extends the already-open replay session, guarded by the reviewed opt-in authorization +mechanism. An independent telemetry connection must not create and inspect a fresh +instance while claiming evidence about the original. Existing clients need not +request or receive these additional messages; agents retain only their MCP boundary. + +The provider owns readiness, bounded exec/logs, fresh launches, inspected settings +and idempotent teardown. It never mounts a Docker socket or credentials into the +subject. Use a non-privileged launch, no host namespaces or host-directory mounts, +dropped capabilities, `no-new-privileges`, read-only image and explicitly bounded +writable areas. Exposed control ports bind only to loopback with allocated ports. +Unsupported storage/identity restrictions produce explicit capability evidence; +there is no automatic retry with weaker isolation. + +`--network none` leaves only loopback, so publishing port 8000 does not solve +orchestrator access. Implement an isolated trusted probe worker sharing only the +subject network namespace and using the ordinary loopback OpenEnv protocol; the +host controls it over bounded runtime exec. It has its own image/filesystem, no +Docker socket, no externally routed interface and no endpoint through which the +subject can request arbitrary outbound traffic. Review this arrangement in PR 1, +implement it in PR 8, and test that it cannot bridge egress. + +An HTTP proxy setting alone does not implement allowlisting. The subject must lack +network-administration authority; trusted default-deny enforcement and DNS policy +remain outside its control. Record provider qualification separately from each +subject's sampled measurements. A blocked connection to an offline endpoint is +not evidence of isolation. + +Memory settings must include swap behavior. A tmpfs limit consumes memory and is +not by itself a general filesystem quota; all writable paths must fit the reviewed +aggregate budget, or disk enforcement remains unsupported. Do not infer effective +limits solely from Docker CLI arguments. + +Independent deadlines cover dependency acquisition, build, start, readiness, +request, exec, episode and whole run. Killing a Docker exec client does not prove +its child stopped. Terminate the process scope or destroy the subject, then verify +cleanup. Label every resource with a run ID and provide a scoped stale-run reaper; +never use a global Docker prune. + +## Reproducible build and run recipe + +Pin a dedicated validation test project instead of changing the whole repository's +dependency policy. Commit its lock, Python patch version, uv version, build +dependencies, image digest and reference architecture. Build a wheel from the exact +PR head and install that wheel both in the fixture and in the installed-package +test. Do not fetch `main`, use `:latest`, or depend on an existing editable install. + +Acquire dependencies as an explicit stage; retain a hash-checked wheelhouse. +Fixture image installation is network-free after that acquisition. Cache keys +include source wheel, fixture, lock and base-image hashes. A clean cache must work. +Arbitrary subject image builds are separately bounded and their build-network +policy is explicit; runtime network settings do not retroactively isolate builds. + +Stage a stable source/build context before hashing and building. Capture generated +wheel/helper inputs as additional digests so the reported source revision actually +matches the tested artifact. Local dirty runs record a patch/content digest; +review evidence uses a clean exact commit. Symlinks or referenced probe paths must +not escape the subject package. Keep generated outputs outside build inputs. + +These are **proposed commands**, to be implemented by the stack: + +```bash +uv sync --project tests/validation_runtime --frozen + +# Pure logic and real local protocol tests; no Docker required. +uv run --project tests/validation_runtime --frozen \ + python scripts/validation/reproduce.py --suite fast +uv run --project tests/validation_runtime --frozen \ + python scripts/validation/reproduce.py --suite protocol + +# Build the exact-head wheel and shared image, exercise all required Docker cases, +# repeat from clean subjects, verify check inventory and cleanup, write artifacts. +uv run --project tests/validation_runtime --frozen \ + python scripts/validation/reproduce.py --suite docker \ + --require-complete --output outputs/validation-runtime +``` + +`--require-complete` is a test-harness flag, not a change to public CLI exit codes. +During development it requires the inventory committed for that PR; the final +reference inventory includes every original L2 ID and the startup outcome. + +The Docker suite must invoke the public command on staged packages, with explicit +local selection, runtime level and pinned policy; it must not validate only private +Python helpers. `--local` is part of the RFC but not the current CLI and lands in +PR 3. Runtime tests must not infer remote execution from an ambient HF token. + +Each case writes its own standard report. The suite writes: + +```text +outputs/validation-runtime// + run-manifest.json + coverage.json + cases.jsonl + junit.xml + cases//report.json + trajectories/ # collector traces and separate emitted records + logs/ # bounded, redacted + cleanup.json + SHA256SUMS +``` + +Record head/base SHA, dirty-state digest, exact argv, source/fixture/runtime-plan/ +policy/lock/wheel hashes, base digest and resolved image ID, platform/kernel/ +cgroup/storage/Docker/Python/uv versions, seeds, replay identities, effective +provider features and all artifact hashes. The final checksum file covers the +other artifacts, not itself. Redact secrets before retention and enforce size +limits, including on malformed output. + +Reproducibility means fixed inputs and equivalent findings/traces under the +qualified platform. Do not promise byte-identical images or duration/log equality. +Preserve raw evidence and compare a narrowly defined normalized view; do not erase +meaningful nondeterminism to make golden files pass. + +## Testing and CI + +1. **Fast, every PR:** scheduler DAG/cycles/missing dependencies, capability and + applicability decisions, skip-build, unexpected grader IDs, exceptions, + cancellation/timeout/cleanup, policy and schema compatibility. Fakes test + orchestration behavior; they do not establish Docker enforcement. +2. **Real protocol, relevant PRs:** a real test server and supported production + transport, raw serialization faults, session identity, discovery and telemetry. + Prefer protocol faults over monkeypatches that skip the actual wire path. +3. **Required Linux Docker, starting PR 2:** exact-head wheel/image, all implemented + positive and single-fault cases, enforced limits/network modes, two clean repeat + runs, no resource leaks. Missing Docker or required platform capability fails + preflight; it must not silently skip the merge gate. +4. **Installed-package compatibility:** run outside the checkout with `PYTHONPATH` + unset; verify schemas, policy files and trusted helpers are included in the + wheel. Keep the existing suite and add an `echo_env` smoke without claiming it + covers every optional capability. + +Use ephemeral Linux CI workers, read-only repository permissions, checkout without +persisted credentials, ordinary PR events, and no publish tokens. Do not run fork +code through a privileged `pull_request_target` path or approve fork workflows +automatically. Keep the required workflow visible on every PR; an internal path +filter can decide whether to run expensive cases. Relevant core protocol/provider +changes must trigger them too. + +Always retain the evidence bundle on failures. Explicitly distinguish expected +subject FAIL fixtures from harness/test failures. CI verifies the expected ID set, +statuses, applicability and evidence; CLI exit zero alone is insufficient because +WARN also exits zero. Test CI for this feature is within this plan; enforcing +validation on user publication remains outside it. + +Initial performance targets, to measure rather than claim as achieved: fast suite +under one minute, protocol suite under two minutes, warm-cache reference Docker +suite under ten minutes. Give the Docker job a separate hard deadline and preserve +partial artifacts on timeout. Cold dependency acquisition/build time is recorded +separately; do not hide it in a warm-cache timing claim. + +## Review and completion gates + +Each PR has one short purpose, incremental diff against its parent, focused files +and an acceptance case in the shared catalog. Keep one stack checklist mapping +check IDs to PRs and evidence. Put detailed run recipes and evidence in the shared +assets/CI artifacts; follow the repository's short PR-description convention. +Avoid copied fixture directories and generated multi-thousand-line snapshots. + +Review three waves: (1) contracts, provider and first real CLI path; (2) telemetry, +replay and discovery; (3) isolation, network enforcement and complete coverage. +Security review remains required even when the fixture suite passes. + +Completion requires all of the following: + +- All 15 existing L2 checks implemented, plus the approved startup outcome. +- A reference fixture declaring all optional features exercises every check; each + check has an independently specified positive and targeted negative case. +- Required reference cases have no unexplained SKIP/ERROR, omitted check or missing + evidence. Unsupported platforms/features report their actual limitations. +- Build/readiness failures, stuck requests, cancellation and process descendants + leave correct reports and no run-owned resources. +- Two clean runs from the same commit and pins produce equivalent findings and + deterministic traces, with recorded differences limited to approved metadata. +- Installed-wheel tests and the existing validation tests pass; schemas and package + data are verified, not just editable-source imports. +- Documentation distinguishes tested runtime behavior from Level 3 semantics, + statistical trainability and broad security certification. + +## Sources + +- [RFC 008 at the planning revision](https://github.com/huggingface/OpenEnv/blob/e3eb3fa5bc11ff0a019c30c7a7b1b7039eb8b615/rfcs/008-environment-auto-validation.md) +- [Validation provider contract](https://github.com/huggingface/OpenEnv/blob/e3eb3fa5bc11ff0a019c30c7a7b1b7039eb8b615/src/openenv/validation/providers/__init__.py) +- [Grader/Subject contract](https://github.com/huggingface/OpenEnv/blob/e3eb3fa5bc11ff0a019c30c7a7b1b7039eb8b615/src/openenv/validation/graders/__init__.py) +- [Core Docker provider](https://github.com/huggingface/OpenEnv/blob/e3eb3fa5bc11ff0a019c30c7a7b1b7039eb8b615/src/openenv/core/containers/runtime/providers.py) +- [Server transport and schemas](https://github.com/huggingface/OpenEnv/blob/e3eb3fa5bc11ff0a019c30c7a7b1b7039eb8b615/src/openenv/core/env_server/http_server.py) +- [Task API and bounded previews](https://github.com/huggingface/OpenEnv/blob/e3eb3fa5bc11ff0a019c30c7a7b1b7039eb8b615/docs/source/guides/task-api.md) +- [Docker network-none semantics](https://docs.docker.com/engine/network/drivers/none/) +- [Docker resource constraints](https://docs.docker.com/engine/containers/resource_constraints/) +- [Docker tmpfs behavior](https://docs.docker.com/engine/storage/tmpfs/) +- [Docker build and pinning guidance](https://docs.docker.com/build/building/best-practices/) diff --git a/rfcs/008-environment-auto-validation.md b/rfcs/008-environment-auto-validation.md index b70bf3fd13..63405af7e6 100644 --- a/rfcs/008-environment-auto-validation.md +++ b/rfcs/008-environment-auto-validation.md @@ -459,6 +459,163 @@ Stacked PRs, each vertically testable: The PostTrain `task.md` parser and the LLM-judged grader path (variance-mode determinism, rubric deepening) are fully specified with contracts and fixtures in PR2 and implemented separately. +## Level 2 execution amendment + +This amendment specifies the missing execution contracts for the Docker-local +runtime implementation. It preserves the existing parser → normalized manifest → +grader → policy → report architecture. The first delivery wave consists of three +stacked slices: versioned contracts/shared assets; reproducible Docker supervision; +and a runnable CLI with startup, reward, observation and state checks. Later slices +add the remaining runtime checks. A partial implementation must expose the missing +requested checks as named SKIPs, never imply full Level 2 coverage from a PASS on +the implemented subset. Existing WARN exit semantics remain unchanged. + +### Versioned declarations and public probe inputs + +`validation.execution` opts a package into normalized manifest schema **2**: + +```yaml +validation: + execution: + kind: openenv_ws + probe_path: validation/runtime.json + dockerfile: Dockerfile + context: . + agent_boundary: api +``` + +The other manifest sections remain authoritative for capabilities, resources, +network policy, reward bounds and grader applicability. The initial execution +binding is `openenv_ws` with API-only agent access. Process/filesystem agent +identities require a later explicit declaration; privileged provider exec is not +evidence of agent access. Paths are package-relative and portable. Readers reject +parent traversal, absolute paths and resolved symlink escapes. + +Packages without `validation.execution` continue to produce the unchanged schema-1 +manifest and static report. `NormalizedManifestV2` and `ValidationReportV2` have +separate committed JSON schemas. Report 2 accepts manifest 1, manifest 2, or null +after a parse failure so a runtime request can report missing prerequisites without +dropping diagnostics. No field is silently added to schema 1. + +The public sidecar is data, not imported Python or a second capability manifest: + +```json +{ + "plan_schema_version": "1", + "reset": {"episode_id": "validation-probe", "seed": 42, "options": {}}, + "actions": [{"increment": 1}, {"increment": 1}] +} +``` + +The trusted loader bounds the file at 65,536 bytes, JSON nesting at 32 levels, +node count at 10,000 and actions at 1–100. It rejects duplicate keys, non-finite +numbers, non-JSON objects and reset options overriding `seed` or `episode_id`. +The seed is an explicit unsigned 32-bit integer. Missing or invalid probe inputs +are reported as an unmet prerequisite or an input failure; they never deselect +graders. The plan's content digest belongs in the reproduction bundle. Neither +privileged oracle inputs nor host callbacks may be substituted for public actions. + +One collector owns reset, ordered actions until termination, and state reads on +**one** WebSocket session. It records immutable raw request/response strings before +client defaults or Pydantic coercion can hide malformed responses. Graders consume +that evidence and cannot mutate the measured episode. The advertised observation +schema is recorded alongside the transcript; reconstruct observation plus the +separate reward/done envelope before validating it. Reset reward may be null; every +step reward must be a finite JSON number, excluding booleans, within the declared +range. State must retain the requested episode identity, reset its step count to +zero and advance it coherently for successful steps. + +### Startup, policy and provider supervision + +Policy **v2** retains every v1 entry and bound and adds `runtime.startup` at level 2, +local lane, failure severity. Runtime defaults to v2 and rejects an explicitly +incompatible policy before executing a subject. Static v1 remains supported. +Build, start or readiness failure caused by the subject yields startup FAIL with +the failed phase; a validator/provider defect yields ERROR. A missing prerequisite +or unsupported enforcement capability yields a named SKIP. One successful build +does not establish `static.reproducible_build`. + +The validation-specific `LaunchSpec` names an immutable image ID/digest, requested +resources and network policy, an explicit run identity, startup deadline and +explicit environment variables. It inherits no host environment or credentials. +The provider owns build, readiness, bounded exec/logs, inspected settings and +idempotent cleanup on success, failure, timeout and cancellation. Cleanup evidence +must establish that no run-owned subject remains. Core provider ABCs are unchanged. + +Launches use no privilege, host namespaces, host-directory mounts, Docker socket +or forwarded credentials. They drop Linux capabilities, enable no-new-privileges, +use a read-only image and explicitly bounded writable roots, and expose only a +loopback control port. CPU is an allocation ceiling; memory includes an explicit +swap setting. `disk_mb` means aggregate writable subject storage, excluding image +layers. The initial provider splits this allowance between bounded writable `/tmp` +and `/dev/shm`; another writable root requires its own accounted budget. Episode +deadlines are externally enforced and +cover descendants; PID, log and build limits are additional supervisor budgets. +Unsupported enforcement is disclosed and must never trigger a weaker retry. + +The initial provider may support only `public` networking and CPU subjects. +`no-network`, `allowlist` and GPU requests must then be refused before launch with +their missing capability named. Public networking permits egress; it does not +establish a host/private-address deny policy. A later no-network provider uses a +trusted helper sharing only the subject network namespace, with separate image and +filesystem, no external interface, Docker socket or outbound-proxy API. Merely +publishing a port on a no-network container is insufficient for control access. + +Allowlist semantics are **destination-address enforcement**: exact DNS hostnames, +`*.example.com` subdomain patterns (not the bare apex), and IPv4/IPv6 CIDRs resolve +through a controlled resolver; matching DNS requests add their recorded addresses +to the run-owned allow set. Direct addresses are allowed only if included in a CIDR +or the recorded set. Entries allow all ports/protocols at those destinations; they +do not authenticate TLS/HTTP host identity and therefore disclose shared-IP +limitations. Alternate DNS, unmatched addresses and unsupported address families +must fail closed. This requires a separate reviewed enforcement implementation; +parsing these declarations does not claim they are enforced. + +### Later runtime evidence contracts + +Seed acceptance and empirical determinism are separate findings. A reset that +silently drops its seed does not establish seed control, while a deterministic +environment may legitimately behave identically under different seeds. Replay +comparison includes observation, reward, done, tool output and state; only +policy-owned volatile metadata can be excluded. Authors cannot exclude fields. +For `llm_judged`, the bounded variance path uses 20 completed identical-input fresh +replays and population reward variance in reward-squared units, compared to the +declared bound. The total run budget bounds all samples; fewer than 20 is incomplete. +This procedure is a runtime check, not a statistical confidence claim. + +Session telemetry for seed handling, named rubric/configuration, child attribution +and subject-emitted record references is orchestrator-only. A future protocol +slice must authorize access with an opt-in, random per-run/per-session capability +attached to the **same** replay connection, reject unauthorized/cross-session +reads and never expose telemetry as agent MCP tools. A second WebSocket creates +another environment and cannot supply evidence for the measured instance. +A validator transcript alone cannot pass subject-emitted trajectory recording. +These authorization requirements do not add new public wire messages in this slice. + +Applicability predicates must distinguish empty declared sets from absent +capabilities. Missing subject features, missing provider support and checks whose +implementation has not shipped are distinct outcomes. Independent graders must +not step a shared environment; invasive containment and resource probes use +disposable subjects. Episode-isolation checks require same-server reset evidence, +not just a new container, and oracle access checks use the declared agent boundary. + +### Shared reproducibility and review assets + +`tests/fixtures/validation/runtime/cases.json` is the versioned acceptance catalog. +Every runtime policy ID has positive/negative expectations, applicability, required +provider features and its implementation slice. Future cases are labelled planned; +they do not count as measured coverage. A small shared served subject supplies +deterministic public actions and test-only faults. Preserve the existing static +fixtures and run their schema-1 round trips alongside schema-2 tests. + +The Docker/reference lane and local reproduction command use the same locked test +project, digest-pinned base, exact-revision wheel and case catalog. An evidence +bundle records wheel/source/plan/image identities, platform details, reports, +transcripts and cleanup findings. CI separately asserts inventory and expected +findings: a WARN exit 0 does not prove completeness. Product publish gating and +operator certification remain outside this feature; the test workflow only proves +the implementation's own acceptance cases. + ## Explicitly out of scope No hub/runner/queue/coordinator; no submission or auth APIs; no statistical-level implementations diff --git a/scripts/sync_validation_schemas.py b/scripts/sync_validation_schemas.py index fb124e8d41..6281a104ce 100644 --- a/scripts/sync_validation_schemas.py +++ b/scripts/sync_validation_schemas.py @@ -25,12 +25,16 @@ def rendered_schemas(): """Return {filename: rendered JSON text} for every exported schema.""" - from openenv.validation.manifest import NormalizedManifest - from openenv.validation.report import ValidationReport + from openenv.validation.manifest import NormalizedManifest, NormalizedManifestV2 + from openenv.validation.report import ValidationReport, ValidationReportV2 + from openenv.validation.runtime.contracts import RuntimePlan exports = { "manifest.schema.json": NormalizedManifest, "report.schema.json": ValidationReport, + "manifest-v2.schema.json": NormalizedManifestV2, + "report-v2.schema.json": ValidationReportV2, + "runtime-plan.schema.json": RuntimePlan, } return { fname: json.dumps(model.model_json_schema(), indent=2, sort_keys=True) + "\n" diff --git a/src/openenv/validation/manifest.py b/src/openenv/validation/manifest.py index 2e6adbc2b8..de5a143330 100644 --- a/src/openenv/validation/manifest.py +++ b/src/openenv/validation/manifest.py @@ -5,9 +5,11 @@ report provenance only — grader selection never reads it. """ +import math +from pathlib import PurePosixPath from typing import Annotated, Any, Literal -from pydantic import BaseModel, ConfigDict, Field, model_validator +from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator from .types import SignatureKind @@ -304,3 +306,81 @@ def _judge_pin_iff_llm_judged(self) -> "NormalizedManifest": "a judge pin is declared but capabilities.llm_judged is false" ) return self + + +class ExecutionDeclaration(BaseModel): + """ + Author-declared runtime binding, introduced by manifest schema 2. + + All paths are portable package-relative paths. Resolving the probe or build + context must additionally reject symlink escapes before reading source files. + The data-only probe supplies actions; it cannot change declared capabilities. + + Attributes: + kind (`str`): + The implemented transport binding, `"openenv_ws"`. + probe_path (`str`): + JSON file containing the versioned runtime plan. + dockerfile (`str`): + Dockerfile relative to the package root. + context (`str`): + Build context relative to the package root. + agent_boundary (`str`): + The access granted to an agent; currently only API access is supported. + """ + + model_config = ConfigDict(extra="forbid", frozen=True) + + kind: Literal["openenv_ws"] = "openenv_ws" + probe_path: str = "validation/runtime.json" + dockerfile: str = "Dockerfile" + context: str = "." + agent_boundary: Literal["api"] = "api" + + @field_validator("probe_path", "dockerfile", "context") + @classmethod + def _package_relative(cls, value: str) -> str: + path = PurePosixPath(value) + if ( + not value + or "\\" in value + or ":" in value + or "\x00" in value + or path.is_absolute() + or ".." in path.parts + ): + raise ValueError("must be a portable package-relative path without '..'") + return value + + @field_validator("probe_path", "dockerfile") + @classmethod + def _file_path(cls, value: str) -> str: + if PurePosixPath(value) == PurePosixPath("."): + raise ValueError("must name a file") + return value + + +class NormalizedManifestV2(NormalizedManifest): + """ + Version 2 adds the runtime execution declaration without changing schema 1. + + Static packages without `validation.execution` continue to normalize to + [`~openenv.validation.manifest.NormalizedManifest`]. An explicit execution + declaration opts into this model and the corresponding version 2 report. + """ + + manifest_schema_version: Literal["2"] + execution: ExecutionDeclaration + + @model_validator(mode="after") + def _finite_declarations(self) -> "NormalizedManifestV2": + pending = [self.model_dump()] + while pending: + value = pending.pop() + if isinstance(value, float) and not math.isfinite(value): + raise ValueError("schema 2 numeric declarations must be finite") + if isinstance(value, dict): + pending.extend(value.values()) + elif isinstance(value, (list, tuple)): + pending.extend(value) + return self diff --git a/src/openenv/validation/parsers/openenv_yaml.py b/src/openenv/validation/parsers/openenv_yaml.py index 39a27ab3ee..6d82d3cacc 100644 --- a/src/openenv/validation/parsers/openenv_yaml.py +++ b/src/openenv/validation/parsers/openenv_yaml.py @@ -5,7 +5,7 @@ import yaml from pydantic import ValidationError -from ..manifest import ManifestError, NormalizedManifest +from ..manifest import ManifestError, NormalizedManifest, NormalizedManifestV2 from ..types import SignatureKind VALIDATION_BLOCK_REMEDIATION = ( @@ -41,7 +41,7 @@ class OpenEnvYamlParser: signature = SignatureKind.OPENENV_SERVED - def parse(self, package_root: Path) -> NormalizedManifest: + def parse(self, package_root: Path) -> NormalizedManifest | NormalizedManifestV2: """ Parse a served-environment package. @@ -69,8 +69,9 @@ def parse(self, package_root: Path) -> NormalizedManifest: if not isinstance(validation, dict): raise ManifestError(["`validation:` must be a mapping"]) + has_execution = "execution" in validation data: dict = { - "manifest_schema_version": "1", + "manifest_schema_version": "2" if has_execution else "1", "signature": SignatureKind.OPENENV_SERVED, "version": raw.get("version"), "judge": validation.get("judge"), @@ -87,8 +88,12 @@ def parse(self, package_root: Path) -> NormalizedManifest: if value is not None: data[key] = value + if has_execution: + data["execution"] = validation["execution"] + try: - return NormalizedManifest.model_validate(data) + model = NormalizedManifestV2 if has_execution else NormalizedManifest + return model.model_validate(data) except ValidationError as exc: raise ManifestError( _format_validation_error(exc), diff --git a/src/openenv/validation/policies/severity-v2.json b/src/openenv/validation/policies/severity-v2.json new file mode 100644 index 0000000000..5bfa5fc6c2 --- /dev/null +++ b/src/openenv/validation/policies/severity-v2.json @@ -0,0 +1,57 @@ +{ + "policy_version": "v2", + "bounds": { + "max_oracle_tolerance": 0.1, + "min_floor_margin": 0.1, + "max_variance_tolerance": 0.2, + "max_episode_timeout_s": 3600.0 + }, + "entries": [ + {"check_id": "static.manifest", "level": 1, "lane": "local", "severity": "fail"}, + {"check_id": "static.reproducible_build", "level": 1, "lane": "local", "severity": "fail"}, + {"check_id": "static.image_hygiene", "level": 1, "lane": "local", "severity": "warn"}, + {"check_id": "static.layout", "level": 1, "lane": "local", "severity": "warn"}, + {"check_id": "static.sbom", "level": 1, "lane": "local", "severity": "fail"}, + {"check_id": "static.oci_labels", "level": 1, "lane": "local", "severity": "fail"}, + {"check_id": "static.resource_declaration", "level": 1, "lane": "local", "severity": "fail"}, + {"check_id": "static.timeout_ceiling", "level": 1, "lane": "local", "severity": "fail"}, + {"check_id": "static.dependency_pinning", "level": 1, "lane": "local", "severity": "fail"}, + {"check_id": "static.task_distribution_pinning", "level": 1, "lane": "local", "severity": "fail"}, + {"check_id": "runtime.startup", "level": 2, "lane": "local", "severity": "fail"}, + {"check_id": "runtime.reward_well_formed", "level": 2, "lane": "local", "severity": "fail"}, + {"check_id": "runtime.rubric_introspectable", "level": 2, "lane": "local", "severity": "warn"}, + {"check_id": "runtime.observation_schema", "level": 2, "lane": "local", "severity": "fail"}, + {"check_id": "runtime.state_contract", "level": 2, "lane": "local", "severity": "fail"}, + {"check_id": "runtime.trajectory_record", "level": 2, "lane": "local", "severity": "warn"}, + {"check_id": "runtime.reward_attribution", "level": 2, "lane": "local", "severity": "warn"}, + {"check_id": "runtime.tool_declaration_accuracy", "level": 2, "lane": "local", "severity": "fail"}, + {"check_id": "runtime.task_declaration_accuracy", "level": 2, "lane": "local", "severity": "fail"}, + {"check_id": "runtime.seed_control", "level": 2, "lane": "local", "severity": "fail"}, + {"check_id": "runtime.episode_determinism", "level": 2, "lane": "local", "severity": "fail"}, + {"check_id": "runtime.network_policy", "level": 2, "lane": "local", "severity": "fail"}, + {"check_id": "runtime.host_containment", "level": 2, "lane": "local", "severity": "fail"}, + {"check_id": "runtime.resource_bounds", "level": 2, "lane": "local", "severity": "fail"}, + {"check_id": "runtime.episode_isolation", "level": 2, "lane": "local", "severity": "fail"}, + {"check_id": "runtime.oracle_containment", "level": 2, "lane": "local", "severity": "fail"}, + {"check_id": "semantic.oracle_max", "level": 3, "lane": "local", "severity": "fail"}, + {"check_id": "semantic.floor_gap", "level": 3, "lane": "local", "severity": "fail"}, + {"check_id": "semantic.canary_floor", "level": 3, "lane": "local", "severity": "fail"}, + {"check_id": "semantic.no_solution_leakage", "level": 3, "lane": "local", "severity": "fail"}, + {"check_id": "semantic.verifier_determinism", "level": 3, "lane": "local", "severity": "fail"}, + {"check_id": "semantic.verifier_portability", "level": 3, "lane": "local", "severity": "fail"}, + {"check_id": "semantic.resource_envelope", "level": 3, "lane": "local", "severity": "fail"}, + {"check_id": "semantic.replayability", "level": 3, "lane": "local", "severity": "fail"}, + {"check_id": "hub.layer_isolation", "level": 1, "lane": "hub", "severity": "warn"}, + {"check_id": "hub.time_to_first_work", "level": 2, "lane": "hub", "severity": "warn"}, + {"check_id": "hub.cosign_signature", "level": 1, "lane": "hub", "severity": "warn"}, + {"check_id": "hub.cross_host_determinism", "level": 2, "lane": "hub", "severity": "fail"}, + {"check_id": "hub.immutable_versioning", "level": 1, "lane": "hub", "severity": "fail"}, + {"check_id": "statistical.reward_reachability", "level": 4, "lane": "hub", "severity": "fail"}, + {"check_id": "statistical.difficulty_separation", "level": 4, "lane": "hub", "severity": "fail"}, + {"check_id": "statistical.headroom", "level": 4, "lane": "hub", "severity": "warn"}, + {"check_id": "statistical.variance_bound", "level": 4, "lane": "hub", "severity": "fail"}, + {"check_id": "statistical.training_signal", "level": 4, "lane": "hub", "severity": "advisory"}, + {"check_id": "statistical.adversarial_floor", "level": 4, "lane": "hub", "severity": "fail"}, + {"check_id": "statistical.gameability_gap", "level": 4, "lane": "hub", "severity": "warn"} + ] +} diff --git a/src/openenv/validation/report.py b/src/openenv/validation/report.py index 3fc1d0ad19..407229def0 100644 --- a/src/openenv/validation/report.py +++ b/src/openenv/validation/report.py @@ -11,7 +11,7 @@ from pydantic import BaseModel, ConfigDict, Field -from .manifest import NormalizedManifest +from .manifest import NormalizedManifest, NormalizedManifestV2 from .types import CheckStatus, Lane, Level, SignatureKind, Verdict @@ -71,6 +71,19 @@ class ValidationReport(BaseModel): verdict: Verdict +class ValidationReportV2(ValidationReport): + """ + Report version 2 preserves runtime declarations and parse-failure reporting. + + A runtime request may target a version 1 package without an execution binding, + or fail before producing any manifest. Both remain representable so the report + can explain the unmet prerequisite or manifest error. + """ + + report_schema_version: Literal["2"] + manifest: NormalizedManifestV2 | NormalizedManifest | None + + def write_report(report: ValidationReport, path: Path | None = None) -> str: """ Serialize a validation report to schema-versioned JSON. diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index 187010882e..7f40e9fc91 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -6,11 +6,11 @@ from .graders import GraderRegistry, Subject from .graders.static import StaticManifestGrader -from .manifest import ManifestError, NormalizedManifest +from .manifest import ManifestError, NormalizedManifest, NormalizedManifestV2 from .parsers import ParserRegistry from .parsers.openenv_yaml import OpenEnvYamlParser from .policy import apply_policy, load_policy, SeverityPolicy -from .report import CheckResult, ValidationReport +from .report import CheckResult, ValidationReport, ValidationReportV2 from .signature import detect_signature from .types import CheckStatus, Lane, Level @@ -136,8 +136,15 @@ def run_validation( for grader in graders.select(manifest, max_level) ) - return ValidationReport( - report_schema_version=REPORT_SCHEMA_VERSION, + report_type = ( + ValidationReportV2 + if isinstance(manifest, NormalizedManifestV2) + else ValidationReport + ) + return report_type( + report_schema_version=( + "2" if report_type is ValidationReportV2 else REPORT_SCHEMA_VERSION + ), target=str(target), source_digest=source_digest(target), signature=signature, diff --git a/src/openenv/validation/runtime/__init__.py b/src/openenv/validation/runtime/__init__.py new file mode 100644 index 0000000000..1f8aba662a --- /dev/null +++ b/src/openenv/validation/runtime/__init__.py @@ -0,0 +1 @@ +"""Runtime validation inputs and immutable protocol evidence.""" diff --git a/src/openenv/validation/runtime/contracts.py b/src/openenv/validation/runtime/contracts.py new file mode 100644 index 0000000000..dd32e42ad2 --- /dev/null +++ b/src/openenv/validation/runtime/contracts.py @@ -0,0 +1,238 @@ +"""Versioned runtime inputs and execution evidence, independent of a provider.""" + +import json +import math +import re +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Literal + +from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator + +from ..manifest import ExecutionDeclaration, NetworkPolicy, ResourceDeclaration + +MAX_PLAN_BYTES = 65_536 +MAX_PLAN_ACTIONS = 100 +MAX_JSON_DEPTH = 32 +MAX_JSON_NODES = 10_000 + + +class RuntimePlanError(ValueError): + """A public runtime plan cannot be read safely or violates its data contract.""" + + +def _check_json(value: Any, *, depth: int = 0, budget: list[int] | None = None) -> None: + """Reject executable objects, non-finite numbers and excessive nesting.""" + if budget is None: + budget = [MAX_JSON_NODES] + budget[0] -= 1 + if budget[0] < 0 or depth > MAX_JSON_DEPTH: + raise ValueError("runtime JSON exceeds the node or nesting limit") + if value is None or type(value) in (bool, int): + return + if type(value) is float: + if not math.isfinite(value): + raise ValueError("runtime JSON numbers must be finite") + return + if type(value) is str: + return + if type(value) is dict: + if not all(type(key) is str for key in value): + raise ValueError("runtime JSON object keys must be strings") + children = value.values() + elif type(value) is list: + children = value + else: + raise ValueError("runtime inputs must contain only JSON data") + for child in children: + _check_json(child, depth=depth + 1, budget=budget) + + +class RuntimeReset(BaseModel): + """ + Explicit reset inputs for one measured episode. + + Attributes: + episode_id (`str`): + Requested episode identity; verified by the state contract grader. + seed (`int`): + Requested seed. Acceptance alone does not establish determinism. + options (`dict[str, Any]`, *optional*): + Additional public reset arguments, never replacements for the seed or ID. + """ + + model_config = ConfigDict(extra="forbid", strict=True, frozen=True) + + episode_id: str = Field(min_length=1, max_length=128) + seed: int = Field(ge=0, le=2**32 - 1) + options: dict[str, Any] = Field(default_factory=dict) + + @field_validator("options") + @classmethod + def _public_options(cls, value: dict[str, Any]) -> dict[str, Any]: + if {"episode_id", "seed"} & value.keys(): + raise ValueError("reset options cannot override episode_id or seed") + _check_json(value) + return value + + +class RuntimePlan(BaseModel): + """ + Bounded, data-only runtime inputs. Capability declarations stay in the manifest. + + Attributes: + plan_schema_version (`str`): + The pinned sidecar schema version, `"1"`. + reset ([`~openenv.validation.runtime.contracts.RuntimeReset`]): + Inputs to the first reset on the measured WebSocket session. + actions (`list[dict[str, Any]]`): + Between one and 100 public actions, applied in order until termination. + """ + + model_config = ConfigDict(extra="forbid", strict=True, frozen=True) + + plan_schema_version: Literal["1"] + reset: RuntimeReset + actions: list[dict[str, Any]] = Field(min_length=1, max_length=MAX_PLAN_ACTIONS) + + @field_validator("actions") + @classmethod + def _data_actions(cls, value: list[dict[str, Any]]) -> list[dict[str, Any]]: + _check_json(value) + return value + + @model_validator(mode="after") + def _total_size(self) -> "RuntimePlan": + size = len(self.model_dump_json().encode("utf-8")) + if size > MAX_PLAN_BYTES: + raise ValueError(f"runtime plan exceeds {MAX_PLAN_BYTES} bytes") + return self + + +def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: + result = {} + for key, value in pairs: + if key in result: + raise ValueError(f"duplicate runtime JSON key: {key}") + result[key] = value + return result + + +def load_runtime_plan(root: Path, execution: ExecutionDeclaration) -> RuntimePlan: + """ + Read a bounded runtime plan without importing any subject code. + + Args: + root (`Path`): + Package root; the resolved plan must stay inside this directory. + execution ([`~openenv.validation.manifest.ExecutionDeclaration`]): + Manifest-owned location of the JSON sidecar. + + Returns: + [`~openenv.validation.runtime.contracts.RuntimePlan`]: validated public inputs. + + Raises: + [`~openenv.validation.runtime.contracts.RuntimePlanError`]: + Missing, escaped, oversized, ambiguous or invalid input. + """ + try: + package_root = Path(root).resolve(strict=True) + path = (package_root / execution.probe_path).resolve(strict=True) + if not path.is_relative_to(package_root) or not path.is_file(): + raise ValueError("runtime plan must be a regular file inside the package") + with path.open("rb") as source: + payload = source.read(MAX_PLAN_BYTES + 1) + if len(payload) > MAX_PLAN_BYTES: + raise ValueError(f"runtime plan exceeds {MAX_PLAN_BYTES} bytes") + raw = json.loads(payload, object_pairs_hook=_unique_object) + _check_json(raw) + return RuntimePlan.model_validate(raw) + except (OSError, ValueError, RecursionError) as exc: + raise RuntimePlanError(f"invalid runtime plan: {exc}") from exc + + +class LaunchSpec(BaseModel): + """ + Explicit validation-only launch request; no caller environment is inherited. + + Attributes: + image_ref (`str`): + Immutable local image ID or repository digest. + resources ([`~openenv.validation.manifest.ResourceDeclaration`]): + Subject CPU, memory, writable-storage and episode budgets. + network ([`~openenv.validation.manifest.NetworkPolicy`]): + Requested policy; unsupported enforcement must be refused. + run_id (`str`): + Unique run-owned resource label, safe for provider names. + startup_timeout_s (`float`, *optional*, defaults to 30): + Independent readiness deadline, at most 300 seconds. + env_vars (`dict[str, str]`, *optional*): + Explicit environment only; providers do not forward host credentials. + """ + + model_config = ConfigDict(extra="forbid", frozen=True) + + image_ref: str = Field(pattern=r"^(?:[^@\s]+@)?sha256:[0-9a-f]{64}$") + resources: ResourceDeclaration + network: NetworkPolicy + run_id: str = Field(pattern=r"^[a-z0-9][a-z0-9-]{0,62}$") + startup_timeout_s: float = Field(default=30.0, gt=0.0, le=300.0) + env_vars: dict[str, str] = Field(default_factory=dict, max_length=64) + + @field_validator("env_vars") + @classmethod + def _explicit_environment(cls, value: dict[str, str]) -> dict[str, str]: + for name, content in value.items(): + if re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", name) is None: + raise ValueError("invalid environment variable name") + if "\x00" in content or len(content.encode("utf-8")) > 8192: + raise ValueError("environment value contains NUL or exceeds 8192 bytes") + return value + + @field_validator("resources") + @classmethod + def _finite_resources(cls, value: ResourceDeclaration) -> ResourceDeclaration: + if not math.isfinite(value.cpu) or not math.isfinite(value.episode_timeout_s): + raise ValueError("runtime resource limits must be finite") + return value + + +@dataclass(frozen=True) +class WireExchange: + """ + One immutable raw exchange, captured before client defaults or type coercion. + + Attributes: + operation (`str`): + The reset, step or state operation performed on the measured session. + request_json (`str`): + Original serialized request; parse only when grading. + response_json (`str`): + Original serialized response, including malformed JSON for diagnostics. + """ + + operation: Literal["reset", "step", "state"] + request_json: str + response_json: str + + +@dataclass(frozen=True) +class RuntimeEvidence: + """ + Collector evidence shared by graders without permitting mutation of the episode. + + Attributes: + exchanges (`tuple[WireExchange, ...]`): + Ordered operations on one WebSocket session. + observation_schema_json (`str`, *optional*): + The advertised observation schema, exactly `/schema`'s observation value. + failure_phase (`str`, *optional*): + Collector phase that failed; a truncated transcript cannot pass silently. + failure_reason (`str`, *optional*): + Credential-safe explanation of the collection failure. + """ + + exchanges: tuple[WireExchange, ...] = () + observation_schema_json: str | None = None + failure_phase: str | None = None + failure_reason: str | None = None diff --git a/src/openenv/validation/schemas/manifest-v2.schema.json b/src/openenv/validation/schemas/manifest-v2.schema.json new file mode 100644 index 0000000000..ab8bc785fa --- /dev/null +++ b/src/openenv/validation/schemas/manifest-v2.schema.json @@ -0,0 +1,469 @@ +{ + "$defs": { + "CapabilitiesSpec": { + "additionalProperties": false, + "description": "Declared capabilities; these select the contract graders.\n\nA `None` oracle is a valid manifest \u2014 the missing oracle is a graded FAIL on\n`semantic.oracle_max` (it appears in the report with remediation), not a parse\nerror.\n\nAttributes:\n oracle ([`~openenv.validation.manifest.OracleDeclaration`], *optional*):\n The oracle declaration; absence FAILs `semantic.oracle_max`.\n verifier ([`~openenv.validation.manifest.VerifierBinding`]):\n How the verifier is invoked.\n set_state (`bool`, *optional*, defaults to `False`):\n Whether the environment accepts an `injected_state` kwarg on `reset()`.\n `False` with `oracle.form == \"injected_state\"` is a manifest schema error.\n llm_judged (`bool`, *optional*, defaults to `False`):\n Reward involves an LLM judge; requires a judge pin and variance tolerance.\n rubric_tree (`bool`, *optional*, defaults to `False`):\n OpenEnv format: an RFC 004 `Rubric` is present.\n task_api (`bool`, *optional*, defaults to `False`):\n `TaskProvider` implemented.\n canaries (`str`, *optional*):\n Package-relative path to shipped canary trajectories (#778 test #28).\n declared_tools (`list[str]`, *optional*):\n Checked against runtime discovery (#778 test #34).\n declared_task_count (`dict[str, int]`, *optional*):\n Split name to task count, checked against discovery (#778 test #35).", + "properties": { + "canaries": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Canaries" + }, + "declared_task_count": { + "additionalProperties": { + "type": "integer" + }, + "title": "Declared Task Count", + "type": "object" + }, + "declared_tools": { + "items": { + "type": "string" + }, + "title": "Declared Tools", + "type": "array" + }, + "llm_judged": { + "default": false, + "title": "Llm Judged", + "type": "boolean" + }, + "oracle": { + "anyOf": [ + { + "$ref": "#/$defs/OracleDeclaration" + }, + { + "type": "null" + } + ], + "default": null + }, + "rubric_tree": { + "default": false, + "title": "Rubric Tree", + "type": "boolean" + }, + "set_state": { + "default": false, + "title": "Set State", + "type": "boolean" + }, + "task_api": { + "default": false, + "title": "Task Api", + "type": "boolean" + }, + "verifier": { + "$ref": "#/$defs/VerifierBinding" + } + }, + "required": [ + "verifier" + ], + "title": "CapabilitiesSpec", + "type": "object" + }, + "ExecutionDeclaration": { + "additionalProperties": false, + "description": "Author-declared runtime binding, introduced by manifest schema 2.\n\nAll paths are portable package-relative paths. Resolving the probe or build\ncontext must additionally reject symlink escapes before reading source files.\nThe data-only probe supplies actions; it cannot change declared capabilities.\n\nAttributes:\n kind (`str`):\n The implemented transport binding, `\"openenv_ws\"`.\n probe_path (`str`):\n JSON file containing the versioned runtime plan.\n dockerfile (`str`):\n Dockerfile relative to the package root.\n context (`str`):\n Build context relative to the package root.\n agent_boundary (`str`):\n The access granted to an agent; currently only API access is supported.", + "properties": { + "agent_boundary": { + "const": "api", + "default": "api", + "title": "Agent Boundary", + "type": "string" + }, + "context": { + "default": ".", + "title": "Context", + "type": "string" + }, + "dockerfile": { + "default": "Dockerfile", + "title": "Dockerfile", + "type": "string" + }, + "kind": { + "const": "openenv_ws", + "default": "openenv_ws", + "title": "Kind", + "type": "string" + }, + "probe_path": { + "default": "validation/runtime.json", + "title": "Probe Path", + "type": "string" + } + }, + "title": "ExecutionDeclaration", + "type": "object" + }, + "JudgePin": { + "additionalProperties": false, + "description": "Pinned judge configuration. Required iff `capabilities.llm_judged`.\n\nThe pin lives in the manifest, not the rubric object: `LLMJudge.state_dict()` does\nnot serialize model/version/params, so validation checks the declared pin.", + "properties": { + "model": { + "title": "Model", + "type": "string" + }, + "params": { + "additionalProperties": true, + "title": "Params", + "type": "object" + }, + "version": { + "title": "Version", + "type": "string" + } + }, + "required": [ + "model", + "version" + ], + "title": "JudgePin", + "type": "object" + }, + "NetworkPolicy": { + "additionalProperties": false, + "description": "Declared network posture, following the Harbor `task.toml` (schema 1.4)\n`network_mode` precedent.\n\nAttributes:\n mode (`str`, *optional*, defaults to `\"public\"`):\n `\"public\"` (egress allowed \u2014 the default), `\"no-network\"`, or\n `\"allowlist\"`.\n allowed_hosts (`list[str]`, *optional*):\n Permitted destinations (exact hostnames, CIDR ranges, wildcards).\n Only valid with mode `\"allowlist\"`.", + "properties": { + "allowed_hosts": { + "items": { + "type": "string" + }, + "title": "Allowed Hosts", + "type": "array" + }, + "mode": { + "default": "public", + "enum": [ + "public", + "no-network", + "allowlist" + ], + "title": "Mode", + "type": "string" + } + }, + "title": "NetworkPolicy", + "type": "object" + }, + "OracleDeclaration": { + "additionalProperties": false, + "description": "How the package demonstrates max reward.\n\nAttributes:\n form (`str`):\n `\"injected_state\"` (preferred \u2014 most deterministic; requires the\n `set_state` capability) or `\"script\"` (Harbor `solution/solve.sh`\n precedent, executed inside the sandbox).\n location (`str`):\n Package-relative path to the oracle artifact; containment-checked.", + "properties": { + "form": { + "enum": [ + "injected_state", + "script" + ], + "title": "Form", + "type": "string" + }, + "location": { + "title": "Location", + "type": "string" + } + }, + "required": [ + "form", + "location" + ], + "title": "OracleDeclaration", + "type": "object" + }, + "ResourceDeclaration": { + "additionalProperties": false, + "description": "Declared resource budget; measured usage must fall within it (#778 test #10).\n\nGPU needs are declarations, not disqualifiers: a package declaring `gpus`\nvalidates on a sandbox provider with the GPU capability; on providers without\nit, runtime+ checks SKIP with the capability named.", + "properties": { + "cpu": { + "exclusiveMinimum": 0.0, + "title": "Cpu", + "type": "number" + }, + "disk_mb": { + "exclusiveMinimum": 0, + "title": "Disk Mb", + "type": "integer" + }, + "episode_timeout_s": { + "exclusiveMinimum": 0.0, + "title": "Episode Timeout S", + "type": "number" + }, + "gpu_types": { + "items": { + "type": "string" + }, + "title": "Gpu Types", + "type": "array" + }, + "gpus": { + "default": 0, + "minimum": 0, + "title": "Gpus", + "type": "integer" + }, + "memory_mb": { + "exclusiveMinimum": 0, + "title": "Memory Mb", + "type": "integer" + } + }, + "required": [ + "cpu", + "memory_mb", + "disk_mb", + "episode_timeout_s" + ], + "title": "ResourceDeclaration", + "type": "object" + }, + "RewardDeclaration": { + "additionalProperties": false, + "description": "Author-declared reward contract. Graders normalize measurements against it.\n\nDeclared tolerances are bounded by the severity policy; reports carry the declared\nvalues so hubs can apply stricter ceilings.\n\nAttributes:\n range (`tuple[float, float]`):\n Declared reward range, defaults to `(0.0, 1.0)`. Must be strictly increasing.\n oracle_tolerance (`float`):\n The oracle check passes when `reward(oracle) >= max - oracle_tolerance`.\n floor_margin (`float`):\n The gap check requires `max - measured_floor >= floor_margin`.\n variance_tolerance (`float`, *optional*):\n Required iff the environment declares `llm_judged`; bounds reward variance\n for variance-mode determinism checks.", + "properties": { + "floor_margin": { + "minimum": 0.0, + "title": "Floor Margin", + "type": "number" + }, + "oracle_tolerance": { + "default": 0.0, + "minimum": 0.0, + "title": "Oracle Tolerance", + "type": "number" + }, + "range": { + "default": [ + 0.0, + 1.0 + ], + "maxItems": 2, + "minItems": 2, + "prefixItems": [ + { + "type": "number" + }, + { + "type": "number" + } + ], + "title": "Range", + "type": "array" + }, + "variance_tolerance": { + "anyOf": [ + { + "minimum": 0.0, + "type": "number" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Variance Tolerance" + } + }, + "required": [ + "floor_margin" + ], + "title": "RewardDeclaration", + "type": "object" + }, + "SignatureKind": { + "description": "Recognized package formats, named by their well-known file.\n\nA manifest's signature is provenance for the report only \u2014 no grader, core or\nthird-party, may branch on it.", + "enum": [ + "openenv.yaml", + "task.toml", + "task.md" + ], + "title": "SignatureKind", + "type": "string" + }, + "TaskDistributionPin": { + "additionalProperties": false, + "description": "Pins for generated or externally sourced task distributions (#778 test #42).", + "properties": { + "dataset_version": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Dataset Version" + }, + "generator_model": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Generator Model" + }, + "generator_seed": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Generator Seed" + } + }, + "title": "TaskDistributionPin", + "type": "object" + }, + "TypeSpec": { + "additionalProperties": false, + "description": "Type tags; these select domain graders.\n\nBare tags are shared (`swe`); prefixed tags are hub-scoped (`hf:agentic-swe`).\npass@k comparisons are valid only within a tag.", + "properties": { + "tags": { + "items": { + "minLength": 1, + "type": "string" + }, + "minItems": 1, + "title": "Tags", + "type": "array" + } + }, + "required": [ + "tags" + ], + "title": "TypeSpec", + "type": "object" + }, + "VerifierBinding": { + "additionalProperties": false, + "description": "How \"evaluate this state\" is invoked.\n\nAttributes:\n kind (`str`):\n `\"reward_channel\"` (served reward) or `\"script\"` (Harbor-style\n `tests/test.sh`).\n entry (`str`, *optional*):\n Script path; required when `kind == \"script\"`, forbidden otherwise.", + "properties": { + "entry": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Entry" + }, + "kind": { + "enum": [ + "reward_channel", + "script" + ], + "title": "Kind", + "type": "string" + } + }, + "required": [ + "kind" + ], + "title": "VerifierBinding", + "type": "object" + } + }, + "additionalProperties": false, + "description": "Version 2 adds the runtime execution declaration without changing schema 1.\n\nStatic packages without `validation.execution` continue to normalize to\n[`~openenv.validation.manifest.NormalizedManifest`]. An explicit execution\ndeclaration opts into this model and the corresponding version 2 report.", + "properties": { + "capabilities": { + "$ref": "#/$defs/CapabilitiesSpec" + }, + "execution": { + "$ref": "#/$defs/ExecutionDeclaration" + }, + "judge": { + "anyOf": [ + { + "$ref": "#/$defs/JudgePin" + }, + { + "type": "null" + } + ], + "default": null + }, + "manifest_schema_version": { + "const": "2", + "title": "Manifest Schema Version", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "network": { + "$ref": "#/$defs/NetworkPolicy" + }, + "resources": { + "$ref": "#/$defs/ResourceDeclaration" + }, + "reward": { + "$ref": "#/$defs/RewardDeclaration" + }, + "signature": { + "$ref": "#/$defs/SignatureKind" + }, + "task_distribution": { + "anyOf": [ + { + "$ref": "#/$defs/TaskDistributionPin" + }, + { + "type": "null" + } + ], + "default": null + }, + "types": { + "$ref": "#/$defs/TypeSpec" + }, + "version": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Version" + } + }, + "required": [ + "manifest_schema_version", + "name", + "signature", + "reward", + "resources", + "capabilities", + "types", + "execution" + ], + "title": "NormalizedManifestV2", + "type": "object" +} diff --git a/src/openenv/validation/schemas/report-v2.schema.json b/src/openenv/validation/schemas/report-v2.schema.json new file mode 100644 index 0000000000..d480354632 --- /dev/null +++ b/src/openenv/validation/schemas/report-v2.schema.json @@ -0,0 +1,712 @@ +{ + "$defs": { + "CapabilitiesSpec": { + "additionalProperties": false, + "description": "Declared capabilities; these select the contract graders.\n\nA `None` oracle is a valid manifest \u2014 the missing oracle is a graded FAIL on\n`semantic.oracle_max` (it appears in the report with remediation), not a parse\nerror.\n\nAttributes:\n oracle ([`~openenv.validation.manifest.OracleDeclaration`], *optional*):\n The oracle declaration; absence FAILs `semantic.oracle_max`.\n verifier ([`~openenv.validation.manifest.VerifierBinding`]):\n How the verifier is invoked.\n set_state (`bool`, *optional*, defaults to `False`):\n Whether the environment accepts an `injected_state` kwarg on `reset()`.\n `False` with `oracle.form == \"injected_state\"` is a manifest schema error.\n llm_judged (`bool`, *optional*, defaults to `False`):\n Reward involves an LLM judge; requires a judge pin and variance tolerance.\n rubric_tree (`bool`, *optional*, defaults to `False`):\n OpenEnv format: an RFC 004 `Rubric` is present.\n task_api (`bool`, *optional*, defaults to `False`):\n `TaskProvider` implemented.\n canaries (`str`, *optional*):\n Package-relative path to shipped canary trajectories (#778 test #28).\n declared_tools (`list[str]`, *optional*):\n Checked against runtime discovery (#778 test #34).\n declared_task_count (`dict[str, int]`, *optional*):\n Split name to task count, checked against discovery (#778 test #35).", + "properties": { + "canaries": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Canaries" + }, + "declared_task_count": { + "additionalProperties": { + "type": "integer" + }, + "title": "Declared Task Count", + "type": "object" + }, + "declared_tools": { + "items": { + "type": "string" + }, + "title": "Declared Tools", + "type": "array" + }, + "llm_judged": { + "default": false, + "title": "Llm Judged", + "type": "boolean" + }, + "oracle": { + "anyOf": [ + { + "$ref": "#/$defs/OracleDeclaration" + }, + { + "type": "null" + } + ], + "default": null + }, + "rubric_tree": { + "default": false, + "title": "Rubric Tree", + "type": "boolean" + }, + "set_state": { + "default": false, + "title": "Set State", + "type": "boolean" + }, + "task_api": { + "default": false, + "title": "Task Api", + "type": "boolean" + }, + "verifier": { + "$ref": "#/$defs/VerifierBinding" + } + }, + "required": [ + "verifier" + ], + "title": "CapabilitiesSpec", + "type": "object" + }, + "CheckResult": { + "additionalProperties": false, + "description": "Outcome of one check, as emitted by its grader.\n\nAttributes:\n check_id (`str`):\n Stable, policy-addressed id. Local runs use `static.*`/`runtime.*`/\n `semantic.*`; the `hub.*` and `statistical.*` namespaces are reserved for\n operator lanes.\n status ([`~openenv.validation.types.CheckStatus`]):\n The grader's finding. Severity is assigned by the policy, never here.\n measured (`dict[str, Any]`, *optional*):\n Measured values, e.g. `{\"oracle_reward\": 0.98, \"declared_tolerance\": 0.05}`.\n evidence (`list[str]`, *optional*):\n Human-readable, credential-safe evidence lines.\n remediation (`str`, *optional*):\n What the author changes to go green.\n duration_s (`float`):\n Wall-clock time this check took.", + "properties": { + "check_id": { + "pattern": "^[a-z_]+\\.[a-z_]+$", + "title": "Check Id", + "type": "string" + }, + "duration_s": { + "minimum": 0.0, + "title": "Duration S", + "type": "number" + }, + "evidence": { + "items": { + "type": "string" + }, + "title": "Evidence", + "type": "array" + }, + "measured": { + "additionalProperties": true, + "title": "Measured", + "type": "object" + }, + "remediation": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Remediation" + }, + "status": { + "$ref": "#/$defs/CheckStatus" + } + }, + "required": [ + "check_id", + "status", + "duration_s" + ], + "title": "CheckResult", + "type": "object" + }, + "CheckStatus": { + "description": "What a grader may return for a single check.\n\nSKIP requires a reason (unmet dependency or missing capability). ERROR means the\ngrader crashed or its evidence was malformed; the policy fails ERROR closed.", + "enum": [ + "pass", + "fail", + "skip", + "error" + ], + "title": "CheckStatus", + "type": "string" + }, + "ExecutionDeclaration": { + "additionalProperties": false, + "description": "Author-declared runtime binding, introduced by manifest schema 2.\n\nAll paths are portable package-relative paths. Resolving the probe or build\ncontext must additionally reject symlink escapes before reading source files.\nThe data-only probe supplies actions; it cannot change declared capabilities.\n\nAttributes:\n kind (`str`):\n The implemented transport binding, `\"openenv_ws\"`.\n probe_path (`str`):\n JSON file containing the versioned runtime plan.\n dockerfile (`str`):\n Dockerfile relative to the package root.\n context (`str`):\n Build context relative to the package root.\n agent_boundary (`str`):\n The access granted to an agent; currently only API access is supported.", + "properties": { + "agent_boundary": { + "const": "api", + "default": "api", + "title": "Agent Boundary", + "type": "string" + }, + "context": { + "default": ".", + "title": "Context", + "type": "string" + }, + "dockerfile": { + "default": "Dockerfile", + "title": "Dockerfile", + "type": "string" + }, + "kind": { + "const": "openenv_ws", + "default": "openenv_ws", + "title": "Kind", + "type": "string" + }, + "probe_path": { + "default": "validation/runtime.json", + "title": "Probe Path", + "type": "string" + } + }, + "title": "ExecutionDeclaration", + "type": "object" + }, + "JudgePin": { + "additionalProperties": false, + "description": "Pinned judge configuration. Required iff `capabilities.llm_judged`.\n\nThe pin lives in the manifest, not the rubric object: `LLMJudge.state_dict()` does\nnot serialize model/version/params, so validation checks the declared pin.", + "properties": { + "model": { + "title": "Model", + "type": "string" + }, + "params": { + "additionalProperties": true, + "title": "Params", + "type": "object" + }, + "version": { + "title": "Version", + "type": "string" + } + }, + "required": [ + "model", + "version" + ], + "title": "JudgePin", + "type": "object" + }, + "Lane": { + "description": "Who runs a check.\n\nLOCAL checks run in the author's `openenv validate`. HUB checks are operator-run\nand are never referenced in local reports \u2014 an author is never shown a check they\ncannot red-to-green.", + "enum": [ + "local", + "hub" + ], + "title": "Lane", + "type": "string" + }, + "Level": { + "description": "Validation levels, ordered by cost budget.\n\nSTATISTICAL is reserved: its check ids exist in the report schema and severity\npolicy, but no local implementation ships in this repo.", + "enum": [ + 1, + 2, + 3, + 4 + ], + "title": "Level", + "type": "integer" + }, + "NetworkPolicy": { + "additionalProperties": false, + "description": "Declared network posture, following the Harbor `task.toml` (schema 1.4)\n`network_mode` precedent.\n\nAttributes:\n mode (`str`, *optional*, defaults to `\"public\"`):\n `\"public\"` (egress allowed \u2014 the default), `\"no-network\"`, or\n `\"allowlist\"`.\n allowed_hosts (`list[str]`, *optional*):\n Permitted destinations (exact hostnames, CIDR ranges, wildcards).\n Only valid with mode `\"allowlist\"`.", + "properties": { + "allowed_hosts": { + "items": { + "type": "string" + }, + "title": "Allowed Hosts", + "type": "array" + }, + "mode": { + "default": "public", + "enum": [ + "public", + "no-network", + "allowlist" + ], + "title": "Mode", + "type": "string" + } + }, + "title": "NetworkPolicy", + "type": "object" + }, + "NormalizedManifest": { + "additionalProperties": false, + "description": "The one document every grader reads. Produced only by parsers.\n\nAfter a parser returns this, the package format has left the pipeline: `signature`\nis provenance for the report and nothing else.", + "properties": { + "capabilities": { + "$ref": "#/$defs/CapabilitiesSpec" + }, + "judge": { + "anyOf": [ + { + "$ref": "#/$defs/JudgePin" + }, + { + "type": "null" + } + ], + "default": null + }, + "manifest_schema_version": { + "const": "1", + "title": "Manifest Schema Version", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "network": { + "$ref": "#/$defs/NetworkPolicy" + }, + "resources": { + "$ref": "#/$defs/ResourceDeclaration" + }, + "reward": { + "$ref": "#/$defs/RewardDeclaration" + }, + "signature": { + "$ref": "#/$defs/SignatureKind" + }, + "task_distribution": { + "anyOf": [ + { + "$ref": "#/$defs/TaskDistributionPin" + }, + { + "type": "null" + } + ], + "default": null + }, + "types": { + "$ref": "#/$defs/TypeSpec" + }, + "version": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Version" + } + }, + "required": [ + "manifest_schema_version", + "name", + "signature", + "reward", + "resources", + "capabilities", + "types" + ], + "title": "NormalizedManifest", + "type": "object" + }, + "NormalizedManifestV2": { + "additionalProperties": false, + "description": "Version 2 adds the runtime execution declaration without changing schema 1.\n\nStatic packages without `validation.execution` continue to normalize to\n[`~openenv.validation.manifest.NormalizedManifest`]. An explicit execution\ndeclaration opts into this model and the corresponding version 2 report.", + "properties": { + "capabilities": { + "$ref": "#/$defs/CapabilitiesSpec" + }, + "execution": { + "$ref": "#/$defs/ExecutionDeclaration" + }, + "judge": { + "anyOf": [ + { + "$ref": "#/$defs/JudgePin" + }, + { + "type": "null" + } + ], + "default": null + }, + "manifest_schema_version": { + "const": "2", + "title": "Manifest Schema Version", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "network": { + "$ref": "#/$defs/NetworkPolicy" + }, + "resources": { + "$ref": "#/$defs/ResourceDeclaration" + }, + "reward": { + "$ref": "#/$defs/RewardDeclaration" + }, + "signature": { + "$ref": "#/$defs/SignatureKind" + }, + "task_distribution": { + "anyOf": [ + { + "$ref": "#/$defs/TaskDistributionPin" + }, + { + "type": "null" + } + ], + "default": null + }, + "types": { + "$ref": "#/$defs/TypeSpec" + }, + "version": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Version" + } + }, + "required": [ + "manifest_schema_version", + "name", + "signature", + "reward", + "resources", + "capabilities", + "types", + "execution" + ], + "title": "NormalizedManifestV2", + "type": "object" + }, + "OracleDeclaration": { + "additionalProperties": false, + "description": "How the package demonstrates max reward.\n\nAttributes:\n form (`str`):\n `\"injected_state\"` (preferred \u2014 most deterministic; requires the\n `set_state` capability) or `\"script\"` (Harbor `solution/solve.sh`\n precedent, executed inside the sandbox).\n location (`str`):\n Package-relative path to the oracle artifact; containment-checked.", + "properties": { + "form": { + "enum": [ + "injected_state", + "script" + ], + "title": "Form", + "type": "string" + }, + "location": { + "title": "Location", + "type": "string" + } + }, + "required": [ + "form", + "location" + ], + "title": "OracleDeclaration", + "type": "object" + }, + "ResourceDeclaration": { + "additionalProperties": false, + "description": "Declared resource budget; measured usage must fall within it (#778 test #10).\n\nGPU needs are declarations, not disqualifiers: a package declaring `gpus`\nvalidates on a sandbox provider with the GPU capability; on providers without\nit, runtime+ checks SKIP with the capability named.", + "properties": { + "cpu": { + "exclusiveMinimum": 0.0, + "title": "Cpu", + "type": "number" + }, + "disk_mb": { + "exclusiveMinimum": 0, + "title": "Disk Mb", + "type": "integer" + }, + "episode_timeout_s": { + "exclusiveMinimum": 0.0, + "title": "Episode Timeout S", + "type": "number" + }, + "gpu_types": { + "items": { + "type": "string" + }, + "title": "Gpu Types", + "type": "array" + }, + "gpus": { + "default": 0, + "minimum": 0, + "title": "Gpus", + "type": "integer" + }, + "memory_mb": { + "exclusiveMinimum": 0, + "title": "Memory Mb", + "type": "integer" + } + }, + "required": [ + "cpu", + "memory_mb", + "disk_mb", + "episode_timeout_s" + ], + "title": "ResourceDeclaration", + "type": "object" + }, + "RewardDeclaration": { + "additionalProperties": false, + "description": "Author-declared reward contract. Graders normalize measurements against it.\n\nDeclared tolerances are bounded by the severity policy; reports carry the declared\nvalues so hubs can apply stricter ceilings.\n\nAttributes:\n range (`tuple[float, float]`):\n Declared reward range, defaults to `(0.0, 1.0)`. Must be strictly increasing.\n oracle_tolerance (`float`):\n The oracle check passes when `reward(oracle) >= max - oracle_tolerance`.\n floor_margin (`float`):\n The gap check requires `max - measured_floor >= floor_margin`.\n variance_tolerance (`float`, *optional*):\n Required iff the environment declares `llm_judged`; bounds reward variance\n for variance-mode determinism checks.", + "properties": { + "floor_margin": { + "minimum": 0.0, + "title": "Floor Margin", + "type": "number" + }, + "oracle_tolerance": { + "default": 0.0, + "minimum": 0.0, + "title": "Oracle Tolerance", + "type": "number" + }, + "range": { + "default": [ + 0.0, + 1.0 + ], + "maxItems": 2, + "minItems": 2, + "prefixItems": [ + { + "type": "number" + }, + { + "type": "number" + } + ], + "title": "Range", + "type": "array" + }, + "variance_tolerance": { + "anyOf": [ + { + "minimum": 0.0, + "type": "number" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Variance Tolerance" + } + }, + "required": [ + "floor_margin" + ], + "title": "RewardDeclaration", + "type": "object" + }, + "SignatureKind": { + "description": "Recognized package formats, named by their well-known file.\n\nA manifest's signature is provenance for the report only \u2014 no grader, core or\nthird-party, may branch on it.", + "enum": [ + "openenv.yaml", + "task.toml", + "task.md" + ], + "title": "SignatureKind", + "type": "string" + }, + "TaskDistributionPin": { + "additionalProperties": false, + "description": "Pins for generated or externally sourced task distributions (#778 test #42).", + "properties": { + "dataset_version": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Dataset Version" + }, + "generator_model": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Generator Model" + }, + "generator_seed": { + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Generator Seed" + } + }, + "title": "TaskDistributionPin", + "type": "object" + }, + "TypeSpec": { + "additionalProperties": false, + "description": "Type tags; these select domain graders.\n\nBare tags are shared (`swe`); prefixed tags are hub-scoped (`hf:agentic-swe`).\npass@k comparisons are valid only within a tag.", + "properties": { + "tags": { + "items": { + "minLength": 1, + "type": "string" + }, + "minItems": 1, + "title": "Tags", + "type": "array" + } + }, + "required": [ + "tags" + ], + "title": "TypeSpec", + "type": "object" + }, + "Verdict": { + "description": "Overall outcome of a validation run. WARN means only warn/advisory findings.", + "enum": [ + "pass", + "warn", + "fail" + ], + "title": "Verdict", + "type": "string" + }, + "VerifierBinding": { + "additionalProperties": false, + "description": "How \"evaluate this state\" is invoked.\n\nAttributes:\n kind (`str`):\n `\"reward_channel\"` (served reward) or `\"script\"` (Harbor-style\n `tests/test.sh`).\n entry (`str`, *optional*):\n Script path; required when `kind == \"script\"`, forbidden otherwise.", + "properties": { + "entry": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Entry" + }, + "kind": { + "enum": [ + "reward_channel", + "script" + ], + "title": "Kind", + "type": "string" + } + }, + "required": [ + "kind" + ], + "title": "VerifierBinding", + "type": "object" + } + }, + "additionalProperties": false, + "description": "Report version 2 preserves runtime declarations and parse-failure reporting.\n\nA runtime request may target a version 1 package without an execution binding,\nor fail before producing any manifest. Both remain representable so the report\ncan explain the unmet prerequisite or manifest error.", + "properties": { + "lane": { + "$ref": "#/$defs/Lane" + }, + "levels_run": { + "items": { + "$ref": "#/$defs/Level" + }, + "title": "Levels Run", + "type": "array" + }, + "manifest": { + "anyOf": [ + { + "$ref": "#/$defs/NormalizedManifestV2" + }, + { + "$ref": "#/$defs/NormalizedManifest" + }, + { + "type": "null" + } + ], + "title": "Manifest" + }, + "policy_version": { + "title": "Policy Version", + "type": "string" + }, + "report_schema_version": { + "const": "2", + "title": "Report Schema Version", + "type": "string" + }, + "results": { + "items": { + "$ref": "#/$defs/CheckResult" + }, + "title": "Results", + "type": "array" + }, + "signature": { + "$ref": "#/$defs/SignatureKind" + }, + "source_digest": { + "title": "Source Digest", + "type": "string" + }, + "target": { + "title": "Target", + "type": "string" + }, + "verdict": { + "$ref": "#/$defs/Verdict" + } + }, + "required": [ + "report_schema_version", + "target", + "source_digest", + "signature", + "manifest", + "policy_version", + "lane", + "levels_run", + "results", + "verdict" + ], + "title": "ValidationReportV2", + "type": "object" +} diff --git a/src/openenv/validation/schemas/runtime-plan.schema.json b/src/openenv/validation/schemas/runtime-plan.schema.json new file mode 100644 index 0000000000..210ed3407e --- /dev/null +++ b/src/openenv/validation/schemas/runtime-plan.schema.json @@ -0,0 +1,62 @@ +{ + "$defs": { + "RuntimeReset": { + "additionalProperties": false, + "description": "Explicit reset inputs for one measured episode.\n\nAttributes:\n episode_id (`str`):\n Requested episode identity; verified by the state contract grader.\n seed (`int`):\n Requested seed. Acceptance alone does not establish determinism.\n options (`dict[str, Any]`, *optional*):\n Additional public reset arguments, never replacements for the seed or ID.", + "properties": { + "episode_id": { + "maxLength": 128, + "minLength": 1, + "title": "Episode Id", + "type": "string" + }, + "options": { + "additionalProperties": true, + "title": "Options", + "type": "object" + }, + "seed": { + "maximum": 4294967295, + "minimum": 0, + "title": "Seed", + "type": "integer" + } + }, + "required": [ + "episode_id", + "seed" + ], + "title": "RuntimeReset", + "type": "object" + } + }, + "additionalProperties": false, + "description": "Bounded, data-only runtime inputs. Capability declarations stay in the manifest.\n\nAttributes:\n plan_schema_version (`str`):\n The pinned sidecar schema version, `\"1\"`.\n reset ([`~openenv.validation.runtime.contracts.RuntimeReset`]):\n Inputs to the first reset on the measured WebSocket session.\n actions (`list[dict[str, Any]]`):\n Between one and 100 public actions, applied in order until termination.", + "properties": { + "actions": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "maxItems": 100, + "minItems": 1, + "title": "Actions", + "type": "array" + }, + "plan_schema_version": { + "const": "1", + "title": "Plan Schema Version", + "type": "string" + }, + "reset": { + "$ref": "#/$defs/RuntimeReset" + } + }, + "required": [ + "plan_schema_version", + "reset", + "actions" + ], + "title": "RuntimePlan", + "type": "object" +} diff --git a/tests/fixtures/validation/runtime/README.md b/tests/fixtures/validation/runtime/README.md new file mode 100644 index 0000000000..e70c052feb --- /dev/null +++ b/tests/fixtures/validation/runtime/README.md @@ -0,0 +1,23 @@ +# Shared Level 2 validation assets + +`cases.json` is the acceptance catalog for the RFC 008 runtime stack. Expected +findings are written from the RFC; they are not generated from grader output. +Every policy-v2 runtime check has a positive and a negative case, an applicability +description, required provider evidence and the PR that implements it. + +Cases whose `implementation_pr` exceeds the current slice are planned acceptance +work, not evidence that a grader exists. During the first three slices, the +`good`, `startup_failure`, `bad_reward`, `bad_observation` and `bad_state` cases +exercise startup and the basic runtime contracts. `good` means those four checks +pass; it does not mean all Level 2 checks have been implemented. + +The `served_probe/` subject is shared by real-protocol, Docker and installed-wheel +tests. Its fault modes are test-only and each introduces one intentional defect. +The public `validation/runtime.json` file contains only reset inputs and actions. +Capabilities, resource budgets and network policy remain in `openenv.yaml`. + +Preserve the existing manifest-only fixture directories. Schema 1 fixtures remain +the static compatibility baseline; adding runtime declarations selects schema 2. +Fast contract tests and the catalog inventory run without Docker. Docker evidence +must record an exact wheel and image identity, expected check inventory and verified +cleanup; a CLI zero exit status alone is insufficient because incomplete runs warn. diff --git a/tests/fixtures/validation/runtime/cases.json b/tests/fixtures/validation/runtime/cases.json new file mode 100644 index 0000000000..d32cd8cdb9 --- /dev/null +++ b/tests/fixtures/validation/runtime/cases.json @@ -0,0 +1,421 @@ +{ + "catalog_schema_version": "1", + "checks": [ + { + "check_id": "runtime.startup", + "implementation_pr": 3, + "applicability": "all runtime declarations", + "required_provider_features": [ + "docker_local" + ], + "positive_case": "good", + "negative_case": "startup_failure" + }, + { + "check_id": "runtime.reward_well_formed", + "implementation_pr": 3, + "applicability": "all runtime declarations", + "required_provider_features": [], + "positive_case": "good", + "negative_case": "bad_reward" + }, + { + "check_id": "runtime.observation_schema", + "implementation_pr": 3, + "applicability": "all runtime declarations", + "required_provider_features": [], + "positive_case": "good", + "negative_case": "bad_observation" + }, + { + "check_id": "runtime.state_contract", + "implementation_pr": 3, + "applicability": "all runtime declarations", + "required_provider_features": [], + "positive_case": "good", + "negative_case": "bad_state" + }, + { + "check_id": "runtime.trajectory_record", + "implementation_pr": 5, + "applicability": "all runtime declarations", + "required_provider_features": [ + "session_telemetry" + ], + "positive_case": "good_trajectory_record", + "negative_case": "missing_record" + }, + { + "check_id": "runtime.reward_attribution", + "implementation_pr": 6, + "applicability": "rubric_tree", + "required_provider_features": [ + "session_telemetry" + ], + "positive_case": "good_reward_attribution", + "negative_case": "bad_attribution" + }, + { + "check_id": "runtime.rubric_introspectable", + "implementation_pr": 6, + "applicability": "rubric_tree", + "required_provider_features": [ + "session_telemetry" + ], + "positive_case": "good_rubric_introspectable", + "negative_case": "missing_rubric" + }, + { + "check_id": "runtime.tool_declaration_accuracy", + "implementation_pr": 6, + "applicability": "declared_tools including empty declarations", + "required_provider_features": [], + "positive_case": "good_tool_declaration_accuracy", + "negative_case": "extra_tool" + }, + { + "check_id": "runtime.task_declaration_accuracy", + "implementation_pr": 6, + "applicability": "task_api", + "required_provider_features": [], + "positive_case": "good_task_declaration_accuracy", + "negative_case": "bad_task_count" + }, + { + "check_id": "runtime.seed_control", + "implementation_pr": 5, + "applicability": "all runtime declarations", + "required_provider_features": [ + "session_telemetry" + ], + "positive_case": "good_seed_control", + "negative_case": "ignored_seed" + }, + { + "check_id": "runtime.episode_determinism", + "implementation_pr": 5, + "applicability": "all runtime declarations", + "required_provider_features": [ + "fresh_container" + ], + "positive_case": "good_episode_determinism", + "negative_case": "nondeterministic" + }, + { + "check_id": "runtime.network_policy", + "implementation_pr": 8, + "applicability": "all runtime declarations", + "required_provider_features": [ + "network_policy" + ], + "positive_case": "good_network_policy", + "negative_case": "network_leak" + }, + { + "check_id": "runtime.host_containment", + "implementation_pr": 7, + "applicability": "declared agent boundary", + "required_provider_features": [ + "inspected_isolation" + ], + "positive_case": "good_host_containment", + "negative_case": "unsafe_inspection" + }, + { + "check_id": "runtime.resource_bounds", + "implementation_pr": 7, + "applicability": "all runtime declarations", + "required_provider_features": [ + "resource_enforcement" + ], + "positive_case": "good_resource_bounds", + "negative_case": "missing_limit" + }, + { + "check_id": "runtime.episode_isolation", + "implementation_pr": 10, + "applicability": "all runtime declarations", + "required_provider_features": [ + "session_telemetry" + ], + "positive_case": "good_episode_isolation", + "negative_case": "sticky_episode" + }, + { + "check_id": "runtime.oracle_containment", + "implementation_pr": 10, + "applicability": "oracle declared", + "required_provider_features": [ + "agent_identity" + ], + "positive_case": "good_oracle_containment", + "negative_case": "exposed_oracle" + } + ], + "cases": [ + { + "case_id": "good", + "implementation_pr": 3, + "requires_docker": true, + "expected": { + "runtime.startup": "pass", + "runtime.reward_well_formed": "pass", + "runtime.observation_schema": "pass", + "runtime.state_contract": "pass" + }, + "evidence_predicate": "one session has valid reset/step/state responses and finite in-range rewards" + }, + { + "case_id": "startup_failure", + "implementation_pr": 3, + "requires_docker": true, + "expected": { + "runtime.startup": "fail" + }, + "evidence_predicate": "readiness must fail before collection starts" + }, + { + "case_id": "bad_reward", + "implementation_pr": 3, + "requires_docker": true, + "expected": { + "runtime.reward_well_formed": "fail" + }, + "evidence_predicate": "raw reward is bool, nonfinite, wrong typed or outside the declared range" + }, + { + "case_id": "bad_observation", + "implementation_pr": 3, + "requires_docker": true, + "expected": { + "runtime.observation_schema": "fail" + }, + "evidence_predicate": "raw envelope or reconstructed observation violates advertised schema" + }, + { + "case_id": "bad_state", + "implementation_pr": 3, + "requires_docker": true, + "expected": { + "runtime.state_contract": "fail" + }, + "evidence_predicate": "episode identity or step count differs from the requested sequence" + }, + { + "case_id": "good_trajectory_record", + "implementation_pr": 5, + "requires_docker": true, + "expected": { + "runtime.trajectory_record": "pass" + }, + "evidence_predicate": "trajectory_record measured evidence satisfies the RFC contract" + }, + { + "case_id": "missing_record", + "implementation_pr": 5, + "requires_docker": true, + "expected": { + "runtime.trajectory_record": "fail" + }, + "evidence_predicate": "subject emitted trajectory is absent or differs from the independent transcript" + }, + { + "case_id": "good_reward_attribution", + "implementation_pr": 6, + "requires_docker": true, + "expected": { + "runtime.reward_attribution": "pass" + }, + "evidence_predicate": "reward_attribution measured evidence satisfies the RFC contract" + }, + { + "case_id": "bad_attribution", + "implementation_pr": 6, + "requires_docker": true, + "expected": { + "runtime.reward_attribution": "fail" + }, + "evidence_predicate": "child scores disagree with the declared aggregation semantics" + }, + { + "case_id": "good_rubric_introspectable", + "implementation_pr": 6, + "requires_docker": true, + "expected": { + "runtime.rubric_introspectable": "pass" + }, + "evidence_predicate": "rubric_introspectable measured evidence satisfies the RFC contract" + }, + { + "case_id": "missing_rubric", + "implementation_pr": 6, + "requires_docker": true, + "expected": { + "runtime.rubric_introspectable": "fail" + }, + "evidence_predicate": "declared rubric has no named serializable session tree" + }, + { + "case_id": "good_tool_declaration_accuracy", + "implementation_pr": 6, + "requires_docker": true, + "expected": { + "runtime.tool_declaration_accuracy": "pass" + }, + "evidence_predicate": "tool_declaration_accuracy measured evidence satisfies the RFC contract" + }, + { + "case_id": "extra_tool", + "implementation_pr": 6, + "requires_docker": true, + "expected": { + "runtime.tool_declaration_accuracy": "fail" + }, + "evidence_predicate": "discovered tool names differ from the declared set" + }, + { + "case_id": "good_task_declaration_accuracy", + "implementation_pr": 6, + "requires_docker": true, + "expected": { + "runtime.task_declaration_accuracy": "pass" + }, + "evidence_predicate": "task_declaration_accuracy measured evidence satisfies the RFC contract" + }, + { + "case_id": "bad_task_count", + "implementation_pr": 6, + "requires_docker": true, + "expected": { + "runtime.task_declaration_accuracy": "fail" + }, + "evidence_predicate": "num_tasks differs from the declared split count" + }, + { + "case_id": "good_seed_control", + "implementation_pr": 5, + "requires_docker": true, + "expected": { + "runtime.seed_control": "pass" + }, + "evidence_predicate": "seed_control measured evidence satisfies the RFC contract" + }, + { + "case_id": "ignored_seed", + "implementation_pr": 5, + "requires_docker": true, + "expected": { + "runtime.seed_control": "fail" + }, + "evidence_predicate": "framework reports discarded or rejected seed" + }, + { + "case_id": "good_episode_determinism", + "implementation_pr": 5, + "requires_docker": true, + "expected": { + "runtime.episode_determinism": "pass" + }, + "evidence_predicate": "episode_determinism measured evidence satisfies the RFC contract" + }, + { + "case_id": "nondeterministic", + "implementation_pr": 5, + "requires_docker": true, + "expected": { + "runtime.episode_determinism": "fail" + }, + "evidence_predicate": "same-input replay diverges outside policy-owned volatile fields" + }, + { + "case_id": "good_network_policy", + "implementation_pr": 8, + "requires_docker": true, + "expected": { + "runtime.network_policy": "pass" + }, + "evidence_predicate": "network_policy measured evidence satisfies the RFC contract" + }, + { + "case_id": "network_leak", + "implementation_pr": 8, + "requires_docker": true, + "expected": { + "runtime.network_policy": "fail" + }, + "evidence_predicate": "subject reaches a prohibited controlled destination" + }, + { + "case_id": "good_host_containment", + "implementation_pr": 7, + "requires_docker": true, + "expected": { + "runtime.host_containment": "pass" + }, + "evidence_predicate": "host_containment measured evidence satisfies the RFC contract" + }, + { + "case_id": "unsafe_inspection", + "implementation_pr": 7, + "requires_docker": false, + "expected": { + "runtime.host_containment": "fail" + }, + "evidence_predicate": "synthetic inspected launch settings violate the isolation contract" + }, + { + "case_id": "good_resource_bounds", + "implementation_pr": 7, + "requires_docker": true, + "expected": { + "runtime.resource_bounds": "pass" + }, + "evidence_predicate": "resource_bounds measured evidence satisfies the RFC contract" + }, + { + "case_id": "missing_limit", + "implementation_pr": 7, + "requires_docker": true, + "expected": { + "runtime.resource_bounds": "fail" + }, + "evidence_predicate": "inspected or measured enforcement fails a declared budget" + }, + { + "case_id": "good_episode_isolation", + "implementation_pr": 10, + "requires_docker": true, + "expected": { + "runtime.episode_isolation": "pass" + }, + "evidence_predicate": "episode_isolation measured evidence satisfies the RFC contract" + }, + { + "case_id": "sticky_episode", + "implementation_pr": 10, + "requires_docker": true, + "expected": { + "runtime.episode_isolation": "fail" + }, + "evidence_predicate": "episode B observes a marker from episode A after same-server reset" + }, + { + "case_id": "good_oracle_containment", + "implementation_pr": 10, + "requires_docker": true, + "expected": { + "runtime.oracle_containment": "pass" + }, + "evidence_predicate": "oracle_containment measured evidence satisfies the RFC contract" + }, + { + "case_id": "exposed_oracle", + "implementation_pr": 10, + "requires_docker": true, + "expected": { + "runtime.oracle_containment": "fail" + }, + "evidence_predicate": "harmless oracle canary is readable through the declared agent boundary" + } + ] +} diff --git a/tests/test_validation/support/__init__.py b/tests/test_validation/support/__init__.py new file mode 100644 index 0000000000..f61cf1121b --- /dev/null +++ b/tests/test_validation/support/__init__.py @@ -0,0 +1 @@ +"""Shared test-only validation evidence builders and provider fakes.""" diff --git a/tests/test_validation/support/runtime.py b/tests/test_validation/support/runtime.py new file mode 100644 index 0000000000..0b984ad13b --- /dev/null +++ b/tests/test_validation/support/runtime.py @@ -0,0 +1,69 @@ +"""Small reusable fakes; real protocol and Docker tests establish execution evidence.""" + +import json +from dataclasses import dataclass, field +from pathlib import Path + +from openenv.validation.manifest import ExecutionDeclaration +from openenv.validation.providers import ExecResult +from openenv.validation.runtime.contracts import ( + LaunchSpec, + RuntimeEvidence, + WireExchange, +) +from openenv.validation.types import ProviderCapability + + +def exchange(operation, request, response): + """Build immutable evidence without hiding malformed wire values via coercion.""" + return WireExchange(operation, json.dumps(request), json.dumps(response)) + + +def evidence(*exchanges, observation_schema=None): + """Build one collector result from exact test exchanges and schema data.""" + schema = None if observation_schema is None else json.dumps(observation_schema) + return RuntimeEvidence(tuple(exchanges), observation_schema_json=schema) + + +@dataclass +class FakeRunningSubject: + """Inert subject recording teardown and exec, with no network or Docker calls.""" + + base_url: str = "http://127.0.0.1:1" + stopped: bool = False + commands: list[list[str]] = field(default_factory=list) + + def exec(self, argv: list[str], timeout_s: float) -> ExecResult: + self.commands.append(list(argv)) + return ExecResult(0, "", "", 0.0) + + def inspect(self) -> dict: + return {"test_only": True} + + def logs(self, max_bytes: int = 65_536) -> str: + return "" + + def stop(self) -> None: + self.stopped = True + + +@dataclass +class FakeRuntimeProvider: + """Validation-only launcher fake that records explicit launch requests.""" + + name: str = "fake-runtime" + capabilities: frozenset = frozenset( + {ProviderCapability.IMAGE_BUILD, ProviderCapability.EXEC} + ) + supported_network_modes: frozenset[str] = frozenset({"public"}) + subject: FakeRunningSubject = field(default_factory=FakeRunningSubject) + launches: list[LaunchSpec] = field(default_factory=list) + builds: list[Path] = field(default_factory=list) + + def build(self, root: Path, execution: ExecutionDeclaration) -> str: + self.builds.append(root) + return "sha256:" + "a" * 64 + + def start(self, spec: LaunchSpec) -> FakeRunningSubject: + self.launches.append(spec) + return self.subject diff --git a/tests/test_validation/test_runtime_contracts.py b/tests/test_validation/test_runtime_contracts.py new file mode 100644 index 0000000000..295838affc --- /dev/null +++ b/tests/test_validation/test_runtime_contracts.py @@ -0,0 +1,339 @@ +import json +from dataclasses import FrozenInstanceError + +import pytest +import yaml +from conftest import FIXTURES, load_fixture_manifest +from openenv.validation.manifest import ( + ExecutionDeclaration, + ManifestError, + NetworkPolicy, + NormalizedManifest, + NormalizedManifestV2, + ResourceDeclaration, +) +from openenv.validation.parsers.openenv_yaml import OpenEnvYamlParser +from openenv.validation.policy import load_policy +from openenv.validation.report import ValidationReport, ValidationReportV2 +from openenv.validation.runner import run_validation +from openenv.validation.runtime.contracts import ( + LaunchSpec, + load_runtime_plan, + MAX_PLAN_BYTES, + RuntimeEvidence, + RuntimePlan, + RuntimePlanError, + WireExchange, +) +from openenv.validation.types import Lane, Level, SignatureKind, Verdict +from pydantic import ValidationError + + +def plan_data(): + return { + "plan_schema_version": "1", + "reset": {"episode_id": "probe-episode", "seed": 42}, + "actions": [{"increment": 1}], + } + + +def write_plan(root, payload): + directory = root / "validation" + directory.mkdir(exist_ok=True) + (directory / "runtime.json").write_text(payload) + + +def test_runtime_plan_roundtrip_and_pure_read(tmp_path): + data = plan_data() + write_plan(tmp_path, json.dumps(data)) + (tmp_path / "server.py").write_text("raise RuntimeError('must not import')") + plan = load_runtime_plan(tmp_path, ExecutionDeclaration()) + assert RuntimePlan.model_validate_json(plan.model_dump_json()) == plan + assert plan.reset.episode_id == "probe-episode" + + +@pytest.mark.parametrize( + "path", ["../secret", "/secret", "a/../../b", "a\\b", "C:/x", ""] +) +@pytest.mark.parametrize("field", ["probe_path", "dockerfile", "context"]) +def test_execution_paths_reject_nonportable_or_escaped_locations(field, path): + with pytest.raises(ValidationError, match="package-relative"): + ExecutionDeclaration(**{field: path}) + + +def test_plan_symlink_cannot_escape_package(tmp_path): + package = tmp_path / "package" + package.mkdir() + (package / "validation").symlink_to(tmp_path, target_is_directory=True) + (tmp_path / "runtime.json").write_text(json.dumps(plan_data())) + with pytest.raises(RuntimePlanError, match="inside the package"): + load_runtime_plan(package, ExecutionDeclaration()) + + +@pytest.mark.parametrize( + "payload", ["x" * (MAX_PLAN_BYTES + 1), " " * MAX_PLAN_BYTES + "{}"] +) +def test_plan_reader_enforces_bytes_before_json_parsing(tmp_path, payload): + write_plan(tmp_path, payload) + with pytest.raises(RuntimePlanError, match="exceeds"): + load_runtime_plan(tmp_path, ExecutionDeclaration()) + + +@pytest.mark.parametrize("number", ["NaN", "Infinity", "-Infinity", "1e10000"]) +def test_nonfinite_json_is_rejected_before_pydantic_coercion(tmp_path, number): + payload = json.dumps(plan_data()).replace( + '"increment": 1', f'"increment": {number}' + ) + write_plan(tmp_path, payload) + with pytest.raises(RuntimePlanError, match="finite"): + load_runtime_plan(tmp_path, ExecutionDeclaration()) + + +def test_duplicate_json_fields_are_not_ambiguous(tmp_path): + write_plan(tmp_path, '{"plan_schema_version":"1","plan_schema_version":"2"}') + with pytest.raises(RuntimePlanError, match="duplicate"): + load_runtime_plan(tmp_path, ExecutionDeclaration()) + + +@pytest.mark.parametrize("override", ["episode_id", "seed"]) +def test_reset_options_cannot_replace_reproducibility_inputs(override): + data = plan_data() + data["reset"]["options"] = {override: "other"} + with pytest.raises(ValidationError, match="cannot override"): + RuntimePlan.model_validate(data) + + +@pytest.mark.parametrize("seed", [True, "42", -1, 2**32]) +def test_reset_seed_is_explicit_bounded_integer(seed): + data = plan_data() + data["reset"]["seed"] = seed + with pytest.raises(ValidationError): + RuntimePlan.model_validate(data) + + +@pytest.mark.parametrize("actions", [[], [{}] * 101, [{"callback": object()}]]) +def test_actions_are_bounded_nonempty_json_data(actions): + data = plan_data() + data["actions"] = actions + with pytest.raises(ValidationError): + RuntimePlan.model_validate(data) + + +def test_input_cannot_override_manifest_capabilities(): + data = plan_data() + data["capabilities"] = {"set_state": True} + with pytest.raises(ValidationError, match="Extra inputs"): + RuntimePlan.model_validate(data) + + +def test_nested_inputs_are_bounded(): + data = plan_data() + nested = {} + for _ in range(40): + nested = {"child": nested} + data["actions"] = [nested] + with pytest.raises(ValidationError, match="nesting"): + RuntimePlan.model_validate(data) + + +def test_direct_plan_construction_also_has_size_bound(): + data = plan_data() + data["actions"] = [{"text": "x" * MAX_PLAN_BYTES}] + with pytest.raises(ValidationError, match="exceeds"): + RuntimePlan.model_validate(data) + + +def test_parser_opts_into_v2_only_when_execution_is_present(tmp_path): + source = yaml.safe_load((FIXTURES / "served_min_pass" / "openenv.yaml").read_text()) + path = tmp_path / "openenv.yaml" + path.write_text(yaml.safe_dump(source)) + v1 = OpenEnvYamlParser().parse(tmp_path) + assert type(v1) is NormalizedManifest + source["validation"]["execution"] = {"kind": "openenv_ws"} + path.write_text(yaml.safe_dump(source)) + v2 = OpenEnvYamlParser().parse(tmp_path) + assert type(v2) is NormalizedManifestV2 + assert v2.manifest_schema_version == "2" + assert NormalizedManifestV2.model_validate_json(v2.model_dump_json()) == v2 + assert v2.model_dump( + exclude={"execution", "manifest_schema_version"} + ) == v1.model_dump(exclude={"manifest_schema_version"}) + + +@pytest.mark.parametrize( + "section,field,value", + [ + ("reward", "range", [0.0, float("inf")]), + ("reward", "range", [float("-inf"), 1.0]), + ("reward", "floor_margin", float("inf")), + ("reward", "oracle_tolerance", float("inf")), + ("reward", "variance_tolerance", float("inf")), + ("resources", "cpu", float("inf")), + ("resources", "episode_timeout_s", float("inf")), + ], +) +def test_v2_rejects_nonfinite_declarations_without_changing_v1(section, field, value): + data = load_fixture_manifest("served_min_pass") + data[section][field] = value + assert NormalizedManifest.model_validate(data).manifest_schema_version == "1" + data.update(manifest_schema_version="2", execution={}) + with pytest.raises(ValidationError, match="numeric declarations must be finite"): + NormalizedManifestV2.model_validate(data) + + +def test_v2_rejects_nonfinite_judge_parameters(): + data = load_fixture_manifest("served_min_pass") + data.update( + manifest_schema_version="2", + execution={}, + judge={ + "model": "judge", + "version": "1", + "params": {"temperature": float("nan")}, + }, + ) + data["capabilities"]["llm_judged"] = True + data["reward"]["variance_tolerance"] = 0.1 + with pytest.raises(ValidationError, match="numeric declarations must be finite"): + NormalizedManifestV2.model_validate(data) + + +def test_null_execution_is_not_silently_downgraded(tmp_path): + source = yaml.safe_load((FIXTURES / "served_min_pass" / "openenv.yaml").read_text()) + source["validation"]["execution"] = None + (tmp_path / "openenv.yaml").write_text(yaml.safe_dump(source)) + with pytest.raises(ManifestError, match="execution"): + OpenEnvYamlParser().parse(tmp_path) + + +def test_static_validation_preserves_v2_execution_in_versioned_report(tmp_path): + source = yaml.safe_load((FIXTURES / "served_min_pass" / "openenv.yaml").read_text()) + source["validation"]["execution"] = {"kind": "openenv_ws"} + (tmp_path / "openenv.yaml").write_text(yaml.safe_dump(source)) + report = run_validation(tmp_path, max_level=Level.STATIC) + payload = json.loads(report.model_dump_json()) + assert payload["report_schema_version"] == "2" + assert payload["manifest"]["execution"]["probe_path"] == "validation/runtime.json" + assert ValidationReportV2.model_validate_json(report.model_dump_json()) == report + from jsonschema import validate + + schema_path = ( + FIXTURES.parents[2] / "src/openenv/validation/schemas/report-v2.schema.json" + ) + validate(payload, json.loads(schema_path.read_text())) + + +@pytest.mark.parametrize("version", [None, "1", "2"]) +def test_v2_report_roundtrips_null_and_both_manifest_versions(version): + manifest = None + if version is not None: + data = load_fixture_manifest("served_min_pass") + data["manifest_schema_version"] = version + model = NormalizedManifest + if version == "2": + data["execution"] = {} + model = NormalizedManifestV2 + manifest = model.model_validate(data) + report = ValidationReportV2( + report_schema_version="2", + target="subject", + source_digest="0" * 64, + signature=SignatureKind.OPENENV_SERVED, + manifest=manifest, + policy_version="v2", + lane=Lane.LOCAL, + levels_run=[Level.STATIC], + results=[], + verdict=Verdict.WARN, + ) + roundtrip = ValidationReportV2.model_validate_json(report.model_dump_json()) + assert roundtrip == report + assert type(roundtrip.manifest) is type(manifest) + with pytest.raises(ValidationError): + ValidationReport.model_validate_json(report.model_dump_json()) + + +def launch_data(): + return { + "image_ref": "sha256:" + "a" * 64, + "run_id": "validation-contract-test", + "resources": ResourceDeclaration( + cpu=1, memory_mb=256, disk_mb=64, episode_timeout_s=30 + ), + "network": NetworkPolicy(), + } + + +@pytest.mark.parametrize( + "image", ["probe:latest", "sha256:abc", "repo@sha256:" + "z" * 64] +) +def test_launch_rejects_mutable_or_invalid_image_identity(image): + data = launch_data() + data["image_ref"] = image + with pytest.raises(ValidationError): + LaunchSpec(**data) + + +def test_launch_defaults_do_not_inherit_host_credentials(monkeypatch): + monkeypatch.setenv("HF_TOKEN", "not-a-real-token") + spec = LaunchSpec(**launch_data()) + assert spec.env_vars == {} + assert spec.startup_timeout_s == 30 + with pytest.raises(ValidationError, match="frozen"): + spec.image_ref = "sha256:" + "b" * 64 + + +@pytest.mark.parametrize("env", [{"BAD=NAME": "x"}, {"NAME": "a\0b"}]) +def test_launch_rejects_unsafe_explicit_environment(env): + with pytest.raises(ValidationError): + LaunchSpec(**launch_data(), env_vars=env) + + +def test_launch_rejects_infinite_budget(): + data = launch_data() + data["resources"].cpu = float("inf") + with pytest.raises(ValidationError, match="finite"): + LaunchSpec(**data) + + +def test_evidence_records_preserve_raw_wire_and_are_immutable(): + exchange = WireExchange("step", '{"increment":1}', '{"reward":true}') + evidence = RuntimeEvidence(exchanges=(exchange,)) + assert evidence.exchanges[0].response_json == '{"reward":true}' + with pytest.raises(FrozenInstanceError): + exchange.response_json = '{"reward":1}' + with pytest.raises(FrozenInstanceError): + evidence.failure_reason = "altered" + + +def test_policy_v2_adds_only_startup_and_preserves_every_v1_rule(): + v1 = load_policy("v1") + v2 = load_policy("v2") + assert v2.bounds == v1.bounds + assert [ + entry for entry in v2.entries if entry.check_id != "runtime.startup" + ] == v1.entries + startup = v2.entries_for_lane(Lane.LOCAL)["runtime.startup"] + assert startup.level is Level.RUNTIME + assert startup.severity.value == "fail" + + +def test_acceptance_catalog_covers_exact_policy_runtime_inventory(): + catalog = json.loads((FIXTURES / "runtime" / "cases.json").read_text()) + assert catalog["catalog_schema_version"] == "1" + checks = catalog["checks"] + assert len({check["check_id"] for check in checks}) == len(checks) + expected = { + check_id + for check_id, entry in load_policy("v2").entries_for_lane(Lane.LOCAL).items() + if entry.level is Level.RUNTIME + } + assert {check["check_id"] for check in checks} == expected + cases = {case["case_id"]: case for case in catalog["cases"]} + for check in checks: + for case_id, status in ( + (check["positive_case"], "pass"), + (check["negative_case"], "fail"), + ): + assert cases[case_id]["expected"][check["check_id"]] == status + assert cases[case_id]["evidence_predicate"] diff --git a/tests/test_validation/test_schema_sync.py b/tests/test_validation/test_schema_sync.py index 5b6598fa91..9b30282e30 100644 --- a/tests/test_validation/test_schema_sync.py +++ b/tests/test_validation/test_schema_sync.py @@ -2,8 +2,9 @@ from pathlib import Path import pytest -from openenv.validation.manifest import NormalizedManifest -from openenv.validation.report import ValidationReport +from openenv.validation.manifest import NormalizedManifest, NormalizedManifestV2 +from openenv.validation.report import ValidationReport, ValidationReportV2 +from openenv.validation.runtime.contracts import RuntimePlan SCHEMAS_DIR = ( Path(__file__).parent.parent.parent / "src" / "openenv" / "validation" / "schemas" @@ -12,6 +13,9 @@ EXPORTS = { "manifest.schema.json": NormalizedManifest, "report.schema.json": ValidationReport, + "manifest-v2.schema.json": NormalizedManifestV2, + "report-v2.schema.json": ValidationReportV2, + "runtime-plan.schema.json": RuntimePlan, }