From 7d3480cf8bb49a43792eeca2f437890c21e6454f Mon Sep 17 00:00:00 2001 From: sneakocom <192013763+sneakocom@users.noreply.github.com> Date: Mon, 31 Aug 2026 16:25:00 +0000 Subject: [PATCH] research: record AV-EXP-003 boundary repair --- PROGRAM_STATUS.md | 4 +- program/experiments.json | 111 ++++++++++++++++++++++++++++++++- program/registry.json | 28 ++++----- tests/test_validate_program.py | 33 +++++++++- 4 files changed, 157 insertions(+), 19 deletions(-) diff --git a/PROGRAM_STATUS.md b/PROGRAM_STATUS.md index 82aff81..ae7db81 100644 --- a/PROGRAM_STATUS.md +++ b/PROGRAM_STATUS.md @@ -4,7 +4,7 @@ **Coverage: 21/21 expected repositories; duplicates: 0.** -Last verified: `2026-08-31T12:52:33Z`. HEADs are the verified default-branch revisions, not an assumption about later changes. +Last verified: `2026-08-31T16:17:40Z`. HEADs are the verified default-branch revisions, not an assumption about later changes. ## Portfolio totals @@ -43,7 +43,7 @@ Program state totals: active 5; waiting 16; complete 0. | 15 | [agent-recovery-policy](https://github.com/opsle/agent-recovery-policy) | concept | `1b733a111e26` | `THEORY` | [none; placeholder source directory only](https://github.com/opsle/agent-recovery-policy/blob/1b733a111e26e0a409fee3b96f627048531daefe/THEORY.md); placeholder only; no automated tests | No shared failure schema, attempt ledger, route evaluator, or comparative fixture set. | After decision evidence and route schemas stabilize, define same-failure convergence on synthetic failures. | `agent-routing-policy`, `agent-state-ledger`, `decision-evidence-protocol` | waiting | | 16 | [ephemeral-agent-workers](https://github.com/opsle/ephemeral-agent-workers) | concept | `ad96fcfdfac0` | `THEORY` | [none; placeholder source directory only](https://github.com/opsle/ephemeral-agent-workers/blob/ad96fcfdfac06d340b5e96d369634980cee78ef4/THEORY.md); placeholder only; no automated tests | Portable authority, claim, and handoff contracts are not ready; no safe synthetic containment harness exists. | Wait for prerequisite contracts, then define a fake worker adapter and destruction receipt without infrastructure changes. | `agent-execution-authorization`, `agent-resource-claims`, `verifiable-agent-handoff` | waiting | | 17 | [gearbox](https://github.com/opsle/gearbox) | concept | `f3fab9f292cf` | `PROTOTYPED` | [provider-free Python reference core with strict authority-policy admission, exact deterministic argv execution, content-addressed staged helper context, injected one-shot helper transport, passive process waiting, compact results, raw-artifact accounting, fail-closed budgets, and Visible Value receipts](https://github.com/opsle/gearbox/blob/f3fab9f292cf4eabd7200615d444f98881f57d55/src/opsle_gearbox/core.py); 19 of 19 provider-free automated tests passed locally, in PR #1 CI, and in final-main CI; ruff, shellcheck, actionlint, gitleaks, wheel build, receipt validation, and public raw-locator/hash checks passed | A production-quality bounded helper transport, independently verified isolation and termination, full Context Firewall integration, and a frozen comparative benchmark remain missing. | Freeze a provider-free deterministic-versus-direct baseline and helper-transport conformance corpus before considering any live model/provider run. | `context-firewall`, `decision-evidence-protocol`, `agent-trajectory-profiler`, `agent-routing-policy`, `agent-execution-authorization` | waiting | -| 18 | [affected-verification](https://github.com/opsle/affected-verification) | concept | `3ff41688dded` | `VERIFIED` | [dependency-free Node.js deterministic planner plus benchmark-only Git/catalog/source-graph adapters for AV-EXP-001 Vitest and AV-EXP-002 Python/pytest-testmon, identity-bound SHADOW result validation, complete frozen-oracle harnesses, explainable skip records, fail-closed uncertainty handling, and opsle.value-receipt.v1 telemetry](https://github.com/opsle/affected-verification/blob/3ff41688dded6e96e65da7cc44fe2608cf86d073/benchmark/av-exp-002/REPORT.md); 86 of 86 automated tests, 15 of 15 conformance scenarios, and 7 of 7 determinism checks passed locally, in PR #3 CI, and from a fresh detached worktree at exact main; the AV-EXP-002 result validator also passed at exact main, and invalid-state coverage includes Python/toolchain, target, patch, catalog, selector state/version/output, dynamic/conftest uncertainty, baseline, oracle, scenario, skip, policy, trust-state, and result tampering | AV-EXP-002 observed a false targeted-sufficiency claim for runtime/subprocess import behavior; historical real-change replay, a production-quality evidence adapter, and independent qualifying replication also remain missing. AV remains OBSERVE/SHADOW and no TRUSTED_BOUNDED change class is authorized. | Preregister and execute a selection-miss repair for AV2-006 that represents runtime/subprocess import uncertainty without changing the preserved AV-EXP-002 result. | — | active | +| 18 | [affected-verification](https://github.com/opsle/affected-verification) | concept | `97f490a67337` | `VERIFIED` | [dependency-free Node.js deterministic plan-v2 planner with check-level dependency-completeness states, mechanism and boundary evidence, check-local fail-closed forced selection, explainable skips, bounded deterministic Python boundary inspection, identity-bound SHADOW validation, frozen-oracle repair/replay harnesses, and opsle.value-receipt.v1 safety-cost telemetry](https://github.com/opsle/affected-verification/blob/97f490a67337552fee25757266f3dc034660dca0/benchmark/av-exp-003/REPORT.md); 107 of 107 automated tests, 15 of 15 conformance scenarios, and 10 of 10 determinism checks passed locally, in PR #4 CI, in exact final-main CI, and from a fresh detached worktree at 97f490a67337552fee25757266f3dc034660dca0; the AV-EXP-003 verifier and full deterministic repair reproduction also passed with identical result and regression-matrix identities | The bounded repair does not close opaque boundaries or constitute dynamic analysis; historical real-change replay, a production-quality evidence adapter, and independent qualifying replication remain missing. AV remains OBSERVE/SHADOW and no TRUSTED_BOUNDED change class is authorized. | Preregister a bounded child-process import-tracing evidence-provider experiment to test whether selected opaque checks can regain precision without weakening the fail-closed rule. | — | active | | 19 | [research](https://github.com/opsle/research) | program infrastructure | `9ee43197880c` | `PROTOTYPED` | [authoritative 21-repository ledger, machine-readable 18-concept theory registry including Affected Verification, canonical theory map, normative Visible Value controls, and provider-free EXP-001 benchmark, launch, one-block coordinator, external four-label LIVE_PROVIDER_RUN authorization, and current catalogue/pricing preflight artifacts with six content-addressed tasks, deterministic oracle, four arm contracts, sealed blinded allocation, exact subject configuration and adapter, exact authorization admission, private boundaries, receipts, mutation tests, and integrity CI](program/THEORY_MAP.md); 88 of 88 repository tests pass locally after deliberate migration to the 21-repository, 18-concept anti-forgetting set, including 13 authorization validations and two byte-identical replays; generated status and registry validation pass | The exact live authorization set remains unconsumed and unreleased, account-specific API entitlement is unverified under the zero-provider-call policy, no immutable dated model snapshot is documented, and the program has no canonical measured concept experiment. | Independently review and release the provider-free live-authorization and catalogue/pricing preflight; do not consume authorization or launch a provider/model subject. | — | active | | 20 | [site](https://github.com/opsle/site) | program infrastructure | `28ad65be4750` | `PROTOTYPED` | [React/Vinext source implementation with content routes](https://github.com/opsle/site/blob/28ad65be4750dc849976fbf5c9eae9501c6bbb25/README.md); automated build/render tests present; not rerun because this reconciliation kept other repositories read-only | Wait for validated registry data and measured research; deployment requires separate authorization. | After registry merge, add a read-only registry ingestion design without deploying the site. | `research` | waiting | | 21 | [.github](https://github.com/opsle/.github) | program infrastructure | `01c38e726db7` | `THEORY` | [documentation-only organization profile](https://github.com/opsle/.github/blob/01c38e726db7c3e45059d25fccce55e071e35938/profile/README.md); not applicable to current single Markdown profile; consistency is unverified | No mechanical registry consistency check exists in this repository. | After registry merge, design a read-only consistency check for organization-profile repository links. | `research` | waiting | diff --git a/program/experiments.json b/program/experiments.json index 5312ead..deb8d03 100644 --- a/program/experiments.json +++ b/program/experiments.json @@ -1,6 +1,6 @@ { "schema_version": 1, - "last_verified_at": "2026-08-31T12:52:33Z", + "last_verified_at": "2026-08-31T16:17:40Z", "experiments": [ { "id": "EXP-001", @@ -456,6 +456,115 @@ "lifecycle_impact": "REMAIN_VERIFIED: the run adds revision-bound falsification evidence and failure modes, but the observed selection miss and explicit run cap do not establish BENCHMARK_READY or EXPERIMENTED lifecycle promotion for the repository.", "next_task": "Preregister and execute a selection-miss repair for AV2-006 that represents runtime/subprocess import uncertainty without changing the preserved AV-EXP-002 result." }, + { + "id": "AV-EXP-003", + "title": "Opaque Dependency Boundary Repair", + "status": "RECORDED", + "hypothesis": "Affected Verification may skip a check only when available evidence defends completeness for every declared dependency mechanism capable of connecting the change to that check; an unmodeled, incomplete, unknown, or opaque boundary forces selection of that check unless identified evidence closes it.", + "participating_repositories": [ + "affected-verification", + "research" + ], + "roles": { + "primary": "affected-verification", + "expected_support": [ + "research" + ], + "potential_support": [] + }, + "baseline": "The permanently preserved AV-EXP-002 FAIL and exact frozen AV2-006 Click scenario, plus frozen AV-EXP-001/002 selector/full-oracle evidence and ten preregistered generalized repair cases.", + "experimental_arms": [ + "historical AV-EXP-001 and AV-EXP-002 AV selections preserved as the pre-repair baseline", + "Affected Verification plan v2 with check-level dependency completeness and deterministic boundary evidence", + "FULL frozen oracle for every repair case and preserved complete FULL evidence for every prior-corpus replay" + ], + "primary_metric": "Zero repaired selection misses, including selection of the exact known AV2-006 check for a generalized dependency-completeness reason.", + "secondary_metrics": [ + "prior and repaired test checks or executions selected in compatible exact units", + "additional verification introduced by dependency-safety policy", + "tests still skipped, scenarios remaining targeted, scenarios broadened, and FULL escalations", + "check-level boundary provenance and forced-selection explanations", + "non-test check differences by class" + ], + "correctness_gate": "FULL remains authoritative in SHADOW; every adversarial catalog is completely executed, every frozen prior relevant set is replayed, and any newly missed relevant check fails the experiment.", + "failure_classifications": [ + "known regression remains omitted", + "new regression miss", + "target-specific special case", + "malformed or unsupported boundary evidence accepted", + "incomplete full oracle", + "unmeasured or hidden precision cost", + "historical result mutation", + "unsafe trust promotion" + ], + "dataset_fixture_identity": "Affected Verification final 97f490a67337552fee25757266f3dc034660dca0; AV-EXP-003 preregistration 7aa4d13e42d6a547973d7f2a6b330821145cedc2; Click 36baa15ff831b939a22bc527cd76ce653ef6f66d; result sha256:03b2f7d6a380c84f6a1749531067cf8b87404c879f42380de8f07cce48251519; regression matrix sha256:7260c2d3476a6e78323e75d36c54c8409ea4cb18fa3a8f9a76b5533e1df08615.", + "model_provider_configuration": "NONE: one interactive Codex session used native shell, patch, Git, and Graphify facilities; no Codex children, external model/provider workloads, or production systems were used.", + "run_identities": [ + "sha256:03b2f7d6a380c84f6a1749531067cf8b87404c879f42380de8f07cce48251519", + "sha256:7260c2d3476a6e78323e75d36c54c8409ea4cb18fa3a8f9a76b5533e1df08615" + ], + "result_artifacts": [ + "https://github.com/opsle/affected-verification/blob/97f490a67337552fee25757266f3dc034660dca0/benchmark/av-exp-003/REPORT.md", + "https://github.com/opsle/affected-verification/blob/97f490a67337552fee25757266f3dc034660dca0/benchmark/av-exp-003/results-v1/summary.json", + "https://github.com/opsle/affected-verification/blob/97f490a67337552fee25757266f3dc034660dca0/benchmark/av-exp-003/results-v1/repair-regression-matrix.json", + "https://github.com/opsle/affected-verification/blob/97f490a67337552fee25757266f3dc034660dca0/benchmark/av-exp-003/results-v1/boundary-evidence.json", + "https://github.com/opsle/affected-verification/blob/3ff41688dded6e96e65da7cc44fe2608cf86d073/benchmark/av-exp-002/REPORT.md" + ], + "target": { + "repository": "https://github.com/pallets/click.git", + "sha": "36baa15ff831b939a22bc527cd76ce653ef6f66d", + "license": "BSD-3-Clause" + }, + "preregistration": { + "commit_sha": "7aa4d13e42d6a547973d7f2a6b330821145cedc2", + "known_failure_outcome_disclosed": true, + "repaired_outcomes_observed_before_commit": false + }, + "benchmark_result": { + "affected_verification_main_sha": "97f490a67337552fee25757266f3dc034660dca0", + "result_identity": "sha256:03b2f7d6a380c84f6a1749531067cf8b87404c879f42380de8f07cce48251519", + "regression_matrix_identity": "sha256:7260c2d3476a6e78323e75d36c54c8409ea4cb18fa3a8f9a76b5533e1df08615", + "evidence_manifest_identity": "sha256:932292a4f47372cab963db85772bb3d6c1e4fa536edd7b99f9686b72afd7e93c", + "known_replay": { + "scenario_id": "AV2-006", + "check_id": "pytest:tests/test_imports.py::test_light_imports", + "historical_outcome": "MISS", + "repaired_outcome": "SELECTED", + "additional_test_executions_per_av_arm": 1 + }, + "adversarial_scenario_count": 10, + "adversarial_repaired_miss_count": 0, + "adversarial_prior_tests_selected": 10, + "adversarial_repaired_tests_selected": 17, + "adversarial_additional_tests_due_to_repair": 7, + "adversarial_tests_still_skipped": 13, + "adversarial_targeted_scenarios_retained": 10, + "adversarial_broadened_scenarios": 7, + "adversarial_full_escalations": 0, + "av_exp_001_new_misses": 0, + "av_exp_001_additional_test_executions": 0, + "av_exp_002_av_core_new_misses": 0, + "av_exp_002_av_core_additional_test_executions": 6, + "av_exp_002_with_selector_new_misses": 0, + "av_exp_002_with_selector_additional_test_executions": 7, + "non_test_check_differences": 0 + }, + "major_findings": [ + "Evidence-source agreement and evidence completeness are independent; static and native selector omission cannot prove irrelevance across an unmodeled boundary.", + "Check-local fail-closed selection repaired the known replay without global FULL broadening; all ten repair scenarios remained targeted.", + "The bounded Python inspector found two open subprocess/child-interpreter checks among 2,016 Click pytest checks and retained provenance without claiming it closes those boundaries.", + "AV-EXP-002 remains a permanent FAIL and no trust promotion follows from repairing its known miss." + ], + "replication_status": "SAME_HOST_DETERMINISTIC_REPLAY_ONLY", + "verdict": "PASS only for the preregistered defect-repair claim: the known AV2-006 check is selected, ten generalized cases have zero misses, no frozen prior AV-relevant check becomes newly missed, and exact precision cost is measured. This does not solve dynamic dependencies, prove general safety, or authorize trusted execution.", + "blockers": [ + "The static boundary inspector identifies but does not close child-process, runtime import, arbitrary plugin, or reflection boundaries.", + "Historical real-change replay, a production-quality evidence adapter, and independent qualifying replication remain missing.", + "Affected Verification remains OBSERVE/SHADOW; no TRUSTED_BOUNDED class is authorized." + ], + "lifecycle_impact": "REMAIN_VERIFIED: the run repairs and measures one defect under SHADOW but is explicitly capped below lifecycle or trust promotion.", + "next_task": "Preregister a bounded child-process import-tracing evidence-provider experiment to test whether selected opaque checks can regain precision without weakening the fail-closed rule." + }, { "id": "LEGACY-001", "title": "Graphify plus Antigravity semantic adapter integration observation", diff --git a/program/registry.json b/program/registry.json index 22a74e8..7e4f40d 100644 --- a/program/registry.json +++ b/program/registry.json @@ -38,7 +38,7 @@ }, "current_highest_priority_workstream": "Independently review and release the provider-free EXP-001 live-authorization and current catalogue/pricing preflight; do not consume authorization or launch any provider/model subject.", "recommended_next_execution": "In opsle/research, create and provider-free validate one exact four-label LIVE_PROVIDER_RUN authorization set plus a model catalogue/pricing preflight artifact; do not consume authorization or launch a provider/model subject.", - "last_verified_at": "2026-08-31T12:52:33Z", + "last_verified_at": "2026-08-31T16:17:40Z", "repositories": [ { "name": "agent-trajectory-profiler", @@ -554,31 +554,31 @@ "name": "affected-verification", "github_url": "https://github.com/opsle/affected-verification", "default_branch": "main", - "last_verified_head_sha": "3ff41688dded6e96e65da7cc44fe2608cf86d073", + "last_verified_head_sha": "97f490a67337552fee25757266f3dc034660dca0", "project_type": "concept", "purpose": "Select the smallest verification workload whose sufficiency can be defended from available change-impact, dependency, coverage, policy, and risk evidence.", "lifecycle_stage": "VERIFIED", - "implementation_status": "dependency-free Node.js deterministic planner plus benchmark-only Git/catalog/source-graph adapters for AV-EXP-001 Vitest and AV-EXP-002 Python/pytest-testmon, identity-bound SHADOW result validation, complete frozen-oracle harnesses, explainable skip records, fail-closed uncertainty handling, and opsle.value-receipt.v1 telemetry", + "implementation_status": "dependency-free Node.js deterministic plan-v2 planner with check-level dependency-completeness states, mechanism and boundary evidence, check-local fail-closed forced selection, explainable skips, bounded deterministic Python boundary inspection, identity-bound SHADOW validation, frozen-oracle repair/replay harnesses, and opsle.value-receipt.v1 safety-cost telemetry", "implementation_requirement": "A runnable deterministic planner, input validator, plan contract, conformance fixtures, and shadow classifier are sufficient for the prototype gate; real adapters and comparative evidence are later gates.", - "specification_status": "versioned input and opsle.affected-verification.plan.v1 contracts define catalog entries, evidence providers, policy matching, selection and skip reasons, provenance, explicit sufficiency/uncertainty/escalation states, failure behavior, Visible Value limits, and shadow observations", - "test_status": "86 of 86 automated tests, 15 of 15 conformance scenarios, and 7 of 7 determinism checks passed locally, in PR #3 CI, and from a fresh detached worktree at exact main; the AV-EXP-002 result validator also passed at exact main, and invalid-state coverage includes Python/toolchain, target, patch, catalog, selector state/version/output, dynamic/conftest uncertainty, baseline, oracle, scenario, skip, policy, trust-state, and result tampering", - "benchmark_status": "AV-EXP-002 preregistered and recorded an eleven-scenario Click 36baa15ff831b939a22bc527cd76ce653ef6f66d SHADOW calibration with pytest-testmon 2.2.0, a 2,024-check full catalog, three stable clean baselines, four arms, fail-closed attacks, individual miss classification, raw evidence, and cross-experiment normalization; FULL selected 87/87 relevant checks, ECOSYSTEM_SELECTOR 77/87, and both AV arms 86/87, with the same runtime/subprocess import test omitted in AV2-006", - "measured_experiment_status": "AV-EXP-001 and AV-EXP-002 RECORDED; exact compatible-unit plan counts and observed runtime evidence exist for pinned JavaScript/Vitest and Python/pytest-testmon targets; AV-EXP-002 falsified perfect AV recall on its frozen corpus with one miss in each AV arm; zero provider/model runs", - "reproducibility_status": "the public AV-EXP-002 bounded harness and result verifier reproduce pinned target validation, locked environment, catalog, selector baseline, scenarios, four shadow arms, FULL oracle, receipts, semantic result, and aggregate report; the published bundle and exact merged SHA were independently validated on this host, but no independent qualifying replication exists", + "specification_status": "versioned input-v2 and opsle.affected-verification.plan.v2 contracts define check-level dependency mechanisms, completeness states, opaque-boundary provenance, forced selection, evidence coverage, policy matching, selection and skip reasons, deterministic identities, Visible Value safety additions, and SHADOW observations; plan v1 remains immutable historical evidence", + "test_status": "107 of 107 automated tests, 15 of 15 conformance scenarios, and 10 of 10 determinism checks passed locally, in PR #4 CI, in exact final-main CI, and from a fresh detached worktree at 97f490a67337552fee25757266f3dc034660dca0; the AV-EXP-003 verifier and full deterministic repair reproduction also passed with identical result and regression-matrix identities", + "benchmark_status": "AV-EXP-003 preregistered and recorded an opaque-boundary SHADOW repair: the permanently preserved AV-EXP-002 AV2-006 miss is selected in both repaired AV arms for generalized subprocess/child-interpreter completeness evidence; ten adversarial cases have zero misses, 7 added checks, 13 checks still skipped, 10/10 targeted scenarios, and zero FULL escalations; frozen AV-EXP-001 adds 0 test executions and has 0 repaired misses, while AV-EXP-002 adds 6 AV_CORE and 7 AV_WITH_SELECTOR executions and has 0 repaired misses", + "measured_experiment_status": "AV-EXP-001, AV-EXP-002, and AV-EXP-003 RECORDED; AV-EXP-002 permanently remains FAIL with one miss in each AV arm; AV-EXP-003 result sha256:03b2f7d6a380c84f6a1749531067cf8b87404c879f42380de8f07cce48251519 and regression matrix sha256:7260c2d3476a6e78323e75d36c54c8409ea4cb18fa3a8f9a76b5533e1df08615 measure repair selection and exact precision cost; zero provider/model runs", + "reproducibility_status": "the public AV-EXP-003 harness deterministically replays frozen AV-EXP-001/002 evidence, reruns ten adversarial full-oracle SHADOW cases, inspects the pinned Click source, validates ten Visible Value receipts, and reproduces identical result and regression-matrix identities from a fresh exact-main worktree on this host; no independent qualifying replication exists", "documentation_status": "public canonical definition, normative specification, source-linked prior-art audit, architecture and independence boundaries, verification catalog and policy semantics, trust ramp, benchmark plan, limitations, usage, security, schema, and fixtures are present", "site_publication_status": "GitHub documentation only; no evidence-backed site publication", - "known_limitations": ["Two pinned repositories and ecosystems have synthetic shadow evidence, but the adapters remain benchmark-only. AV-EXP-002 exposed a runtime/subprocess import dependency miss that both the static Python graph and pytest-testmon omitted. There is no historical real-change replay, production-quality adapter, independent qualifying replication, general safety proof, correctness equivalence, causal time or cost saving, or production trust."], + "known_limitations": ["AV-EXP-002 permanently remains a FAIL. AV-EXP-003 repairs the known skip in frozen replay but its bounded static inspector does not trace child-process imports, close arbitrary plugin/reflection behavior, solve dynamic Python dependencies, provide a production adapter, add historical real-change evidence, establish independent replication, prove general safety or correctness equivalence, claim causal savings, or authorize production trust."], "dependencies": [], "dependents": [], - "active_experiment_ids": ["AV-EXP-001", "AV-EXP-002"], - "blockers": ["AV-EXP-002 observed a false targeted-sufficiency claim for runtime/subprocess import behavior; historical real-change replay, a production-quality evidence adapter, and independent qualifying replication also remain missing. AV remains OBSERVE/SHADOW and no TRUSTED_BOUNDED change class is authorized."], - "next_task": "Preregister and execute a selection-miss repair for AV2-006 that represents runtime/subprocess import uncertainty without changing the preserved AV-EXP-002 result.", - "evidence": ["https://github.com/opsle/affected-verification/blob/3ff41688dded6e96e65da7cc44fe2608cf86d073/benchmark/av-exp-002/REPORT.md", "https://github.com/opsle/affected-verification/blob/f8a183c460535f3352fad2fb4990b0c54818d623/benchmark/av-exp-002/preregistration-v1/preregistration.json", "https://github.com/opsle/affected-verification/blob/3ff41688dded6e96e65da7cc44fe2608cf86d073/benchmark/av-exp-002/results-v1/summary.json", "https://github.com/opsle/affected-verification/blob/3ff41688dded6e96e65da7cc44fe2608cf86d073/benchmark/av-exp-002/results-v1/cross-experiment.json", "https://github.com/opsle/affected-verification/blob/3ff41688dded6e96e65da7cc44fe2608cf86d073/benchmark/av-exp-002/results-v1/evidence-manifest.json", "https://github.com/opsle/affected-verification/blob/3ff41688dded6e96e65da7cc44fe2608cf86d073/benchmark/av-exp-002/verify-results.mjs", "https://github.com/opsle/affected-verification/pull/3", "https://github.com/opsle/affected-verification/actions/runs/33393661890", "https://github.com/opsle/affected-verification/blob/641aee9d29a89e2a8819f00817ccee8e5d234dcb/benchmark/av-exp-001/REPORT.md"], + "active_experiment_ids": ["AV-EXP-001", "AV-EXP-002", "AV-EXP-003"], + "blockers": ["The bounded repair does not close opaque boundaries or constitute dynamic analysis; historical real-change replay, a production-quality evidence adapter, and independent qualifying replication remain missing. AV remains OBSERVE/SHADOW and no TRUSTED_BOUNDED change class is authorized."], + "next_task": "Preregister a bounded child-process import-tracing evidence-provider experiment to test whether selected opaque checks can regain precision without weakening the fail-closed rule.", + "evidence": ["https://github.com/opsle/affected-verification/blob/97f490a67337552fee25757266f3dc034660dca0/benchmark/av-exp-003/REPORT.md", "https://github.com/opsle/affected-verification/blob/7aa4d13e42d6a547973d7f2a6b330821145cedc2/benchmark/av-exp-003/preregistration-v1/preregistration.json", "https://github.com/opsle/affected-verification/blob/97f490a67337552fee25757266f3dc034660dca0/benchmark/av-exp-003/results-v1/summary.json", "https://github.com/opsle/affected-verification/blob/97f490a67337552fee25757266f3dc034660dca0/benchmark/av-exp-003/results-v1/repair-regression-matrix.json", "https://github.com/opsle/affected-verification/blob/97f490a67337552fee25757266f3dc034660dca0/benchmark/av-exp-003/results-v1/evidence-manifest.json", "https://github.com/opsle/affected-verification/blob/3ff41688dded6e96e65da7cc44fe2608cf86d073/benchmark/av-exp-002/REPORT.md", "https://github.com/opsle/affected-verification/pull/4", "https://github.com/opsle/affected-verification/actions/runs/33412942072"], "completion_criteria": ["Publish production evidence adapters and independently validate their completeness boundaries.", "Run correctness-first comparisons against full verification and native selectors with durable shadow miss evidence.", "Replicate bounded workload and safety claims on independent repositories or environments."], "completion_evidence": [], "completion_status": "INCOMPLETE", "program_state": "active", - "last_verified_at": "2026-08-31T12:52:33Z" + "last_verified_at": "2026-08-31T16:17:40Z" }, { "name": "research", diff --git a/tests/test_validate_program.py b/tests/test_validate_program.py index a5425bd..04bb53a 100644 --- a/tests/test_validate_program.py +++ b/tests/test_validate_program.py @@ -47,11 +47,11 @@ def test_affected_verification_is_the_twenty_first_repository(self): self.assertEqual(affected["lifecycle_stage"], "VERIFIED") self.assertEqual( affected["last_verified_head_sha"], - "3ff41688dded6e96e65da7cc44fe2608cf86d073", + "97f490a67337552fee25757266f3dc034660dca0", ) self.assertEqual( affected["active_experiment_ids"], - ["AV-EXP-001", "AV-EXP-002"], + ["AV-EXP-001", "AV-EXP-002", "AV-EXP-003"], ) experiment = next( @@ -85,6 +85,35 @@ def test_affected_verification_is_the_twenty_first_repository(self): experiment["benchmark_result"]["av_miss"]["check_id"], "pytest:tests/test_imports.py::test_light_imports", ) + self.assertIn("FAIL", experiment["verdict"]) + + experiment = next( + item for item in self.experiments["experiments"] + if item["id"] == "AV-EXP-003" + ) + self.assertEqual(experiment["status"], "RECORDED") + self.assertEqual( + experiment["benchmark_result"]["affected_verification_main_sha"], + "97f490a67337552fee25757266f3dc034660dca0", + ) + self.assertEqual( + experiment["benchmark_result"]["known_replay"]["historical_outcome"], + "MISS", + ) + self.assertEqual( + experiment["benchmark_result"]["known_replay"]["repaired_outcome"], + "SELECTED", + ) + self.assertEqual( + experiment["benchmark_result"]["av_exp_002_av_core_new_misses"], + 0, + ) + self.assertEqual( + experiment["benchmark_result"]["av_exp_002_av_core_additional_test_executions"], + 6, + ) + self.assertEqual(experiment["lifecycle_impact"].split(":", 1)[0], "REMAIN_VERIFIED") + self.assertTrue(any("OBSERVE/SHADOW" in item for item in experiment["blockers"])) def test_missing_expected_repository_fails(self): registry = copy.deepcopy(self.registry)