diff --git a/README.md b/README.md index e986459..7b43085 100644 --- a/README.md +++ b/README.md @@ -164,6 +164,36 @@ joint outcome. Cloud Run traversal through Interlock, a receipt-bound protected mutation, independently read back and correlated in Cloud Logging. +**Bounded operational utility (HAC-343)** — one frozen sixteen-scenario corpus +run through four coordination strategies. Exact counts, because the corpus is an +exhaustive enumeration rather than a sample: + +| Strategy | Hazards unsafe | Independent opportunities parallel | +| --- | --- | --- | +| Uncoordinated | 2/2 | 2/2 | +| Global lock | 0/2 | 0/2 | +| Per-target lock | 2/2 | 2/2 | +| Interlock | 0/2 | 2/2 | + +The per-target lock is a real lock, not a straw man: it serialized same-target +contention 2/2 and parallelised cross-target pairs 4/4, and still missed +cross-target hazards 2/2 — a composition hazard spanning two lock keys is +invisible to any per-key discipline. + +The safety is the evidence's, not the engine's. Removing the coupling evidence +reverses the decision: + +| Condition | Invalid outcomes | +| --- | --- | +| Interlock + coupling evidence present | 0/2 | +| Interlock + coupling evidence removed | 2/2 | + +Interlock is **not** 0% unsafe — it produced invalid joint states in both +ablation scenarios by design — and it is **not** "safer than locking": per-target +locking is correct for the hazard it addresses. Every figure is read from +[`experiments/hac-343/evidence/judge-export.json`](./experiments/hac-343/evidence/judge-export.json), +anchored at canonical result `7ede0f9`. + **Not claimed.** HAC-330 did not run on Google Cloud, and HAC-340 does not reproduce the 140/120 counterfactual there. Agent Runtime and Agent Gateway did not participate. Wrong-audience token rejection is controlled local parity @@ -173,9 +203,8 @@ evidence, not a cloud result. `ALLOW` is not `VERIFIED`; `OBSERVED` is not guarantee. No safety, security, verification or production-readiness guarantee. No fleet-scale readiness and no universal collision prevention. -Evaluation (HAC-319) is **not yet bound**: no SPR, precision, recall, -false-block or useful-concurrency number exists in this package, and none is -shown. +The broader evaluation (HAC-319) — precision, recall, fleet-scale behaviour — is +**not bound**. HAC-343 below is a bounded child of it, not a substitute. [`DISCLOSURE.md`](./DISCLOSURE.md) is the full provenance statement. diff --git a/media/hac-334/evidence/visual-model.json b/media/hac-334/evidence/visual-model.json index 5e88393..c07285b 100644 --- a/media/hac-334/evidence/visual-model.json +++ b/media/hac-334/evidence/visual-model.json @@ -857,20 +857,18 @@ ], "composition": { "state": "evaluation-not-yet-bound", - "message": "Evaluation not yet bound.", + "message": "Three-regime evaluation not yet bound.", "regimes": [ "Regime 1", "Regime 2", "Regime 3" ], "metricsWithheld": [ - "SPR", "precision", "recall", - "false-block rate", - "useful-concurrency" + "fleet-scale behaviour" ], - "rule": "Labels only. No value, no mark, no proportional geometry until HAC-319 supplies a frozen evaluation packet.", + "rule": "Labels only. No value, no mark, no proportional geometry until HAC-319 supplies a frozen three-regime evaluation packet.", "marks": [] }, "exports": [ diff --git a/media/hac-334/masters/IL-DIAG-013-evaluation-shell.svg b/media/hac-334/masters/IL-DIAG-013-evaluation-shell.svg index 45fb860..fe3c2dc 100644 --- a/media/hac-334/masters/IL-DIAG-013-evaluation-shell.svg +++ b/media/hac-334/masters/IL-DIAG-013-evaluation-shell.svg @@ -1 +1 @@ -IL-DIAG-013 HAC-319 anti-global-mutex evaluation shellDESIGN SHELL - AWAITING HAC-319. Frozen evidence: none - evaluation not yet bound Non-claim: no value, no mark and no proportional geometry until a frozen evaluation packet existsDESIGN SHELL - AWAITING HAC-319IL-DIAG-013HAC-319 anti-global-mutex evaluation shellEVALUATION NOT YET BOUND.Regime 1NOT BOUNDRegime 2NOT BOUNDRegime 3NOT BOUNDMETRICS WITHHELDSPR precision recall false-block rate useful-concurrencyLabels only. No value, no mark, no proportional geometry until HAC-319 supplies a frozen evaluation packet.Frozen evidence: none - evaluation not yet boundNon-claim: no value, no mark and no proportional geometry until a frozen evaluation packet exists +IL-DIAG-013 HAC-319 anti-global-mutex evaluation shellDESIGN SHELL - AWAITING HAC-319. Frozen evidence: none - evaluation not yet bound Non-claim: no value, no mark and no proportional geometry until a frozen evaluation packet existsDESIGN SHELL - AWAITING HAC-319IL-DIAG-013HAC-319 anti-global-mutex evaluation shellTHREE-REGIME EVALUATION NOT YET BOUND.Regime 1NOT BOUNDRegime 2NOT BOUNDRegime 3NOT BOUNDMETRICS WITHHELDprecision recall fleet-scale behaviourLabels only. No value, no mark, no proportional geometry until HAC-319 supplies a frozen three-regime evaluation packet.Frozen evidence: none - evaluation not yet boundNon-claim: no value, no mark and no proportional geometry until a frozen evaluation packet exists diff --git a/media/hac-335/README.md b/media/hac-335/README.md index f56fd0b..86dc21a 100644 --- a/media/hac-335/README.md +++ b/media/hac-335/README.md @@ -133,8 +133,11 @@ fails. into `evidencePublicationSha`; - `sourcePacketSha256` is never described as reader-recomputable; - no unevidenced deployment revision is named; -- HAC-319 is unbound: no metric, no value, and `IL-DIAG-013` is out of the - judge-facing registry and sequence with its seam recorded; +- every HAC-343 figure matches a frozen `display` value in the judge export, + Panel 1 never travels without Panel 2, the A3 credibility strip is present, + and every `mustNotClaim` reading is refused; +- HAC-319 proper stays unbound: no precision, recall or fleet-scale value, and + `IL-DIAG-013` stays out of the judge-facing registry with its seam recorded; - every capture's proof class agrees with its URL and its semantic state; - every filename passes the frozen HAC-332 naming grammar; - no derivative is stale relative to its source, checked by digest and PNG @@ -147,13 +150,31 @@ fails. **33 negative cases** in `test/hac-335-package-gates.test.mjs` prove each of these still fails when violated. -## HAC-319 +## The evaluation: HAC-343 bound, HAC-319 still not -Not bound. No SPR, precision, recall, false-block or useful-concurrency value -appears anywhere in this package. `IL-DIAG-013` stays in the HAC-334 registry as -the reserved evaluation shell and is deliberately excluded here, recorded in -both the sequence and the registry with `seamPreserved: true`, so binding it -later is an addition rather than an excavation. +These are two different things and the package keeps them apart. + +**HAC-343 is bound.** The bounded four-arm evaluation has a frozen canonical +result at `7ede0f9`, and every judge-facing figure in this package is read from +`experiments/hac-343/evidence/judge-export.json` — never recalculated here. +HAC-335 authors no evaluation facts; it decides where the comparison sits in the +judge path and what it is allowed to claim. + +The gate enforces four properties that prose alone cannot hold: + +- every HAC-343 figure in judge-facing copy matches a frozen `display` value in + the export, so a number cannot drift or be invented; +- Panel 1 never appears without Panel 2 in the same file — the four-strategy + comparison and the evidence ablation travel together, because Panel 1 alone + reads as "Interlock is the safe one" and the export forbids that reading; +- the A3 credibility strip (`2/2`, `4/4`, `2/2`) is present wherever the + comparison is, so the per-target lock cannot be quietly reduced to a straw man; +- every entry in the export's own `mustNotClaim` list is checked against the + prose, so the forbidden readings fail the build rather than a review. + +**HAC-319 is still not bound.** Precision, recall and fleet-scale behaviour have +no frozen packet. HAC-343 is a bounded child of HAC-319, not a substitute, and +the package says so rather than letting the bound child imply the unbound parent. ## Still open diff --git a/media/hac-335/bin/build-registry.mjs b/media/hac-335/bin/build-registry.mjs index 9401db9..c20704d 100644 --- a/media/hac-335/bin/build-registry.mjs +++ b/media/hac-335/bin/build-registry.mjs @@ -435,9 +435,11 @@ const registry = { { assetId: 'IL-DIAG-013', reason: - 'HAC-319 evaluation is not bound. The asset stays in the HAC-334 registry as the reserved ' - + 'evaluation shell and is deliberately absent from every judge-facing surface here, so an ' - + 'unavailable evaluation cannot be read as a pending result. The integration seam is preserved.', + 'HAC-319 proper — precision, recall, fleet-scale behaviour — is not bound. The asset stays in ' + + 'the HAC-334 registry as the reserved shell for that evaluation and is deliberately absent ' + + 'from every judge-facing surface here, so an unavailable evaluation cannot be read as a ' + + 'pending result. The bounded HAC-343 four-arm evaluation IS bound and is rendered from its ' + + 'frozen judge export; it is a child of HAC-319, not a substitute. The seam is preserved.', seamPreserved: true, }, { diff --git a/media/hac-335/bin/verify-package.mjs b/media/hac-335/bin/verify-package.mjs index da13926..3dcc843 100644 --- a/media/hac-335/bin/verify-package.mjs +++ b/media/hac-335/bin/verify-package.mjs @@ -79,6 +79,10 @@ function loadContext(root) { sequence: readJson('media/hac-335/evidence/judge-sequence.json'), captures: readJson('media/hac-335/evidence/capture-manifest.json'), cockpit: readJson('media/hac-341/evidence/view-model.json'), + /* The sole source of every HAC-343 figure this package renders. Read here + so the gate compares prose against the frozen export rather than against + a number somebody typed twice. */ + judgeExport: readJson('experiments/hac-343/evidence/judge-export.json'), shots, prose, allProse: Object.values(prose).join('\n'), @@ -265,27 +269,129 @@ function checkRevisions({ cloud, prose, registry }, fail) { } } -/* -- 13/14. HAC-319 stays unbound and unrendered --------------------------- */ +/* -- 13/14. the HAC-343 evaluation is bound, and stays bounded ------------- */ + +/** + * HAC-343 went from "no packet exists" to a frozen canonical result, so the + * rule this gate enforces inverted. It used to prove the evaluation was absent. + * Absence is no longer the truth, and a gate that still enforced it would keep + * the package stating something false — which is exactly what it did until this + * check was rewritten. + * + * What replaces it is stricter, not looser. "No number" is trivially checkable; + * "every number is the frozen one, and none of them travels alone" is the + * property that actually protects a judge, and it needs the export in hand. + */ +function checkEvaluationBound({ prose, judgeExport, registry, sequence }, fail) { + const p1 = judgeExport.panel1.rows; + const cred = judgeExport.panel1.perTargetLockCredibility; + const p2 = judgeExport.panel2.rows; + + /* Every count the export blesses. A count-shaped token inside a file that + renders the comparison must be one of these, so a figure cannot be + mistyped, rounded, or quietly recomputed from the raw records. */ + const frozen = new Set([ + ...p1.flatMap((r) => [r.coupledUnsafe.display, r.safeParallelism.display]), + cred.serializedSameTargetContention.display, + cred.parallelisedCrossTarget.display, + cred.missedCrossTargetHazards.display, + ...p2.map((r) => r.invalidOutcomes.display), + judgeExport.provenance.matrix.display, + ]); + /* Counts this package already renders for other, separately gated evidence. + Listed rather than pattern-matched so adding one is a deliberate edit. */ + const OTHER_EVIDENCE = new Set([ + '24/24', '9/9', '3/3', '2/3', + /* HAC-330's counterfactual is written "the 140/120 counterfactual". It is a + pair of bound values from a different experiment, not a count, and + checkNumerals already holds it against the frozen arms. */ + '140/120', + ]); + + /* A file "renders the comparison" when it names every strategy the export + names. Anything less is prose mentioning an arm, not a comparison table. */ + const labels = p1.map((r) => r.label); + const rendersComparison = (text) => labels.every((l) => text.includes(l)); -function checkEvaluationUnbound({ prose, registry, sequence }, fail) { - const metrics = /\bSPR\b|useful[- ]concurrency|false[- ]block/gi; for (const [file, text] of Object.entries(prose)) { - for (const m of text.matchAll(metrics)) { - if (!disclaimed(text, m.index, /not|no |none|unbound|not yet bound|withheld/i, 200)) { - fail(`${file}: mentions a HAC-319 metric without marking it unbound`); + if (!rendersComparison(text)) continue; + + for (const m of text.matchAll(/\b\d+\/\d+\b/g)) { + if (!frozen.has(m[0]) && !OTHER_EVIDENCE.has(m[0])) { + fail(`${file}: renders ${m[0]}, which is not a frozen HAC-343 display value`); } - // A metric adjacent to a number is a rendered value, disclaimed or not. - if (new RegExp(String.raw`${m[0]}[^.\n]{0,24}\d`, 'i').test(text.slice(m.index))) { - fail(`${file}: a HAC-319 metric appears next to a numeric value`); + } + + /* Panel 1 alone reads as "Interlock is the safe one". The export forbids + that reading, and the only thing that refutes it is Panel 2 in the same + place a judge is already looking. */ + for (const row of p2) { + if (!text.includes(row.condition)) { + fail(`${file}: shows the four-strategy comparison without the evidence-ablation condition "${row.condition}"`); } } + + /* Without the strip, A3 is a straw man: a lock that missed the hazards and + is never shown to have locked anything. */ + for (const [name, fig] of [ + ['same-target contention serialized', cred.serializedSameTargetContention.display], + ['cross-target pairs parallelised', cred.parallelisedCrossTarget.display], + ['cross-target hazards missed', cred.missedCrossTargetHazards.display], + ]) { + if (!text.includes(fig)) { + fail(`${file}: shows the comparison without the A3 credibility figure for ${name} (${fig})`); + } + } + + if (!/sixteen|16[- ]scenario/i.test(text)) { + fail(`${file}: renders the comparison without stating the corpus it is bounded to`); + } } + + /* The export names the readings it must never produce. Each one is checked + against the assembled judge-facing copy, so a forbidden claim fails the + build rather than a reviewer's attention. */ + /* Every occurrence, not the first. These phrases legitimately appear in this + package as the negations the export requires ("Interlock is **not** 0% + unsafe"), so a check that stopped at the first match would find the + disclaimed one, pass, and never look at the undisclaimed claim below it. */ + const FORBIDDEN = [ + [/\b(0|zero)\s*%?\s*unsafe\b/gi, 'describes Interlock as 0% unsafe'], + [/safer than (locking|locks|a lock)/gi, 'claims Interlock is safer than locking'], + [/prevents (all )?(composition|collision)/gi, 'claims Interlock prevents composition hazards'], + [/statistical(ly)? significan|confidence interval|\bp\s*<\s*0\./gi, 'claims statistical significance'], + ]; + for (const [file, text] of Object.entries(prose)) { + for (const [re, why] of FORBIDDEN) { + for (const m of text.matchAll(re)) { + if (!disclaimed(text, m.index, /\*\*not\*\*|is not|are not|never|must not|cannot|no confidence interval|no interval|not a sample|exhaustive/i, 90)) { + fail(`${file}: ${why} — forbidden by judge-export mustNotClaim`); + } + } + } + } + + /* The statements this package used to make. They are false now, and a revert + that reintroduced one would otherwise pass every other check here. */ + const STALE = [ + [/no SPR[^.]{0,90}(exists|appears|is shown)/i, 'asserts no SPR value exists'], + [/evaluation is \*\*not yet bound\*\*/i, 'asserts the evaluation is not yet bound'], + [/HAC-343[^.]{0,40}not bound/i, 'asserts HAC-343 is not bound'], + ]; + for (const [file, text] of Object.entries(prose)) { + for (const [re, why] of STALE) { + if (re.test(text)) fail(`${file}: ${why}, contradicting the frozen HAC-343 result`); + } + } + + /* HAC-319 proper is still unbound, and the bound child must not be allowed to + imply the unbound parent. IL-DIAG-013 is HAC-319's reserved shell. */ if (registry.assets.some((a) => a.assetId === 'IL-DIAG-013')) { - fail('IL-DIAG-013 is in the judge-facing registry; the evaluation shell is not bound'); + fail('IL-DIAG-013 is in the judge-facing registry; HAC-319 proper is still unbound'); } for (const s of sequence.steps) { if ([s.primaryAsset, ...(s.supportingAssets || [])].includes('IL-DIAG-013')) { - fail(`judge sequence step ${s.stepId} points at the unbound evaluation shell`); + fail(`judge sequence step ${s.stepId} points at the unbound HAC-319 shell`); } } if (!sequence.excludedFromJudgePath.some((e) => e.assetId === 'IL-DIAG-013' && e.seamPreserved)) { @@ -550,7 +656,7 @@ const CHECKS = [ checkEvidenceUrls, checkWithheldEvidence, checkRevisions, - checkEvaluationUnbound, + checkEvaluationBound, checkCaptures, checkCaptureFreshness, checkNamingAndFreshness, @@ -597,5 +703,6 @@ if (invokedDirectly) { console.log(` judge sequence ${sequence.steps.length} steps, hero ${sequence.steps[0].primaryAsset}, reset at step ${resetStep.order}`); console.log(` registry ${registry.assets.length} assets, ${exportCount} exports, naming contract clean`); console.log(` claim ledger ${ledger.claims.length} claims, every cited id resolved`); - console.log(' proof classes separate, evidence links commit-pinned, HAC-319 unbound and unrendered'); + console.log(' proof classes separate, evidence links commit-pinned'); + console.log(' HAC-343 bound to the frozen judge export, panels adjacent; HAC-319 proper still unbound'); } diff --git a/media/hac-335/devpost/02-narrative.md b/media/hac-335/devpost/02-narrative.md index 8036396..634514f 100644 --- a/media/hac-335/devpost/02-narrative.md +++ b/media/hac-335/devpost/02-narrative.md @@ -108,8 +108,45 @@ No exactly-once, restart-safety or recovery guarantee. No safety, security, verification or production-readiness guarantee. No fleet-scale readiness, and no universal collision prevention. -Evaluation is **not yet bound**: no SPR, precision, recall, false-block or -useful-concurrency number exists in this submission, and none is shown. +## Compared with what? + +A bounded four-arm evaluation (HAC-343) runs one frozen sixteen-scenario corpus +through four mechanically distinct coordination strategies. Exact counts, not +percentages — the corpus is an exhaustive enumeration, not a sample. + +| Strategy | Hazards unsafe | Independent opportunities parallel | +| --- | --- | --- | +| Uncoordinated | 2/2 | 2/2 | +| Global lock | 0/2 | 0/2 | +| Per-target lock | 2/2 | 2/2 | +| Interlock | 0/2 | 2/2 | + +Global locking bought safety by eliminating concurrency. Per-target locking kept +the concurrency and missed both hazards. + +**Is that per-target lock credible?** It has to be, or the comparison is a straw +man. It serialized same-target contention 2/2, parallelised cross-target pairs +4/4, and still missed cross-target hazards 2/2. It locked exactly what a lock can +see; a composition hazard spanning two lock keys is invisible to any per-key +discipline. + +**Is the safety from the evidence, or from Interlock?** Remove the evidence and +find out: + +| Condition | Invalid outcomes | +| --- | --- | +| Interlock + coupling evidence present | 0/2 | +| Interlock + coupling evidence removed | 2/2 | + +The decision reverses. The safety is evidence-derived, not a property of the +engine. + +**What this is not.** Interlock is not 0% unsafe — it produced invalid joint +states in the two ablation scenarios by design. It is not "safer than locking": +per-target locking is correct for the hazard it addresses. The 0/2 is bounded to +the coupled scenarios of this corpus and must not be collapsed into a single +rate over all sixteen. No interval or significance is claimed, and no +exactly-once, restart-safety or production-readiness result was tested here. ## What is new here diff --git a/media/hac-335/devpost/06-limitations.md b/media/hac-335/devpost/06-limitations.md index 6ec35f2..edf0fd8 100644 --- a/media/hac-335/devpost/06-limitations.md +++ b/media/hac-335/devpost/06-limitations.md @@ -42,14 +42,31 @@ timeline links their events. - No universal collision prevention. - No complete co-change coupling recall. -## Not yet bound - -**Evaluation (HAC-319)** has no frozen packet. There is no SPR, precision, -recall, false-block rate or useful-concurrency number in this submission, and -none is shown — not as a value, not as a bar, not as proportional geometry. The -surface is reserved and labelled `EVALUATION NOT YET BOUND`; it is deliberately -kept out of the judge-facing sequence so that an unavailable evaluation cannot -read as a pending result. +## What the evaluation does and does not establish + +**The bounded evaluation (HAC-343)** is frozen and rerunnable. Its metric +definitions, corpus and execution semantics were each committed and tagged +before any result existed, and the canonical result is anchored at +`7ede0f9`. Every judge-facing figure is read from its deterministic judge +export rather than restated. + +What it does not establish: + +- **It is not a general result.** Sixteen scenarios across two fixture + families. Every count is a property of that corpus and does not extrapolate. +- **Interlock is not 0% unsafe.** It produced invalid joint states in both + evidence-ablation scenarios by design — that is the point of Panel 2. +- **It is not "safer than locking".** Per-target locking is correct for the + same-target hazard it addresses. Interlock is safe against a hazard class + per-key locking cannot see. Those are different statements. +- **No interval, no significance.** The corpus is an exhaustive deterministic + enumeration, not a sample, so no confidence interval is meaningful. +- **No lifecycle claim.** Exactly-once, restart safety, target-side atomicity + and production readiness were not tested and are not claimed. + +**The broader evaluation (HAC-319)** — precision, recall and fleet-scale +behaviour — remains unbound. HAC-343 is a bounded child of it, not a +substitute for it. ## Scale of the evidence diff --git a/media/hac-335/devpost/screenshot-order.json b/media/hac-335/devpost/screenshot-order.json index dd4af94..bbd5c59 100644 --- a/media/hac-335/devpost/screenshot-order.json +++ b/media/hac-335/devpost/screenshot-order.json @@ -95,7 +95,7 @@ "excluded": [ { "assetId": "IL-DIAG-013", - "reason": "HAC-319 evaluation is not bound. Excluded so an unavailable evaluation cannot read as a pending result." + "reason": "HAC-319 proper — precision, recall, fleet-scale behaviour — is not bound. The shell stays in the internal registry as EVALUATION NOT YET BOUND and is deliberately kept out of README, Devpost and video assets so no judge reads an unavailable evaluation as a pending result. The bounded HAC-343 four-arm evaluation IS bound and appears in the judge path, rendered from its frozen judge export." }, { "assetId": "IL-COCK-011", diff --git a/media/hac-335/evidence/asset-registry.json b/media/hac-335/evidence/asset-registry.json index 03192a2..cf9c1ce 100644 --- a/media/hac-335/evidence/asset-registry.json +++ b/media/hac-335/evidence/asset-registry.json @@ -42,7 +42,7 @@ "excluded": [ { "assetId": "IL-DIAG-013", - "reason": "HAC-319 evaluation is not bound. The asset stays in the HAC-334 registry as the reserved evaluation shell and is deliberately absent from every judge-facing surface here, so an unavailable evaluation cannot be read as a pending result. The integration seam is preserved.", + "reason": "HAC-319 proper — precision, recall, fleet-scale behaviour — is not bound. The asset stays in the HAC-334 registry as the reserved shell for that evaluation and is deliberately absent from every judge-facing surface here, so an unavailable evaluation cannot be read as a pending result. The bounded HAC-343 four-arm evaluation IS bound and is rendered from its frozen judge export; it is a child of HAC-319, not a substitute. The seam is preserved.", "seamPreserved": true }, { @@ -496,7 +496,7 @@ "canonicalMasterIssue": null, "authoredBy": "HAC-335", "capturedFrom": "HAC-341 merged cockpit", - "capturedFromSha": "c4654821b6e1a443ab4ef6218ddde4daaf1ee20c", + "capturedFromSha": "7d62f6fe169cf354a548447dae0ce1ea7e210a08", "sourceUrl": "/media/hac-341/cockpit.html?run=hac330-local&proof=local&state=run.local.treatment&static=1", "sourceFormat": "live surface (html)", "exportFormat": "png", @@ -540,7 +540,7 @@ "canonicalMasterIssue": null, "authoredBy": "HAC-335", "capturedFrom": "HAC-341 merged cockpit", - "capturedFromSha": "c4654821b6e1a443ab4ef6218ddde4daaf1ee20c", + "capturedFromSha": "7d62f6fe169cf354a548447dae0ce1ea7e210a08", "sourceUrl": "/media/hac-341/cockpit.html?run=hac330-local&proof=local&state=run.local.perturbed&static=1", "sourceFormat": "live surface (html)", "exportFormat": "png", @@ -584,7 +584,7 @@ "canonicalMasterIssue": null, "authoredBy": "HAC-335", "capturedFrom": "HAC-341 merged cockpit", - "capturedFromSha": "c4654821b6e1a443ab4ef6218ddde4daaf1ee20c", + "capturedFromSha": "7d62f6fe169cf354a548447dae0ce1ea7e210a08", "sourceUrl": "/media/hac-341/cockpit.html?run=hac340-cloud&proof=cloud&state=run.cloud.overview&static=1", "sourceFormat": "live surface (html)", "exportFormat": "png", @@ -644,7 +644,7 @@ "canonicalMasterIssue": null, "authoredBy": "HAC-335", "capturedFrom": "HAC-341 merged cockpit", - "capturedFromSha": "c4654821b6e1a443ab4ef6218ddde4daaf1ee20c", + "capturedFromSha": "7d62f6fe169cf354a548447dae0ce1ea7e210a08", "sourceUrl": "/media/hac-341/cockpit.html?run=hac340-cloud&proof=cloud&state=run.cloud.overview&static=1", "sourceFormat": "live surface (html)", "exportFormat": "png", diff --git a/media/hac-335/evidence/capture-manifest.json b/media/hac-335/evidence/capture-manifest.json index dc5617f..f31b9ce 100644 --- a/media/hac-335/evidence/capture-manifest.json +++ b/media/hac-335/evidence/capture-manifest.json @@ -2,11 +2,11 @@ "manifestId": "HAC-335-capture-manifest", "issue": "HAC-335", "generator": "media/hac-335/bin/capture-cockpit.mjs", - "capturedFromSha": "12969351d9c7d21288a601ec35c83d5a70156a3e", + "capturedFromSha": "7d62f6fe169cf354a548447dae0ce1ea7e210a08", "capturedSurface": "media/hac-341/cockpit.html (merged executable surface)", "servedFrom": "repository root — the cockpit resolves shared identity from /assets", "viewport": "1440x900", - "captureSourceDigest": "1f47bf95ae8eecfc0ea47a68cf6c18c4e12ae12d292d128061fdd8abd6aa9f30", + "captureSourceDigest": "24924971fd814b5f1baa460e420ebe3f9c9c9d09ec125ee616354f01d557819a", "captureSourceFiles": [ "assets/fonts/geist-mono-variable.woff2", "assets/fonts/geist-variable.woff2", diff --git a/media/hac-335/evidence/claim-ledger.json b/media/hac-335/evidence/claim-ledger.json index a83f789..9d0fbfe 100644 --- a/media/hac-335/evidence/claim-ledger.json +++ b/media/hac-335/evidence/claim-ledger.json @@ -251,13 +251,49 @@ }, { "id": "CL-025", - "text": "HAC-319 evaluation is not bound. No SPR, precision, recall, false-block or useful-concurrency value exists in this package.", + "text": "HAC-319 proper — precision, recall and fleet-scale behaviour — is not bound. No such value exists in this package. HAC-343 is a bounded child of HAC-319, not a substitute for it.", "classification": "NOT YET BOUND", "proofClass": "none", - "proofSource": "media/hac-341/evidence/view-model.json (reserved); media/hac-334/masters/IL-DIAG-013-evaluation-shell.svg", + "proofSource": "media/hac-334/masters/IL-DIAG-013-evaluation-shell.svg (reserved shell, excluded from the judge path)", "surfaces": ["readme", "devpost-limitations"], "status": "OPEN — awaiting a frozen HAC-319 evaluation packet" }, + { + "id": "CL-028", + "text": "On a frozen sixteen-scenario corpus, four coordination strategies compare as: Uncoordinated 2/2 hazards unsafe and 2/2 independent opportunities parallel; Global lock 0/2 and 0/2; Per-target lock 2/2 and 2/2; Interlock 0/2 and 2/2.", + "classification": "EVIDENCED", + "proofClass": "controlled local evaluation (HAC-343)", + "proofSource": "experiments/hac-343/evidence/judge-export.json#panel1.rows", + "surfaces": ["readme", "devpost-narrative", "cockpit-compare-drawer"], + "status": "FROZEN — canonical result 7ede0f97e55685c16e5bb762b5e7fbe471a6e8b0" + }, + { + "id": "CL-029", + "text": "The per-target lock baseline is credible rather than a straw man: it serialized same-target contention 2/2, parallelised cross-target pairs 4/4, and missed cross-target hazards 2/2. A composition hazard spanning two lock keys is not visible to any per-key discipline.", + "classification": "EVIDENCED", + "proofClass": "controlled local evaluation (HAC-343)", + "proofSource": "experiments/hac-343/evidence/judge-export.json#panel1.perTargetLockCredibility", + "surfaces": ["readme", "devpost-narrative", "cockpit-compare-drawer"], + "status": "FROZEN — canonical result 7ede0f97e55685c16e5bb762b5e7fbe471a6e8b0" + }, + { + "id": "CL-030", + "text": "Interlock's safety in this corpus is evidence-derived: with coupling evidence present it produced 0/2 invalid outcomes, and with that evidence removed the same core produced 2/2 invalid outcomes.", + "classification": "EVIDENCED", + "proofClass": "controlled local evaluation (HAC-343)", + "proofSource": "experiments/hac-343/evidence/judge-export.json#panel2.rows", + "surfaces": ["readme", "devpost-narrative", "cockpit-compare-drawer"], + "status": "FROZEN — canonical result 7ede0f97e55685c16e5bb762b5e7fbe471a6e8b0" + }, + { + "id": "CL-031", + "text": "Interlock is not 0% unsafe, is not safer than locking, and the sixteen-scenario corpus is not collapsed into one denominator. No interval or statistical significance is claimed, and no exactly-once, restart-safety or production-readiness result was tested.", + "classification": "NOT CLAIMED", + "proofClass": "controlled local evaluation (HAC-343)", + "proofSource": "experiments/hac-343/evidence/judge-export.json#mustNotClaim", + "surfaces": ["readme", "devpost-narrative", "devpost-limitations"], + "status": "ENFORCED — media/hac-335/bin/verify-package.mjs checkEvaluationBound" + }, { "id": "CL-026", "text": "Interlock is new work created during the contest, built on the pre-existing open-source workspace.json specification and toolchain consumed at pinned revisions and never copied.", diff --git a/media/hac-335/evidence/judge-sequence.json b/media/hac-335/evidence/judge-sequence.json index 71678c9..ec1f0f9 100644 --- a/media/hac-335/evidence/judge-sequence.json +++ b/media/hac-335/evidence/judge-sequence.json @@ -128,7 +128,7 @@ "excludedFromJudgePath": [ { "assetId": "IL-DIAG-013", - "reason": "HAC-319 evaluation is not bound. The shell stays in the internal registry as EVALUATION NOT YET BOUND and is deliberately kept out of README, Devpost and video assets so no judge reads an unavailable evaluation as a pending result.", + "reason": "HAC-319 proper — precision, recall, fleet-scale behaviour — is not bound. The shell stays in the internal registry as EVALUATION NOT YET BOUND and is deliberately kept out of README, Devpost and video assets so no judge reads an unavailable evaluation as a pending result. The bounded HAC-343 four-arm evaluation IS bound and appears in the judge path, rendered from its frozen judge export.", "seamPreserved": true }, { diff --git a/media/hac-341/bin/build-view-model.mjs b/media/hac-341/bin/build-view-model.mjs index 34b2050..d9410e1 100644 --- a/media/hac-341/bin/build-view-model.mjs +++ b/media/hac-341/bin/build-view-model.mjs @@ -273,15 +273,26 @@ cloudRun.rawProof = { /* --- reserved surface: HAC-319 ----------------------------------------- */ +/** + * HAC-343 froze a bounded four-arm result, and this shell had to narrow. + * + * It previously withheld `SPR` — while the comparison panel beside it now + * renders SPR from HAC-343's frozen export. A surface that says a metric is + * withheld next to a panel showing that metric is not being careful, it is + * being wrong. What is still genuinely unbound is HAC-319 *proper*: the + * three-regime anti-global-mutex evaluation with precision and recall over a + * population this corpus does not sample. + */ const reserved = { semanticStateId: 'evaluation.unbound', label: 'Anti-global-mutex evaluation', sourceIssue: 'HAC-319', degradedState: 'evaluation-not-yet-bound', - message: 'Evaluation not yet bound.', + message: 'Three-regime evaluation not yet bound.', regimes: ['Regime 1', 'Regime 2', 'Regime 3'], - metricsWithheld: ['SPR', 'precision', 'recall', 'false-block rate', 'useful-concurrency'], - rule: 'Labels only. No value, no mark, no proportional geometry until HAC-319 supplies a frozen evaluation packet.', + metricsWithheld: ['precision', 'recall', 'fleet-scale behaviour'], + boundElsewhere: 'HAC-343 binds a bounded four-arm comparison, including SPR, over its own sixteen-scenario corpus. It is a child of HAC-319, not a substitute: it samples no population and reports no precision or recall.', + rule: 'Labels only. No value, no mark, no proportional geometry until HAC-319 supplies a frozen three-regime evaluation packet.', }; /* --- coordination-strategy comparison: bound to HAC-343 ---------------- */ diff --git a/media/hac-341/cockpit.html b/media/hac-341/cockpit.html index c6aa157..3a528d0 100644 --- a/media/hac-341/cockpit.html +++ b/media/hac-341/cockpit.html @@ -1227,8 +1227,18 @@

Decision, effect, observation

${esc(cell.source)}

`).join('')} ${s.armId === 'A3_per_target_lock' ? `

Is this lock credible?

-

Same-target contention serialized ${esc(c.perTargetLockCredibility.display)}. - ${esc(c.perTargetLockCredibility.note)}

` : ''} +
+
same-target contention serialized
${esc(c.perTargetLockCredibility.serializedSameTargetContention)}
+
cross-target pairs parallelised
${esc(c.perTargetLockCredibility.parallelisedCrossTarget)}
+
cross-target hazards missed
${esc(c.perTargetLockCredibility.missedCrossTargetHazards)}
+
+

${esc(c.perTargetLockCredibility.claim)} ${esc(c.perTargetLockCredibility.note)}

` : ''} +

${esc(c.evidenceAblation.question)}

+
+ ${c.evidenceAblation.rows.map((r) => `
${esc(r.condition)}
${esc(r.invalidOutcomes)} invalid
`).join('')} +
+

${esc(c.evidenceAblation.reading)}

+

${esc(c.evidenceAblation.forbiddenRendering)}

Scope

${esc(c.scopeNote)}

Source

${c.artifacts.map((a) => `
artifact
${esc(a)}
`).join('')} diff --git a/media/hac-341/evidence/view-model.json b/media/hac-341/evidence/view-model.json index 42c8ef5..3ea2d0f 100644 --- a/media/hac-341/evidence/view-model.json +++ b/media/hac-341/evidence/view-model.json @@ -796,9 +796,27 @@ } ], "perTargetLockCredibility": { - "display": "2/2", + "claim": "A3 is a real lock, so its misses are blindness rather than absence of a lock.", + "serializedSameTargetContention": "2/2", + "parallelisedCrossTarget": "4/4", + "missedCrossTargetHazards": "2/2", "note": "It locked exactly what a lock can see. A composition hazard spanning two lock keys is not visible to any per-key discipline." }, + "evidenceAblation": { + "question": "Is Interlock’s safety derived from the evidence, or from something else?", + "rows": [ + { + "condition": "Interlock + coupling evidence present", + "invalidOutcomes": "0/2" + }, + { + "condition": "Interlock + coupling evidence removed", + "invalidOutcomes": "2/2" + } + ], + "reading": "Interlock’s safety is evidence-derived. With revision-bound composition evidence present it withheld both hazardous compositions while retaining both safe parallel opportunities. When that evidence was deliberately removed, the decision reversed and both invariants failed.", + "forbiddenRendering": "A4 must not be described as globally 0% unsafe, and the sixteen-scenario corpus must not be collapsed into a single unsafe-rate denominator. The 0/2 in Panel 1 is bounded to COUPLED scenarios and means nothing without Panel 2 beside it." + }, "unresolved": [], "resolved": true, "unresolvedLabel": "Unresolved binding scaffold · not evidence", @@ -809,20 +827,19 @@ "label": "Anti-global-mutex evaluation", "sourceIssue": "HAC-319", "degradedState": "evaluation-not-yet-bound", - "message": "Evaluation not yet bound.", + "message": "Three-regime evaluation not yet bound.", "regimes": [ "Regime 1", "Regime 2", "Regime 3" ], "metricsWithheld": [ - "SPR", "precision", "recall", - "false-block rate", - "useful-concurrency" + "fleet-scale behaviour" ], - "rule": "Labels only. No value, no mark, no proportional geometry until HAC-319 supplies a frozen evaluation packet." + "boundElsewhere": "HAC-343 binds a bounded four-arm comparison, including SPR, over its own sixteen-scenario corpus. It is a child of HAC-319, not a substitute: it samples no population and reports no precision or recall.", + "rule": "Labels only. No value, no mark, no proportional geometry until HAC-319 supplies a frozen three-regime evaluation packet." }, "degradedStates": [ { diff --git a/media/hac-341/lib/comparison.mjs b/media/hac-341/lib/comparison.mjs index ff2533a..b16c05d 100644 --- a/media/hac-341/lib/comparison.mjs +++ b/media/hac-341/lib/comparison.mjs @@ -125,13 +125,41 @@ export function buildComparison(sources = {}) { // The experiment names itself; typing "HAC-343" here would be the one value // on this panel that came from nowhere. const experiment = bind(sources, EXPORT, 'experiment'); + const credibilityClaim = bind(sources, EXPORT, 'panel1.perTargetLockCredibility.claim'); const credibility = bind(sources, EXPORT, 'panel1.perTargetLockCredibility.serializedSameTargetContention.display'); + const parallelised = bind(sources, EXPORT, 'panel1.perTargetLockCredibility.parallelisedCrossTarget.display'); + const missed = bind(sources, EXPORT, 'panel1.perTargetLockCredibility.missedCrossTargetHazards.display'); const credibilityNote = bind(sources, EXPORT, 'panel1.perTargetLockCredibility.note'); const scope = bind(sources, EXPORT, 'panel1.scope'); const commit = bind(sources, EXPORT, 'derivedFrom.canonicalResultCommit'); + /* Panel 2 travels with Panel 1 or not at all. + * + * Panel 1 alone reads as "Interlock is the safe one". The frozen export + * forbids exactly that reading: the 0/2 is bounded to COUPLED scenarios and + * is a property of the evidence being present, not of Interlock. Panel 2 is + * the arm of the same experiment that shows the dependency — remove the + * coupling evidence and the same core produces 2/2 invalid. Binding them in + * one object is what makes "adjacent" a mechanical property of the panel + * rather than a layout habit a later edit can quietly separate. */ + const panel2Rows = [0, 1].map((i) => ({ + condition: bind(sources, EXPORT, `panel2.rows.${i}.condition`), + invalidOutcomes: bind(sources, EXPORT, `panel2.rows.${i}.invalidOutcomes.display`), + })); + const panel2 = { + question: bind(sources, EXPORT, 'panel2.question'), + reading: bind(sources, EXPORT, 'panel2.reading'), + forbiddenRendering: bind(sources, EXPORT, 'panel2.forbiddenRendering'), + }; + const cells = strategies.flatMap((s) => s.cells); - const unresolved = [...cells, ...captions, experiment, credibility, credibilityNote, scope, commit] + const unresolved = [ + ...cells, ...captions, experiment, + credibilityClaim, credibility, parallelised, missed, credibilityNote, + scope, commit, + ...Object.values(panel2), + ...panel2Rows.flatMap((r) => [r.condition, r.invalidOutcomes]), + ] .filter((c) => !c.resolved) .map((c) => c.source); for (const s of strategies) if (String(s.label).startsWith('[BIND:')) unresolved.push(s.labelSource); @@ -149,7 +177,29 @@ export function buildComparison(sources = {}) { dimensions: DIMENSIONS, captions, strategies, - perTargetLockCredibility: { display: credibility.value, note: credibilityNote.value }, + /* Three figures, not one. `2/2 serialized` alone shows the lock ran; it does + not show what the lock cost or what it still missed. The judge's question + is whether A3 is a credible alternative to Interlock, and that is only + answerable with all three: it serialized every same-target contention, + kept every cross-target pair concurrent, and still missed both + cross-target hazards. Dropping either of the last two lets the strip read + as a clean bill of health for per-target locking. */ + perTargetLockCredibility: { + claim: credibilityClaim.value, + serializedSameTargetContention: credibility.value, + parallelisedCrossTarget: parallelised.value, + missedCrossTargetHazards: missed.value, + note: credibilityNote.value, + }, + evidenceAblation: { + question: panel2.question.value, + rows: panel2Rows.map((r) => ({ + condition: r.condition.value, + invalidOutcomes: r.invalidOutcomes.value, + })), + reading: panel2.reading.value, + forbiddenRendering: panel2.forbiddenRendering.value, + }, unresolved, resolved: unresolved.length === 0, /* Shown verbatim whenever anything is unbound. Never shown when everything @@ -187,6 +237,7 @@ export function judgeFacing(comparison) { captions: comparison.captions, strategies: comparison.strategies, perTargetLockCredibility: comparison.perTargetLockCredibility, + evidenceAblation: comparison.evidenceAblation, resolved: comparison.resolved, unresolved: comparison.unresolved, unresolvedLabel: comparison.unresolvedLabel, @@ -202,5 +253,6 @@ export function judgeFacing(comparison) { export const JUDGE_FACING_FIELDS = [ 'title', 'sourceIssue', 'separateExperiment', 'artifacts', 'canonicalResultCommit', 'scopeNote', 'dimensions', 'captions', 'strategies', 'perTargetLockCredibility', + 'evidenceAblation', 'resolved', 'unresolved', 'unresolvedLabel', 'unresolvedNote', ]; diff --git a/test/hac-335-package-gates.test.mjs b/test/hac-335-package-gates.test.mjs index c424d43..b36b69f 100644 --- a/test/hac-335-package-gates.test.mjs +++ b/test/hac-335-package-gates.test.mjs @@ -32,6 +32,8 @@ const NEEDED = [ 'experiments/hac-342/evidence/cloud-run.public.json', 'experiments/hac-342/evidence/publication-bindings.json', 'experiments/hac-342/evidence/runtime-source-snapshot.json', + // The frozen source of every HAC-343 figure the package renders. + 'experiments/hac-343/evidence/judge-export.json', ]; const GATE = 'media/hac-335/bin/verify-package.mjs'; @@ -232,14 +234,62 @@ describe('public evidence integrity', () => { }); }); -describe('HAC-319 stays unbound', () => { - it('fails when a metric appears next to a number', () => { - const r = perturbed((p) => p.append('README.md', 'Measured SPR: 0.94 across the evaluation set.')); +describe('the HAC-343 evaluation stays bound and bounded', () => { + /* The gate used to prove the evaluation was absent. It now proves every + figure is the frozen one and that no figure travels alone, so each rule + below is exercised by breaking exactly one of those properties. */ + + it('fails when a rendered count is not a frozen judge-export value', () => { + const r = perturbed((p) => p.edit('README.md', '| Interlock | 0/2 | 2/2 |', '| Interlock | 0/7 | 2/2 |')); + expect(r.code).toBe(1); + expect(r.out).toMatch(/0\/7, which is not a frozen HAC-343 display value/); + }); + + it('fails when Panel 1 is shown without the evidence ablation', () => { + const r = perturbed((p) => + p.edit('README.md', '| Interlock + coupling evidence removed | 2/2 |', '')); + expect(r.code).toBe(1); + expect(r.out).toMatch(/without the evidence-ablation condition/); + }); + + it('fails when the A3 credibility strip is dropped', () => { + const r = perturbed((p) => + p.edit('README.md', 'parallelised cross-target pairs 4/4', 'parallelised cross-target pairs')); + expect(r.code).toBe(1); + expect(r.out).toMatch(/A3 credibility figure/); + }); + + it('fails when the comparison is shown without its corpus bound', () => { + const r = perturbed((p) => p.edit('README.md', 'one frozen sixteen-scenario corpus', 'a corpus')); + expect(r.code).toBe(1); + expect(r.out).toMatch(/without stating the corpus it is bounded to/); + }); + + it('fails when copy claims Interlock is 0% unsafe', () => { + const r = perturbed((p) => p.append('README.md', '\n\nIn practice Interlock is 0% unsafe.\n')); + expect(r.code).toBe(1); + expect(r.out).toMatch(/0% unsafe/); + }); + + it('fails when copy claims Interlock is safer than locking', () => { + const r = perturbed((p) => p.append('README.md', '\n\nInterlock is safer than locking.\n')); + expect(r.code).toBe(1); + expect(r.out).toMatch(/safer than locking/); + }); + + it('fails when copy claims statistical significance', () => { + const r = perturbed((p) => p.append('README.md', '\n\nThe difference is statistically significant.\n')); + expect(r.code).toBe(1); + expect(r.out).toMatch(/statistical significance/); + }); + + it('fails when the superseded "no SPR exists" claim is reintroduced', () => { + const r = perturbed((p) => p.append('README.md', '\n\nNo SPR value exists in this package.\n')); expect(r.code).toBe(1); - expect(r.out).toMatch(/HAC-319 metric/); + expect(r.out).toMatch(/asserts no SPR value exists/); }); - it('fails when the evaluation shell enters the judge registry', () => { + it('fails when HAC-319 proper enters the judge registry', () => { const r = perturbed((p) => { const reg = p.json('media/hac-335/evidence/asset-registry.json'); reg.assets.push({ assetId: 'IL-DIAG-013', exports: [], claimIds: [] }); @@ -249,14 +299,14 @@ describe('HAC-319 stays unbound', () => { expect(r.out).toMatch(/IL-DIAG-013 is in the judge-facing registry/); }); - it('fails when the judge sequence points at the unbound shell', () => { + it('fails when the judge sequence points at the unbound HAC-319 shell', () => { const r = perturbed((p) => { const seq = p.json('media/hac-335/evidence/judge-sequence.json'); seq.steps[1].supportingAssets = ['IL-DIAG-013']; p.writeJson('media/hac-335/evidence/judge-sequence.json', seq); }); expect(r.code).toBe(1); - expect(r.out).toMatch(/points at the unbound evaluation shell/); + expect(r.out).toMatch(/points at the unbound HAC-319 shell/); }); });