From 39aac5e527907c520a81bf4347d1b1d07c270844 Mon Sep 17 00:00:00 2001 From: "Jonathan D.A. Jewell" <6759885+hyperpolymath@users.noreply.github.com> Date: Mon, 24 Aug 2026 07:59:17 +0100 Subject: [PATCH] refactor: migrate repository documentation from Markdown to AsciiDoc --- .claude/PROJECT.adoc | 34 + .claude/PROJECT.md | 37 - CHANGELOG.adoc | 371 ++++++ CHANGELOG.md | 343 ----- CODE_OF_CONDUCT.adoc | 129 ++ CODE_OF_CONDUCT.md | 129 -- CONTRIBUTING.adoc | 83 ++ CONTRIBUTING.md | 73 - DEBT.adoc | 360 +++++ DEBT.md | 158 --- GOVERNANCE.adoc | 60 + GOVERNANCE.md | 60 - PROOF-NEEDS.adoc | 325 +++++ PROOF-NEEDS.md | 238 ---- SECURITY.adoc | 783 +++++++++++ SECURITY.md | 714 ---------- TEST-NEEDS.adoc | 143 ++ TEST-NEEDS.md | 130 -- TOPOLOGY.adoc | 223 ++++ TOPOLOGY.md | 172 --- audits/audit-ffi-2026-05-26.adoc | 61 + audits/audit-ffi-2026-05-26.md | 46 - docs/backend-assurance/README.adoc | 74 ++ docs/backend-assurance/README.md | 67 - docs/backend-assurance/prim__eqChar.adoc | 134 ++ docs/backend-assurance/prim__eqChar.md | 134 -- docs/backend-assurance/prim__strAppend.adoc | 132 ++ docs/backend-assurance/prim__strAppend.md | 139 -- docs/backend-assurance/prim__strSubstr.adoc | 143 ++ docs/backend-assurance/prim__strSubstr.md | 152 --- .../prim__strToCharList.adoc | 142 ++ docs/backend-assurance/prim__strToCharList.md | 149 --- docs/decisions/0001-adopt-rsr-standard.adoc | 94 ++ docs/decisions/0001-adopt-rsr-standard.md | 88 -- .../0002-align-unified-zig-api-stack.adoc | 127 ++ .../0002-align-unified-zig-api-stack.md | 123 -- ...003-extract-cartridge-spec-standalone.adoc | 110 ++ .../0003-extract-cartridge-spec-standalone.md | 101 -- .../0004-adopt-http-capability-gateway.adoc | 184 +++ .../0004-adopt-http-capability-gateway.md | 167 --- .../0005-elixir-to-zig-ffi-transport.adoc | 151 +++ .../0005-elixir-to-zig-ffi-transport.md | 160 --- docs/decisions/0006-cartridge-invoke-abi.adoc | 169 +++ docs/decisions/0006-cartridge-invoke-abi.md | 167 --- .../decisions/0007-trust-tier-policy-dsl.adoc | 271 ++++ docs/decisions/0007-trust-tier-policy-dsl.md | 178 --- .../decisions/0008-cartridge-marketplace.adoc | 251 ++++ docs/decisions/0008-cartridge-marketplace.md | 161 --- docs/decisions/0009-sandbox-cartridge.adoc | 275 ++++ docs/decisions/0009-sandbox-cartridge.md | 173 --- .../0010-cross-machine-coord-federation.adoc | 286 ++++ .../0010-cross-machine-coord-federation.md | 180 --- ...11-webhooks-inbound-mcp-notifications.adoc | 313 +++++ ...0011-webhooks-inbound-mcp-notifications.md | 201 --- .../0012-server-initiated-sampling.adoc | 310 +++++ .../0012-server-initiated-sampling.md | 202 --- .../0013-streamable-http-transport.adoc | 316 +++++ .../0013-streamable-http-transport.md | 187 --- ...14-cross-cartridge-composition-safety.adoc | 251 ++++ ...0014-cross-cartridge-composition-safety.md | 252 ---- .../0015-backend-file-lock-primitive.adoc | 139 ++ .../0015-backend-file-lock-primitive.md | 128 -- .../0016-mtls-federation-stopgap.adoc | 171 +++ .../decisions/0016-mtls-federation-stopgap.md | 147 -- docs/decisions/README.adoc | 18 + docs/decisions/README.md | 19 - docs/glama/CAPABILITIES.adoc | 226 ++++ docs/glama/CAPABILITIES.md | 197 --- docs/glama/PROMPTS.adoc | 202 +++ docs/glama/PROMPTS.md | 200 --- docs/glama/RESOURCES.adoc | 372 ++++++ docs/glama/RESOURCES.md | 369 ----- docs/glama/SERVER_CONFIGURATION.adoc | 199 +++ docs/glama/SERVER_CONFIGURATION.md | 176 --- .../QED-AXIOM-AUDIT-2026-04-19.adoc | 43 + docs/governance/QED-AXIOM-AUDIT-2026-04-19.md | 32 - docs/handover/COORD-MCP-DESIGN-LOG.adoc | 1184 +++++++++++++++++ docs/handover/COORD-MCP-DESIGN-LOG.md | 736 ---------- ...MPTS.md => COORD-MCP-HANDOFF-PROMPTS.adoc} | 125 +- ...COORD-MCP-PROMPT3-HANDOVER-2026-04-20.adoc | 131 ++ .../COORD-MCP-PROMPT3-HANDOVER-2026-04-20.md | 121 -- docs/handover/COORD-MCP-STATE.adoc | 254 ++++ docs/handover/COORD-MCP-STATE.md | 198 --- docs/handover/COORD-MCP-TODO.adoc | 239 ++++ docs/handover/COORD-MCP-TODO.md | 111 -- docs/handover/HAIKU-SCOUT-PASS.adoc | 58 + docs/handover/HAIKU-SCOUT-PASS.md | 61 - docs/handover/README.adoc | 83 ++ docs/handover/README.md | 72 - .../boj-side-observability-spec.adoc | 504 +++++++ .../boj-side-observability-spec.md | 264 ---- docs/integration/gateway-load-profile.adoc | 465 +++++++ docs/integration/gateway-load-profile.md | 226 ---- .../gateway-observability-spec.adoc | 686 ++++++++++ .../integration/gateway-observability-spec.md | 412 ------ .../hcg-tier2-rollout-runbook.adoc | 636 +++++++++ docs/integration/hcg-tier2-rollout-runbook.md | 321 ----- .../http-capability-gateway-audit.adoc | 628 +++++++++ .../http-capability-gateway-audit.md | 515 ------- .../http-capability-gateway-boj-contract.adoc | 265 ++++ .../http-capability-gateway-boj-contract.md | 214 --- .../http-capability-gateway-plan.adoc | 500 +++++++ .../http-capability-gateway-plan.md | 475 ------- ...p-capability-gateway-policy-authoring.adoc | 149 +++ ...ttp-capability-gateway-policy-authoring.md | 141 -- docs/maintenance/MAINTENANCE-CHECKLIST.adoc | 671 ++++++++++ docs/maintenance/MAINTENANCE-CHECKLIST.md | 572 -------- ...ions.md => awesome-list-descriptions.adoc} | 87 +- docs/outreach/blog-post-draft.adoc | 286 ++++ docs/outreach/blog-post-draft.md | 203 --- docs/outreach/show-hn-draft.adoc | 41 + docs/outreach/show-hn-draft.md | 31 - docs/outreach/show-hn-post.adoc | 43 + docs/outreach/show-hn-post.md | 29 - docs/papers/boj-architecture-paper.adoc | 847 ++++++++++++ docs/papers/boj-architecture-paper.md | 748 ----------- docs/papers/umoja-federation-draft.adoc | 773 +++++++++++ docs/papers/umoja-federation-draft.md | 790 ----------- .../boj-server-proof-story-2026-06-01.adoc | 581 ++++++++ .../boj-server-proof-story-2026-06-01.md | 298 ----- .../cartridge-catalogue-2026-06-01.adoc | 631 +++++++++ .../cartridge-catalogue-2026-06-01.md | 286 ---- docs/practice/TESTS-AND-BENCHES.adoc | 124 ++ docs/practice/TESTS-AND-BENCHES.md | 85 -- docs/proof-debt.adoc | 133 ++ docs/proof-debt.md | 119 -- .../specification/cartridge-tools/README.adoc | 575 ++++++++ docs/specification/cartridge-tools/README.md | 429 ------ docs/specification/cartridges/README.adoc | 872 ++++++++++++ docs/specification/cartridges/README.md | 731 ---------- docs/tech-debt-2026-05-26.adoc | 80 ++ docs/tech-debt-2026-05-26.md | 70 - schemas/SCHEMA-MIRROR.adoc | 46 + schemas/SCHEMA-MIRROR.md | 31 - setup-scripts/onboard-son.adoc | 268 ++++ setup-scripts/onboard-son.md | 233 ---- .../{README.md => README.adoc} | 140 +- tests/backend-assurance/README.adoc | 33 + tests/backend-assurance/README.md | 36 - tools/cartridge-configurator/README.adoc | 67 + tools/cartridge-configurator/README.md | 62 - tools/cartridge-provisioner/README.adoc | 75 ++ tools/cartridge-provisioner/README.md | 66 - tools/panel-harness/README.adoc | 70 + tools/panel-harness/README.md | 64 - 145 files changed, 19858 insertions(+), 15266 deletions(-) create mode 100644 .claude/PROJECT.adoc delete mode 100644 .claude/PROJECT.md create mode 100644 CHANGELOG.adoc delete mode 100644 CHANGELOG.md create mode 100644 CODE_OF_CONDUCT.adoc delete mode 100644 CODE_OF_CONDUCT.md create mode 100644 CONTRIBUTING.adoc delete mode 100644 CONTRIBUTING.md create mode 100644 DEBT.adoc delete mode 100644 DEBT.md create mode 100644 GOVERNANCE.adoc delete mode 100644 GOVERNANCE.md create mode 100644 PROOF-NEEDS.adoc delete mode 100644 PROOF-NEEDS.md create mode 100644 SECURITY.adoc delete mode 100644 SECURITY.md create mode 100644 TEST-NEEDS.adoc delete mode 100644 TEST-NEEDS.md create mode 100644 TOPOLOGY.adoc delete mode 100644 TOPOLOGY.md create mode 100644 audits/audit-ffi-2026-05-26.adoc delete mode 100644 audits/audit-ffi-2026-05-26.md create mode 100644 docs/backend-assurance/README.adoc delete mode 100644 docs/backend-assurance/README.md create mode 100644 docs/backend-assurance/prim__eqChar.adoc delete mode 100644 docs/backend-assurance/prim__eqChar.md create mode 100644 docs/backend-assurance/prim__strAppend.adoc delete mode 100644 docs/backend-assurance/prim__strAppend.md create mode 100644 docs/backend-assurance/prim__strSubstr.adoc delete mode 100644 docs/backend-assurance/prim__strSubstr.md create mode 100644 docs/backend-assurance/prim__strToCharList.adoc delete mode 100644 docs/backend-assurance/prim__strToCharList.md create mode 100644 docs/decisions/0001-adopt-rsr-standard.adoc delete mode 100644 docs/decisions/0001-adopt-rsr-standard.md create mode 100644 docs/decisions/0002-align-unified-zig-api-stack.adoc delete mode 100644 docs/decisions/0002-align-unified-zig-api-stack.md create mode 100644 docs/decisions/0003-extract-cartridge-spec-standalone.adoc delete mode 100644 docs/decisions/0003-extract-cartridge-spec-standalone.md create mode 100644 docs/decisions/0004-adopt-http-capability-gateway.adoc delete mode 100644 docs/decisions/0004-adopt-http-capability-gateway.md create mode 100644 docs/decisions/0005-elixir-to-zig-ffi-transport.adoc delete mode 100644 docs/decisions/0005-elixir-to-zig-ffi-transport.md create mode 100644 docs/decisions/0006-cartridge-invoke-abi.adoc delete mode 100644 docs/decisions/0006-cartridge-invoke-abi.md create mode 100644 docs/decisions/0007-trust-tier-policy-dsl.adoc delete mode 100644 docs/decisions/0007-trust-tier-policy-dsl.md create mode 100644 docs/decisions/0008-cartridge-marketplace.adoc delete mode 100644 docs/decisions/0008-cartridge-marketplace.md create mode 100644 docs/decisions/0009-sandbox-cartridge.adoc delete mode 100644 docs/decisions/0009-sandbox-cartridge.md create mode 100644 docs/decisions/0010-cross-machine-coord-federation.adoc delete mode 100644 docs/decisions/0010-cross-machine-coord-federation.md create mode 100644 docs/decisions/0011-webhooks-inbound-mcp-notifications.adoc delete mode 100644 docs/decisions/0011-webhooks-inbound-mcp-notifications.md create mode 100644 docs/decisions/0012-server-initiated-sampling.adoc delete mode 100644 docs/decisions/0012-server-initiated-sampling.md create mode 100644 docs/decisions/0013-streamable-http-transport.adoc delete mode 100644 docs/decisions/0013-streamable-http-transport.md create mode 100644 docs/decisions/0014-cross-cartridge-composition-safety.adoc delete mode 100644 docs/decisions/0014-cross-cartridge-composition-safety.md create mode 100644 docs/decisions/0015-backend-file-lock-primitive.adoc delete mode 100644 docs/decisions/0015-backend-file-lock-primitive.md create mode 100644 docs/decisions/0016-mtls-federation-stopgap.adoc delete mode 100644 docs/decisions/0016-mtls-federation-stopgap.md create mode 100644 docs/decisions/README.adoc delete mode 100644 docs/decisions/README.md create mode 100644 docs/glama/CAPABILITIES.adoc delete mode 100644 docs/glama/CAPABILITIES.md create mode 100644 docs/glama/PROMPTS.adoc delete mode 100644 docs/glama/PROMPTS.md create mode 100644 docs/glama/RESOURCES.adoc delete mode 100644 docs/glama/RESOURCES.md create mode 100644 docs/glama/SERVER_CONFIGURATION.adoc delete mode 100644 docs/glama/SERVER_CONFIGURATION.md create mode 100644 docs/governance/QED-AXIOM-AUDIT-2026-04-19.adoc delete mode 100644 docs/governance/QED-AXIOM-AUDIT-2026-04-19.md create mode 100644 docs/handover/COORD-MCP-DESIGN-LOG.adoc delete mode 100644 docs/handover/COORD-MCP-DESIGN-LOG.md rename docs/handover/{COORD-MCP-HANDOFF-PROMPTS.md => COORD-MCP-HANDOFF-PROMPTS.adoc} (71%) create mode 100644 docs/handover/COORD-MCP-PROMPT3-HANDOVER-2026-04-20.adoc delete mode 100644 docs/handover/COORD-MCP-PROMPT3-HANDOVER-2026-04-20.md create mode 100644 docs/handover/COORD-MCP-STATE.adoc delete mode 100644 docs/handover/COORD-MCP-STATE.md create mode 100644 docs/handover/COORD-MCP-TODO.adoc delete mode 100644 docs/handover/COORD-MCP-TODO.md create mode 100644 docs/handover/HAIKU-SCOUT-PASS.adoc delete mode 100644 docs/handover/HAIKU-SCOUT-PASS.md create mode 100644 docs/handover/README.adoc delete mode 100644 docs/handover/README.md create mode 100644 docs/integration/boj-side-observability-spec.adoc delete mode 100644 docs/integration/boj-side-observability-spec.md create mode 100644 docs/integration/gateway-load-profile.adoc delete mode 100644 docs/integration/gateway-load-profile.md create mode 100644 docs/integration/gateway-observability-spec.adoc delete mode 100644 docs/integration/gateway-observability-spec.md create mode 100644 docs/integration/hcg-tier2-rollout-runbook.adoc delete mode 100644 docs/integration/hcg-tier2-rollout-runbook.md create mode 100644 docs/integration/http-capability-gateway-audit.adoc delete mode 100644 docs/integration/http-capability-gateway-audit.md create mode 100644 docs/integration/http-capability-gateway-boj-contract.adoc delete mode 100644 docs/integration/http-capability-gateway-boj-contract.md create mode 100644 docs/integration/http-capability-gateway-plan.adoc delete mode 100644 docs/integration/http-capability-gateway-plan.md create mode 100644 docs/integration/http-capability-gateway-policy-authoring.adoc delete mode 100644 docs/integration/http-capability-gateway-policy-authoring.md create mode 100644 docs/maintenance/MAINTENANCE-CHECKLIST.adoc delete mode 100644 docs/maintenance/MAINTENANCE-CHECKLIST.md rename docs/outreach/{awesome-list-descriptions.md => awesome-list-descriptions.adoc} (74%) create mode 100644 docs/outreach/blog-post-draft.adoc delete mode 100644 docs/outreach/blog-post-draft.md create mode 100644 docs/outreach/show-hn-draft.adoc delete mode 100644 docs/outreach/show-hn-draft.md create mode 100644 docs/outreach/show-hn-post.adoc delete mode 100644 docs/outreach/show-hn-post.md create mode 100644 docs/papers/boj-architecture-paper.adoc delete mode 100644 docs/papers/boj-architecture-paper.md create mode 100644 docs/papers/umoja-federation-draft.adoc delete mode 100644 docs/papers/umoja-federation-draft.md create mode 100644 docs/planning/boj-server-proof-story-2026-06-01.adoc delete mode 100644 docs/planning/boj-server-proof-story-2026-06-01.md create mode 100644 docs/planning/cartridge-catalogue-2026-06-01.adoc delete mode 100644 docs/planning/cartridge-catalogue-2026-06-01.md create mode 100644 docs/practice/TESTS-AND-BENCHES.adoc delete mode 100644 docs/practice/TESTS-AND-BENCHES.md create mode 100644 docs/proof-debt.adoc delete mode 100644 docs/proof-debt.md create mode 100644 docs/specification/cartridge-tools/README.adoc delete mode 100644 docs/specification/cartridge-tools/README.md create mode 100644 docs/specification/cartridges/README.adoc delete mode 100644 docs/specification/cartridges/README.md create mode 100644 docs/tech-debt-2026-05-26.adoc delete mode 100644 docs/tech-debt-2026-05-26.md create mode 100644 schemas/SCHEMA-MIRROR.adoc delete mode 100644 schemas/SCHEMA-MIRROR.md create mode 100644 setup-scripts/onboard-son.adoc delete mode 100644 setup-scripts/onboard-son.md rename templates/cartridge-template/{README.md => README.adoc} (73%) create mode 100644 tests/backend-assurance/README.adoc delete mode 100644 tests/backend-assurance/README.md create mode 100644 tools/cartridge-configurator/README.adoc delete mode 100644 tools/cartridge-configurator/README.md create mode 100644 tools/cartridge-provisioner/README.adoc delete mode 100644 tools/cartridge-provisioner/README.md create mode 100644 tools/panel-harness/README.adoc delete mode 100644 tools/panel-harness/README.md diff --git a/.claude/PROJECT.adoc b/.claude/PROJECT.adoc new file mode 100644 index 00000000..71c5a839 --- /dev/null +++ b/.claude/PROJECT.adoc @@ -0,0 +1,34 @@ +== BOJ Server - Claude Code Instructions + +This repository contains the BOJ (Battle of the Judges) server +application. + +=== Project Structure + +.... +boj-server/ +├── .claude/ # AI assistant instructions +├── .git/ # Version control +├── .gitignore # Git ignore rules +├── .editorconfig # Editor configuration +└── ... # Server files +.... + +=== Build Commands + +Refer to project-specific documentation. + +=== Coding Conventions + +* Follow hyperpolymath standards +* All code must have SPDX license headers +* Use approved languages only (see CLAUDE.md) +* Document all non-obvious decisions + +=== Security + +* No hardcoded secrets +* All secrets through environment variables or secret management +* SHA-pinned dependencies where applicable +* HTTPS only, no HTTP URLs +* No MD5/SHA1 for security purposes diff --git a/.claude/PROJECT.md b/.claude/PROJECT.md deleted file mode 100644 index 072fdb51..00000000 --- a/.claude/PROJECT.md +++ /dev/null @@ -1,37 +0,0 @@ - -# BOJ Server - Claude Code Instructions - -This repository contains the BOJ (Battle of the Judges) server application. - -## Project Structure - -``` -boj-server/ -├── .claude/ # AI assistant instructions -├── .git/ # Version control -├── .gitignore # Git ignore rules -├── .editorconfig # Editor configuration -└── ... # Server files -``` - -## Build Commands - -Refer to project-specific documentation. - -## Coding Conventions - -- Follow hyperpolymath standards -- All code must have SPDX license headers -- Use approved languages only (see CLAUDE.md) -- Document all non-obvious decisions - -## Security - -- No hardcoded secrets -- All secrets through environment variables or secret management -- SHA-pinned dependencies where applicable -- HTTPS only, no HTTP URLs -- No MD5/SHA1 for security purposes diff --git a/CHANGELOG.adoc b/CHANGELOG.adoc new file mode 100644 index 00000000..b8ff0b11 --- /dev/null +++ b/CHANGELOG.adoc @@ -0,0 +1,371 @@ +== Changelog + +All notable changes to Bundle of Joy Server are documented here. + +=== [0.4.7] — 2026-05-20 + +==== Changed + +* *README install section expanded* to cover every major MCP client: +Claude Code, Claude Desktop, Gemini CLI, GitHub Copilot (VS Code), +Cursor, Cline, Windsurf, Continue.dev, Zed, plus a generic stdio +template. Copy-paste-ready snippets for each client’s config-file path. +* *Runtime documentation corrected*: clone-and-configure path now lists +Deno (preferred per CLAUDE.md policy), Bun, and Node as equally valid +runtimes for `+mcp-bridge/main.js+`. Spurious `+npm install+` step +removed — `+package.json+` declares zero runtime dependencies, so no +install is ever required. + +==== Notes + +* This is the release that publishes the *AAA-tier tool descriptions* to +npm. The description rewrite landed in `+25887157+` / `+ad837abe+` after +0.4.1 but was never npm-published; downstream MCP clients (and Glama’s +quality scoring) were running 0.4.1 with the older one-liner +descriptions. Republishing as 0.4.7 ships the rich Purpose / Behavior / +Returns / Errors / Usage text on every tool and the per-parameter +`+description+` fields with patterns/enums. + +=== [Unreleased] + +==== Added + +* *`+k8s/networkpolicy.yaml+`* — defence-in-depth `+NetworkPolicy+` +restricting BoJ pod ingress to pods labelled +`+app: http-capability-gateway+`. Stacks on top of the ClusterIP Service +(#131) and Cowboy/Zig loopback binds (#130/#132): three independent +layers must be violated before BoJ’s back-side surface is reachable from +anywhere other than HCG. Optional — Phase E acceptance does not require +it; CNI plugins without NetworkPolicy enforcement (e.g. flannel without +VXLAN) silently treat it as a no-op. Override pattern documented in the +manifest header for non-HCG-fronted deployments. Closes #135. Refs +https://github.com/hyperpolymath/standards/issues/100[`+hyperpolymath/standards#100+`], +https://github.com/hyperpolymath/standards/issues/91[`+#91+`]. + +==== Documentation + +* *HCG tier-2 rollout runbook refreshed (v0.1 → v0.2)* in +`+docs/integration/hcg-tier2-rollout-runbook.md+` to reflect the +post-Phase-D state of the single-lane channel rooted at +https://github.com/hyperpolymath/standards/issues/91[`+hyperpolymath/standards#91+`]. +§1.1 (Phase D deliverables) ticks D-1/D-2/D-3 + D-4 bootstrap + the +cross-repo D-1 load-profile (boj-server#168) with PR references; calls +out the remaining owner-driven D-4 rebaseline + `+_status+` flip as the +single open item. §1.4 (BoJ-side prereqs) ticks the loopback bind layers +(#130/#131/#132), the Phase C `+TrustPolicy.satisfies?/3+` clause +(#106), the NetworkPolicy (#173), and the SSE-route policy coverage +(#165). §1.5 (gateway-side prereqs) ticks the new +`+container/gateway-deploy.k9.ncl+` (http-capability-gateway#38) and +records what’s still placeholder until cerro-torre signing runs. Header +banner replaces the stale Phase-D-scaffold-only note with current state. +Refs +https://github.com/hyperpolymath/standards/issues/100[`+hyperpolymath/standards#100+`], +https://github.com/hyperpolymath/standards/issues/91[`+#91+`]. +* *Repository documentation reorganised to match rsr-template-repo +taxonomy.* Root-level `+.adoc+` clutter eliminated; all docs now live +under `+docs/+` subdirectories clustered by purpose (`+quickstarts/+`, +`+wikis/+`, `+architecture/+`, `+status/+`, `+developer/+`, +`+governance/+`, `+decisions/+`, `+specification/+`, `+integration/+`, +`+backend-assurance/+`, `+compliance/+`, `+practice/+`, `+proposals/+`, +`+attribution/+`, `+accessibility/+`, `+papers/+`, `+examples/+`, +`+glama/+`, `+outreach/+`, `+handover/+`, `+maintenance/+`). Each +subdirectory has its own `+README.adoc+` index. High-coupling root files +(`+PROOF-NEEDS.md+`, `+TOPOLOGY.md+`, `+TEST-NEEDS.md+`) stay at root +pending follow-up PR to update their 16/11/5 cross-references. +* *`+README.md+` merged into `+README.adoc+` and dropped.* Single +canonical root README in AsciiDoc. Preserves the full 11-client MCP +install matrix (Claude Code, Claude Desktop, Gemini CLI, Copilot, +Cursor, Cline, Windsurf, Continue.dev, Zed, generic stdio) and +collapsible cartridge tables via `+[%collapsible]+` blocks. +* *All `+docs/*.md+` files converted to `+.adoc+`* (format only; content +preserved). Affected files: `+ARCHITECTURE.md+` → +`+docs/architecture/README.adoc+`, `+DEVELOPERS.md+` → +`+docs/developer/README.adoc+`, `+OPERATOR-QUICKSTART.md+` → +`+docs/quickstarts/MAINTAINER.adoc+`, `+DEVELOPER-QUICKSTART.md+` → +`+docs/quickstarts/DEV.adoc+`, and 8 others. `+docs/wikis/+` sources +fully converted and expanded (Home, User-Guide, Operator-Guide, +Developer-Guide, FAQ all in `+.adoc+`). +* *`+docs/README.adoc+` rewritten* with reading-order-by-audience table, +full directory taxonomy, standalone-docs table, and rationale for the +three high-coupling deferred moves. +* *New subdir index files*: `+docs/quickstarts/README.adoc+`, +`+docs/wikis/README.adoc+`, `+docs/status/README.adoc+` — each explains +its subdirectory’s scope and contents. +* *STATE.a2ml cartridge count corrected*: was 112, actual is 125 +(verified by `+find cartridges -name cartridge.json | wc -l+`). Session +log entry added documenting all 2026-05-26 work. + +==== Changed + +* *Cowboy listener now binds to `+127.0.0.1+` by default* (was: all +interfaces). Configurable via the `+BOJ_BIND_IP+` environment variable; +invalid values fail-fast at boot rather than silently falling back to +`+0.0.0.0+`. This is the code-enforced expression of the ADR-0004 §1 +invariant that BoJ’s back-side bind is not externally routable in +deployments fronted by `+http-capability-gateway+` (HCG tier-2). Phase E +rollout-runbook §1.4 prerequisite #6. Legacy/standalone deployments that +want all-interfaces exposure must now opt in explicitly +(`+BOJ_BIND_IP=0.0.0.0+` or `+BOJ_BIND_IP=::+`). Refs +https://github.com/hyperpolymath/standards/issues/100[`+hyperpolymath/standards#100+`], +https://github.com/hyperpolymath/standards/issues/91[`+#91+`]. +* *k8s Service for BoJ is now `+type: ClusterIP+`* (was: +`+LoadBalancer+`). Per ADR-0004 §1 and the Phase E rollout-runbook §1.4 +prereq #8, BoJ must not be externally addressable when fronted by +`+http-capability-gateway+` (HCG tier-2). External clients reach HCG; +HCG forwards to BoJ over the pod-network loopback. Legacy/standalone +deployments that need BoJ exposed externally should override `+type+` in +a kustomize/helm overlay rather than reverting the canonical manifest +(see header comment in `+k8s/service.yaml+`). Adds +`+hyperpolymath.dev/exposure: "internal-only"+` and +`+hyperpolymath.dev/external-via: "http-capability-gateway (tier-2)"+` +annotations so the posture is discoverable from `+kubectl describe+`. +Refs +https://github.com/hyperpolymath/standards/issues/100[`+hyperpolymath/standards#100+`], +https://github.com/hyperpolymath/standards/issues/91[`+#91+`]. +* *Container `+APP_HOST+` default is now `+127.0.0.1+`* (was: `+"[::]"+` +IPv6 all-interfaces). Tightens three sites that feed the Zig adapter +binary’s `+--host+` flag: `+stapeln.toml [targets.production]+`, +`+container/entrypoint.sh+`, and `+container/compose.prod.yaml+`. Same +Phase E posture as the Cowboy bind change in the Elixir path: BoJ binds +loopback by default when fronted by `+http-capability-gateway+` (HCG +tier-2). Legacy/standalone deployments without HCG in front should +override `+APP_HOST=0.0.0.0+` (IPv4 all-interfaces) or `+APP_HOST=::+` +(IPv6 all-interfaces) in their deployment config. Phase E +rollout-runbook §1.4 prereq #7. Refs +https://github.com/hyperpolymath/standards/issues/100[`+hyperpolymath/standards#100+`], +https://github.com/hyperpolymath/standards/issues/91[`+#91+`]. + +==== Added + +* *ADR-0014 — cross-cartridge composition safety (RFC)* — frames the +unresolved research question that the per-cartridge ABI proofs do not +compose automatically across `+boj_cartridge_invoke+`. Defines +composition safety as a two-level contract: a static Idris2 envelope +(`+Boj.Composition.InvocationOf+` lifting `+IsUnbreakable+` + +`+ProtocolMatch+` +** per-cartridge `+ArgsContract+` into the inter-cartridge call) and a +dynamic Nickel `+compositions+` block in ADR-0007’s `+policy-mcp+` PDP. +First proof pair is `+panic-attack-mcp → vordr-mcp+` (both cartridges +exist on disk); the prompt-suggested `+panic-attack → sandbox → vordr+` +chain is parked behind ADR-0009’s `+sandbox-mcp+` build-out. +* *README "`Formal verification`" section* — surfaces the audited +posture outside `+PROOF-NEEDS.md+` so external readers can see, without +digging, that all P1/P2 obligations are closed with constructive proofs +and that the remaining `+believe_me+` invocations are _principled +assumptions over Idris2 primitives_, not unproven debt. +* *Streamable HTTP transport (ADR-0013, PR1 of 2)* — MCP bridge now +selects between stdio (default), `+http+`, and `+both+` via +`+BOJ_TRANSPORT+`. HTTP endpoints: `+POST /mcp+` for JSON-RPC, +`+GET /mcp+` for the server-initiated SSE notifications stream, +`+DELETE /mcp+` for explicit session teardown, `+GET /healthz+` for +liveness. Sessions are server-issued UUIDs in the `+Mcp-Session-Id+` +header; the manager expires idle sessions after 30 min and fans events +out across attached SSE streams. Auth: `+none+` (loopback only — refuses +non-loopback binds) or `+bearer+` (token list via +`+BOJ_HTTP_AUTH_TOKENS+`). The same `+hardeningGate+` runs on every +request. Zero new deps — built on `+Deno.serve+` and `+node:http+`. mTLS +/ OIDC auth and the Cloudflare Workers / Durable-Objects shim are owed +in PR2. +* *`+boj://capabilities/deployment+` resource* — reports per-deployment +cartridge availability so clients can avoid invoking host-local-only +cartridges (browser-mcp, container-mcp, local-coord-mcp, sandbox-mcp, +ffmpeg-mcp) against a Worker / remote-HTTP deployment. +* *k9iser-mcp cartridge* — reference implementation of the `+-iser+` +regeneration-cartridge pattern (central K9 contract regeneration), +mirroring ssg-mcp: `+cartridge.json+`, `+mod.js+`, Idris2 ABI, Zig FFI, +panels. +* *Unified transaction-gated adapter*: one internal/loopback listener, +protocol-routed REST + SSE + GraphQL + gRPC-compat → single dispatch → +one Zig ABI. Replaces the ssg-era 3-parallel-port anti-pattern; the +trust gate runs before every dispatch, mirroring the Idris2 +`+exposureSatisfied+` contract (no gatekeeperless path). Internal-only +behind `+http-capability-gateway+` per ADR-0004. +* *boj-rest SSE surface*: `+POST /cartridge/:name/sse+` on the same +single Cowboy listener and trust-gated dispatch, `+text/event-stream+`. + +==== Changed + +* *Doc reconciliation to ADR-0004*: `+elixir/README.adoc+`, +`+mcp-bridge/api-clients.js+`, and `+OPERATOR-QUICKSTART.md+` corrected +to the verified runtime + ADR-0004 tiered model (they previously and +wrongly described it as "`skeleton/501/pending rewrite`"). + +==== Fixed + +* *`+Boj.SafeAPIKey.logSafeBounded+` rebuilt for Idris2 0.8.0.* The pre- +existing proof did not type-check on `+main+`; the 2026-05-18 audit’s +claim that `+SafeAPIKey+` carried constructive proofs closing +BJ2-partial was a desk-read, not a build. Three independent defects: (1) +removed the redundant local `+plusLteMonotone+` helper (called now-gone +`+lteTransitive+` and used wrong arg order on +`+plusLteMonotoneRight+`/`+Left+`; stdlib’s `+Data.Nat.plusLteMonotone+` +has exactly the needed shape); (2) lifted both short and long paths out +of the `+with+`-block (the elaborator doesn’t reduce `+length "***"+` at +type level inside a `+with+`-block — goal stays as +`+LTE (integerToNat (prim__cast_IntInteger (prim__strLength (if ...)))) 11+` +with the `+if+`-arm unreduced); (3) right-associated the long-path proof +to match `++++`’s associativity (`+a ++ b ++ c = a ++ (b ++ c)+`). Plus +two bound-name typos in `+toLogSafeShortEq+`/`+toLogSafeLongEq+`. All 12 +safety modules now build green via per-module `+idris2 --check+`. No new +`+believe_me+` axioms. +* *`+tests/aspect_tests.sh+` grep-count bash bug.* +`+Aspect — Thread Safety + ABI Contract + SPDX+` had been red on +`+main+`, gating every PR with +`+tests/aspect_tests.sh: line 77: [[: 0\n0: syntax error in expression+`. +Root cause: `+grep -c 'pattern' file 2>/dev/null || echo "0"+`. +`+grep -c+` always prints the count (including `+0+`) *and* exits +non-zero on no-match, so `+|| echo "0"+` also fires — `+has_export+` +ends up `+"0\n0"+` and `+[[ "0\n0" -gt 0 ]]+` chokes on the newline in +arithmetic context. Swapped `+|| echo "0"+` → `+|| true+` on all four +call-sites. +* *Honest framing of the ABI axiom count.* +`+src/abi/Boj/SafetyLemmas.idr+`’s module docstring claimed "`Three +axiomatic `+believe_me+` primitives`" while five live in the file. +Docstring now enumerates all five and tags each to its underlying +`+prim__*+` primitive. `+appendLengthSum+` and `+substrLengthBound+` +also had `+(x y : T)+` multi-binder syntax that Idris2 0.8.0 rejects at +parse time — comma-separated form `+(x, y : T)+` restores parsability. +Types and proof terms unchanged. The 2026-05-18 `+PROOF-NEEDS.md+` audit +(5 axioms, all class (J) — irreducible over Idris2 primitives, +principled assumptions not unproven debt) is now consistent with the +source and surfaced via the new README "`Formal verification`" section. +* *`+dogfood-gate.yml+` failed YAML validation at startup* (0 s, no +jobs) on every branch including `+main+`: an inline `+python3 -c "+` +block placed Python source at column 1 inside a `+run: |+` block scalar, +terminating the scalar early. Because *Dogfood Gate* is a required +status check, this silently blocked every PR in the repo. The validator +now lives in `+.github/scripts/validate-eclexiaiser.py+` and is invoked +from the workflow. + +____ +Verification (k9iser-mcp): Elixir suite 177/177 (incl. 2 SSE tests); Zig +ffi 16/16 and unified adapter 5/5 (exposure-gate truth table mirroring +the Idris2 contract); `+idris2 --check K9iserMcp/SafeK9iser.idr+` +passes. http-capability-gateway production-wiring (ADR-0004 tier-2) and +the iseriser-scaffold rollout remain out of scope and separately +tracked. +____ + +=== [0.4.0] — 2026-04-17 + +==== Changed + +* *zig banned estate-wide (2026-04-10)*: Adapter layer language policy +updated. zig is no longer an accepted cartridge adapter language. Zig is +the default replacement for the adapter tier (`+ffi/zig/+` remains; V +adapter files were swept in commit c4674f8). Historical zig API +interfaces have been moved to +`+developer-ecosystem/v-ecosystem/v-api-interfaces/v-/+` for +potential donation to the V community — they are not HP infrastructure. +* *Cartridge manifests = Nickel* (prior closed decision +`+boj-cartridge-manifest-format-dd.md+`): The authoritative cartridge +manifest format is Nickel (`+.ncl+`). Current on-disk manifests are +`+cartridge.json+`; migration to Nickel is tracked as future work (see +open question in ADR-0002). +* *BoJ-only MCP rule* (standing estate policy): All MCP access to +hyperpolymath services MUST route through BoJ. Standalone MCPs outside +BoJ are not permitted. Added explicit citation in +`+docs/FEDERATION.md+`. +* *Unified-zig-api stack alignment* (planned): BoJ will consume +`+developer-ecosystem/zig-api/+` — the unified Idris2 ABI + Zig runtime ++ C adaptor +** proven-backed path safety stack. `+UNIFIED-ZIG-API-STACK.adoc+` in +`+developer-ecosystem/+` is the canonical reference. BoJ does *not yet* +call `+libzig_api+` in code; alignment is tracked in ADR-0002 as future +work. First estate consumers wired on 2026-04-17: lol-gateway (commits +dbb475f/26b6b8c), aerie (e0b17f8), emergency-button/emergency-room +(4bd070b), proven→zig-api path-safety wiring (6663956), gen-header CI +drift check (0d6a814). +* *ADR-0002 added*: Documents the decision to align BoJ with the +unified-zig-api stack, with explicit status of current zig adapter +retirement and Zig migration. + +=== [0.3.0] — 2026-03-20 + +==== Added + +* Consolidated boj-server-mistral and boj-server-gemini into unified +repo +* PanLL AffineScript/TEA UI components (BojModel, BojEngine, Boj, +BojModule) +* Gemini CLI extension support (gemini-extension.json, GEMINI.md) +* 9 architecture docs: Quantum Security, HSM Integration, Cartridge +Marketplace, BoJ OS, Formal Verification, Type Safety, Zero Trust, SDP +Architecture, Gossip Protocol +* Cartridge tools specification (Minter, Provisioner, Configurator, +Panel Harness) +* Intentfile and Mustfile (contractile invariant declarations) +* Farm/fleet enrollment configs +* EXHIBIT-A (Ethical Use) and EXHIBIT-B (Quantum-Safe Provenance) +* Hypatia vulnerability-scanning and dependency-update rules + +==== Fixed + +* Constant-time comparison in webhook HMAC verification (timing attack +prevention) +* .mcp.json version aligned to 0.3.0 +* package.json license corrected to MPL-2.0 +* SPDX headers added to all new files + +==== Removed + +* boj-server-gemini repo (consolidated, deleted from GitHub) +* boj-server-mistral repo (consolidated, deleted locally) + +=== [0.2.0] — 2026-03-09 + +==== Added + +* Thread-safety hardening: `+std.Thread.Mutex+` on all 9 FFI modules (55 +globals, ~120 exports) +* 2 thread-safety seam checks (concurrent register+query, concurrent +mount+unmount) +* panic-attack assail validation (1 expected weak point in QUIC crypto, +0 critical) +* Third-axis extensibility (backend/provider dimension) with +extension.a2ml template +* MCP stdio bridge (`+boj-server --mcp+`, JSON-RPC 2.0, all 18 +cartridges as MCP tools) +* Seam checks module (15 panic-attack-style integration contract tests) +* SLA monitoring (3-tier: community/standard/premium, percentile +tracking, 11 tests) +* Community cartridge submissions (Ayo tier, review state machine, 11 +tests) +* Auto-SDP perimeter (zero-trust, allow-list, auto-ban, 10 tests) +* 4-continent seed node configuration (EU-West, EU-Central, US-East, +AP-South) +* QUIC-first transport (X25519+ChaCha20-Poly1305, backward compatible, +10 tests) +* Multi-node federation testing (11 tests, REST API peering) +* Coprocessor dispatch (Axiom.jl-style: detect→select→dispatch→fallback, +14 tests) +* Podman secure instance (quadlet + seccomp + read-only rootfs) +* docs/API-CONTRACT.md — stable API surface +* docs/GETTING-STARTED.md — clone→build→run→test→extend +* docs/EXTENSIBILITY.md — third axis and extension mechanism + +==== Fixed + +* V 0.5.0 http.Server auto-bind broken → pre-bind with net.listen_tcp +* Duplicate linker symbols (loader includes catalogue transitively) +* Deadlock in coprocessor select_by_name (calls selectDevice directly +under mutex) + +=== [0.1.0] — 2026-03-08 + +==== Added + +* Core catalogue ABI (Idris2) with IsUnbreakable proof gate +* Core catalogue FFI (Zig) with C-ABI exports +* Dynamic loader with SHA-256 hash verification +* Guardian resource-aware failure tolerance (12 tests) +* zig triple adapter (REST 7700 + gRPC 7701 + GraphQL 7702) +* 18 cartridges: database, fleet, nesy, agent, cloud, container, k8s, +git, secrets, queues, iac, observe, ssg, proof, lsp, dap, bsp, feedback +* All 18 cartridges with ABI + FFI + Adapter + .so shared library builds +* Umoja federation with QUIC+UDP gossip protocol (40 tests) +* VeriSimDB backing store integration (7 tests) +* PanLL BoJ panel (887 lines, 5 tabs) +* Containerfile (Chainguard base), compose.toml, vordr.toml +* CI pipeline (zig-test.yml) +* Configurable ports via environment variables diff --git a/CHANGELOG.md b/CHANGELOG.md deleted file mode 100644 index 7a328559..00000000 --- a/CHANGELOG.md +++ /dev/null @@ -1,343 +0,0 @@ - -# Changelog - -All notable changes to Bundle of Joy Server are documented here. - -## [0.4.7] — 2026-05-20 - -### Changed - -- **README install section expanded** to cover every major MCP client: Claude Code, - Claude Desktop, Gemini CLI, GitHub Copilot (VS Code), Cursor, Cline, Windsurf, - Continue.dev, Zed, plus a generic stdio template. Copy-paste-ready snippets - for each client's config-file path. -- **Runtime documentation corrected**: clone-and-configure path now lists Deno - (preferred per CLAUDE.md policy), Bun, and Node as equally valid runtimes for - `mcp-bridge/main.js`. Spurious `npm install` step removed — `package.json` - declares zero runtime dependencies, so no install is ever required. - -### Notes - -- This is the release that publishes the **AAA-tier tool descriptions** to npm. - The description rewrite landed in `25887157` / `ad837abe` after 0.4.1 but was - never npm-published; downstream MCP clients (and Glama's quality scoring) - were running 0.4.1 with the older one-liner descriptions. Republishing as - 0.4.7 ships the rich Purpose / Behavior / Returns / Errors / Usage text on - every tool and the per-parameter `description` fields with patterns/enums. - -## [Unreleased] - -### Added - -- **`k8s/networkpolicy.yaml`** — defence-in-depth `NetworkPolicy` restricting - BoJ pod ingress to pods labelled `app: http-capability-gateway`. Stacks on - top of the ClusterIP Service (#131) and Cowboy/Zig loopback binds (#130/#132): - three independent layers must be violated before BoJ's back-side surface is - reachable from anywhere other than HCG. Optional — Phase E acceptance does - not require it; CNI plugins without NetworkPolicy enforcement (e.g. - flannel without VXLAN) silently treat it as a no-op. Override pattern - documented in the manifest header for non-HCG-fronted deployments. Closes - #135. Refs [`hyperpolymath/standards#100`](https://github.com/hyperpolymath/standards/issues/100), - [`#91`](https://github.com/hyperpolymath/standards/issues/91). - -### Documentation - -- **HCG tier-2 rollout runbook refreshed (v0.1 → v0.2)** in - `docs/integration/hcg-tier2-rollout-runbook.md` to reflect the - post-Phase-D state of the single-lane channel rooted at - [`hyperpolymath/standards#91`](https://github.com/hyperpolymath/standards/issues/91). - §1.1 (Phase D deliverables) ticks D-1/D-2/D-3 + D-4 bootstrap + the - cross-repo D-1 load-profile (boj-server#168) with PR references; calls - out the remaining owner-driven D-4 rebaseline + `_status` flip as the - single open item. §1.4 (BoJ-side prereqs) ticks the loopback bind layers - (#130/#131/#132), the Phase C `TrustPolicy.satisfies?/3` clause (#106), - the NetworkPolicy (#173), and the SSE-route policy coverage (#165). - §1.5 (gateway-side prereqs) ticks the new `container/gateway-deploy.k9.ncl` - (http-capability-gateway#38) and records what's still placeholder until - cerro-torre signing runs. Header banner replaces the stale Phase-D-scaffold-only - note with current state. Refs - [`hyperpolymath/standards#100`](https://github.com/hyperpolymath/standards/issues/100), - [`#91`](https://github.com/hyperpolymath/standards/issues/91). - -- **Repository documentation reorganised to match rsr-template-repo taxonomy.** - Root-level `.adoc` clutter eliminated; all docs now live under `docs/` - subdirectories clustered by purpose (`quickstarts/`, `wikis/`, `architecture/`, - `status/`, `developer/`, `governance/`, `decisions/`, `specification/`, - `integration/`, `backend-assurance/`, `compliance/`, `practice/`, - `proposals/`, `attribution/`, `accessibility/`, `papers/`, `examples/`, - `glama/`, `outreach/`, `handover/`, `maintenance/`). Each subdirectory has - its own `README.adoc` index. High-coupling root files (`PROOF-NEEDS.md`, - `TOPOLOGY.md`, `TEST-NEEDS.md`) stay at root pending follow-up PR to update - their 16/11/5 cross-references. - -- **`README.md` merged into `README.adoc` and dropped.** Single canonical - root README in AsciiDoc. Preserves the full 11-client MCP install matrix - (Claude Code, Claude Desktop, Gemini CLI, Copilot, Cursor, Cline, Windsurf, - Continue.dev, Zed, generic stdio) and collapsible cartridge tables via - `[%collapsible]` blocks. - -- **All `docs/*.md` files converted to `.adoc`** (format only; content - preserved). Affected files: `ARCHITECTURE.md` → `docs/architecture/README.adoc`, - `DEVELOPERS.md` → `docs/developer/README.adoc`, `OPERATOR-QUICKSTART.md` → - `docs/quickstarts/MAINTAINER.adoc`, `DEVELOPER-QUICKSTART.md` → - `docs/quickstarts/DEV.adoc`, and 8 others. `docs/wikis/` sources fully - converted and expanded (Home, User-Guide, Operator-Guide, Developer-Guide, - FAQ all in `.adoc`). - -- **`docs/README.adoc` rewritten** with reading-order-by-audience table, - full directory taxonomy, standalone-docs table, and rationale for the three - high-coupling deferred moves. - -- **New subdir index files**: `docs/quickstarts/README.adoc`, - `docs/wikis/README.adoc`, `docs/status/README.adoc` — each explains its - subdirectory's scope and contents. - -- **STATE.a2ml cartridge count corrected**: was 112, actual is 125 (verified - by `find cartridges -name cartridge.json | wc -l`). Session log entry added - documenting all 2026-05-26 work. - -### Changed - -- **Cowboy listener now binds to `127.0.0.1` by default** (was: all - interfaces). Configurable via the `BOJ_BIND_IP` environment variable; - invalid values fail-fast at boot rather than silently falling back to - `0.0.0.0`. This is the code-enforced expression of the ADR-0004 §1 - invariant that BoJ's back-side bind is not externally routable in - deployments fronted by `http-capability-gateway` (HCG tier-2). Phase E - rollout-runbook §1.4 prerequisite #6. Legacy/standalone deployments - that want all-interfaces exposure must now opt in explicitly - (`BOJ_BIND_IP=0.0.0.0` or `BOJ_BIND_IP=::`). Refs - [`hyperpolymath/standards#100`](https://github.com/hyperpolymath/standards/issues/100), - [`#91`](https://github.com/hyperpolymath/standards/issues/91). - -- **k8s Service for BoJ is now `type: ClusterIP`** (was: `LoadBalancer`). - Per ADR-0004 §1 and the Phase E rollout-runbook §1.4 prereq #8, BoJ - must not be externally addressable when fronted by - `http-capability-gateway` (HCG tier-2). External clients reach HCG; - HCG forwards to BoJ over the pod-network loopback. Legacy/standalone - deployments that need BoJ exposed externally should override `type` - in a kustomize/helm overlay rather than reverting the canonical - manifest (see header comment in `k8s/service.yaml`). Adds - `hyperpolymath.dev/exposure: "internal-only"` and - `hyperpolymath.dev/external-via: "http-capability-gateway (tier-2)"` - annotations so the posture is discoverable from `kubectl describe`. - Refs - [`hyperpolymath/standards#100`](https://github.com/hyperpolymath/standards/issues/100), - [`#91`](https://github.com/hyperpolymath/standards/issues/91). - -- **Container `APP_HOST` default is now `127.0.0.1`** (was: `"[::]"` - IPv6 all-interfaces). Tightens three sites that feed the Zig adapter - binary's `--host` flag: `stapeln.toml [targets.production]`, - `container/entrypoint.sh`, and `container/compose.prod.yaml`. Same - Phase E posture as the Cowboy bind change in the Elixir path: BoJ - binds loopback by default when fronted by `http-capability-gateway` - (HCG tier-2). Legacy/standalone deployments without HCG in front - should override `APP_HOST=0.0.0.0` (IPv4 all-interfaces) or - `APP_HOST=::` (IPv6 all-interfaces) in their deployment config. - Phase E rollout-runbook §1.4 prereq #7. Refs - [`hyperpolymath/standards#100`](https://github.com/hyperpolymath/standards/issues/100), - [`#91`](https://github.com/hyperpolymath/standards/issues/91). - -### Added - -- **ADR-0014 — cross-cartridge composition safety (RFC)** — frames the - unresolved research question that the per-cartridge ABI proofs do not - compose automatically across `boj_cartridge_invoke`. Defines composition - safety as a two-level contract: a static Idris2 envelope - (`Boj.Composition.InvocationOf` lifting `IsUnbreakable` + `ProtocolMatch` - + per-cartridge `ArgsContract` into the inter-cartridge call) and a - dynamic Nickel `compositions` block in ADR-0007's `policy-mcp` PDP. - First proof pair is `panic-attack-mcp → vordr-mcp` (both cartridges - exist on disk); the prompt-suggested `panic-attack → sandbox → vordr` - chain is parked behind ADR-0009's `sandbox-mcp` build-out. - -- **README "Formal verification" section** — surfaces the audited posture - outside `PROOF-NEEDS.md` so external readers can see, without digging, - that all P1/P2 obligations are closed with constructive proofs and that - the remaining `believe_me` invocations are *principled assumptions over - Idris2 primitives*, not unproven debt. - -- **Streamable HTTP transport (ADR-0013, PR1 of 2)** — MCP bridge now selects - between stdio (default), `http`, and `both` via `BOJ_TRANSPORT`. HTTP - endpoints: `POST /mcp` for JSON-RPC, `GET /mcp` for the server-initiated - SSE notifications stream, `DELETE /mcp` for explicit session teardown, - `GET /healthz` for liveness. Sessions are server-issued UUIDs in the - `Mcp-Session-Id` header; the manager expires idle sessions after 30 min - and fans events out across attached SSE streams. Auth: `none` (loopback - only — refuses non-loopback binds) or `bearer` (token list via - `BOJ_HTTP_AUTH_TOKENS`). The same `hardeningGate` runs on every request. - Zero new deps — built on `Deno.serve` and `node:http`. mTLS / OIDC auth - and the Cloudflare Workers / Durable-Objects shim are owed in PR2. -- **`boj://capabilities/deployment` resource** — reports per-deployment - cartridge availability so clients can avoid invoking host-local-only - cartridges (browser-mcp, container-mcp, local-coord-mcp, sandbox-mcp, - ffmpeg-mcp) against a Worker / remote-HTTP deployment. -- **k9iser-mcp cartridge** — reference implementation of the `-iser` - regeneration-cartridge pattern (central K9 contract regeneration), mirroring - ssg-mcp: `cartridge.json`, `mod.js`, Idris2 ABI, Zig FFI, panels. -- **Unified transaction-gated adapter**: one internal/loopback listener, - protocol-routed REST + SSE + GraphQL + gRPC-compat → single dispatch → one - Zig ABI. Replaces the ssg-era 3-parallel-port anti-pattern; the trust gate - runs before every dispatch, mirroring the Idris2 `exposureSatisfied` - contract (no gatekeeperless path). Internal-only behind - `http-capability-gateway` per ADR-0004. -- **boj-rest SSE surface**: `POST /cartridge/:name/sse` on the same single - Cowboy listener and trust-gated dispatch, `text/event-stream`. - -### Changed - -- **Doc reconciliation to ADR-0004**: `elixir/README.adoc`, - `mcp-bridge/api-clients.js`, and `OPERATOR-QUICKSTART.md` corrected to the - verified runtime + ADR-0004 tiered model (they previously and wrongly - described it as "skeleton/501/pending rewrite"). - -### Fixed - -- **`Boj.SafeAPIKey.logSafeBounded` rebuilt for Idris2 0.8.0.** The pre- - existing proof did not type-check on `main`; the 2026-05-18 audit's claim - that `SafeAPIKey` carried constructive proofs closing BJ2-partial was a - desk-read, not a build. Three independent defects: (1) removed the - redundant local `plusLteMonotone` helper (called now-gone `lteTransitive` - and used wrong arg order on `plusLteMonotoneRight`/`Left`; stdlib's - `Data.Nat.plusLteMonotone` has exactly the needed shape); (2) lifted both - short and long paths out of the `with`-block (the elaborator doesn't - reduce `length "***"` at type level inside a `with`-block — goal stays - as `LTE (integerToNat (prim__cast_IntInteger (prim__strLength (if ...)))) - 11` with the `if`-arm unreduced); (3) right-associated the long-path - proof to match `++`'s associativity (`a ++ b ++ c = a ++ (b ++ c)`). - Plus two bound-name typos in `toLogSafeShortEq`/`toLogSafeLongEq`. All - 12 safety modules now build green via per-module `idris2 --check`. No - new `believe_me` axioms. - -- **`tests/aspect_tests.sh` grep-count bash bug.** `Aspect — Thread - Safety + ABI Contract + SPDX` had been red on `main`, gating every PR - with `tests/aspect_tests.sh: line 77: [[: 0\n0: syntax error in - expression`. Root cause: `grep -c 'pattern' file 2>/dev/null || echo - "0"`. `grep -c` always prints the count (including `0`) **and** exits - non-zero on no-match, so `|| echo "0"` also fires — `has_export` ends - up `"0\n0"` and `[[ "0\n0" -gt 0 ]]` chokes on the newline in - arithmetic context. Swapped `|| echo "0"` → `|| true` on all four - call-sites. - -- **Honest framing of the ABI axiom count.** `src/abi/Boj/SafetyLemmas.idr`'s - module docstring claimed "Three axiomatic `believe_me` primitives" while - five live in the file. Docstring now enumerates all five and tags each to - its underlying `prim__*` primitive. `appendLengthSum` and - `substrLengthBound` also had `(x y : T)` multi-binder syntax that Idris2 - 0.8.0 rejects at parse time — comma-separated form `(x, y : T)` restores - parsability. Types and proof terms unchanged. The 2026-05-18 - `PROOF-NEEDS.md` audit (5 axioms, all class (J) — irreducible over Idris2 - primitives, principled assumptions not unproven debt) is now consistent - with the source and surfaced via the new README "Formal verification" - section. - -- **`dogfood-gate.yml` failed YAML validation at startup** (0 s, no jobs) on - every branch including `main`: an inline `python3 -c "` block placed Python - source at column 1 inside a `run: |` block scalar, terminating the scalar - early. Because **Dogfood Gate** is a required status check, this silently - blocked every PR in the repo. The validator now lives in - `.github/scripts/validate-eclexiaiser.py` and is invoked from the workflow. - -> Verification (k9iser-mcp): Elixir suite 177/177 (incl. 2 SSE tests); Zig -> ffi 16/16 and unified adapter 5/5 (exposure-gate truth table mirroring the -> Idris2 contract); `idris2 --check K9iserMcp/SafeK9iser.idr` passes. -> http-capability-gateway production-wiring (ADR-0004 tier-2) and the -> iseriser-scaffold rollout remain out of scope and separately tracked. - -## [0.4.0] — 2026-04-17 - -### Changed - -- **zig banned estate-wide (2026-04-10)**: Adapter layer language policy updated. - zig is no longer an accepted cartridge adapter language. Zig is the default - replacement for the adapter tier (`ffi/zig/` remains; V adapter files were swept - in commit c4674f8). Historical zig API interfaces have been moved to - `developer-ecosystem/v-ecosystem/v-api-interfaces/v-/` for potential - donation to the V community — they are not HP infrastructure. -- **Cartridge manifests = Nickel** (prior closed decision `boj-cartridge-manifest-format-dd.md`): - The authoritative cartridge manifest format is Nickel (`.ncl`). Current on-disk - manifests are `cartridge.json`; migration to Nickel is tracked as future work - (see open question in ADR-0002). -- **BoJ-only MCP rule** (standing estate policy): All MCP access to hyperpolymath - services MUST route through BoJ. Standalone MCPs outside BoJ are not permitted. - Added explicit citation in `docs/FEDERATION.md`. -- **Unified-zig-api stack alignment** (planned): BoJ will consume - `developer-ecosystem/zig-api/` — the unified Idris2 ABI + Zig runtime + C adaptor - + proven-backed path safety stack. `UNIFIED-ZIG-API-STACK.adoc` in - `developer-ecosystem/` is the canonical reference. BoJ does **not yet** call - `libzig_api` in code; alignment is tracked in ADR-0002 as future work. - First estate consumers wired on 2026-04-17: lol-gateway (commits dbb475f/26b6b8c), - aerie (e0b17f8), emergency-button/emergency-room (4bd070b), - proven→zig-api path-safety wiring (6663956), gen-header CI drift check (0d6a814). -- **ADR-0002 added**: Documents the decision to align BoJ with the unified-zig-api - stack, with explicit status of current zig adapter retirement and Zig migration. - -## [0.3.0] — 2026-03-20 - -### Added -- Consolidated boj-server-mistral and boj-server-gemini into unified repo -- PanLL AffineScript/TEA UI components (BojModel, BojEngine, Boj, BojModule) -- Gemini CLI extension support (gemini-extension.json, GEMINI.md) -- 9 architecture docs: Quantum Security, HSM Integration, Cartridge Marketplace, - BoJ OS, Formal Verification, Type Safety, Zero Trust, SDP Architecture, Gossip Protocol -- Cartridge tools specification (Minter, Provisioner, Configurator, Panel Harness) -- Intentfile and Mustfile (contractile invariant declarations) -- Farm/fleet enrollment configs -- EXHIBIT-A (Ethical Use) and EXHIBIT-B (Quantum-Safe Provenance) -- Hypatia vulnerability-scanning and dependency-update rules - -### Fixed -- Constant-time comparison in webhook HMAC verification (timing attack prevention) -- .mcp.json version aligned to 0.3.0 -- package.json license corrected to MPL-2.0 -- SPDX headers added to all new files - -### Removed -- boj-server-gemini repo (consolidated, deleted from GitHub) -- boj-server-mistral repo (consolidated, deleted locally) - -## [0.2.0] — 2026-03-09 - -### Added -- Thread-safety hardening: `std.Thread.Mutex` on all 9 FFI modules (55 globals, ~120 exports) -- 2 thread-safety seam checks (concurrent register+query, concurrent mount+unmount) -- panic-attack assail validation (1 expected weak point in QUIC crypto, 0 critical) -- Third-axis extensibility (backend/provider dimension) with extension.a2ml template -- MCP stdio bridge (`boj-server --mcp`, JSON-RPC 2.0, all 18 cartridges as MCP tools) -- Seam checks module (15 panic-attack-style integration contract tests) -- SLA monitoring (3-tier: community/standard/premium, percentile tracking, 11 tests) -- Community cartridge submissions (Ayo tier, review state machine, 11 tests) -- Auto-SDP perimeter (zero-trust, allow-list, auto-ban, 10 tests) -- 4-continent seed node configuration (EU-West, EU-Central, US-East, AP-South) -- QUIC-first transport (X25519+ChaCha20-Poly1305, backward compatible, 10 tests) -- Multi-node federation testing (11 tests, REST API peering) -- Coprocessor dispatch (Axiom.jl-style: detect→select→dispatch→fallback, 14 tests) -- Podman secure instance (quadlet + seccomp + read-only rootfs) -- docs/API-CONTRACT.md — stable API surface -- docs/GETTING-STARTED.md — clone→build→run→test→extend -- docs/EXTENSIBILITY.md — third axis and extension mechanism - -### Fixed -- V 0.5.0 http.Server auto-bind broken → pre-bind with net.listen_tcp -- Duplicate linker symbols (loader includes catalogue transitively) -- Deadlock in coprocessor select_by_name (calls selectDevice directly under mutex) - -## [0.1.0] — 2026-03-08 - -### Added -- Core catalogue ABI (Idris2) with IsUnbreakable proof gate -- Core catalogue FFI (Zig) with C-ABI exports -- Dynamic loader with SHA-256 hash verification -- Guardian resource-aware failure tolerance (12 tests) -- zig triple adapter (REST 7700 + gRPC 7701 + GraphQL 7702) -- 18 cartridges: database, fleet, nesy, agent, cloud, container, k8s, git, secrets, queues, iac, observe, ssg, proof, lsp, dap, bsp, feedback -- All 18 cartridges with ABI + FFI + Adapter + .so shared library builds -- Umoja federation with QUIC+UDP gossip protocol (40 tests) -- VeriSimDB backing store integration (7 tests) -- PanLL BoJ panel (887 lines, 5 tabs) -- Containerfile (Chainguard base), compose.toml, vordr.toml -- CI pipeline (zig-test.yml) -- Configurable ports via environment variables diff --git a/CODE_OF_CONDUCT.adoc b/CODE_OF_CONDUCT.adoc new file mode 100644 index 00000000..4ded1ed6 --- /dev/null +++ b/CODE_OF_CONDUCT.adoc @@ -0,0 +1,129 @@ +== Contributor Covenant Code of Conduct + +=== Our Pledge + +We as members, contributors, and leaders pledge to make participation in +our community a harassment-free experience for everyone, regardless of +age, body size, visible or invisible disability, ethnicity, sex +characteristics, gender identity and expression, level of experience, +education, socio-economic status, nationality, personal appearance, +race, religion, or sexual identity and orientation. + +We pledge to act and interact in ways that contribute to an open, +welcoming, diverse, inclusive, and healthy community. + +=== Our Standards + +Examples of behavior that contributes to a positive environment for our +community include: + +* Demonstrating empathy and kindness toward other people +* Being respectful of differing opinions, viewpoints, and experiences +* Giving and gracefully accepting constructive feedback +* Accepting responsibility and apologizing to those affected by our +mistakes, and learning from the experience +* Focusing on what is best not just for us as individuals, but for the +overall community + +Examples of unacceptable behavior include: + +* Trolling, insulting or derogatory comments, and personal or political +attacks +* Public or private harassment of any kind +* Publishing others’ private information, such as a physical or email +address, without their explicit permission +* Other conduct which could reasonably be considered inappropriate in a +professional setting + +=== Enforcement Responsibilities + +Community leaders are responsible for clarifying and enforcing our +standards of acceptable behavior and will take appropriate and fair +corrective action in response to any behavior that they deem +inappropriate, threatening, offensive, or harmful. + +Community leaders have the right and responsibility to remove, edit, or +reject comments, commits, code, wiki edits, issues, and other +contributions that are not aligned to this Code of Conduct, and will +communicate reasons for moderation decisions when appropriate. + +=== Scope + +This Code of Conduct applies within all community spaces, and also +applies when an individual is officially representing the community in +public spaces. Examples of representing our community include using an +official e-mail address, posting via an official social media account, +or acting as an appointed representative at an online or offline event. + +=== Enforcement + +Instances of unacceptable behavior may be reported to the community +leaders responsible for enforcement at j.d.a.jewell@open.ac.uk. All +complaints will be reviewed and investigated promptly and fairly. + +All community leaders are obligated to respect the privacy and security +of the reporter of any incident. + +=== Enforcement Guidelines + +Community leaders will follow these Community Impact Guidelines in +determining the consequences for any action they deem in violation of +this Code of Conduct: + +==== 1. Correction + +*Community Impact*: Use of inappropriate language or other behavior +deemed unprofessional or unwelcome in the community. + +*Consequence*: A private, written warning from community leaders, +providing clarity around the nature of the violation and an explanation +of why the behavior was inappropriate. A public apology may be +requested. + +==== 2. Warning + +*Community Impact*: A violation through a single incident or series of +actions. + +*Consequence*: A warning with consequences for continued behavior. No +interaction with the people involved, including unsolicited interaction +with those enforcing the Code of Conduct, for a specified period of +time. This includes avoiding interactions in community spaces as well as +external channels like social media. Violating these terms may lead to a +temporary or permanent ban. + +==== 3. Temporary Ban + +*Community Impact*: A serious violation of community standards, +including sustained inappropriate behavior. + +*Consequence*: A temporary ban from any sort of interaction or public +communication with the community for a specified period of time. No +public or private interaction with the people involved, including +unsolicited interaction with those enforcing the Code of Conduct, is +allowed during this period. Violating these terms may lead to a +permanent ban. + +==== 4. Permanent Ban + +*Community Impact*: Demonstrating a pattern of violation of community +standards, including sustained inappropriate behavior, or aggression +toward or disparagement of classes of individuals. + +*Consequence*: A permanent ban from any sort of public interaction +within the community. + +=== Attribution + +This Code of Conduct is adapted from the +https://www.contributor-covenant.org[Contributor Covenant], version 2.1, +available at +https://www.contributor-covenant.org/version/2/1/code_of_conduct.html. + +Community Impact Guidelines were inspired by +https://github.com/mozilla/diversity[Mozilla’s code of conduct +enforcement ladder]. + +For answers to common questions about this code of conduct, see the FAQ +at https://www.contributor-covenant.org/faq. Translations are available +at https://www.contributor-covenant.org/translations. diff --git a/CODE_OF_CONDUCT.md b/CODE_OF_CONDUCT.md deleted file mode 100644 index 05a6fe18..00000000 --- a/CODE_OF_CONDUCT.md +++ /dev/null @@ -1,129 +0,0 @@ - -# Contributor Covenant Code of Conduct - -## Our Pledge - -We as members, contributors, and leaders pledge to make participation in our -community a harassment-free experience for everyone, regardless of age, body -size, visible or invisible disability, ethnicity, sex characteristics, gender -identity and expression, level of experience, education, socio-economic status, -nationality, personal appearance, race, religion, or sexual identity -and orientation. - -We pledge to act and interact in ways that contribute to an open, welcoming, -diverse, inclusive, and healthy community. - -## Our Standards - -Examples of behavior that contributes to a positive environment for our -community include: - -* Demonstrating empathy and kindness toward other people -* Being respectful of differing opinions, viewpoints, and experiences -* Giving and gracefully accepting constructive feedback -* Accepting responsibility and apologizing to those affected by our mistakes, - and learning from the experience -* Focusing on what is best not just for us as individuals, but for the - overall community - -Examples of unacceptable behavior include: - -* Trolling, insulting or derogatory comments, and personal or political attacks -* Public or private harassment of any kind -* Publishing others' private information, such as a physical or email - address, without their explicit permission -* Other conduct which could reasonably be considered inappropriate in a - professional setting - -## Enforcement Responsibilities - -Community leaders are responsible for clarifying and enforcing our standards of -acceptable behavior and will take appropriate and fair corrective action in -response to any behavior that they deem inappropriate, threatening, offensive, -or harmful. - -Community leaders have the right and responsibility to remove, edit, or reject -comments, commits, code, wiki edits, issues, and other contributions that are -not aligned to this Code of Conduct, and will communicate reasons for moderation -decisions when appropriate. - -## Scope - -This Code of Conduct applies within all community spaces, and also applies when -an individual is officially representing the community in public spaces. -Examples of representing our community include using an official e-mail address, -posting via an official social media account, or acting as an appointed -representative at an online or offline event. - -## Enforcement - -Instances of unacceptable behavior may be reported to the community leaders -responsible for enforcement at j.d.a.jewell@open.ac.uk. -All complaints will be reviewed and investigated promptly and fairly. - -All community leaders are obligated to respect the privacy and security of the -reporter of any incident. - -## Enforcement Guidelines - -Community leaders will follow these Community Impact Guidelines in determining -the consequences for any action they deem in violation of this Code of Conduct: - -### 1. Correction - -**Community Impact**: Use of inappropriate language or other behavior deemed -unprofessional or unwelcome in the community. - -**Consequence**: A private, written warning from community leaders, providing -clarity around the nature of the violation and an explanation of why the -behavior was inappropriate. A public apology may be requested. - -### 2. Warning - -**Community Impact**: A violation through a single incident or series -of actions. - -**Consequence**: A warning with consequences for continued behavior. No -interaction with the people involved, including unsolicited interaction with -those enforcing the Code of Conduct, for a specified period of time. This -includes avoiding interactions in community spaces as well as external channels -like social media. Violating these terms may lead to a temporary or -permanent ban. - -### 3. Temporary Ban - -**Community Impact**: A serious violation of community standards, including -sustained inappropriate behavior. - -**Consequence**: A temporary ban from any sort of interaction or public -communication with the community for a specified period of time. No public or -private interaction with the people involved, including unsolicited interaction -with those enforcing the Code of Conduct, is allowed during this period. -Violating these terms may lead to a permanent ban. - -### 4. Permanent Ban - -**Community Impact**: Demonstrating a pattern of violation of community -standards, including sustained inappropriate behavior, or aggression toward -or disparagement of classes of individuals. - -**Consequence**: A permanent ban from any sort of public interaction within -the community. - -## Attribution - -This Code of Conduct is adapted from the [Contributor Covenant][homepage], -version 2.1, available at -https://www.contributor-covenant.org/version/2/1/code_of_conduct.html. - -Community Impact Guidelines were inspired by [Mozilla's code of conduct -enforcement ladder](https://github.com/mozilla/diversity). - -[homepage]: https://www.contributor-covenant.org - -For answers to common questions about this code of conduct, see the FAQ at -https://www.contributor-covenant.org/faq. Translations are available at -https://www.contributor-covenant.org/translations. diff --git a/CONTRIBUTING.adoc b/CONTRIBUTING.adoc new file mode 100644 index 00000000..94aa289b --- /dev/null +++ b/CONTRIBUTING.adoc @@ -0,0 +1,83 @@ +== Contributing + +Thank you for your interest in contributing! We follow a "`Dual-Track`" +architecture where human-readable documentation lives in the root and +machine-readable policies live in `+.machine_readable/+`. + +=== How to Contribute + +We welcome contributions in many forms: + +* *Code:* Improving the core stack or extensions +* *Documentation:* Enhancing docs or AI manifests +* *Testing:* Adding property-based tests or formal proofs +* *Bug reports:* Filing clear, reproducible issues + +=== Getting Started + +[arabic] +. *Read the AI Manifest:* Start with `+0-AI-MANIFEST.a2ml+` (if present) +to understand the repository structure. +. *Environment:* Use `+guix develop+` or `+direnv allow+` to set up your +tools. +. *Task Runner:* Use `+just+` to see available commands +(`+just --list+`). + +=== Development Workflow + +==== Branch Naming + +.... +docs/short-description # Documentation +test/what-added # Test additions +feat/short-description # New features +fix/issue-number-description # Bug fixes +refactor/what-changed # Code improvements +security/what-fixed # Security fixes +.... + +==== Commit Messages + +We follow https://www.conventionalcommits.org/[Conventional Commits]: + +.... +(): + +[optional body] + +[optional footer] +.... + +Types: `+feat+`, `+fix+`, `+docs+`, `+test+`, `+refactor+`, `+ci+`, +`+chore+`, `+security+` + +==== CI / Required Checks + +Required status-check workflows must *always report*. Never add +`+on.*.paths+` to a required workflow — a path-filtered required check +that doesn’t trigger is reported as permanently "`Expected`" and blocks +the PR even when everything is green. Use the estate pattern: an +always-run `+changes+` job plus heavy jobs gated by +`+if: needs.changes.outputs.run == 'true'+` (a job skipped via `+if:+` +passes the required check). Full rationale in +`+docs/AI-CONVENTIONS.adoc+` §"`CI / Required Status Checks`" and the +`+docs/wikis/CI-and-Required-Checks.adoc+` wiki page. + +=== Reporting Bugs + +Before reporting: 1. Search existing issues 2. Check if it’s already +fixed in `+main+` + +When reporting, include: - Clear, descriptive title - Environment +details (OS, versions, toolchain) - Steps to reproduce - Expected vs +actual behaviour + +=== Code of Conduct + +All contributors are expected to adhere to our +link:CODE_OF_CONDUCT.md[Code of Conduct]. + +=== License + +By contributing, you agree that your contributions will be licensed +under the same license as the project (see LICENSE). diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md deleted file mode 100644 index 3b5442cd..00000000 --- a/CONTRIBUTING.md +++ /dev/null @@ -1,73 +0,0 @@ - -# Contributing - -Thank you for your interest in contributing! We follow a "Dual-Track" architecture where human-readable documentation lives in the root and machine-readable policies live in `.machine_readable/`. - -## How to Contribute - -We welcome contributions in many forms: - -- **Code:** Improving the core stack or extensions -- **Documentation:** Enhancing docs or AI manifests -- **Testing:** Adding property-based tests or formal proofs -- **Bug reports:** Filing clear, reproducible issues - -## Getting Started - -1. **Read the AI Manifest:** Start with `0-AI-MANIFEST.a2ml` (if present) to understand the repository structure. -2. **Environment:** Use `guix develop` or `direnv allow` to set up your tools. -3. **Task Runner:** Use `just` to see available commands (`just --list`). - -## Development Workflow - -### Branch Naming - -``` -docs/short-description # Documentation -test/what-added # Test additions -feat/short-description # New features -fix/issue-number-description # Bug fixes -refactor/what-changed # Code improvements -security/what-fixed # Security fixes -``` - -### Commit Messages - -We follow [Conventional Commits](https://www.conventionalcommits.org/): - -``` -(): - -[optional body] - -[optional footer] -``` - -Types: `feat`, `fix`, `docs`, `test`, `refactor`, `ci`, `chore`, `security` - -### CI / Required Checks - -Required status-check workflows must **always report**. Never add `on.*.paths` to a required workflow — a path-filtered required check that doesn't trigger is reported as permanently "Expected" and blocks the PR even when everything is green. Use the estate pattern: an always-run `changes` job plus heavy jobs gated by `if: needs.changes.outputs.run == 'true'` (a job skipped via `if:` passes the required check). Full rationale in `docs/AI-CONVENTIONS.adoc` §"CI / Required Status Checks" and the `docs/wikis/CI-and-Required-Checks.adoc` wiki page. - -## Reporting Bugs - -Before reporting: -1. Search existing issues -2. Check if it's already fixed in `main` - -When reporting, include: -- Clear, descriptive title -- Environment details (OS, versions, toolchain) -- Steps to reproduce -- Expected vs actual behaviour - -## Code of Conduct - -All contributors are expected to adhere to our [Code of Conduct](CODE_OF_CONDUCT.md). - -## License - -By contributing, you agree that your contributions will be licensed under the same license as the project (see [LICENSE](LICENSE)). diff --git a/DEBT.adoc b/DEBT.adoc new file mode 100644 index 00000000..dd80460f --- /dev/null +++ b/DEBT.adoc @@ -0,0 +1,360 @@ +== Technical debt register + +One index of known debt in this repository, measured 2026-08-07 against +`+78c4a05a+`. Every item carries *the command that produced the +evidence*, so any entry can be re-checked or falsified in one step. +Claims that are not verified are labelled *DIAGNOSIS (unconfirmed)* +rather than asserted. + +This file is an index, not a replacement. The pre-existing registers +remain authoritative in their own domains and are linked, not +duplicated: link:PROOF-NEEDS.md[`+PROOF-NEEDS.md+`] · +link:TEST-NEEDS.md[`+TEST-NEEDS.md+`] · +link:docs/proof-debt.md[`+docs/proof-debt.md+`] · +link:docs/tech-debt-2026-05-26.md[`+docs/tech-debt-2026-05-26.md+`]. + +Severity: *HIGH* — actively misleads, or a gate that cannot fail · +*MEDIUM* — wrong but self-evident on contact · *LOW* — cosmetic or +historical. + +''''' + +=== The single largest item + +*The `+cartridges/+` retirement (#300) is incomplete.* Removing 128 +cartridges took the manifests but left every consumer behind: two +permanently-off workflows, scripts whose loops match nothing, Justfile +recipes, test scripts, count claims in fourteen documents, and 1,346 +files of build residue. Items C-1…C-6, D-1…D-4, T-1 and X-1 below are +all one migration, not ten problems. + +[source,sh] +---- +git grep -nF 'cartridges/' -- ':!docs' ':!*.md' ':!*.adoc' ':!tests/fixtures' | wc -l +---- + +''''' + +=== Licence — L + +[width="100%",cols="16%,20%,24%,40%",options="header",] +|=== +|ID |Sev |Item |Evidence +|L-1 |MEDIUM |`+glama.json+` is the only packaging manifest with *no +`+license+` field at all*. Every sibling declares MPL-2.0 +(`+package.json+`, `+jsr.json+`, `+smithery.yaml+`, `+CITATION.cff+`, +`+guix.scm+`, `+elixir/mix.exs+`). +|`+grep -L '"license"' glama.json package.json jsr.json+` + +|L-2 |LOW |`+ai-plugin.json+` declares licence only as a URL +(`+legal_info_url+`), with no SPDX key — inconsistent with the rest of +the estate. |`+grep -n 'legal_info_url\|license' ai-plugin.json+` + +|L-3 |LOW |Four tracked source files of 500+ carry no +`+SPDX-License-Identifier+`: `+.github/copilot/coding-agent.yml+`, +`+.github/funding.yml+`, +`+.machine_readable/scripts/forge/git-cleanup.sh+`, +`+configs/config.ncl+`. +|`+git ls-files \| xargs grep -L 'SPDX-License-Identifier' 2>/dev/null+` +|=== + +*Not debt, recorded as the positive control:* the dual-licence posture +is correct and documented — `+NOTICE+` explains MPL-2.0 (code) / +CC-BY-SA-4.0 (prose), `+LICENSES/+` holds both texts, and +`+.reuse/dep5+` covers headerless config. 335 MPL / 184 CC-BY-SA +headers, zero third licence, zero unattributed vendored trees. *The +sibling registry has neither `+NOTICE+` nor `+.reuse/+` — see its own +`+DEBT.md+` L-1.* + +''''' + +=== Proof — P + +The proofs themselves are in good order: *4* `+believe_me+` sites, all +inside the sanctioned module, all `+%unsafe+`-tagged; zero +`+postulate+`, `+assert_total+`, `+assert_smaller+`, `+idris_crash+`, +`+sorry+`, `+%default partial+`, `+?hole+` anywhere in `+src/abi/+`. +*The debt is in the gates and the prose, not the proofs.* + +[source,sh] +---- +grep -rn 'believe_me' src/abi --include='*.idr' | grep -v '|||' # 4 sites +grep -n 'EXPECTED_AXIOMS=' scripts/check-trusted-base.sh # 4 +---- + +[width="100%",cols="16%,20%,24%,40%",options="header",] +|=== +|ID |Sev |Item |Evidence +|P-1 |HIGH |*`+proofs.yml+` can report success with no prover having +run.* The `+changes+` job sets `+run=false+` for any PR outside its path +set, and both proof jobs are +`+if: needs.changes.outputs.run == 'true'+`; a skipped job reports +SUCCESS to a required check. The weekly `+cron+` at `+proofs.yml+` is +the only backstop. This is a deliberate design (documented in the +workflow header) — recorded here because the failure mode is invisible +to a reviewer reading a green tick. +|`+grep -n "run=false\|needs.changes.outputs.run" .github/workflows/proofs.yml+` + +|P-2 |MEDIUM |`+scripts/check-trusted-base.sh+` still greps +`+src/ cartridges/ verification/+`; one of the three no longer exists, +so the scan surface is a third smaller than it reads. The axiom count +itself still works. +|`+grep -n 'cartridges/' scripts/check-trusted-base.sh+` + +|P-3 |MEDIUM |`+PROOF-NEEDS.md+` cites +`+cartridges/fleet-mcp/abi/FleetMcp/SafeFleet.idr lines 14 & 34+` — a +proof obligation anchored to a file in the _other_ repo, so the line +numbers cannot be checked from here. +|`+grep -n 'SafeFleet' PROOF-NEEDS.md+` + +|P-4 |LOW |*FIXED 2026-08-07, retained for provenance.* +`+PROOF-NEEDS.md+` asserted `+PASS=105+` and "`**exactly 5**`" axioms; +the enforcing script has said `+EXPECTED_AXIOMS=4+` since `+charEqSym+` +was discharged, and the gate now covers 1 package. Two "`in sync`" +documents disagreed with each other and with the code. +|`+git log -1 --format=%h -- PROOF-NEEDS.md+` +|=== + +*Cross-repo:* `+boj-server-cartridges/scripts/check-trusted-base.sh+` +still says _"`boj-server sanctions EXACTLY 5 class-(J) axioms`"_. Fixing +it there needs this file’s correction to land first. + +''''' + +=== CI/CD — C + +[width="100%",cols="16%,20%,24%,40%",options="header",] +|=== +|ID |Sev |Item |Evidence +|C-1 |HIGH |*`+abi-drift.yml+` is permanently off* — +`+printf 'run=false'+` unconditionally, so job +`+Emit manifest + verify FFI+` never runs while still satisfying its +required check by skipping. Its subject (per-cartridge iseriser drift) +must be ported to the registry before the workflow _and its required +context_ are deleted. The port has not happened. +|`+grep -n "run=false" .github/workflows/abi-drift.yml+` + +|C-2 |HIGH |*`+lsp-dap-bsp.yml+` is permanently off* — same mechanism, +*four* dead jobs: ABI Specification Check, FFI Build & Test, Panel +Manifest Validation, Cartridge Completeness Check. +|`+grep -n "run=false" .github/workflows/lsp-dap-bsp.yml+` + +|C-3 |HIGH |*`+main-estate-audit.yml+` is still untracked here, +deliberately — arming it today would fail `+main+` immediately.* The +referenced suite is now published (`+hyperpolymath/cicd-suite+`, +`+11b5ab51+`, all 26 actions resolve), so the 404 is fixed. But running +its two relevant hard gates against this repo: `+required-files-check+` +fails on 3 missing files (`+CODEOWNERS+`, `+ARCHITECTURE.md+`, +`+MAINTAINERS.adoc+`), and `+code-hygiene-check+` matched *112 files* +before cicd-suite#1; *2* after — including this repo’s four _sanctioned, +documented, CI-counted_ `+believe_me+` axioms, which are its declared +trusted base, not debt. Satisfying `+required-files-check+` means adding +presence-only filler, which is how the template boilerplate on +`+fix/zig-ptr-cast-shim+` was generated. *Fix the gates (see +cicd-suite’s README), then pin to `+11b5ab51+` and commit.* +|`+for f in CODEOWNERS ARCHITECTURE.md MAINTAINERS.adoc; do [ -f $f ] \|\| echo MISSING $f; done+` +· +`+git grep -Eic 'TODO\|FIXME\|STUB\|sorry\|believe_me\|admit' \| wc -l+` +→ 112 + +|C-7 |MEDIUM |*13 of `+cicd-suite+`’s 26 actions cannot fail* — they +emit `+::warning::+` and exit successfully while being named "`Gate`". +_(Was HIGH and three-part; two of the three are fixed — cicd-suite#1 +repaired `+code-hygiene-check+`’s whole-tree grep and +`+required-files-check+`’s presence-only filler, and ended their mutual +contradiction. The advisory/enforcing split is what remains.)_ +Estate-wide: the consuming workflow sits in *199 repos*, untracked in +*198*. +|`+for a in ../cicd-suite/actions/*/; do grep -q 'exit 1' $a/action.yml \|\| echo $a; done \| wc -l+` +→ 13 + +|C-4 |MEDIUM |Five required status-check contexts correspond to jobs +that are green-by-skip (C-1, C-2). A reviewer cannot distinguish +"`passed`" from "`never ran`". +|`+gh api repos/:owner/:repo/branches/main/protection+` + +|C-5 |MEDIUM |`+fuzz.yml+` suppresses failure twice over: `+\|\| true+` +*and* `+continue-on-error: true+`, with stderr sent to `+/dev/null+`. A +crash is invisible rather than merely non-blocking. The bridge probes +(including a `+../../../etc/passwd+` traversal case) assert nothing. +|`+grep -n 'continue-on-error\|\|\| true' .github/workflows/fuzz.yml+` + +|C-6 |LOW |`+pages.yml+` and `+pages-deploy.yml+` both fire on push to +main, publishing different content to two different hosts with no +coordination. +|`+grep -l 'branches: \[main' .github/workflows/pages*.yml+` +|=== + +*FIXED 2026-08-07* (recorded so the pattern is searchable): two gates — +`+tests/truthfulness_check.sh+` and `+scripts/typecheck-proofs.sh+` — +looped over the deleted tree, matched zero files, and exited 0 reporting +success. Both now fail hard on an empty subject. _A gate that cannot +fail is worse than no gate, because it is credited as assurance._ + +''''' + +=== Code — D + +[width="100%",cols="16%,20%,24%,40%",options="header",] +|=== +|ID |Sev |Item |Evidence +|D-1 |MEDIUM |Justfile recipes still operate on the deleted tree; +`+CART_COUNT=$(ls -d cartridges/*-mcp \| wc -l)+` now reports 0 as +though that were a fact about the system. +|`+grep -n 'cartridges/' Justfile+` + +|D-2 |MEDIUM |`+scripts/refresh-bundled-cartridges.sh+` exists solely to +sync the retired tree (it `+rm -rf+`s inside it). +`+scripts/boj-selinux-contexts.sh+` labels `+${BOJ_ROOT}/cartridges/+`. +|`+grep -ln 'cartridges/' scripts/*.sh+` + +|D-3 |MEDIUM |`+mcp-bridge/lib/generate-offline-menu.js+` falls back to +`+../../cartridges+` when `+BOJ_CARTRIDGES_PATH+` is unset — so it +regenerates an *empty menu* silently instead of failing. +|`+grep -n 'cartridges' mcp-bridge/lib/generate-offline-menu.js+` + +|D-4 |MEDIUM |Test scripts still traverse the tree: +`+tests/aspect_tests.sh+`, `+tests/integration.sh+`, +`+tests/federation_multinode.sh+`. +|`+grep -ln 'cartridges/' tests/*.sh+` + +|D-5 |MEDIUM |*Two git worktrees are committed as gitlinks (mode +`+160000+`) with no `+.gitmodules+`.* A fresh clone gets two empty +directories, and both show as permanently modified because neither +matches its recorded commit. |`+git ls-files -s .claude/worktrees/+` · +`+ls .gitmodules+` + +|D-8 |HIGH |*`+container/Containerfile.fly:80+` cannot build.* It does +`+COPY cartridges/ /tmp/carts-meta/+` from the host build context and +never runs `+fetch-cartridges.sh+`; `+COPY+` on a missing source is a +hard failure. (The main `+container/Containerfile+` is *fine* — it +fetches into the builder stage first, so its `+COPY --from=zig-builder+` +is populated. One file, not both.) +|`+grep -n 'COPY cartridges/' container/Containerfile.fly+` · +`+grep -c fetch-cartridges container/Containerfile.fly+` → 0 + +|D-9 |MEDIUM |More empty-set loops outside the fixed set: +`+stapeln.toml:44+` iterates `+cartridges/*/ffi+` *and* suffixes +`+\|\| true+`, so it can never fail; `+coord-tui/install.sh:28+` builds +from a path that no longer exists; `+guix.scm:34+` chdirs into it; +`+elixir/test/js_worker_pool_test.exs:6+` resolves a missing module but +passes today by short-circuiting when Deno is absent. +|`+git grep -n 'cartridges/\*' stapeln.toml guix.scm coord-tui/install.sh+` + +|D-7 |LOW |Dead exemption entries left behind by the retirement: 9 in +`+.hypatia-ignore+`, 5 in `+.gitleaksignore+`, plus `+.dockerignore+` +headers still claiming "`Stage 3 needs `+cartridges/+``". Harmless, but +they make the allowlists look larger than the real exposure. +|`+grep -c cartridges .hypatia-ignore .gitleaksignore+` + +|D-6 |LOW |Machine-specific absolute paths baked into tracked files: +`+generated/alloyiser/run-analysis.sh+` (`+/var/mnt/eclipse/...+`), +`+reports/maintenance/latest.json+`. Unrunnable off the original +machine. |`+git grep -n '/var/mnt/eclipse'+` + +|D-10 |LOW |TODO/FIXME/XXX/HACK density is genuinely near zero — all 63 +matches are policy/tooling references to marker _scanning_, not markers. +Recorded as a positive control. +|`+git grep -nE '\b(TODO\|FIXME\|XXX\|HACK)\b' \| wc -l+` +|=== + +''''' + +=== Test — T + +[width="100%",cols="16%,20%,24%,40%",options="header",] +|=== +|ID |Sev |Item |Evidence +|T-1 |MEDIUM |`+tests/security_test.js+` (313 lines) and +`+tests/federation_multinode.sh+` (170 lines) are referenced by *no +workflow and no Justfile recipe*. They exist and run nowhere. +|`+grep -rn 'security_test\|federation_multinode' .github/ Justfile+` + +|T-2 |MEDIUM |A real `+zig build test+` invocation inside +`+lsp-dap-bsp.yml+` is permanently unreachable behind C-2’s hardcoded +`+run=false+`. +|`+grep -n 'zig build test' .github/workflows/lsp-dap-bsp.yml+` + +|T-3 |LOW |`+TEST-NEEDS.md+` documents that E2E tests skip cleanly when +Deno is absent — a documented silent coverage reduction. +|`+grep -n 'Deno-gated' TEST-NEEDS.md+` +|=== + +''''' + +=== Documentation — X + +Full findings live in the docs refresh; only structural items are +indexed here. + +[width="100%",cols="16%,20%,24%,40%",options="header",] +|=== +|ID |Sev |Item |Evidence +|X-1 |HIGH |*Cartridge counts disagree across the estate.* This repo +asserts 125 in ~14 places; the registry’s README asserts 139; disk says +*142*. `+README.md+`’s own "`Number transparency`" clause makes this +self-refuting. +|`+find ../boj-server-cartridges/cartridges -name cartridge.json \| wc -l+` + +|X-2 |MEDIUM |`+docs/AI-CONVENTIONS.adoc+` opens agent onboarding by +directing every AI agent to read three files that *do not exist* +(`+.machine_readable/STATE.a2ml+`, `+anchors/ANCHOR.a2ml+`, +`+AGENTIC.a2ml+`). +|`+ls .machine_readable/STATE.a2ml .machine_readable/AGENTIC.a2ml+` + +|X-3 |MEDIUM |`+docs/zig-ffi-verification.adoc+` documents a Mutex +migration *backwards* — it recommends `+std.atomic.Mutex+`, the symbol +0.16 removed, and names nine modules that use no such pattern. +|`+grep -n 'atomic.Mutex' docs/zig-ffi-verification.adoc+` + +|X-4 |MEDIUM |`+docs/wikis/+` (7 `+.adoc+`) and the live GitHub wiki (6 +`+.md+`) are *different page sets with no sync mechanism*, while +`+docs/wikis/README.adoc+` claims to be "`the sources for GitHub’s wiki +tab`". +|`+git clone https://github.com/hyperpolymath/boj-server.wiki.git+` + +|X-5 |MEDIUM |`+CHANGELOG.md+` has no `+[0.5.0]+` section though +`+package.json+` says 0.5.0, and its `+[Unreleased]+` heading sits +_below_ the last release. 151 commits since the last dated entry, +including two CWE-tagged security fixes. +|`+git log --oneline --since=2026-05-20 \| wc -l+` + +|X-6 |LOW |`+jsr.json+` still says `+0.4.7+` while `+package.json+` says +`+0.5.0+`. |`+grep -h '"version"' package.json jsr.json+` +|=== + +''''' + +=== Supply chain — S + +[width="100%",cols="16%,20%,24%,40%",options="header",] +|=== +|ID |Sev |Item |Evidence +|S-1 |HIGH |See *C-3* — 27 mutable `+@main+` action pins in an +untracked, permission-less workflow is the repo’s largest supply-chain +exposure. |`+grep -c '@main' .github/workflows/main-estate-audit.yml+` + +|S-2 |MEDIUM |1,346 files of build residue survive under `+cartridges/+` +on disk (226 `+.so+`), untracked and gitignored so `+git status+` stays +silent about them. Confirmed: *no `+.so+` post-dates the retirement*, so +nothing has been rebuilt there since — it is stale, not live. *DIAGNOSIS +(unconfirmed):* a `+local-coord-mcp.service+` user unit was historically +built from this tree and may still point at it; deletion is the owner’s +call. |`+find cartridges -type f \| wc -l+` · +`+find cartridges -name '*.so' -newermt 2026-08-03 \| wc -l+` → 0 + +|S-3 |LOW |Release provenance is SLSA3 via `+slsa-github-generator+`, +SHA-pinned. Recorded as a positive control. +|`+grep -n 'slsa' .github/workflows/release.yml+` +|=== + +''''' + +=== How to use this file + +Add an item when you find debt you are not fixing in the same change. +Give it the next ID in its domain, a severity, and — non-negotiably — *a +command that reproduces the evidence*. An item without a reproducible +check is an opinion, and opinions rot silently. Remove an item only when +its command proves it gone; where the fix is interesting, keep the row +and mark it FIXED with the date, as P-4 and the C-section note do. diff --git a/DEBT.md b/DEBT.md deleted file mode 100644 index 9927a221..00000000 --- a/DEBT.md +++ /dev/null @@ -1,158 +0,0 @@ - - - -# Technical debt register - -One index of known debt in this repository, measured 2026-08-07 against -`78c4a05a`. Every item carries **the command that produced the evidence**, so -any entry can be re-checked or falsified in one step. Claims that are not -verified are labelled **DIAGNOSIS (unconfirmed)** rather than asserted. - -This file is an index, not a replacement. The pre-existing registers remain -authoritative in their own domains and are linked, not duplicated: -[`PROOF-NEEDS.md`](PROOF-NEEDS.md) · [`TEST-NEEDS.md`](TEST-NEEDS.md) · -[`docs/proof-debt.md`](docs/proof-debt.md) · -[`docs/tech-debt-2026-05-26.md`](docs/tech-debt-2026-05-26.md). - -Severity: **HIGH** — actively misleads, or a gate that cannot fail · -**MEDIUM** — wrong but self-evident on contact · **LOW** — cosmetic or -historical. - ---- - -## The single largest item - -**The `cartridges/` retirement (#300) is incomplete.** Removing 128 cartridges -took the manifests but left every consumer behind: two permanently-off -workflows, scripts whose loops match nothing, Justfile recipes, test scripts, -count claims in fourteen documents, and 1,346 files of build residue. Items -C-1…C-6, D-1…D-4, T-1 and X-1 below are all one migration, not ten problems. - -```sh -git grep -nF 'cartridges/' -- ':!docs' ':!*.md' ':!*.adoc' ':!tests/fixtures' | wc -l -``` - ---- - -## Licence — L - -| ID | Sev | Item | Evidence | -|----|-----|------|----------| -| L-1 | MEDIUM | `glama.json` is the only packaging manifest with **no `license` field at all**. Every sibling declares MPL-2.0 (`package.json`, `jsr.json`, `smithery.yaml`, `CITATION.cff`, `guix.scm`, `elixir/mix.exs`). | `grep -L '"license"' glama.json package.json jsr.json` | -| L-2 | LOW | `ai-plugin.json` declares licence only as a URL (`legal_info_url`), with no SPDX key — inconsistent with the rest of the estate. | `grep -n 'legal_info_url\|license' ai-plugin.json` | -| L-3 | LOW | Four tracked source files of 500+ carry no `SPDX-License-Identifier`: `.github/copilot/coding-agent.yml`, `.github/funding.yml`, `.machine_readable/scripts/forge/git-cleanup.sh`, `configs/config.ncl`. | `git ls-files \| xargs grep -L 'SPDX-License-Identifier' 2>/dev/null` | - -**Not debt, recorded as the positive control:** the dual-licence posture is -correct and documented — `NOTICE` explains MPL-2.0 (code) / CC-BY-SA-4.0 -(prose), `LICENSES/` holds both texts, and `.reuse/dep5` covers headerless -config. 335 MPL / 184 CC-BY-SA headers, zero third licence, zero unattributed -vendored trees. **The sibling registry has neither `NOTICE` nor `.reuse/` — -see its own `DEBT.md` L-1.** - ---- - -## Proof — P - -The proofs themselves are in good order: **4** `believe_me` sites, all inside -the sanctioned module, all `%unsafe`-tagged; zero `postulate`, `assert_total`, -`assert_smaller`, `idris_crash`, `sorry`, `%default partial`, `?hole` anywhere -in `src/abi/`. **The debt is in the gates and the prose, not the proofs.** - -```sh -grep -rn 'believe_me' src/abi --include='*.idr' | grep -v '|||' # 4 sites -grep -n 'EXPECTED_AXIOMS=' scripts/check-trusted-base.sh # 4 -``` - -| ID | Sev | Item | Evidence | -|----|-----|------|----------| -| P-1 | HIGH | **`proofs.yml` can report success with no prover having run.** The `changes` job sets `run=false` for any PR outside its path set, and both proof jobs are `if: needs.changes.outputs.run == 'true'`; a skipped job reports SUCCESS to a required check. The weekly `cron` at `proofs.yml` is the only backstop. This is a deliberate design (documented in the workflow header) — recorded here because the failure mode is invisible to a reviewer reading a green tick. | `grep -n "run=false\|needs.changes.outputs.run" .github/workflows/proofs.yml` | -| P-2 | MEDIUM | `scripts/check-trusted-base.sh` still greps `src/ cartridges/ verification/`; one of the three no longer exists, so the scan surface is a third smaller than it reads. The axiom count itself still works. | `grep -n 'cartridges/' scripts/check-trusted-base.sh` | -| P-3 | MEDIUM | `PROOF-NEEDS.md` cites `cartridges/fleet-mcp/abi/FleetMcp/SafeFleet.idr lines 14 & 34` — a proof obligation anchored to a file in the *other* repo, so the line numbers cannot be checked from here. | `grep -n 'SafeFleet' PROOF-NEEDS.md` | -| P-4 | LOW | **FIXED 2026-08-07, retained for provenance.** `PROOF-NEEDS.md` asserted `PASS=105` and "**exactly 5**" axioms; the enforcing script has said `EXPECTED_AXIOMS=4` since `charEqSym` was discharged, and the gate now covers 1 package. Two "in sync" documents disagreed with each other and with the code. | `git log -1 --format=%h -- PROOF-NEEDS.md` | - -**Cross-repo:** `boj-server-cartridges/scripts/check-trusted-base.sh` still -says *"boj-server sanctions EXACTLY 5 class-(J) axioms"*. Fixing it there -needs this file's correction to land first. - ---- - -## CI/CD — C - -| ID | Sev | Item | Evidence | -|----|-----|------|----------| -| C-1 | HIGH | **`abi-drift.yml` is permanently off** — `printf 'run=false'` unconditionally, so job `Emit manifest + verify FFI` never runs while still satisfying its required check by skipping. Its subject (per-cartridge iseriser drift) must be ported to the registry before the workflow *and its required context* are deleted. The port has not happened. | `grep -n "run=false" .github/workflows/abi-drift.yml` | -| C-2 | HIGH | **`lsp-dap-bsp.yml` is permanently off** — same mechanism, **four** dead jobs: ABI Specification Check, FFI Build & Test, Panel Manifest Validation, Cartridge Completeness Check. | `grep -n "run=false" .github/workflows/lsp-dap-bsp.yml` | -| C-3 | HIGH | **`main-estate-audit.yml` is still untracked here, deliberately — arming it today would fail `main` immediately.** The referenced suite is now published (`hyperpolymath/cicd-suite`, `11b5ab51`, all 26 actions resolve), so the 404 is fixed. But running its two relevant hard gates against this repo: `required-files-check` fails on 3 missing files (`CODEOWNERS`, `ARCHITECTURE.md`, `MAINTAINERS.adoc`), and `code-hygiene-check` matched **112 files** before cicd-suite#1; **2** after — including this repo's four *sanctioned, documented, CI-counted* `believe_me` axioms, which are its declared trusted base, not debt. Satisfying `required-files-check` means adding presence-only filler, which is how the template boilerplate on `fix/zig-ptr-cast-shim` was generated. **Fix the gates (see cicd-suite's README), then pin to `11b5ab51` and commit.** | `for f in CODEOWNERS ARCHITECTURE.md MAINTAINERS.adoc; do [ -f $f ] \|\| echo MISSING $f; done` · `git grep -Eic 'TODO\|FIXME\|STUB\|sorry\|believe_me\|admit' \| wc -l` → 112 | -| C-7 | MEDIUM | **13 of `cicd-suite`'s 26 actions cannot fail** — they emit `::warning::` and exit successfully while being named "Gate". *(Was HIGH and three-part; two of the three are fixed — cicd-suite#1 repaired `code-hygiene-check`'s whole-tree grep and `required-files-check`'s presence-only filler, and ended their mutual contradiction. The advisory/enforcing split is what remains.)* Estate-wide: the consuming workflow sits in **199 repos**, untracked in **198**. | `for a in ../cicd-suite/actions/*/; do grep -q 'exit 1' $a/action.yml \|\| echo $a; done \| wc -l` → 13 | -| C-4 | MEDIUM | Five required status-check contexts correspond to jobs that are green-by-skip (C-1, C-2). A reviewer cannot distinguish "passed" from "never ran". | `gh api repos/:owner/:repo/branches/main/protection` | -| C-5 | MEDIUM | `fuzz.yml` suppresses failure twice over: `\|\| true` **and** `continue-on-error: true`, with stderr sent to `/dev/null`. A crash is invisible rather than merely non-blocking. The bridge probes (including a `../../../etc/passwd` traversal case) assert nothing. | `grep -n 'continue-on-error\|\|\| true' .github/workflows/fuzz.yml` | -| C-6 | LOW | `pages.yml` and `pages-deploy.yml` both fire on push to main, publishing different content to two different hosts with no coordination. | `grep -l 'branches: \[main' .github/workflows/pages*.yml` | - -**FIXED 2026-08-07** (recorded so the pattern is searchable): two gates — -`tests/truthfulness_check.sh` and `scripts/typecheck-proofs.sh` — looped over -the deleted tree, matched zero files, and exited 0 reporting success. Both now -fail hard on an empty subject. *A gate that cannot fail is worse than no gate, -because it is credited as assurance.* - ---- - -## Code — D - -| ID | Sev | Item | Evidence | -|----|-----|------|----------| -| D-1 | MEDIUM | Justfile recipes still operate on the deleted tree; `CART_COUNT=$(ls -d cartridges/*-mcp \| wc -l)` now reports 0 as though that were a fact about the system. | `grep -n 'cartridges/' Justfile` | -| D-2 | MEDIUM | `scripts/refresh-bundled-cartridges.sh` exists solely to sync the retired tree (it `rm -rf`s inside it). `scripts/boj-selinux-contexts.sh` labels `${BOJ_ROOT}/cartridges/`. | `grep -ln 'cartridges/' scripts/*.sh` | -| D-3 | MEDIUM | `mcp-bridge/lib/generate-offline-menu.js` falls back to `../../cartridges` when `BOJ_CARTRIDGES_PATH` is unset — so it regenerates an **empty menu** silently instead of failing. | `grep -n 'cartridges' mcp-bridge/lib/generate-offline-menu.js` | -| D-4 | MEDIUM | Test scripts still traverse the tree: `tests/aspect_tests.sh`, `tests/integration.sh`, `tests/federation_multinode.sh`. | `grep -ln 'cartridges/' tests/*.sh` | -| D-5 | MEDIUM | **Two git worktrees are committed as gitlinks (mode `160000`) with no `.gitmodules`.** A fresh clone gets two empty directories, and both show as permanently modified because neither matches its recorded commit. | `git ls-files -s .claude/worktrees/` · `ls .gitmodules` | -| D-8 | HIGH | **`container/Containerfile.fly:80` cannot build.** It does `COPY cartridges/ /tmp/carts-meta/` from the host build context and never runs `fetch-cartridges.sh`; `COPY` on a missing source is a hard failure. (The main `container/Containerfile` is **fine** — it fetches into the builder stage first, so its `COPY --from=zig-builder` is populated. One file, not both.) | `grep -n 'COPY cartridges/' container/Containerfile.fly` · `grep -c fetch-cartridges container/Containerfile.fly` → 0 | -| D-9 | MEDIUM | More empty-set loops outside the fixed set: `stapeln.toml:44` iterates `cartridges/*/ffi` **and** suffixes `\|\| true`, so it can never fail; `coord-tui/install.sh:28` builds from a path that no longer exists; `guix.scm:34` chdirs into it; `elixir/test/js_worker_pool_test.exs:6` resolves a missing module but passes today by short-circuiting when Deno is absent. | `git grep -n 'cartridges/\*' stapeln.toml guix.scm coord-tui/install.sh` | -| D-7 | LOW | Dead exemption entries left behind by the retirement: 9 in `.hypatia-ignore`, 5 in `.gitleaksignore`, plus `.dockerignore` headers still claiming "Stage 3 needs `cartridges/`". Harmless, but they make the allowlists look larger than the real exposure. | `grep -c cartridges .hypatia-ignore .gitleaksignore` | -| D-6 | LOW | Machine-specific absolute paths baked into tracked files: `generated/alloyiser/run-analysis.sh` (`/var/mnt/eclipse/...`), `reports/maintenance/latest.json`. Unrunnable off the original machine. | `git grep -n '/var/mnt/eclipse'` | -| D-10 | LOW | TODO/FIXME/XXX/HACK density is genuinely near zero — all 63 matches are policy/tooling references to marker *scanning*, not markers. Recorded as a positive control. | `git grep -nE '\b(TODO\|FIXME\|XXX\|HACK)\b' \| wc -l` | - ---- - -## Test — T - -| ID | Sev | Item | Evidence | -|----|-----|------|----------| -| T-1 | MEDIUM | `tests/security_test.js` (313 lines) and `tests/federation_multinode.sh` (170 lines) are referenced by **no workflow and no Justfile recipe**. They exist and run nowhere. | `grep -rn 'security_test\|federation_multinode' .github/ Justfile` | -| T-2 | MEDIUM | A real `zig build test` invocation inside `lsp-dap-bsp.yml` is permanently unreachable behind C-2's hardcoded `run=false`. | `grep -n 'zig build test' .github/workflows/lsp-dap-bsp.yml` | -| T-3 | LOW | `TEST-NEEDS.md` documents that E2E tests skip cleanly when Deno is absent — a documented silent coverage reduction. | `grep -n 'Deno-gated' TEST-NEEDS.md` | - ---- - -## Documentation — X - -Full findings live in the docs refresh; only structural items are indexed here. - -| ID | Sev | Item | Evidence | -|----|-----|------|----------| -| X-1 | HIGH | **Cartridge counts disagree across the estate.** This repo asserts 125 in ~14 places; the registry's README asserts 139; disk says **142**. `README.md`'s own "Number transparency" clause makes this self-refuting. | `find ../boj-server-cartridges/cartridges -name cartridge.json \| wc -l` | -| X-2 | MEDIUM | `docs/AI-CONVENTIONS.adoc` opens agent onboarding by directing every AI agent to read three files that **do not exist** (`.machine_readable/STATE.a2ml`, `anchors/ANCHOR.a2ml`, `AGENTIC.a2ml`). | `ls .machine_readable/STATE.a2ml .machine_readable/AGENTIC.a2ml` | -| X-3 | MEDIUM | `docs/zig-ffi-verification.adoc` documents a Mutex migration **backwards** — it recommends `std.atomic.Mutex`, the symbol 0.16 removed, and names nine modules that use no such pattern. | `grep -n 'atomic.Mutex' docs/zig-ffi-verification.adoc` | -| X-4 | MEDIUM | `docs/wikis/` (7 `.adoc`) and the live GitHub wiki (6 `.md`) are **different page sets with no sync mechanism**, while `docs/wikis/README.adoc` claims to be "the sources for GitHub's wiki tab". | `git clone https://github.com/hyperpolymath/boj-server.wiki.git` | -| X-5 | MEDIUM | `CHANGELOG.md` has no `[0.5.0]` section though `package.json` says 0.5.0, and its `[Unreleased]` heading sits *below* the last release. 151 commits since the last dated entry, including two CWE-tagged security fixes. | `git log --oneline --since=2026-05-20 \| wc -l` | -| X-6 | LOW | `jsr.json` still says `0.4.7` while `package.json` says `0.5.0`. | `grep -h '"version"' package.json jsr.json` | - ---- - -## Supply chain — S - -| ID | Sev | Item | Evidence | -|----|-----|------|----------| -| S-1 | HIGH | See **C-3** — 27 mutable `@main` action pins in an untracked, permission-less workflow is the repo's largest supply-chain exposure. | `grep -c '@main' .github/workflows/main-estate-audit.yml` | -| S-2 | MEDIUM | 1,346 files of build residue survive under `cartridges/` on disk (226 `.so`), untracked and gitignored so `git status` stays silent about them. Confirmed: **no `.so` post-dates the retirement**, so nothing has been rebuilt there since — it is stale, not live. **DIAGNOSIS (unconfirmed):** a `local-coord-mcp.service` user unit was historically built from this tree and may still point at it; deletion is the owner's call. | `find cartridges -type f \| wc -l` · `find cartridges -name '*.so' -newermt 2026-08-03 \| wc -l` → 0 | -| S-3 | LOW | Release provenance is SLSA3 via `slsa-github-generator`, SHA-pinned. Recorded as a positive control. | `grep -n 'slsa' .github/workflows/release.yml` | - ---- - -## How to use this file - -Add an item when you find debt you are not fixing in the same change. Give it -the next ID in its domain, a severity, and — non-negotiably — **a command that -reproduces the evidence**. An item without a reproducible check is an opinion, -and opinions rot silently. Remove an item only when its command proves it gone; -where the fix is interesting, keep the row and mark it FIXED with the date, as -P-4 and the C-section note do. diff --git a/GOVERNANCE.adoc b/GOVERNANCE.adoc new file mode 100644 index 00000000..9b836fb2 --- /dev/null +++ b/GOVERNANCE.adoc @@ -0,0 +1,60 @@ +== Governance + +=== Overview + +This project is governed by the following principles and structures to +ensure transparent, inclusive, and effective decision-making. + +=== Roles and Responsibilities + +==== Maintainers + +Maintainers are responsible for: - Reviewing and merging pull requests - +Managing releases and versioning - Ensuring code quality and standards - +Triaging issues and bug reports - Community engagement and support + +==== Contributors + +Contributors are expected to: - Follow the code of conduct - Submit +well-documented pull requests - Write tests for new functionality - +Maintain existing tests - Update documentation as needed + +=== Decision Making + +==== Minor Changes + +* Can be made by any maintainer +* Include bug fixes, documentation updates, dependency updates + +==== Major Changes + +* Require discussion in issues or pull requests +* Include new features, architectural changes, API changes +* Need approval from at least 2 maintainers + +==== Breaking Changes + +* Require RFC (Request for Comments) process +* Need approval from majority of maintainers +* Must include migration guide + +=== Code of Conduct + +All participants are expected to follow our Code of Conduct. Violations +can be reported to the maintainers. + +=== Communication + +* *Issues*: For bug reports and feature requests +* *Discussions*: For questions and general discussion +* *Pull Requests*: For code contributions + +=== Licensing + +All contributions are made under the terms of the repository’s LICENSE +file. By submitting a pull request, you agree to license your +contributions accordingly. + +''''' + +_Last updated: 2026-07-18_ diff --git a/GOVERNANCE.md b/GOVERNANCE.md deleted file mode 100644 index e27364c7..00000000 --- a/GOVERNANCE.md +++ /dev/null @@ -1,60 +0,0 @@ -# Governance - -## Overview - -This project is governed by the following principles and structures to ensure transparent, inclusive, and effective decision-making. - -## Roles and Responsibilities - -### Maintainers - -Maintainers are responsible for: -- Reviewing and merging pull requests -- Managing releases and versioning -- Ensuring code quality and standards -- Triaging issues and bug reports -- Community engagement and support - -### Contributors - -Contributors are expected to: -- Follow the code of conduct -- Submit well-documented pull requests -- Write tests for new functionality -- Maintain existing tests -- Update documentation as needed - -## Decision Making - -### Minor Changes -- Can be made by any maintainer -- Include bug fixes, documentation updates, dependency updates - -### Major Changes -- Require discussion in issues or pull requests -- Include new features, architectural changes, API changes -- Need approval from at least 2 maintainers - -### Breaking Changes -- Require RFC (Request for Comments) process -- Need approval from majority of maintainers -- Must include migration guide - -## Code of Conduct - -All participants are expected to follow our Code of Conduct. Violations can be reported to the maintainers. - -## Communication - -- **Issues**: For bug reports and feature requests -- **Discussions**: For questions and general discussion -- **Pull Requests**: For code contributions - -## Licensing - -All contributions are made under the terms of the repository's LICENSE file. -By submitting a pull request, you agree to license your contributions accordingly. - ---- - -*Last updated: 2026-07-18* diff --git a/PROOF-NEEDS.adoc b/PROOF-NEEDS.adoc new file mode 100644 index 00000000..9012b4dc --- /dev/null +++ b/PROOF-NEEDS.adoc @@ -0,0 +1,325 @@ +== PROOF-NEEDS.md — boj-server + +____ +*See also*: link:docs/proof-debt.md[`+docs/proof-debt.md+`] is the +schema-conformant per-repo index under the estate +https://github.com/hyperpolymath/standards/blob/main/docs/TRUSTED-BASE-REDUCTION-POLICY.adoc[Trusted-Base +Reduction Policy] (standards#203, enforcement standards#211). +PROOF-NEEDS.md is the _strategic-goals_ narrative — full audit table, +classification rationale, and external-validation pointers. +`+docs/proof-debt.md+` is the _machine-checked_ index that +`+scripts/check-trusted-base.sh+` greps. Keep both in sync when the +marker count in `+src/abi/Boj/SafetyLemmas.idr+` changes. +____ + +=== Discharge (2026-06-24) — `+charEqSym+` axiom → constructive theorem + +`+charEqSym : (x, y : Char) -> (x == y) = (y == x)+` is no longer a +`+believe_me+` axiom. It is now *derived constructively from +`+charEqSound+`* in `+src/abi/Boj/SafetyLemmas.idr+`: a `+True+` result +forces propositional equality (`+charEqSound+`), collapsing both sides +to the same expression; a mixed `+True+`/`+False+` split is impossible +under soundness. This is the first *§(a) DISCHARGED* entry under the +trusted-base reduction policy (standards#203) and drops the sanctioned +class-(J) count *5 → 4* (`+charEqSound+`, `+unpackLength+`, +`+appendLengthSum+`, `+substrLengthBound+` remain — all genuinely +irreducible over opaque `+Char+`/`+String+` primitives). + +Verified locally: `+cd src/abi && idris2 --typecheck boj.ipkg+` → 17/17 +modules clean (Idris2 0.8.0, Chez 9.5.8). +`+scripts/check-trusted-base.sh+` updated to `+EXPECTED_AXIOMS=4+` and +passes. The dated audits below are retained as history; where they say +"`5`", read "`4 since 2026-06-24`". + +=== Build Verification (2026-06-13) — full sweep, freshly built toolchain + +Re-verified end-to-end on a clean machine with a from-source toolchain +(Idris2 0.8.0 bootstrapped via Chez Scheme 9.5.8 + libgmp), not a +desk-read of the prior entry. Three checks, all green: + +* *Core ABI:* `+cd src/abi && idris2 --typecheck boj.ipkg+` → 17/17 +modules clean, exit 0 (~6.5 s). +* *Full proof gate:* `+bash scripts/typecheck-proofs.sh+` → *PASS=1 +FAIL=0* (the core package). Historical note: this read PASS=105 (core + +104 cartridge ABIs) until the bundled `+cartridges/+` tree was retired +in #300. The 104 cartridge ABIs are now type-checked by +`+hyperpolymath/boj-server-cartridges+` in its own proofs gate; this +repo no longer has them to check. The script grew a vacuous-pass guard +at the same time, so PASS=0 is now a hard failure rather than a green +run over nothing. +* *Trusted base:* `+bash scripts/check-trusted-base.sh+` → PASS, +*exactly 4* sanctioned class-(J) axioms in `+SafetyLemmas.idr+`, zero +undocumented unsound constructs. (`+charEqSym+` was discharged +2026-06-24, taking the count 5 → 4; `+EXPECTED_AXIOMS=4+` in the +enforcing script is the source of truth, and `+docs/proof-debt.md+` +agrees.) Independently corroborated by a `+panic-attack assail+` scan +(MPL-2.0, built from source), which flags the same `+believe_me+` sites +as `+ProofDrift+` and nothing else proof-shaped. + +The axiom budget and theorem statements are unchanged from the +2026-06-03 entry; this checkpoint records that the *full cartridge +sweep* (not just the core package) compiles, so a future reader sees a +build-backed number rather than an inherited claim. + +=== Build Verification (2026-06-03) + +The full core ABI package now type-checks under the pinned toolchain +(*Idris2 0.8.0*, Chez backend) — +`+cd src/abi && idris2 --typecheck boj.ipkg+` builds all *17* modules +clean. A re-verification on this date found six modules that the package +build had never actually exercised (the `+just typecheck+` recipe used +an invalid `+--check --package boj boj.ipkg+` invocation, and +`+CartridgeDispatch+`/`+APIContractCoverage+` were absent from +`+boj.ipkg+`). The fixes were purely in the proofs’ construction — +*theorem statements and the axiom budget are unchanged*: + +* `+CartridgeDispatch+` — `+with+`-clause syntax (0.8.0 needs the full +LHS or `+_ |+`); `+dispatch+` factored through a reducible helper so +refused- completeness is provable; conjunction/`+Uninhabited+` lemmas +replace the earlier `+absurd+` shortcuts. +* `+SafePromptInjection+`, `+SafeCORS+` — `+with+`-abstraction rewrites +the goal to `+True = True+`, so the residual obligations are `+Refl+`. +* `+SafeHTTP+` — missing `+Data.List.Elem+`/`+Data.Maybe+` imports, +`+IsJust+`→ `+isJust+`, `+all+`→`+allRec+` (to match the +`+SafetyLemmas+` lemmas), and explicit `+{xs, ys}+` binders +(auto-generalised implicits are erased). +* `+SafeWebSocket+` — `+FrameSizeSafe+` is now +`+FrameSizeSafeUpTo maxFrameSize+` over a new bound-parameterised +family. Baking the 16 MiB `+maxFrameSize+` literal directly into a +constructor’s `+LTE+` index forced it into a unary Nat and exhausted the +elaborator; keeping the bound symbolic in the constructor (and applying +it in a synonym) fixes this with no loss of strength. +* `+APIContractCoverage+` — `+representativeCatalogue+` in a signature +was being auto-bound as a fresh implicit (shadowing the global); fully +qualified. +* `+SafetyLemmas+` — added the constructive lemma `+allTake+` (used by +the header/delimiter `+take+` proofs). No new axioms. + +`+just verify-no-believe-me+` was also reconciled: it had enforced +_zero_ `+believe_me+`, which contradicts the sanctioned 5-axiom trusted +base; it now permits exactly the 5 `+%unsafe+` class-(J) axioms in +`+SafetyLemmas.idr+` and fails on anything else (or on axiom-count +drift). + +=== Current State (Updated 2026-06-03) + +* *src/abi/Boj/*: 17 Idris2 ABI files +* *Dangerous patterns*: *4* `+believe_me+` invocations, *all* in +`+src/abi/Boj/SafetyLemmas.idr+`, *all* classified `+(J)+` genuinely +unavoidable (see audit table below; `+charEqSym+` was a 5th, discharged +to a theorem 2026-06-24). `+logSafeBounded+` (SafeAPIKey.idr) consumes +these structurally and contains no direct `+believe_me+` — down from 31 +historically. +** Note: a raw `+grep believe_me *.idr+` returns *9* hits. 4 of these +are documentation/comment mentions of the word (1 in SafeHTTP.idr line +15, 1 in SafetyLemmas.idr line 9, 2 in +cartridges/fleet-mcp/abi/FleetMcp/SafeFleet.idr lines 14 & 34) and are +*not* axiom uses. The audited true count is 5. +* *LOC*: ~5,400 (Elixir + Idris2) +* *ABI layer*: Comprehensive dependent-type ABI + +=== Axiom Audit (2026-05-18) + +Every real `+believe_me+` invocation, with its classification. +Classification key: *(J)* genuinely unavoidable, documented as an axiom; +*(R)* replaceable with a real proof / total definition; *(S)* +stub/laziness to fix. + +[width="100%",cols="10%,13%,23%,13%,16%,25%",options="header",] +|=== +|# |Site |Function |Type |Class |Rationale +|1 |`+SafetyLemmas.idr:60+` |`+charEqSound+` +|`+(c1,c2:Char) -> c1 == c2 = True -> c1 = c2+` |*J ✓* |`+Char+` is an +opaque primitive; `+==+` is `+prim__eqChar+` (foreign `+Bool+`). Idris2 +0.8.0 has no in-language soundness principle for primitive equality. +Standard, well-understood axiom. *Externally validated* via +backend-assurance harness (`+docs/backend-assurance/prim__eqChar.md+` + +BEAM property test). + +|2 |`+SafetyLemmas.idr+` |`+charEqSym+` +|`+(x,y:Char) -> (x == y) = (y == x)+` |*DISCHARGED 2026-06-24* +|[line-through]#Symmetry of `+prim__eqChar+`# — superseded. NOT a +necessary axiom: derived constructively from `+charEqSound+` (a `+True+` +result forces `+x = y+`, collapsing both sides; mixed `+True+`/`+False+` +impossible under soundness). Now a theorem in `+SafetyLemmas.idr+`; no +longer counts against the trusted base. + +|3 |`+SafetyLemmas.idr:218+` |`+unpackLength+` +|`+length (unpack s) = length s+` |*J ✓* |`+unpack+` = +`+prim__strToCharList+` (foreign). `+String+` is opaque with no +induction principle; the relation between primitive `+String+` length +and `+List Char+` length is not reducible in-language. *Externally +validated* via backend-assurance harness +(`+docs/backend-assurance/prim__strToCharList.md+` + BEAM property +test). + +|4 |`+SafetyLemmas.idr:226+` |`+appendLengthSum+` +|`+length (s ++ t) = length s + length t+` |*J ✓* |`++++` on `+String+` += `+prim__strAppend+` (foreign). Length additivity is a +backend-semantics guarantee, not type-level reducible. *Externally +validated* via backend-assurance harness +(`+docs/backend-assurance/prim__strAppend.md+` + BEAM property test). + +|5 |`+SafetyLemmas.idr:233+` |`+substrLengthBound+` +|`+LTE (length (substr start len s)) len+` |*J ✓* |`+substr+` = +`+prim__strSubstr+` (foreign). The "`result no longer than `+len+``" +bound is a primitive-semantics guarantee with no in-language proof. +*Externally validated* via backend-assurance harness +(`+docs/backend-assurance/prim__strSubstr.md+` + BEAM property test). +|=== + +*Verdict (revised 2026-06-24): 4 class (J) + 1 discharged.* +`+charEqSym+` (row 2) is no longer an axiom — it was found derivable +from `+charEqSound+` and is now a constructive theorem (see the +Discharge entry at the top of this file). The remaining four reduce to +the same root cause: Idris2 treats `+Char+` and `+String+` as opaque +primitive types whose operations are foreign functions with no +constructors and no induction principle. There is no constructive +in-language proof for any of them; a `+believe_me+` (or an equivalent +`+%foreign+` postulate) is the only option short of changing the trusted +computing base. They are correctly marked `+%unsafe+`, individually +documented, and isolated in one module. + +The *J ✓* marker indicates the axiom is class (J) *and* externally +validated by the backend-assurance harness (see +`+docs/backend-assurance/+`). The validation does not change the +in-language proof — the `+believe_me+` sites stay in source. It shrinks +the trusted base from "`we trust the backend`" to "`we have read the +backend lowering and randomly tested the operation against the claimed +property`". A bare *J* indicates a class-(J) axiom whose harness has not +yet landed. + +No *(R)* or *(S)* sites were found. The audit’s "`9`" was a raw text +count conflating 5 real axioms with 4 comment mentions. + +=== Completed Proofs + +[width="100%",cols="16%,21%,63%",options="header",] +|=== +|File |Covers |REQUIREMENTS-MASTER.md +|`+src/abi/Boj/SafePromptInjection.idr+` |6 properties preventing LLM +prompt escape |— + +|`+src/abi/Boj/SafeCORS.idr+` |Mutually exclusive wildcard/credentials; +origin char validation |— + +|`+src/abi/Boj/SafeAPIKey.idr+` |Entropy bounds, format safety, +log-masking, timing-safe checks |BJ2 partial ✅ + +|`+src/abi/Boj/SafeWebSocket.idr+` |Frame length bounds, opcode +validation |— + +|`+src/abi/Boj/SafeHTTP.idr+` |Path traversal prevention, header +sanitisation |BJ2 partial ✅ + +|`+src/abi/Boj/Federation.idr+` |Handshake authenticity and +non-replayability |— + +|`+src/abi/Boj/Catalogue.idr+` |IsUnbreakable: only Ready cartridges +mountable |— + +|`+src/abi/Boj/CartridgeDispatch.idr+` |Dispatch type safety: +protocol-match + readiness guard + disjointness |BJ1 ✅ + +|`+src/abi/Boj/CredentialIsolation.idr+` |Per-cartridge vault +partitioning; cross-partition non-interference (`+vaultIsolation+`) |BJ2 +✅ CLOSED + +|`+src/abi/Boj/APIContractCoverage.idr+` |Protocol compliance (NonEmpty +protocols), 15-domain coverage, 7-protocol coverage |BJ3 ✅ CLOSED +|=== + +=== What Still Needs Proving + +All P1/P2 proof obligations are now closed. + +[width="100%",cols="14%,50%,36%",options="header",] +|=== +|# |Component |Status +|BJ2 |Auth/credential handling — full isolation model |✅ DONE +(`+CredentialIsolation.idr+`) + +|BJ3 |API contract compliance — protocol/domain coverage |✅ DONE +(`+APIContractCoverage.idr+`) +|=== + +==== Remaining axiomatic surface (4 believe_me, all in SafetyLemmas.idr) + +All four are class *(J)* — genuinely unavoidable in Idris2 0.8.0 (opaque +`+Char+`/`+String+` primitives, foreign-backed operations). See the +*Axiom Audit* table above for per-site detail. (`+charEqSym+` was a 5th +here until 2026-06-24, now discharged to a theorem — see the Discharge +entry at the top of this file.) + +[width="100%",cols="14%,10%,26%,50%",options="header",] +|=== +|Axiom |Site |Justification |Backend-assurance evidence +|`+charEqSound+` |`+SafetyLemmas.idr+` |Soundness of `+prim__eqChar+` — +backend primitive correctness, externally validated +|`+docs/backend-assurance/prim__eqChar.md+` + +`+elixir/test/backend_assurance/prim_eq_char_test.exs+` + +|`+unpackLength+` |`+SafetyLemmas.idr+` |`+prim__strToCharList+` +preserves length — backend primitive correctness, externally validated +|`+docs/backend-assurance/prim__strToCharList.md+` + +`+elixir/test/backend_assurance/prim_str_to_char_list_test.exs+` + +|`+appendLengthSum+` |`+SafetyLemmas.idr+` |`+prim__strAppend+` length +semantics — not reducible at type level, externally validated +|`+docs/backend-assurance/prim__strAppend.md+` + +`+elixir/test/backend_assurance/prim_str_append_test.exs+` + +|`+substrLengthBound+` |`+SafetyLemmas.idr+` |`+prim__strSubstr+` length +bound — not reducible at type level, externally validated +|`+docs/backend-assurance/prim__strSubstr.md+` + +`+elixir/test/backend_assurance/prim_str_substr_test.exs+` +|=== + +Note: `+logSafeBounded+` in SafeAPIKey.idr no longer uses `+believe_me+` +directly; it calls the documented SafetyLemmas axioms via structural +proof. `+charEqSound+` is consumed by `+Safety.idr+` (`+shellContra+`, +`+sqlContra+`); `+unpackLength+` by `+SafeAPIKey.idr+` +(`+sufficientEntropyNonEmpty+`). + +==== Backend-assurance harness + +The "`Backend-assurance evidence`" column above cites the external +evidence reducing each class-(J) axiom’s trusted base. Two artefact +shapes per primitive: + +* *Trusted-extraction validation* under +`+docs/backend-assurance/.md+` — prose argument citing each +shipping backend’s lowering of the primitive (Idris2 0.8.0 sources for +Chez, OTP/R6RS for the runtime operations) and why the lowering +satisfies the axiom. +* *Property-test harness* under `+elixir/test/backend_assurance/+` — +BEAM-native property tests via `+stream_data+` covering the codepoint / +string spaces. Run via `+mix test --only backend_assurance+` and gated +in CI by `+.github/workflows/backend-assurance.yml+`. + +This harness does not change the in-language proof — the `+believe_me+` +sites stay in `+SafetyLemmas.idr+`. The harness shrinks the trusted base +from "`we trust the backend`" to "`we read the lowering and randomly +tested the operation`". + +Primitives validated so far: `+prim__eqChar+` (covering `+charEqSound+`; +formerly also `+charEqSym+`, now a derived theorem), +`+prim__strToCharList+` (covering `+unpackLength+`), `+prim__strAppend+` +(covering `+appendLengthSum+`), and `+prim__strSubstr+` (covering +`+substrLengthBound+`). Tracked under epic #87 Tier C backend-assurance +campaign. + +=== Priority Going Forward + +*LOW* — All named proof obligations closed. Future work: - Proofs for +Dap/Bsp/CodeIntel domains when those cartridges reach Ready status - +Proof that `+dispatch+` is surjective onto all BJ3-covered (domain, +protocol) pairs - The 4 string/char-primitive axioms are *irreducible +within Idris2* (opaque primitive types). They cannot be "`proved away`" +in-language; the only reduction path is to shrink the trusted base by +validating `+prim__eqChar+` / `+prim__strToCharList+` / +`+prim__strAppend+` / `+prim__strSubstr+` against the chosen backend +(Chez/BEAM) via an external trusted-extraction or property-test harness, +then citing that evidence here. This is a backend-assurance task, not a +proof obligation — keep it tracked but do not expect a constructive +proof. diff --git a/PROOF-NEEDS.md b/PROOF-NEEDS.md deleted file mode 100644 index 015e8d14..00000000 --- a/PROOF-NEEDS.md +++ /dev/null @@ -1,238 +0,0 @@ - -# PROOF-NEEDS.md — boj-server - -> **See also**: [`docs/proof-debt.md`](docs/proof-debt.md) is the -> schema-conformant per-repo index under the estate -> [Trusted-Base Reduction Policy](https://github.com/hyperpolymath/standards/blob/main/docs/TRUSTED-BASE-REDUCTION-POLICY.adoc) -> (standards#203, enforcement standards#211). PROOF-NEEDS.md is the -> *strategic-goals* narrative — full audit table, classification -> rationale, and external-validation pointers. `docs/proof-debt.md` is -> the *machine-checked* index that `scripts/check-trusted-base.sh` -> greps. Keep both in sync when the marker count in -> `src/abi/Boj/SafetyLemmas.idr` changes. - -## Discharge (2026-06-24) — `charEqSym` axiom → constructive theorem - -`charEqSym : (x, y : Char) -> (x == y) = (y == x)` is no longer a -`believe_me` axiom. It is now **derived constructively from `charEqSound`** -in `src/abi/Boj/SafetyLemmas.idr`: a `True` result forces propositional -equality (`charEqSound`), collapsing both sides to the same expression; a -mixed `True`/`False` split is impossible under soundness. This is the first -**§(a) DISCHARGED** entry under the trusted-base reduction policy -(standards#203) and drops the sanctioned class-(J) count **5 → 4** -(`charEqSound`, `unpackLength`, `appendLengthSum`, `substrLengthBound` -remain — all genuinely irreducible over opaque `Char`/`String` primitives). - -Verified locally: `cd src/abi && idris2 --typecheck boj.ipkg` → 17/17 -modules clean (Idris2 0.8.0, Chez 9.5.8). `scripts/check-trusted-base.sh` -updated to `EXPECTED_AXIOMS=4` and passes. The dated audits below are -retained as history; where they say "5", read "4 since 2026-06-24". - -## Build Verification (2026-06-13) — full sweep, freshly built toolchain - -Re-verified end-to-end on a clean machine with a from-source toolchain -(Idris2 0.8.0 bootstrapped via Chez Scheme 9.5.8 + libgmp), not a -desk-read of the prior entry. Three checks, all green: - -- **Core ABI:** `cd src/abi && idris2 --typecheck boj.ipkg` → 17/17 - modules clean, exit 0 (~6.5 s). -- **Full proof gate:** `bash scripts/typecheck-proofs.sh` → **PASS=1 - FAIL=0** (the core package). Historical note: this read PASS=105 - (core + 104 cartridge ABIs) until the bundled `cartridges/` tree was - retired in #300. The 104 cartridge ABIs are now type-checked by - `hyperpolymath/boj-server-cartridges` in its own proofs gate; this - repo no longer has them to check. The script grew a vacuous-pass - guard at the same time, so PASS=0 is now a hard failure rather than - a green run over nothing. -- **Trusted base:** `bash scripts/check-trusted-base.sh` → PASS, - **exactly 4** sanctioned class-(J) axioms in `SafetyLemmas.idr`, zero - undocumented unsound constructs. (`charEqSym` was discharged - 2026-06-24, taking the count 5 → 4; `EXPECTED_AXIOMS=4` in the - enforcing script is the source of truth, and `docs/proof-debt.md` - agrees.) Independently corroborated by a `panic-attack assail` scan - (MPL-2.0, built from source), which flags the same `believe_me` sites - as `ProofDrift` and nothing else proof-shaped. - -The axiom budget and theorem statements are unchanged from the -2026-06-03 entry; this checkpoint records that the **full cartridge -sweep** (not just the core package) compiles, so a future reader sees a -build-backed number rather than an inherited claim. - -## Build Verification (2026-06-03) - -The full core ABI package now type-checks under the pinned toolchain -(**Idris2 0.8.0**, Chez backend) — `cd src/abi && idris2 --typecheck boj.ipkg` -builds all **17** modules clean. A re-verification on this date found six -modules that the package build had never actually exercised (the `just -typecheck` recipe used an invalid `--check --package boj boj.ipkg` -invocation, and `CartridgeDispatch`/`APIContractCoverage` were absent from -`boj.ipkg`). The fixes were purely in the proofs' construction — **theorem -statements and the axiom budget are unchanged**: - -- `CartridgeDispatch` — `with`-clause syntax (0.8.0 needs the full LHS or - `_ |`); `dispatch` factored through a reducible helper so refused- - completeness is provable; conjunction/`Uninhabited` lemmas replace the - earlier `absurd` shortcuts. -- `SafePromptInjection`, `SafeCORS` — `with`-abstraction rewrites the goal - to `True = True`, so the residual obligations are `Refl`. -- `SafeHTTP` — missing `Data.List.Elem`/`Data.Maybe` imports, `IsJust`→ - `isJust`, `all`→`allRec` (to match the `SafetyLemmas` lemmas), and - explicit `{xs, ys}` binders (auto-generalised implicits are erased). -- `SafeWebSocket` — `FrameSizeSafe` is now `FrameSizeSafeUpTo maxFrameSize` - over a new bound-parameterised family. Baking the 16 MiB `maxFrameSize` - literal directly into a constructor's `LTE` index forced it into a unary - Nat and exhausted the elaborator; keeping the bound symbolic in the - constructor (and applying it in a synonym) fixes this with no loss of - strength. -- `APIContractCoverage` — `representativeCatalogue` in a signature was being - auto-bound as a fresh implicit (shadowing the global); fully qualified. -- `SafetyLemmas` — added the constructive lemma `allTake` (used by the - header/delimiter `take` proofs). No new axioms. - -`just verify-no-believe-me` was also reconciled: it had enforced *zero* -`believe_me`, which contradicts the sanctioned 5-axiom trusted base; it now -permits exactly the 5 `%unsafe` class-(J) axioms in `SafetyLemmas.idr` and -fails on anything else (or on axiom-count drift). - -## Current State (Updated 2026-06-03) - -- **src/abi/Boj/**: 17 Idris2 ABI files -- **Dangerous patterns**: **4** `believe_me` invocations, **all** in - `src/abi/Boj/SafetyLemmas.idr`, **all** classified `(J)` genuinely - unavoidable (see audit table below; `charEqSym` was a 5th, discharged to a - theorem 2026-06-24). `logSafeBounded` (SafeAPIKey.idr) - consumes these structurally and contains no direct `believe_me` — - down from 31 historically. - - Note: a raw `grep believe_me *.idr` returns **9** hits. 4 of these - are documentation/comment mentions of the word (1 in SafeHTTP.idr - line 15, 1 in SafetyLemmas.idr line 9, 2 in - cartridges/fleet-mcp/abi/FleetMcp/SafeFleet.idr lines 14 & 34) and - are **not** axiom uses. The audited true count is 5. -- **LOC**: ~5,400 (Elixir + Idris2) -- **ABI layer**: Comprehensive dependent-type ABI - -## Axiom Audit (2026-05-18) - -Every real `believe_me` invocation, with its classification. -Classification key: **(J)** genuinely unavoidable, documented as an axiom; -**(R)** replaceable with a real proof / total definition; -**(S)** stub/laziness to fix. - -| # | Site | Function | Type | Class | Rationale | -|---|------|----------|------|-------|-----------| -| 1 | `SafetyLemmas.idr:60` | `charEqSound` | `(c1,c2:Char) -> c1 == c2 = True -> c1 = c2` | **J ✓** | `Char` is an opaque primitive; `==` is `prim__eqChar` (foreign `Bool`). Idris2 0.8.0 has no in-language soundness principle for primitive equality. Standard, well-understood axiom. **Externally validated** via backend-assurance harness (`docs/backend-assurance/prim__eqChar.md` + BEAM property test). | -| 2 | `SafetyLemmas.idr` | `charEqSym` | `(x,y:Char) -> (x == y) = (y == x)` | **DISCHARGED 2026-06-24** | ~~Symmetry of `prim__eqChar`~~ — superseded. NOT a necessary axiom: derived constructively from `charEqSound` (a `True` result forces `x = y`, collapsing both sides; mixed `True`/`False` impossible under soundness). Now a theorem in `SafetyLemmas.idr`; no longer counts against the trusted base. | -| 3 | `SafetyLemmas.idr:218` | `unpackLength` | `length (unpack s) = length s` | **J ✓** | `unpack` = `prim__strToCharList` (foreign). `String` is opaque with no induction principle; the relation between primitive `String` length and `List Char` length is not reducible in-language. **Externally validated** via backend-assurance harness (`docs/backend-assurance/prim__strToCharList.md` + BEAM property test). | -| 4 | `SafetyLemmas.idr:226` | `appendLengthSum` | `length (s ++ t) = length s + length t` | **J ✓** | `++` on `String` = `prim__strAppend` (foreign). Length additivity is a backend-semantics guarantee, not type-level reducible. **Externally validated** via backend-assurance harness (`docs/backend-assurance/prim__strAppend.md` + BEAM property test). | -| 5 | `SafetyLemmas.idr:233` | `substrLengthBound` | `LTE (length (substr start len s)) len` | **J ✓** | `substr` = `prim__strSubstr` (foreign). The "result no longer than `len`" bound is a primitive-semantics guarantee with no in-language proof. **Externally validated** via backend-assurance harness (`docs/backend-assurance/prim__strSubstr.md` + BEAM property test). | - -**Verdict (revised 2026-06-24): 4 class (J) + 1 discharged.** `charEqSym` -(row 2) is no longer an axiom — it was found derivable from `charEqSound` -and is now a constructive theorem (see the Discharge entry at the top of -this file). The remaining four reduce to the same root cause: -Idris2 treats `Char` and `String` as opaque primitive types whose -operations are foreign functions with no constructors and no induction -principle. There is no constructive in-language proof for any of them; a -`believe_me` (or an equivalent `%foreign` postulate) is the only option -short of changing the trusted computing base. They are correctly marked -`%unsafe`, individually documented, and isolated in one module. - -The **J ✓** marker indicates the axiom is class (J) **and** -externally validated by the backend-assurance harness (see -`docs/backend-assurance/`). The validation does not change the -in-language proof — the `believe_me` sites stay in source. It shrinks -the trusted base from "we trust the backend" to "we have read the -backend lowering and randomly tested the operation against the -claimed property". A bare **J** indicates a class-(J) axiom whose -harness has not yet landed. - -No **(R)** or **(S)** sites were found. The audit's "9" was a raw text -count conflating 5 real axioms with 4 comment mentions. - -## Completed Proofs - -| File | Covers | REQUIREMENTS-MASTER.md | -|------|--------|------------------------| -| `src/abi/Boj/SafePromptInjection.idr` | 6 properties preventing LLM prompt escape | — | -| `src/abi/Boj/SafeCORS.idr` | Mutually exclusive wildcard/credentials; origin char validation | — | -| `src/abi/Boj/SafeAPIKey.idr` | Entropy bounds, format safety, log-masking, timing-safe checks | BJ2 partial ✅ | -| `src/abi/Boj/SafeWebSocket.idr` | Frame length bounds, opcode validation | — | -| `src/abi/Boj/SafeHTTP.idr` | Path traversal prevention, header sanitisation | BJ2 partial ✅ | -| `src/abi/Boj/Federation.idr` | Handshake authenticity and non-replayability | — | -| `src/abi/Boj/Catalogue.idr` | IsUnbreakable: only Ready cartridges mountable | — | -| `src/abi/Boj/CartridgeDispatch.idr` | Dispatch type safety: protocol-match + readiness guard + disjointness | BJ1 ✅ | -| `src/abi/Boj/CredentialIsolation.idr` | Per-cartridge vault partitioning; cross-partition non-interference (`vaultIsolation`) | BJ2 ✅ CLOSED | -| `src/abi/Boj/APIContractCoverage.idr` | Protocol compliance (NonEmpty protocols), 15-domain coverage, 7-protocol coverage | BJ3 ✅ CLOSED | - -## What Still Needs Proving - -All P1/P2 proof obligations are now closed. - -| # | Component | Status | -|---|-----------|--------| -| BJ2 | Auth/credential handling — full isolation model | ✅ DONE (`CredentialIsolation.idr`) | -| BJ3 | API contract compliance — protocol/domain coverage | ✅ DONE (`APIContractCoverage.idr`) | - -### Remaining axiomatic surface (4 believe_me, all in SafetyLemmas.idr) - -All four are class **(J)** — genuinely unavoidable in Idris2 0.8.0 -(opaque `Char`/`String` primitives, foreign-backed operations). See the -**Axiom Audit** table above for per-site detail. (`charEqSym` was a 5th here -until 2026-06-24, now discharged to a theorem — see the Discharge entry at -the top of this file.) - -| Axiom | Site | Justification | Backend-assurance evidence | -|-------|------|---------------|----------------------------| -| `charEqSound` | `SafetyLemmas.idr` | Soundness of `prim__eqChar` — backend primitive correctness, externally validated | `docs/backend-assurance/prim__eqChar.md` + `elixir/test/backend_assurance/prim_eq_char_test.exs` | -| `unpackLength` | `SafetyLemmas.idr` | `prim__strToCharList` preserves length — backend primitive correctness, externally validated | `docs/backend-assurance/prim__strToCharList.md` + `elixir/test/backend_assurance/prim_str_to_char_list_test.exs` | -| `appendLengthSum` | `SafetyLemmas.idr` | `prim__strAppend` length semantics — not reducible at type level, externally validated | `docs/backend-assurance/prim__strAppend.md` + `elixir/test/backend_assurance/prim_str_append_test.exs` | -| `substrLengthBound` | `SafetyLemmas.idr` | `prim__strSubstr` length bound — not reducible at type level, externally validated | `docs/backend-assurance/prim__strSubstr.md` + `elixir/test/backend_assurance/prim_str_substr_test.exs` | - -Note: `logSafeBounded` in SafeAPIKey.idr no longer uses `believe_me` directly; -it calls the documented SafetyLemmas axioms via structural proof. -`charEqSound` is consumed by `Safety.idr` (`shellContra`, `sqlContra`); -`unpackLength` by `SafeAPIKey.idr` (`sufficientEntropyNonEmpty`). - -### Backend-assurance harness - -The "Backend-assurance evidence" column above cites the external -evidence reducing each class-(J) axiom's trusted base. Two artefact -shapes per primitive: - -- **Trusted-extraction validation** under `docs/backend-assurance/.md` - — prose argument citing each shipping backend's lowering of the - primitive (Idris2 0.8.0 sources for Chez, OTP/R6RS for the runtime - operations) and why the lowering satisfies the axiom. -- **Property-test harness** under `elixir/test/backend_assurance/` - — BEAM-native property tests via `stream_data` covering the - codepoint / string spaces. Run via `mix test --only backend_assurance` - and gated in CI by `.github/workflows/backend-assurance.yml`. - -This harness does not change the in-language proof — the `believe_me` -sites stay in `SafetyLemmas.idr`. The harness shrinks the trusted base -from "we trust the backend" to "we read the lowering and randomly -tested the operation". - -Primitives validated so far: `prim__eqChar` (covering `charEqSound`; -formerly also `charEqSym`, now a derived theorem), `prim__strToCharList` -(covering `unpackLength`), -`prim__strAppend` (covering `appendLengthSum`), and `prim__strSubstr` -(covering `substrLengthBound`). Tracked under epic #87 Tier C -backend-assurance campaign. - -## Priority Going Forward - -**LOW** — All named proof obligations closed. Future work: -- Proofs for Dap/Bsp/CodeIntel domains when those cartridges reach Ready status -- Proof that `dispatch` is surjective onto all BJ3-covered (domain, protocol) pairs -- The 4 string/char-primitive axioms are **irreducible within Idris2** - (opaque primitive types). They cannot be "proved away" in-language; - the only reduction path is to shrink the trusted base by validating - `prim__eqChar` / `prim__strToCharList` / `prim__strAppend` / - `prim__strSubstr` against the chosen backend (Chez/BEAM) via an - external trusted-extraction or property-test harness, then citing - that evidence here. This is a backend-assurance task, not a proof - obligation — keep it tracked but do not expect a constructive proof. diff --git a/SECURITY.adoc b/SECURITY.adoc new file mode 100644 index 00000000..f4d0e52e --- /dev/null +++ b/SECURITY.adoc @@ -0,0 +1,783 @@ +== Security Policy + +We take security seriously. BoJ-server is an MCP (Model Context +Protocol) server that orchestrates browser automation, cloud provider +access, GitHub/GitLab integrations, cartridge execution, and AI-assisted +research. The attack surface is broad and the trust assumptions are +significant. We appreciate your efforts to responsibly disclose +vulnerabilities and will make every effort to acknowledge your +contributions. + +''''' + +=== Table of Contents + +* link:#supported-versions[Supported Versions] +* link:#reporting-a-vulnerability[Reporting a Vulnerability] +* link:#what-to-include[What to Include] +* link:#response-timeline[Response Timeline] +* link:#severity-classification[Severity Classification] +* link:#incident-response-phases[Incident Response Phases] +* link:#disclosure-policy[Disclosure Policy] +* link:#scope[Scope] +* link:#safe-harbour[Safe Harbour] +* link:#reporter-credits[Reporter Credits] +* link:#security-architecture[Security Architecture] +* link:#formal-verification[Formal Verification] +* link:#container-security[Container Security] +* link:#dependency-management[Dependency Management] +* link:#security-best-practices-for-operators[Security Best Practices +for Operators] +* link:#security-updates[Security Updates] +* link:#contact[Contact] + +''''' + +=== Supported Versions + +[width="100%",cols="35%,40%,25%",options="header",] +|=== +|Version |Supported |Notes +|Latest `+main+` |Yes — full support |All security fixes applied here +first + +|Latest tagged release |Yes — full support |Backported critical and high +fixes + +|Previous minor release |Critical/High only |Patches for P0 and P1 +vulnerabilities + +|Older releases |No |Please upgrade to a supported version +|=== + +We follow semantic versioning. Security fixes are released as patch +versions (e.g., `+1.2.3+` -> `+1.2.4+`). Critical vulnerabilities may +trigger out-of-band releases. + +''''' + +=== Reporting a Vulnerability + +==== Preferred Method: GitHub Security Advisories + +The preferred method for reporting security vulnerabilities is through +GitHub’s Security Advisory feature: + +[arabic] +. Navigate to +https://github.com/hyperpolymath/boj-server/security/advisories/new[Report +a Vulnerability] +. Click *"`Report a vulnerability`"* +. Complete the form with as much detail as possible +. Submit — we will receive a private notification + +This method ensures: + +* End-to-end encryption of your report +* Private discussion space for collaboration +* Coordinated disclosure tooling +* Automatic credit when the advisory is published + +==== Alternative: Encrypted Email + +If you cannot use GitHub Security Advisories, you may email us directly: + +[width="100%",cols="50%,50%",] +|=== +|*Email* |j.d.a.jewell@open.ac.uk +|*PGP Key* |https://github.com/hyperpolymath.gpg[Download Public Key] +|*Fingerprint* |See GitHub profile +|=== + +[source,bash] +---- +# Import our PGP key +curl -sSL https://github.com/hyperpolymath.gpg | gpg --import + +# Verify fingerprint +gpg --fingerprint j.d.a.jewell@open.ac.uk + +# Encrypt your report +gpg --armor --encrypt --recipient j.d.a.jewell@open.ac.uk report.txt +---- + +==== Do NOT Report Via + +* Public GitHub issues +* Pull requests +* GitHub Discussions +* Social media +* Third-party vulnerability disclosure platforms (without prior +coordination) + +''''' + +=== What to Include + +A good vulnerability report helps us understand and reproduce the issue +quickly. + +==== Required Information + +* *Description*: Clear explanation of the vulnerability +* *Impact*: What an attacker could achieve (confidentiality, integrity, +availability) +* *Affected component*: Which part of boj-server is affected (MCP core, +cartridges, browser automation, FFI layer, etc.) +* *Affected versions*: Which versions or commits are affected +* *Reproduction steps*: Detailed steps to reproduce the issue + +==== Helpful Additional Information + +* *Proof of concept*: Code, scripts, MCP tool calls, or screenshots +* *Attack scenario*: Realistic attack scenario showing exploitability +* *CVSS score*: Your assessment of severity (use +https://www.first.org/cvss/calculator/3.1[CVSS 3.1 Calculator]) +* *CWE ID*: Common Weakness Enumeration identifier if known +* *Suggested fix*: If you have ideas for remediation +* *References*: Links to related vulnerabilities, research, or +advisories + +==== Example Report Structure + +[source,markdown] +---- +## Summary +[One-sentence description of the vulnerability] + +## Vulnerability Type +[e.g., Command Injection, SSRF, Path Traversal, Privilege Escalation, etc.] + +## Affected Component +[e.g., ffi/zig/src/, elixir/lib/boj_rest/, mcp-bridge/, or a cartridge in +hyperpolymath/boj-server-cartridges — cartridges are no longer bundled here] + +## Affected Versions +[Version range or specific commits] + +## Severity Assessment +- CVSS 3.1 Score: [X.X] +- CVSS Vector: [CVSS:3.1/AV:X/AC:X/PR:X/UI:X/S:X/C:X/I:X/A:X] + +## Description +[Detailed technical description] + +## Steps to Reproduce +1. [First step] +2. [Second step] +3. [...] + +## Proof of Concept +[Code, MCP tool invocations, curl commands, screenshots, etc.] + +## Impact +[What can an attacker achieve?] + +## Suggested Remediation +[Optional: your ideas for fixing] + +## References +[Links to related issues, CVEs, research] +---- + +''''' + +=== Response Timeline + +We commit to the following response times: + +[width="100%",cols="24%,35%,41%",options="header",] +|=== +|Stage |Timeframe |Description +|*Acknowledgement* |48 hours |We acknowledge receipt and confirm we are +investigating + +|*Triage* |72 hours |We assess severity, confirm the vulnerability, and +classify priority + +|*Status Update* |Every 7 days |Regular updates on remediation progress + +|*Resolution (P0)* |7 days |Critical vulnerabilities — emergency patch + +|*Resolution (P1)* |30 days |High severity — expedited fix + +|*Resolution (P2)* |60 days |Medium severity — scheduled fix + +|*Resolution (P3)* |90 days |Low severity — included in next release + +|*Disclosure* |90 days max |Public disclosure after fix is available +(coordinated with reporter) +|=== + +These are targets, not guarantees. Complex vulnerabilities may require +more time. We will communicate openly about any delays. + +''''' + +=== Severity Classification + +==== P0 — Critical + +*Response time:* < 4 hours acknowledgement, < 7 days fix + +Active exploitation or imminent threat. Requires immediate attention and +potentially an out-of-band release. + +*Examples:* - Remote code execution via MCP tool invocation - Credential +exfiltration from environment variables or secrets store - Browser +automation sandbox escape allowing arbitrary system access - Cartridge +execution allowing arbitrary code outside sandbox - Supply chain +compromise of a direct dependency - Authentication bypass granting full +MCP server control + +==== P1 — High + +*Response time:* < 24 hours acknowledgement, < 30 days fix + +Exploitable vulnerability with significant impact, but no evidence of +active exploitation. + +*Examples:* - Server-side request forgery (SSRF) via browser navigation +tools - Path traversal in cartridge loading or file operations - +Injection attacks in GitHub/GitLab API proxy calls - Privilege +escalation between cartridge isolation boundaries - Unsafe +deserialisation of MCP messages - FFI memory safety violations (buffer +overflow, use-after-free) - Credential leakage in logs or error messages + +==== P2 — Medium + +*Response time:* < 7 days acknowledgement, < 60 days fix + +Vulnerability requiring specific conditions or user interaction to +exploit. + +*Examples:* - Cross-site scripting (XSS) in PanLL tray interface - +Information disclosure through verbose error responses - Denial of +service via malformed MCP requests - Insecure default configuration +options - Missing input validation on non-critical endpoints - Race +conditions in concurrent cartridge execution + +==== P3 — Low + +*Response time:* Next scheduled release + +Minor issues with limited security impact. + +*Examples:* - Missing security headers on informational endpoints - +Software version disclosure in responses - Verbose stack traces in +non-production modes - Suboptimal cryptographic parameter choices (but +still safe) - Best practice deviations without demonstrable impact + +''''' + +=== Incident Response Phases + +==== Phase 1 — Detection and Triage (0-4 hours) + +[arabic] +. Acknowledge the report via GitHub Security Advisory or encrypted email +. Classify severity using the priority table above +. Assign an owner from the response team +. Create a private security advisory on GitHub +. Determine affected components and blast radius + +==== Phase 2 — Containment (4-48 hours) + +[arabic] +. Assess blast radius — which versions, deployments, and users are +affected +. If actively exploited: issue an emergency advisory with mitigation +steps +. Disable affected functionality if necessary (cartridge disable, +feature flag, hotfix) +. Preserve evidence (logs, artefacts) for root cause analysis +. For FFI/ABI issues: verify whether the Idris2 proof obligations still +hold + +==== Phase 3 — Remediation (48 hours - 14 days) + +[arabic] +. Develop fix on a private branch (GitHub Security Advisory fork) +. Write regression tests covering the vulnerability +. If FFI-related: update Zig safety gates and re-verify ABI proofs +. If cartridge-related: audit all cartridges for similar patterns +. Peer review the fix (at least one other contributor) +. Backport to supported versions if applicable +. Run full CI pipeline including Hypatia security scan + +==== Phase 4 — Disclosure and Recovery (14-90 days) + +[arabic] +. Coordinate disclosure date with the reporter +. Publish GitHub Security Advisory with CVE (if applicable) +. Release patched versions to all supported channels +. Update CHANGELOG.md and release notes +. Credit the reporter (unless anonymity requested) +. Notify downstream consumers if the vulnerability affects the MCP +protocol surface + +==== Phase 5 — Post-Incident Review + +[arabic] +. Conduct a blameless post-mortem within 7 days of disclosure +. Document root cause, timeline, and lessons learned +. Identify systemic improvements (CI checks, code patterns, formal +verification gaps) +. Update this security policy if gaps are found +. Feed findings back into Hypatia scan rules + +==== Response Team + +The incident response team consists of the project maintainer (Jonathan +D.A. Jewell) and any active contributors with commit access. + +==== Communication Channels + +During an active incident: + +* *Internal*: GitHub Security Advisory private discussion +* *Reporter*: Direct replies in the advisory or encrypted email +* *Public*: Advisory published only after fix is available +* *Users*: GitHub release notes + repository watch notifications + +''''' + +=== Disclosure Policy + +We follow *coordinated disclosure* (also known as responsible +disclosure): + +[arabic] +. *You report* the vulnerability privately +. *We acknowledge* and begin investigation +. *We develop* a fix and prepare a release +. *We coordinate* disclosure timing with you +. *We publish* security advisory and fix simultaneously +. *You may publish* your research after disclosure + +==== Disclosure Timeline + +.... +Day 0 You report vulnerability +Day 1-2 We acknowledge receipt +Day 3 We confirm vulnerability and share initial assessment +Day 3-90 We develop and test fix +Day 90 Coordinated public disclosure + (earlier if fix is ready; later by mutual agreement) +.... + +If we cannot reach agreement on disclosure timing, we default to 90 days +from your initial report. + +''''' + +=== Scope + +==== In Scope + +The following are within scope for security research: + +* *MCP server core* — message handling, tool dispatch, authentication +* *Browser automation* — Playwright-based navigation, screenshot, JS +execution, tab management +* *Cartridge system* — cartridge loading, isolation, invocation, and the +cartridge SDK +* *Cloud provider integrations* — Cloudflare, Vercel, Verpex API proxies +* *GitHub/GitLab integrations* — API proxying, repository operations, +PR/issue management +* *Communication tools* — Gmail and Calendar integrations +* *FFI layer* — Zig FFI implementations in `+ffi/+` +* *ABI definitions* — Idris2 ABI proofs in `+src/abi/+` +* *MCP bridge* — protocol translation layer in `+mcp-bridge/+` +* *PanLL tray interface* — system tray UI in `+tray/+` +* *Container images* — Containerfile, build scripts, runtime +configuration +* *CI/CD workflows* — GitHub Actions, deployment scripts +* *Configuration files* — `+openapi.yaml+`, `+smithery.yaml+`, +`+ai-plugin.json+`, `+glama.json+` +* *Dependencies* — report here, we will coordinate with upstream +* *Documentation* that could lead to security issues (e.g., unsafe +examples) + +==== Out of Scope + +The following are *not* in scope: + +* Third-party services we integrate with (report directly to them: +GitHub, GitLab, Cloudflare, Vercel, Google) +* Social engineering attacks against maintainers +* Physical security +* Denial of service attacks against production infrastructure +* Spam, phishing, or other non-technical attacks +* Issues already reported or publicly known +* Theoretical vulnerabilities without proof of concept +* The MCP protocol specification itself (report to the MCP specification +maintainers) +* Vulnerabilities in AI/LLM models invoked through boj-server (report to +model providers) + +==== Qualifying Vulnerabilities + +We are particularly interested in: + +* Remote code execution via MCP tool calls +* Command injection through tool parameters +* Sandbox escape from cartridge execution +* Authentication or authorisation bypass +* Server-side request forgery (SSRF) via browser automation +* Path traversal or local file inclusion +* Credential or secret exfiltration +* Memory safety issues in FFI layer (buffer overflows, use-after-free) +* Supply chain vulnerabilities (dependency confusion, typosquatting) +* Cross-site scripting (XSS) in tray or web interfaces +* Deserialisation vulnerabilities in MCP message handling +* Information disclosure (API keys, tokens, PII) +* Cryptographic weaknesses +* Container escape or privilege escalation + +==== Non-Qualifying Issues + +The following generally do not qualify as security vulnerabilities: + +* Missing security headers on non-sensitive endpoints +* Software version disclosure +* Self-XSS (requires user to paste malicious code) +* Missing rate limiting (unless it enables a specific attack) +* Verbose error messages (unless exposing secrets) +* Best practice deviations without demonstrable impact +* Issues that require physical access to the host machine +* Attacks requiring the attacker to already have MCP server credentials + +''''' + +=== Safe Harbour + +We support security research conducted in good faith. + +==== Our Promise + +If you conduct security research in accordance with this policy: + +* We will not initiate legal action against you +* We will not report your activity to law enforcement +* We will work with you in good faith to resolve issues +* We consider your research authorised under the Computer Fraud and +Abuse Act (CFAA), UK Computer Misuse Act 1990, and equivalent +legislation in other jurisdictions +* We waive any potential claim against you for circumvention of security +controls + +==== Good Faith Requirements + +To qualify for safe harbour, you must: + +* Comply with this security policy +* Report vulnerabilities promptly after discovery +* Avoid privacy violations (do not access others’ data) +* Avoid service degradation (no destructive testing) +* Not exploit vulnerabilities beyond proof-of-concept +* Not use vulnerabilities for profit (beyond bug bounties where offered) +* Not access, modify, or delete data beyond what is necessary to +demonstrate the vulnerability + +This safe harbour does not extend to third-party systems. Always check +their policies before testing. + +''''' + +=== Reporter Credits + +==== Hall of Fame + +Researchers who report valid vulnerabilities will be acknowledged in our +link:SECURITY-ACKNOWLEDGMENTS.md[Security Acknowledgments] file (unless +they prefer anonymity). + +Recognition includes: + +* Your name (or chosen alias) +* Link to your website or profile (optional) +* Brief description of the vulnerability class +* Date of report and severity level + +==== What We Offer + +* Public credit in security advisories +* Acknowledgment in release notes and CHANGELOG.md +* Entry in our Hall of Fame +* Reference or recommendation letter upon request (for significant +findings) + +==== What We Do Not Currently Offer + +* Monetary bug bounties +* Hardware or swag +* Paid security research contracts + +We are an open-source project. Your contributions help everyone who uses +this software. + +''''' + +=== Security Architecture + +BoJ-server operates with multiple trust boundaries that security +researchers should understand. + +==== Trust Boundaries + +.... + +---------------------------+ + | LLM / AI Client | + | (Claude, etc.) | + +----------+----------------+ + | + MCP Protocol + | + +----------v----------------+ + | MCP Server Core | + | (tool dispatch, | + | auth, rate limiting) | + +--+------+------+----------+ + | | | + +----------+ +---+---+ +----------+ + | | | | + +-------v------+ +---v----+ +v-----------+ | + | Browser | |Cartridge| |Cloud/API | | + | Automation | |Engine | |Integrations| | + | (Playwright) | |(sandbox)| |(GitHub, | | + +--------------+ +--------+ | GitLab, | | + | Cloudflare)| | + +------------+ | + | + +----------v---+ + | FFI Layer | + | (Zig impl, | + | Idris2 ABI) | + +--------------+ +.... + +==== Key Security Properties + +[arabic] +. *MCP tool dispatch* — All tool invocations pass through a central +dispatcher that validates parameters before execution +. *Cartridge isolation* — Cartridges execute in a controlled environment +with declared capabilities; they cannot access resources outside their +manifest +. *Browser sandbox* — Playwright browser automation runs in a sandboxed +context with configurable navigation restrictions +. *API credential isolation* — Cloud provider and service credentials +are stored separately and accessed only by their respective integration +modules +. *FFI safety gates* — The Zig FFI layer enforces memory safety +invariants proven by the Idris2 ABI definitions + +For a complete threat model, see +link:docs/THREAT-MODEL.adoc[THREAT-MODEL.md] (if available) or request +one via a GitHub issue. + +''''' + +=== Formal Verification + +BoJ-server uses a layered formal verification approach for its ABI and +FFI boundaries. + +==== Idris2 ABI Proofs + +The `+src/abi/+` directory contains Idris2 definitions that provide: + +* *Dependent type proofs* for interface correctness at compile time +* *Memory layout verification* ensuring data structures are correctly +aligned across language boundaries +* *Platform-specific ABI selection* with compile-time guarantees +* *Backward compatibility proofs* when interfaces evolve + +==== Zig FFI Safety Gates + +The `+ffi/zig/+` directory implements C-compatible FFI functions with: + +* *Bounds checking* on all buffer operations +* *Null pointer guards* at every FFI entry point +* *No runtime dependencies* — zero-cost abstractions +* *Cross-compilation support* for reproducible builds + +==== Dangerous Pattern Ban + +The following patterns are banned across the entire codebase and +enforced by CI: + +* `+believe_me+`, `+assert_total+` (Idris2) +* `+unsafeCoerce+`, `+Obj.magic+` (OCaml/AffineScript) +* `+Admitted+`, `+sorry+` (Coq/Lean) +* `+unsafe+` blocks without justification comments (Rust/Zig) + +The Hypatia neurosymbolic scanner checks for these patterns on every +commit. + +''''' + +=== Container Security + +==== Base Images + +All container images use *Chainguard* base images, which provide: + +* Minimal attack surface (no shell, no package manager in production +images) +* Daily CVE scanning and patching by Chainguard +* SBOM (Software Bill of Materials) included with every image +* Signed images with cosign for supply chain integrity + +==== Build Process + +* The `+Containerfile+` (not `+Dockerfile+`) defines the build +* Multi-stage builds separate build dependencies from runtime +* No secrets baked into images — all credentials injected at runtime +* Images are scanned with *Trivy* before release + +==== Runtime Hardening + +* Containers run as non-root user +* Read-only root filesystem where possible +* Seccomp and AppArmor profiles applied +* Network access restricted to declared integrations +* Resource limits (CPU, memory) enforced + +''''' + +=== Dependency Management + +==== Hypatia Neurosymbolic Scanning + +Every commit is scanned by *Hypatia*, our neurosymbolic CI/CD +intelligence system, which checks for: + +* Known CVEs in direct and transitive dependencies +* Licence compliance violations +* Dangerous code patterns (see Formal Verification section) +* Supply chain anomalies (unexpected dependency changes, typosquatting) +* Secret leakage (API keys, tokens, credentials in source) + +==== SHA-Pinned GitHub Actions + +All GitHub Actions workflows use SHA-pinned action references (not +mutable tags) to prevent supply chain attacks via tag mutation: + +[source,yaml] +---- +# Correct — SHA-pinned +- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4 + +# Incorrect — mutable tag (NEVER used) +- uses: actions/checkout@v4 +---- + +==== Additional Scanning + +[width="100%",cols="24%,34%,42%",options="header",] +|=== +|Tool |Purpose |Frequency +|*Hypatia* |Neurosymbolic security analysis |Every commit +|*CodeQL* |Static analysis and variant detection |Every commit +|*TruffleHog* |Secret detection in source and history |Every commit +|*Trivy* |Container image vulnerability scanning |Every build +|*OpenSSF Scorecard* |Supply chain security posture |Weekly +|*Dependabot* |Dependency update alerts |Continuous +|*panic-attacker* |Pre-commit security assurance |Every commit (local) +|=== + +==== Dependency Policy + +* Direct dependencies are reviewed before adoption +* Transitive dependency trees are audited periodically +* Dependencies with known unpatched vulnerabilities are replaced or +vendored with patches +* No dependencies from untrusted registries + +''''' + +=== Security Best Practices for Operators + +If you are deploying boj-server, follow these guidelines: + +==== Credential Management + +* Store API keys and tokens in environment variables or a secrets +manager +* Never commit credentials to the repository +* Rotate credentials regularly (at least quarterly) +* Use scoped API tokens with minimum required permissions + +==== Network Configuration + +* Run boj-server behind a reverse proxy with TLS termination +* Restrict MCP endpoint access to trusted clients only +* Use firewall rules to limit outbound connections to declared +integrations +* Monitor network traffic for anomalous patterns + +==== Logging and Monitoring + +* Enable structured logging for audit trails +* Forward logs to a centralised logging system +* Set up alerts for authentication failures and unusual tool invocations +* Retain logs for at least 90 days + +==== Update Policy + +* Subscribe to GitHub release notifications for this repository +* Apply security patches within the timeframes matching their severity +* Test updates in a staging environment before production deployment + +''''' + +=== Security Updates + +==== Receiving Updates + +To stay informed about security updates: + +* *Watch this repository*: Click "`Watch`" then "`Custom`" then select +"`Security alerts`" +* *GitHub Security Advisories*: Published at +https://github.com/hyperpolymath/boj-server/security/advisories[Security +Advisories] +* *Release notes*: Security fixes noted in CHANGELOG.md + +==== Update Urgency by Severity + +[cols=",",options="header",] +|=== +|Severity |Action Required +|*P0 — Critical* |Patch immediately; out-of-band release issued +|*P1 — High* |Patch within 7 days of release +|*P2 — Medium* |Patch at next maintenance window +|*P3 — Low* |Patch at next scheduled upgrade +|=== + +''''' + +=== Contact + +[width="100%",cols="50%,50%",] +|=== +|*Maintainer* |Jonathan D.A. Jewell + +|*Security email* |j.d.a.jewell@open.ac.uk + +|*GitHub* |https://github.com/hyperpolymath[@hyperpolymath] + +|*PGP Key* |https://github.com/hyperpolymath.gpg + +|*Security Advisories* +|https://github.com/hyperpolymath/boj-server/security/advisories[github.com/hyperpolymath/boj-server/security/advisories] +|=== + +''''' + +_This security policy is reviewed and updated at least annually, or +whenever a significant change to the project’s security posture occurs._ + +_Last updated: 2026-03-23_ diff --git a/SECURITY.md b/SECURITY.md deleted file mode 100644 index 106600df..00000000 --- a/SECURITY.md +++ /dev/null @@ -1,714 +0,0 @@ - - - -# Security Policy - -We take security seriously. BoJ-server is an MCP (Model Context Protocol) server -that orchestrates browser automation, cloud provider access, GitHub/GitLab -integrations, cartridge execution, and AI-assisted research. The attack surface -is broad and the trust assumptions are significant. We appreciate your efforts -to responsibly disclose vulnerabilities and will make every effort to acknowledge -your contributions. - ---- - -## Table of Contents - -- [Supported Versions](#supported-versions) -- [Reporting a Vulnerability](#reporting-a-vulnerability) -- [What to Include](#what-to-include) -- [Response Timeline](#response-timeline) -- [Severity Classification](#severity-classification) -- [Incident Response Phases](#incident-response-phases) -- [Disclosure Policy](#disclosure-policy) -- [Scope](#scope) -- [Safe Harbour](#safe-harbour) -- [Reporter Credits](#reporter-credits) -- [Security Architecture](#security-architecture) -- [Formal Verification](#formal-verification) -- [Container Security](#container-security) -- [Dependency Management](#dependency-management) -- [Security Best Practices for Operators](#security-best-practices-for-operators) -- [Security Updates](#security-updates) -- [Contact](#contact) - ---- - -## Supported Versions - -| Version | Supported | Notes | -|---------|-----------|-------| -| Latest `main` | Yes — full support | All security fixes applied here first | -| Latest tagged release | Yes — full support | Backported critical and high fixes | -| Previous minor release | Critical/High only | Patches for P0 and P1 vulnerabilities | -| Older releases | No | Please upgrade to a supported version | - -We follow semantic versioning. Security fixes are released as patch versions -(e.g., `1.2.3` -> `1.2.4`). Critical vulnerabilities may trigger out-of-band -releases. - ---- - -## Reporting a Vulnerability - -### Preferred Method: GitHub Security Advisories - -The preferred method for reporting security vulnerabilities is through GitHub's -Security Advisory feature: - -1. Navigate to [Report a Vulnerability](https://github.com/hyperpolymath/boj-server/security/advisories/new) -2. Click **"Report a vulnerability"** -3. Complete the form with as much detail as possible -4. Submit — we will receive a private notification - -This method ensures: - -- End-to-end encryption of your report -- Private discussion space for collaboration -- Coordinated disclosure tooling -- Automatic credit when the advisory is published - -### Alternative: Encrypted Email - -If you cannot use GitHub Security Advisories, you may email us directly: - -| | | -|---|---| -| **Email** | j.d.a.jewell@open.ac.uk | -| **PGP Key** | [Download Public Key](https://github.com/hyperpolymath.gpg) | -| **Fingerprint** | See GitHub profile | - -```bash -# Import our PGP key -curl -sSL https://github.com/hyperpolymath.gpg | gpg --import - -# Verify fingerprint -gpg --fingerprint j.d.a.jewell@open.ac.uk - -# Encrypt your report -gpg --armor --encrypt --recipient j.d.a.jewell@open.ac.uk report.txt -``` - -### Do NOT Report Via - -- Public GitHub issues -- Pull requests -- GitHub Discussions -- Social media -- Third-party vulnerability disclosure platforms (without prior coordination) - ---- - -## What to Include - -A good vulnerability report helps us understand and reproduce the issue quickly. - -### Required Information - -- **Description**: Clear explanation of the vulnerability -- **Impact**: What an attacker could achieve (confidentiality, integrity, availability) -- **Affected component**: Which part of boj-server is affected (MCP core, cartridges, browser automation, FFI layer, etc.) -- **Affected versions**: Which versions or commits are affected -- **Reproduction steps**: Detailed steps to reproduce the issue - -### Helpful Additional Information - -- **Proof of concept**: Code, scripts, MCP tool calls, or screenshots -- **Attack scenario**: Realistic attack scenario showing exploitability -- **CVSS score**: Your assessment of severity (use [CVSS 3.1 Calculator](https://www.first.org/cvss/calculator/3.1)) -- **CWE ID**: Common Weakness Enumeration identifier if known -- **Suggested fix**: If you have ideas for remediation -- **References**: Links to related vulnerabilities, research, or advisories - -### Example Report Structure - -```markdown -## Summary -[One-sentence description of the vulnerability] - -## Vulnerability Type -[e.g., Command Injection, SSRF, Path Traversal, Privilege Escalation, etc.] - -## Affected Component -[e.g., ffi/zig/src/, elixir/lib/boj_rest/, mcp-bridge/, or a cartridge in -hyperpolymath/boj-server-cartridges — cartridges are no longer bundled here] - -## Affected Versions -[Version range or specific commits] - -## Severity Assessment -- CVSS 3.1 Score: [X.X] -- CVSS Vector: [CVSS:3.1/AV:X/AC:X/PR:X/UI:X/S:X/C:X/I:X/A:X] - -## Description -[Detailed technical description] - -## Steps to Reproduce -1. [First step] -2. [Second step] -3. [...] - -## Proof of Concept -[Code, MCP tool invocations, curl commands, screenshots, etc.] - -## Impact -[What can an attacker achieve?] - -## Suggested Remediation -[Optional: your ideas for fixing] - -## References -[Links to related issues, CVEs, research] -``` - ---- - -## Response Timeline - -We commit to the following response times: - -| Stage | Timeframe | Description | -|-------|-----------|-------------| -| **Acknowledgement** | 48 hours | We acknowledge receipt and confirm we are investigating | -| **Triage** | 72 hours | We assess severity, confirm the vulnerability, and classify priority | -| **Status Update** | Every 7 days | Regular updates on remediation progress | -| **Resolution (P0)** | 7 days | Critical vulnerabilities — emergency patch | -| **Resolution (P1)** | 30 days | High severity — expedited fix | -| **Resolution (P2)** | 60 days | Medium severity — scheduled fix | -| **Resolution (P3)** | 90 days | Low severity — included in next release | -| **Disclosure** | 90 days max | Public disclosure after fix is available (coordinated with reporter) | - -These are targets, not guarantees. Complex vulnerabilities may require more time. -We will communicate openly about any delays. - ---- - -## Severity Classification - -### P0 — Critical - -**Response time:** < 4 hours acknowledgement, < 7 days fix - -Active exploitation or imminent threat. Requires immediate attention and -potentially an out-of-band release. - -**Examples:** -- Remote code execution via MCP tool invocation -- Credential exfiltration from environment variables or secrets store -- Browser automation sandbox escape allowing arbitrary system access -- Cartridge execution allowing arbitrary code outside sandbox -- Supply chain compromise of a direct dependency -- Authentication bypass granting full MCP server control - -### P1 — High - -**Response time:** < 24 hours acknowledgement, < 30 days fix - -Exploitable vulnerability with significant impact, but no evidence of active -exploitation. - -**Examples:** -- Server-side request forgery (SSRF) via browser navigation tools -- Path traversal in cartridge loading or file operations -- Injection attacks in GitHub/GitLab API proxy calls -- Privilege escalation between cartridge isolation boundaries -- Unsafe deserialisation of MCP messages -- FFI memory safety violations (buffer overflow, use-after-free) -- Credential leakage in logs or error messages - -### P2 — Medium - -**Response time:** < 7 days acknowledgement, < 60 days fix - -Vulnerability requiring specific conditions or user interaction to exploit. - -**Examples:** -- Cross-site scripting (XSS) in PanLL tray interface -- Information disclosure through verbose error responses -- Denial of service via malformed MCP requests -- Insecure default configuration options -- Missing input validation on non-critical endpoints -- Race conditions in concurrent cartridge execution - -### P3 — Low - -**Response time:** Next scheduled release - -Minor issues with limited security impact. - -**Examples:** -- Missing security headers on informational endpoints -- Software version disclosure in responses -- Verbose stack traces in non-production modes -- Suboptimal cryptographic parameter choices (but still safe) -- Best practice deviations without demonstrable impact - ---- - -## Incident Response Phases - -### Phase 1 — Detection and Triage (0-4 hours) - -1. Acknowledge the report via GitHub Security Advisory or encrypted email -2. Classify severity using the priority table above -3. Assign an owner from the response team -4. Create a private security advisory on GitHub -5. Determine affected components and blast radius - -### Phase 2 — Containment (4-48 hours) - -1. Assess blast radius — which versions, deployments, and users are affected -2. If actively exploited: issue an emergency advisory with mitigation steps -3. Disable affected functionality if necessary (cartridge disable, feature flag, hotfix) -4. Preserve evidence (logs, artefacts) for root cause analysis -5. For FFI/ABI issues: verify whether the Idris2 proof obligations still hold - -### Phase 3 — Remediation (48 hours - 14 days) - -1. Develop fix on a private branch (GitHub Security Advisory fork) -2. Write regression tests covering the vulnerability -3. If FFI-related: update Zig safety gates and re-verify ABI proofs -4. If cartridge-related: audit all cartridges for similar patterns -5. Peer review the fix (at least one other contributor) -6. Backport to supported versions if applicable -7. Run full CI pipeline including Hypatia security scan - -### Phase 4 — Disclosure and Recovery (14-90 days) - -1. Coordinate disclosure date with the reporter -2. Publish GitHub Security Advisory with CVE (if applicable) -3. Release patched versions to all supported channels -4. Update CHANGELOG.md and release notes -5. Credit the reporter (unless anonymity requested) -6. Notify downstream consumers if the vulnerability affects the MCP protocol surface - -### Phase 5 — Post-Incident Review - -1. Conduct a blameless post-mortem within 7 days of disclosure -2. Document root cause, timeline, and lessons learned -3. Identify systemic improvements (CI checks, code patterns, formal verification gaps) -4. Update this security policy if gaps are found -5. Feed findings back into Hypatia scan rules - -### Response Team - -The incident response team consists of the project maintainer -(Jonathan D.A. Jewell) and any active contributors with commit access. - -### Communication Channels - -During an active incident: - -- **Internal**: GitHub Security Advisory private discussion -- **Reporter**: Direct replies in the advisory or encrypted email -- **Public**: Advisory published only after fix is available -- **Users**: GitHub release notes + repository watch notifications - ---- - -## Disclosure Policy - -We follow **coordinated disclosure** (also known as responsible disclosure): - -1. **You report** the vulnerability privately -2. **We acknowledge** and begin investigation -3. **We develop** a fix and prepare a release -4. **We coordinate** disclosure timing with you -5. **We publish** security advisory and fix simultaneously -6. **You may publish** your research after disclosure - -### Disclosure Timeline - -``` -Day 0 You report vulnerability -Day 1-2 We acknowledge receipt -Day 3 We confirm vulnerability and share initial assessment -Day 3-90 We develop and test fix -Day 90 Coordinated public disclosure - (earlier if fix is ready; later by mutual agreement) -``` - -If we cannot reach agreement on disclosure timing, we default to 90 days from -your initial report. - ---- - -## Scope - -### In Scope - -The following are within scope for security research: - -- **MCP server core** — message handling, tool dispatch, authentication -- **Browser automation** — Playwright-based navigation, screenshot, JS execution, tab management -- **Cartridge system** — cartridge loading, isolation, invocation, and the cartridge SDK -- **Cloud provider integrations** — Cloudflare, Vercel, Verpex API proxies -- **GitHub/GitLab integrations** — API proxying, repository operations, PR/issue management -- **Communication tools** — Gmail and Calendar integrations -- **FFI layer** — Zig FFI implementations in `ffi/` -- **ABI definitions** — Idris2 ABI proofs in `src/abi/` -- **MCP bridge** — protocol translation layer in `mcp-bridge/` -- **PanLL tray interface** — system tray UI in `tray/` -- **Container images** — Containerfile, build scripts, runtime configuration -- **CI/CD workflows** — GitHub Actions, deployment scripts -- **Configuration files** — `openapi.yaml`, `smithery.yaml`, `ai-plugin.json`, `glama.json` -- **Dependencies** — report here, we will coordinate with upstream -- **Documentation** that could lead to security issues (e.g., unsafe examples) - -### Out of Scope - -The following are **not** in scope: - -- Third-party services we integrate with (report directly to them: GitHub, GitLab, Cloudflare, Vercel, Google) -- Social engineering attacks against maintainers -- Physical security -- Denial of service attacks against production infrastructure -- Spam, phishing, or other non-technical attacks -- Issues already reported or publicly known -- Theoretical vulnerabilities without proof of concept -- The MCP protocol specification itself (report to the MCP specification maintainers) -- Vulnerabilities in AI/LLM models invoked through boj-server (report to model providers) - -### Qualifying Vulnerabilities - -We are particularly interested in: - -- Remote code execution via MCP tool calls -- Command injection through tool parameters -- Sandbox escape from cartridge execution -- Authentication or authorisation bypass -- Server-side request forgery (SSRF) via browser automation -- Path traversal or local file inclusion -- Credential or secret exfiltration -- Memory safety issues in FFI layer (buffer overflows, use-after-free) -- Supply chain vulnerabilities (dependency confusion, typosquatting) -- Cross-site scripting (XSS) in tray or web interfaces -- Deserialisation vulnerabilities in MCP message handling -- Information disclosure (API keys, tokens, PII) -- Cryptographic weaknesses -- Container escape or privilege escalation - -### Non-Qualifying Issues - -The following generally do not qualify as security vulnerabilities: - -- Missing security headers on non-sensitive endpoints -- Software version disclosure -- Self-XSS (requires user to paste malicious code) -- Missing rate limiting (unless it enables a specific attack) -- Verbose error messages (unless exposing secrets) -- Best practice deviations without demonstrable impact -- Issues that require physical access to the host machine -- Attacks requiring the attacker to already have MCP server credentials - ---- - -## Safe Harbour - -We support security research conducted in good faith. - -### Our Promise - -If you conduct security research in accordance with this policy: - -- We will not initiate legal action against you -- We will not report your activity to law enforcement -- We will work with you in good faith to resolve issues -- We consider your research authorised under the Computer Fraud and Abuse Act - (CFAA), UK Computer Misuse Act 1990, and equivalent legislation in other - jurisdictions -- We waive any potential claim against you for circumvention of security controls - -### Good Faith Requirements - -To qualify for safe harbour, you must: - -- Comply with this security policy -- Report vulnerabilities promptly after discovery -- Avoid privacy violations (do not access others' data) -- Avoid service degradation (no destructive testing) -- Not exploit vulnerabilities beyond proof-of-concept -- Not use vulnerabilities for profit (beyond bug bounties where offered) -- Not access, modify, or delete data beyond what is necessary to demonstrate - the vulnerability - -This safe harbour does not extend to third-party systems. Always check their -policies before testing. - ---- - -## Reporter Credits - -### Hall of Fame - -Researchers who report valid vulnerabilities will be acknowledged in our -[Security Acknowledgments](SECURITY-ACKNOWLEDGMENTS.md) file (unless they -prefer anonymity). - -Recognition includes: - -- Your name (or chosen alias) -- Link to your website or profile (optional) -- Brief description of the vulnerability class -- Date of report and severity level - -### What We Offer - -- Public credit in security advisories -- Acknowledgment in release notes and CHANGELOG.md -- Entry in our Hall of Fame -- Reference or recommendation letter upon request (for significant findings) - -### What We Do Not Currently Offer - -- Monetary bug bounties -- Hardware or swag -- Paid security research contracts - -We are an open-source project. Your contributions help everyone who uses this -software. - ---- - -## Security Architecture - -BoJ-server operates with multiple trust boundaries that security researchers -should understand. - -### Trust Boundaries - -``` - +---------------------------+ - | LLM / AI Client | - | (Claude, etc.) | - +----------+----------------+ - | - MCP Protocol - | - +----------v----------------+ - | MCP Server Core | - | (tool dispatch, | - | auth, rate limiting) | - +--+------+------+----------+ - | | | - +----------+ +---+---+ +----------+ - | | | | - +-------v------+ +---v----+ +v-----------+ | - | Browser | |Cartridge| |Cloud/API | | - | Automation | |Engine | |Integrations| | - | (Playwright) | |(sandbox)| |(GitHub, | | - +--------------+ +--------+ | GitLab, | | - | Cloudflare)| | - +------------+ | - | - +----------v---+ - | FFI Layer | - | (Zig impl, | - | Idris2 ABI) | - +--------------+ -``` - -### Key Security Properties - -1. **MCP tool dispatch** — All tool invocations pass through a central dispatcher - that validates parameters before execution -2. **Cartridge isolation** — Cartridges execute in a controlled environment with - declared capabilities; they cannot access resources outside their manifest -3. **Browser sandbox** — Playwright browser automation runs in a sandboxed - context with configurable navigation restrictions -4. **API credential isolation** — Cloud provider and service credentials are - stored separately and accessed only by their respective integration modules -5. **FFI safety gates** — The Zig FFI layer enforces memory safety invariants - proven by the Idris2 ABI definitions - -For a complete threat model, see [THREAT-MODEL.md](docs/THREAT-MODEL.adoc) (if -available) or request one via a GitHub issue. - ---- - -## Formal Verification - -BoJ-server uses a layered formal verification approach for its ABI and FFI -boundaries. - -### Idris2 ABI Proofs - -The `src/abi/` directory contains Idris2 definitions that provide: - -- **Dependent type proofs** for interface correctness at compile time -- **Memory layout verification** ensuring data structures are correctly aligned - across language boundaries -- **Platform-specific ABI selection** with compile-time guarantees -- **Backward compatibility proofs** when interfaces evolve - -### Zig FFI Safety Gates - -The `ffi/zig/` directory implements C-compatible FFI functions with: - -- **Bounds checking** on all buffer operations -- **Null pointer guards** at every FFI entry point -- **No runtime dependencies** — zero-cost abstractions -- **Cross-compilation support** for reproducible builds - -### Dangerous Pattern Ban - -The following patterns are banned across the entire codebase and enforced by CI: - -- `believe_me`, `assert_total` (Idris2) -- `unsafeCoerce`, `Obj.magic` (OCaml/AffineScript) -- `Admitted`, `sorry` (Coq/Lean) -- `unsafe` blocks without justification comments (Rust/Zig) - -The Hypatia neurosymbolic scanner checks for these patterns on every commit. - ---- - -## Container Security - -### Base Images - -All container images use **Chainguard** base images, which provide: - -- Minimal attack surface (no shell, no package manager in production images) -- Daily CVE scanning and patching by Chainguard -- SBOM (Software Bill of Materials) included with every image -- Signed images with cosign for supply chain integrity - -### Build Process - -- The `Containerfile` (not `Dockerfile`) defines the build -- Multi-stage builds separate build dependencies from runtime -- No secrets baked into images — all credentials injected at runtime -- Images are scanned with **Trivy** before release - -### Runtime Hardening - -- Containers run as non-root user -- Read-only root filesystem where possible -- Seccomp and AppArmor profiles applied -- Network access restricted to declared integrations -- Resource limits (CPU, memory) enforced - ---- - -## Dependency Management - -### Hypatia Neurosymbolic Scanning - -Every commit is scanned by **Hypatia**, our neurosymbolic CI/CD intelligence -system, which checks for: - -- Known CVEs in direct and transitive dependencies -- Licence compliance violations -- Dangerous code patterns (see Formal Verification section) -- Supply chain anomalies (unexpected dependency changes, typosquatting) -- Secret leakage (API keys, tokens, credentials in source) - -### SHA-Pinned GitHub Actions - -All GitHub Actions workflows use SHA-pinned action references (not mutable tags) -to prevent supply chain attacks via tag mutation: - -```yaml -# Correct — SHA-pinned -- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4 - -# Incorrect — mutable tag (NEVER used) -- uses: actions/checkout@v4 -``` - -### Additional Scanning - -| Tool | Purpose | Frequency | -|------|---------|-----------| -| **Hypatia** | Neurosymbolic security analysis | Every commit | -| **CodeQL** | Static analysis and variant detection | Every commit | -| **TruffleHog** | Secret detection in source and history | Every commit | -| **Trivy** | Container image vulnerability scanning | Every build | -| **OpenSSF Scorecard** | Supply chain security posture | Weekly | -| **Dependabot** | Dependency update alerts | Continuous | -| **panic-attacker** | Pre-commit security assurance | Every commit (local) | - -### Dependency Policy - -- Direct dependencies are reviewed before adoption -- Transitive dependency trees are audited periodically -- Dependencies with known unpatched vulnerabilities are replaced or vendored - with patches -- No dependencies from untrusted registries - ---- - -## Security Best Practices for Operators - -If you are deploying boj-server, follow these guidelines: - -### Credential Management - -- Store API keys and tokens in environment variables or a secrets manager -- Never commit credentials to the repository -- Rotate credentials regularly (at least quarterly) -- Use scoped API tokens with minimum required permissions - -### Network Configuration - -- Run boj-server behind a reverse proxy with TLS termination -- Restrict MCP endpoint access to trusted clients only -- Use firewall rules to limit outbound connections to declared integrations -- Monitor network traffic for anomalous patterns - -### Logging and Monitoring - -- Enable structured logging for audit trails -- Forward logs to a centralised logging system -- Set up alerts for authentication failures and unusual tool invocations -- Retain logs for at least 90 days - -### Update Policy - -- Subscribe to GitHub release notifications for this repository -- Apply security patches within the timeframes matching their severity -- Test updates in a staging environment before production deployment - ---- - -## Security Updates - -### Receiving Updates - -To stay informed about security updates: - -- **Watch this repository**: Click "Watch" then "Custom" then select "Security alerts" -- **GitHub Security Advisories**: Published at [Security Advisories](https://github.com/hyperpolymath/boj-server/security/advisories) -- **Release notes**: Security fixes noted in [CHANGELOG.md](CHANGELOG.md) - -### Update Urgency by Severity - -| Severity | Action Required | -|----------|----------------| -| **P0 — Critical** | Patch immediately; out-of-band release issued | -| **P1 — High** | Patch within 7 days of release | -| **P2 — Medium** | Patch at next maintenance window | -| **P3 — Low** | Patch at next scheduled upgrade | - ---- - -## Contact - -| | | -|---|---| -| **Maintainer** | Jonathan D.A. Jewell | -| **Security email** | j.d.a.jewell@open.ac.uk | -| **GitHub** | [@hyperpolymath](https://github.com/hyperpolymath) | -| **PGP Key** | [https://github.com/hyperpolymath.gpg](https://github.com/hyperpolymath.gpg) | -| **Security Advisories** | [github.com/hyperpolymath/boj-server/security/advisories](https://github.com/hyperpolymath/boj-server/security/advisories) | - ---- - -*This security policy is reviewed and updated at least annually, or whenever -a significant change to the project's security posture occurs.* - -*Last updated: 2026-03-23* diff --git a/TEST-NEEDS.adoc b/TEST-NEEDS.adoc new file mode 100644 index 00000000..1aa6e65c --- /dev/null +++ b/TEST-NEEDS.adoc @@ -0,0 +1,143 @@ +== TEST-NEEDS.md — BoJ Server + +*Last updated:* 2026-04-25 *Stack:* Elixir/OTP REST layer + Deno JS +dispatch + Zig FFI invoker *CRG target:* Grade C (dogfood threshold) +*CRG C standards from:* `+developer-ecosystem/standards/TEST-NEEDS.md+` + +''''' + +=== Current Coverage (173 ExUnit tests, 0 failures) + +10 StreamData property tests + 163 regular ExUnit tests = 173 total. CRG +C threshold met. + +[width="100%",cols="27%,30%,43%",options="header",] +|=== +|File |Tests |Category +|`+elixir/test/catalog_test.exs+` |25 |Unit + schema invariants + +|`+elixir/test/router_test.exs+` |37 |Integration (in-process Plug.Test) ++ auth enforcement + +|`+elixir/test/trust_policy_test.exs+` |11 |Unit — TrustPolicy +required_exposure + satisfies? + +|`+elixir/test/credential_decryptor_test.exs+` |19 |Unit + crypto +round-trip + +|`+elixir/test/node_key_test.exs+` |11 |Unit + X25519 ECDH + +|`+elixir/test/js_invoker_test.exs+` |10 |Unit + E2E (Deno-gated) + +|`+elixir/test/invoker_test.exs+` |15 |Unit + exit-code classification + +|`+elixir/test/js_worker_pool_test.exs+` |9 |Unit + E2E (Deno-gated) + +|`+elixir/test/contract_test.exs+` |15 |Contract (boundary pairs) + +|`+elixir/test/aspect_test.exs+` |15 |Aspect (no-crash, content-type, +security, idempotency) + +|`+elixir/test/catalog_properties_test.exs+` |10 properties |StreamData +property tests + +|`+elixir/benchmarks/boj_bench.exs+` |10 Benchee scenarios |Benchmarks +(dev only) +|=== + +==== What each file covers + +*catalog_test.exs* — `+BojRest.Catalog+` GenServer - `+list/0+` returns +non-empty list of maps - Every cartridge has required string fields: +name, version, domain - Every cartridge has a non-empty tools list - +Every tool has name and description - Cartridge names are unique - +`+get/1+` returns `+:not_found+` for unknown names - `+get/1+` returns +`+{:ok, cart}+` for `+boj-health+` (FFI cartridge) - `+get/1+` returns +`+{:ok, cart}+` for `+model-router-mcp+` (JS cartridge, no FFI) - FFI +cartridges all have `+ffi.so_path+` as string - `+auth.method+` is +always one of the known values - `+tier+` is always one of the known +tier values + +*router_test.exs* — `+BojRest.Router+` HTTP surface (Plug.Test) - +`+GET /health+` — 200, has status/version/cartridges_loaded, no `+mode+` +field - `+GET /menu+` — 200, all entries have name/domain/tier - +`+GET /cartridges+` — 200, count matches list length - +`+GET /cartridge/boj-health+` — 200, has `+ffi+` key - +`+GET /cartridge/model-router-mcp+` — 200, no `+ffi+` key - +`+GET /cartridge/unknown-xyz-999+` — 404 with +`+error: "unknown-cartridge"+` - `+GET /.well-known/boj-node-pubkey+` — +200, 43-char base64url pubkey, algorithm x25519 - +`+POST /cartridge/unknown-xyz/invoke+` — 404 - +`+POST /cartridge/model-router-mcp/invoke+` without `+tool+` — 400 +`+missing-tool-field+` - +`+POST /cartridge/model-router-mcp/invoke classify_task+` — 200 E2E +(Deno-gated, `+@tag :e2e+`) - Unknown route — 404 `+route-not-found+` + +*credential_decryptor_test.exs* — `+BojRest.CredentialDecryptor+` - +Nil/absent credentials → `+{:ok, %{}}+` - Plaintext accepted from +loopback - Plaintext rejected from non-loopback - Plaintext with +non-string values rejected - Multiple plaintext credentials accepted - +ECDH + ChaCha20-Poly1305 round-trip decryption succeeds - Wrong node key +→ `+"decryption failed"+` error - Unsupported version (v99) → version +error - Missing `+caller_pubkey+` field → error - Malformed base64 +`+caller_pubkey+` → error - Wrong pubkey size (16 bytes, not 32) → error +- Wrong nonce size (8 bytes, not 12) → error - Credentials as non-map +string → error + +*node_key_test.exs* — `+BojRest.NodeKey+` GenServer - `+public_key/0+` +returns 32-byte binary - `+private_key/0+` returns 32-byte binary - +Public key stable across calls - Private key stable across calls - +Public key consistent with private key (scalar-mult derivation) - Node +key participates correctly in X25519 ECDH shared-secret derivation - +Public and private keys are distinct + +*js_invoker_test.exs* — `+BojRest.JsInvoker+` - `+deno_path/0+` returns +nil or string - Non-existent mod.js → +`+{:error, %{classification: :mod_missing}}+` - Missing deno binary → +`+{:error, %{classification: :deno_missing}}+` - E2E: `+classify_task+` +via model-router-mcp (Deno-gated, `+@tag :e2e+`) - E2E: +`+estimate_cost+` via model-router-mcp (Deno-gated) - E2E: unknown tool +→ graceful error - E2E: `+extra_env+` forwarding does not crash + +''''' + +=== Coverage vs CRG C Standard + +[cols=",,,",options="header",] +|=== +|Category |Required |Current |Status +|Unit tests |100+ |~110 |✅ +|Smoke tests |9+ |~12 (router happy-path set) |✅ +|End-to-end tests |4+ |5 E2E (Deno-gated) |✅ +|Property tests |10+ |10 StreamData properties |✅ +|Contract tests |13+ |15 (contract_test.exs) |✅ +|Aspect tests |14+ |15 (aspect_test.exs) |✅ +|Benchmarks |10+ |10 Benchee scenarios |✅ +|*Total* |*165+* |*173* |*CRG C ✅* +|=== + +*Current grade: C* (all categories at or above threshold) + +''''' + +=== Path to CRG Grade B + +Grade B requires: - External validation targets (published test results, +CI badge) - Mutation testing (Muzak or equivalent) showing >80% mutation +kill rate - Six Sigma benchmark targets (latency at 99th percentile) - +Security/fuzz corpus for CredentialDecryptor + +''''' + +=== Notes + +* *E2E tests are Deno-gated* — tagged `+@tag :e2e+` and skip cleanly if +`+deno+` is absent. CI must install Deno for E2E coverage. +* *FFI/Zig tests* are not in this suite — they run via `+zig test+` in +`+ffi/zig/+`. +* *115 cartridges loaded* from `+cartridges/+` as of 2026-04-30 (111 +with `+.so+` built; the 4 without are `+database-mcp+`, +`+echidna-llm-mcp+`, `+lang-mcp+`, `+orchestrator-lsp-mcp+`). +* *Panic-attack pre-commit hook* runs `+panic-attack assail+` — check +`+PANIC-ATTACK.a2ml+` for current Clade classification and any open +findings. diff --git a/TEST-NEEDS.md b/TEST-NEEDS.md deleted file mode 100644 index 7b40c456..00000000 --- a/TEST-NEEDS.md +++ /dev/null @@ -1,130 +0,0 @@ - -# TEST-NEEDS.md — BoJ Server - -**Last updated:** 2026-04-25 -**Stack:** Elixir/OTP REST layer + Deno JS dispatch + Zig FFI invoker -**CRG target:** Grade C (dogfood threshold) -**CRG C standards from:** `developer-ecosystem/standards/TEST-NEEDS.md` - ---- - -## Current Coverage (173 ExUnit tests, 0 failures) - -10 StreamData property tests + 163 regular ExUnit tests = 173 total. CRG C threshold met. - -| File | Tests | Category | -|------|-------|----------| -| `elixir/test/catalog_test.exs` | 25 | Unit + schema invariants | -| `elixir/test/router_test.exs` | 37 | Integration (in-process Plug.Test) + auth enforcement | -| `elixir/test/trust_policy_test.exs` | 11 | Unit — TrustPolicy required_exposure + satisfies? | -| `elixir/test/credential_decryptor_test.exs` | 19 | Unit + crypto round-trip | -| `elixir/test/node_key_test.exs` | 11 | Unit + X25519 ECDH | -| `elixir/test/js_invoker_test.exs` | 10 | Unit + E2E (Deno-gated) | -| `elixir/test/invoker_test.exs` | 15 | Unit + exit-code classification | -| `elixir/test/js_worker_pool_test.exs` | 9 | Unit + E2E (Deno-gated) | -| `elixir/test/contract_test.exs` | 15 | Contract (boundary pairs) | -| `elixir/test/aspect_test.exs` | 15 | Aspect (no-crash, content-type, security, idempotency) | -| `elixir/test/catalog_properties_test.exs` | 10 properties | StreamData property tests | -| `elixir/benchmarks/boj_bench.exs` | 10 Benchee scenarios | Benchmarks (dev only) | - -### What each file covers - -**catalog_test.exs** — `BojRest.Catalog` GenServer -- `list/0` returns non-empty list of maps -- Every cartridge has required string fields: name, version, domain -- Every cartridge has a non-empty tools list -- Every tool has name and description -- Cartridge names are unique -- `get/1` returns `:not_found` for unknown names -- `get/1` returns `{:ok, cart}` for `boj-health` (FFI cartridge) -- `get/1` returns `{:ok, cart}` for `model-router-mcp` (JS cartridge, no FFI) -- FFI cartridges all have `ffi.so_path` as string -- `auth.method` is always one of the known values -- `tier` is always one of the known tier values - -**router_test.exs** — `BojRest.Router` HTTP surface (Plug.Test) -- `GET /health` — 200, has status/version/cartridges_loaded, no `mode` field -- `GET /menu` — 200, all entries have name/domain/tier -- `GET /cartridges` — 200, count matches list length -- `GET /cartridge/boj-health` — 200, has `ffi` key -- `GET /cartridge/model-router-mcp` — 200, no `ffi` key -- `GET /cartridge/unknown-xyz-999` — 404 with `error: "unknown-cartridge"` -- `GET /.well-known/boj-node-pubkey` — 200, 43-char base64url pubkey, algorithm x25519 -- `POST /cartridge/unknown-xyz/invoke` — 404 -- `POST /cartridge/model-router-mcp/invoke` without `tool` — 400 `missing-tool-field` -- `POST /cartridge/model-router-mcp/invoke classify_task` — 200 E2E (Deno-gated, `@tag :e2e`) -- Unknown route — 404 `route-not-found` - -**credential_decryptor_test.exs** — `BojRest.CredentialDecryptor` -- Nil/absent credentials → `{:ok, %{}}` -- Plaintext accepted from loopback -- Plaintext rejected from non-loopback -- Plaintext with non-string values rejected -- Multiple plaintext credentials accepted -- ECDH + ChaCha20-Poly1305 round-trip decryption succeeds -- Wrong node key → `"decryption failed"` error -- Unsupported version (v99) → version error -- Missing `caller_pubkey` field → error -- Malformed base64 `caller_pubkey` → error -- Wrong pubkey size (16 bytes, not 32) → error -- Wrong nonce size (8 bytes, not 12) → error -- Credentials as non-map string → error - -**node_key_test.exs** — `BojRest.NodeKey` GenServer -- `public_key/0` returns 32-byte binary -- `private_key/0` returns 32-byte binary -- Public key stable across calls -- Private key stable across calls -- Public key consistent with private key (scalar-mult derivation) -- Node key participates correctly in X25519 ECDH shared-secret derivation -- Public and private keys are distinct - -**js_invoker_test.exs** — `BojRest.JsInvoker` -- `deno_path/0` returns nil or string -- Non-existent mod.js → `{:error, %{classification: :mod_missing}}` -- Missing deno binary → `{:error, %{classification: :deno_missing}}` -- E2E: `classify_task` via model-router-mcp (Deno-gated, `@tag :e2e`) -- E2E: `estimate_cost` via model-router-mcp (Deno-gated) -- E2E: unknown tool → graceful error -- E2E: `extra_env` forwarding does not crash - ---- - -## Coverage vs CRG C Standard - -| Category | Required | Current | Status | -|----------|----------|---------|--------| -| Unit tests | 100+ | ~110 | ✅ | -| Smoke tests | 9+ | ~12 (router happy-path set) | ✅ | -| End-to-end tests | 4+ | 5 E2E (Deno-gated) | ✅ | -| Property tests | 10+ | 10 StreamData properties | ✅ | -| Contract tests | 13+ | 15 (contract_test.exs) | ✅ | -| Aspect tests | 14+ | 15 (aspect_test.exs) | ✅ | -| Benchmarks | 10+ | 10 Benchee scenarios | ✅ | -| **Total** | **165+** | **173** | **CRG C ✅** | - -**Current grade: C** (all categories at or above threshold) - ---- - -## Path to CRG Grade B - -Grade B requires: -- External validation targets (published test results, CI badge) -- Mutation testing (Muzak or equivalent) showing >80% mutation kill rate -- Six Sigma benchmark targets (latency at 99th percentile) -- Security/fuzz corpus for CredentialDecryptor - ---- - -## Notes - -- **E2E tests are Deno-gated** — tagged `@tag :e2e` and skip cleanly if `deno` is absent. - CI must install Deno for E2E coverage. -- **FFI/Zig tests** are not in this suite — they run via `zig test` in `ffi/zig/`. -- **115 cartridges loaded** from `cartridges/` as of 2026-04-30 (111 with `.so` built; the 4 without are `database-mcp`, `echidna-llm-mcp`, `lang-mcp`, `orchestrator-lsp-mcp`). -- **Panic-attack pre-commit hook** runs `panic-attack assail` — check `PANIC-ATTACK.a2ml` - for current Clade classification and any open findings. diff --git a/TOPOLOGY.adoc b/TOPOLOGY.adoc new file mode 100644 index 00000000..cf5f0acc --- /dev/null +++ b/TOPOLOGY.adoc @@ -0,0 +1,223 @@ +== SPDX-License-Identifier: CC-BY-SA-4.0 + +== Copyright (c) 2026 Jonathan D.A. Jewell j.d.a.jewell@open.ac.uk + +== + +== TOPOLOGY.md — BoJ Server Component Matrix + +== + +== Updated 2026-06-01. 125 cartridges (per `+find cartridges -name cartridge.json | wc -l+`). + +=== Ports + +[cols=",,",options="header",] +|=== +|Port |Protocol |Status +|7700 |REST (HTTP) |Running +|7701 |gRPC |Running +|7702 |GraphQL |Running +|7703 |SSE (Server-Sent Events) |Running +|7745 |local-coord-mcp (loopback only) |Running +|=== + +=== Cartridge Matrix (125 total) + +==== Tier 1 — High-Value APIs (11) + +[cols=",",options="header",] +|=== +|Cartridge |Domain +|github-api-mcp |Source control +|gitlab-api-mcp |Source control +|browser-mcp |Web automation +|slack-mcp |Communication +|vault-mcp |Secrets +|linear-mcp |Project management +|notion-mcp |Knowledge base +|jira-mcp |Project management +|discord-mcp |Communication +|telegram-mcp |Communication +|matrix-mcp |Communication +|=== + +==== Tier 2 — Databases (10) + +[cols=",",options="header",] +|=== +|Cartridge |Domain +|postgresql-mcp |Relational DB +|redis-mcp |Cache/KV +|mongodb-mcp |Document DB +|neon-mcp |Serverless Postgres +|turso-mcp |Edge SQLite +|supabase-mcp |Backend-as-Service +|arango-mcp |Multi-model DB +|neo4j-mcp |Graph DB +|clickhouse-mcp |Analytics DB +|duckdb-mcp |Embedded OLAP +|=== + +==== Tier 3 — Cloud Providers (8) + +[cols=",",options="header",] +|=== +|Cartridge |Domain +|aws-mcp |Cloud +|gcp-mcp |Cloud +|hetzner-mcp |Cloud +|fly-mcp |Edge compute +|railway-mcp |PaaS +|render-mcp |PaaS +|digitalocean-mcp |Cloud +|linode-mcp |Cloud +|=== + +==== Tier 4 — Dev Tools & Registries (13) + +[cols=",",options="header",] +|=== +|Cartridge |Domain +|docker-hub-mcp |Container registry +|npm-registry-mcp |JS packages +|crates-mcp |Rust packages +|pypi-mcp |Python packages +|hex-mcp |Elixir packages +|opam-mcp |OCaml packages +|hackage-mcp |Haskell packages +|github-actions-mcp |CI/CD +|buildkite-mcp |CI/CD +|circleci-mcp |CI/CD +|git-mcp |Version control +|lsp-mcp |Language Server Protocol +|dap-mcp |Debug Adapter Protocol +|=== + +==== Tier 5 — Productivity & Comms (8) + +[cols=",",options="header",] +|=== +|Cartridge |Domain +|comms-mcp |Unified messaging +|google-docs-mcp |Documents +|google-sheets-mcp |Spreadsheets +|obsidian-mcp |Notes +|todoist-mcp |Task management +|airtable-mcp |Low-code DB +|zotero-mcp |Research +|feedback-mcp |User feedback +|=== + +==== Tier 6 — Monitoring & Observability (6) + +[cols=",",options="header",] +|=== +|Cartridge |Domain +|grafana-mcp |Dashboards +|prometheus-mcp |Metrics +|sentry-mcp |Error tracking +|observe-mcp |Observability +|laminar-mcp |Flow monitoring +|ums-mcp |Unified monitoring +|=== + +==== Hyperpolymath Ecosystem (27) + +[width="100%",cols="58%,42%",options="header",] +|=== +|Cartridge |Domain +|agent-mcp |Agent orchestration + +|local-coord-mcp |Localhost multi-instance coordination (loopback only, +port 7745) + +|affinescript-mcp |Language tooling + +|aerie-mcp |Deployment + +|bsp-mcp |Build Server Protocol + +|burble-admin-mcp |Voice platform + +|civic-connect-mcp |Civic engagement + +|cloud-mcp |Multi-cloud + +|conflow-mcp |Config flow + +|container-mcp |Container ops + +|database-mcp |Multi-DB + +|echidna-llm-mcp |LLM prover + +|fleet-mcp |Bot fleet + +|game-admin-mcp |Game servers + +|gossamer-mcp |Webview shell + +|hypatia-mcp |Neurosymbolic CI + +|iac-mcp |Infrastructure-as-Code + +|idaptik-admin-mcp |Game admin + +|k8s-mcp |Kubernetes + +|kategoria-mcp |Categorisation + +|lang-mcp |Language services + +|ml-mcp |Machine learning + +|model-router-mcp |Model routing + +|nesy-mcp |Neurosymbolic + +|opsm-mcp |Operations + +|panic-attack-mcp |Security scanning + +|proof-mcp |Formal verification +|=== + +==== Infrastructure & Utility (19) + +[cols=",",options="header",] +|=== +|Cartridge |Domain +|queues-mcp |Message queues +|railway-mcp |PaaS +|render-mcp |PaaS +|reposystem-mcp |Repo management +|research-mcp |Research tools +|rokur-mcp |Deployment +|secrets-mcp |Secret management +|ssg-mcp |Static site gen +|stapeln-mcp |Container stack +|typed-wasm-mcp |WASM types +|verisimdb-mcp |8-modality DB +|vext-mcp |Extensions +|=== + +=== CI/CD + +[cols=",",options="header",] +|=== +|Workflow |Purpose +|zig-test.yml |Zig FFI tests +|release.yml |Release pipeline +|lsp-dap-bsp.yml |Protocol column tests +|=== + +=== PanLL Integration + +[cols=",",options="header",] +|=== +|Module |Location +|BojCmd.res |panll/src/commands/ +|BojLiveCmd.res |panll/src/commands/ +|CartridgeAbi.res |panll/src/generated/ +|=== diff --git a/TOPOLOGY.md b/TOPOLOGY.md deleted file mode 100644 index cebe1fc0..00000000 --- a/TOPOLOGY.md +++ /dev/null @@ -1,172 +0,0 @@ - -# SPDX-License-Identifier: CC-BY-SA-4.0 -# Copyright (c) 2026 Jonathan D.A. Jewell -# -# TOPOLOGY.md — BoJ Server Component Matrix -# -# Updated 2026-06-01. 125 cartridges (per `find cartridges -name cartridge.json | wc -l`). - -## Ports - -| Port | Protocol | Status | -|------|----------|--------| -| 7700 | REST (HTTP) | Running | -| 7701 | gRPC | Running | -| 7702 | GraphQL | Running | -| 7703 | SSE (Server-Sent Events) | Running | -| 7745 | local-coord-mcp (loopback only) | Running | - -## Cartridge Matrix (125 total) - - - -### Tier 1 — High-Value APIs (11) -| Cartridge | Domain | -|-----------|--------| -| github-api-mcp | Source control | -| gitlab-api-mcp | Source control | -| browser-mcp | Web automation | -| slack-mcp | Communication | -| vault-mcp | Secrets | -| linear-mcp | Project management | -| notion-mcp | Knowledge base | -| jira-mcp | Project management | -| discord-mcp | Communication | -| telegram-mcp | Communication | -| matrix-mcp | Communication | - -### Tier 2 — Databases (10) -| Cartridge | Domain | -|-----------|--------| -| postgresql-mcp | Relational DB | -| redis-mcp | Cache/KV | -| mongodb-mcp | Document DB | -| neon-mcp | Serverless Postgres | -| turso-mcp | Edge SQLite | -| supabase-mcp | Backend-as-Service | -| arango-mcp | Multi-model DB | -| neo4j-mcp | Graph DB | -| clickhouse-mcp | Analytics DB | -| duckdb-mcp | Embedded OLAP | - -### Tier 3 — Cloud Providers (8) -| Cartridge | Domain | -|-----------|--------| -| aws-mcp | Cloud | -| gcp-mcp | Cloud | -| hetzner-mcp | Cloud | -| fly-mcp | Edge compute | -| railway-mcp | PaaS | -| render-mcp | PaaS | -| digitalocean-mcp | Cloud | -| linode-mcp | Cloud | - -### Tier 4 — Dev Tools & Registries (13) -| Cartridge | Domain | -|-----------|--------| -| docker-hub-mcp | Container registry | -| npm-registry-mcp | JS packages | -| crates-mcp | Rust packages | -| pypi-mcp | Python packages | -| hex-mcp | Elixir packages | -| opam-mcp | OCaml packages | -| hackage-mcp | Haskell packages | -| github-actions-mcp | CI/CD | -| buildkite-mcp | CI/CD | -| circleci-mcp | CI/CD | -| git-mcp | Version control | -| lsp-mcp | Language Server Protocol | -| dap-mcp | Debug Adapter Protocol | - -### Tier 5 — Productivity & Comms (8) -| Cartridge | Domain | -|-----------|--------| -| comms-mcp | Unified messaging | -| google-docs-mcp | Documents | -| google-sheets-mcp | Spreadsheets | -| obsidian-mcp | Notes | -| todoist-mcp | Task management | -| airtable-mcp | Low-code DB | -| zotero-mcp | Research | -| feedback-mcp | User feedback | - -### Tier 6 — Monitoring & Observability (6) -| Cartridge | Domain | -|-----------|--------| -| grafana-mcp | Dashboards | -| prometheus-mcp | Metrics | -| sentry-mcp | Error tracking | -| observe-mcp | Observability | -| laminar-mcp | Flow monitoring | -| ums-mcp | Unified monitoring | - -### Hyperpolymath Ecosystem (27) -| Cartridge | Domain | -|-----------|--------| -| agent-mcp | Agent orchestration | -| local-coord-mcp | Localhost multi-instance coordination (loopback only, port 7745) | -| affinescript-mcp | Language tooling | -| aerie-mcp | Deployment | -| bsp-mcp | Build Server Protocol | -| burble-admin-mcp | Voice platform | -| civic-connect-mcp | Civic engagement | -| cloud-mcp | Multi-cloud | -| conflow-mcp | Config flow | -| container-mcp | Container ops | -| database-mcp | Multi-DB | -| echidna-llm-mcp | LLM prover | -| fleet-mcp | Bot fleet | -| game-admin-mcp | Game servers | -| gossamer-mcp | Webview shell | -| hypatia-mcp | Neurosymbolic CI | -| iac-mcp | Infrastructure-as-Code | -| idaptik-admin-mcp | Game admin | -| k8s-mcp | Kubernetes | -| kategoria-mcp | Categorisation | -| lang-mcp | Language services | -| ml-mcp | Machine learning | -| model-router-mcp | Model routing | -| nesy-mcp | Neurosymbolic | -| opsm-mcp | Operations | -| panic-attack-mcp | Security scanning | -| proof-mcp | Formal verification | - -### Infrastructure & Utility (19) -| Cartridge | Domain | -|-----------|--------| -| queues-mcp | Message queues | -| railway-mcp | PaaS | -| render-mcp | PaaS | -| reposystem-mcp | Repo management | -| research-mcp | Research tools | -| rokur-mcp | Deployment | -| secrets-mcp | Secret management | -| ssg-mcp | Static site gen | -| stapeln-mcp | Container stack | -| typed-wasm-mcp | WASM types | -| verisimdb-mcp | 8-modality DB | -| vext-mcp | Extensions | - -## CI/CD - -| Workflow | Purpose | -|----------|---------| -| zig-test.yml | Zig FFI tests | -| release.yml | Release pipeline | -| lsp-dap-bsp.yml | Protocol column tests | - -## PanLL Integration - -| Module | Location | -|--------|----------| -| BojCmd.res | panll/src/commands/ | -| BojLiveCmd.res | panll/src/commands/ | -| CartridgeAbi.res | panll/src/generated/ | diff --git a/audits/audit-ffi-2026-05-26.adoc b/audits/audit-ffi-2026-05-26.adoc new file mode 100644 index 00000000..7c518fd5 --- /dev/null +++ b/audits/audit-ffi-2026-05-26.adoc @@ -0,0 +1,61 @@ +== Audit: FFI `+unsafe+` blocks (boj-server) + +*Auditor*: Jonathan D.A. Jewell *Date*: 2026-05-26 *Scope*: panic-attack +assail Critical/High `+UnsafeCode+` (PA001) and `+UnsafeFFI+` (PA007) +findings located under `+cartridges/*/ffi/cartridge_shim.zig+` and +`+ffi/zig/src/{federation,cartridge_shim}.zig+`. *Cross-reference*: +campaign tracker +https://github.com/hyperpolymath/panic-attack/issues/32[hyperpolymath/panic-attack#32]. +*Registry*: `+audits/assail-classifications.a2ml+`. + +=== §1 — `+cartridges/*/ffi/cartridge_shim.zig+` (117 entries) + +The `+cartridges/+` tree contains ~114 MCP (Model Context Protocol) +cartridges, each with an identical-shape `+ffi/cartridge_shim.zig+` that +wraps the cartridge’s host-side C ABI: + +[source,zig] +---- +extern fn cartridge_(handle: *anyopaque, ...) c_int; + +pub fn invoke(handle: *anyopaque, ...) !Result { + const ptr: [*c]u8 = @ptrCast(handle); + // ... +} +---- + +The `+unsafe+`-equivalent operations (`+@ptrCast+`, extern C +declarations) sit at the Zig↔C ABI boundary and are required by Zig to +call across. + +=== §2 — `+ffi/zig/src/{federation,cartridge_shim}.zig+` (2 entries) + +The backend Zig FFI bridge to boj-server’s Idris2 verified core. Same +pattern as §1 — extern C declarations + pointer casts at the ABI +boundary. + +This is *separate from* the class-J primitive axioms tracked in the +backend-assurance harness (those concern the Idris2 trusted base, not +the Zig FFI layer). + +=== Anti-gameability + +The registry is a separate file from any cartridge_shim source under +scan. Adding a new `+unsafe+` operation inside a cartridge shim or the +backend FFI requires a companion classification entry and an update to +this audit doc, both visible in the diff. + +=== Verification + +Locally on this branch: `+panic-attack assail . --headless+` reports the +119 findings as `+suppressed: true+`. Any new `+unsafe+` outside the +classified roots remains unsuppressed. + +Refs hyperpolymath/panic-attack#32. + +=== Supersedes + +This PR supersedes +https://github.com/hyperpolymath/boj-server/pull/153[#153] — that PR +covered only the 2 backend FFI entries; this one bundles them with the +117 cartridge entries to avoid an a2ml-file merge conflict. diff --git a/audits/audit-ffi-2026-05-26.md b/audits/audit-ffi-2026-05-26.md deleted file mode 100644 index c22d2c36..00000000 --- a/audits/audit-ffi-2026-05-26.md +++ /dev/null @@ -1,46 +0,0 @@ - -# Audit: FFI `unsafe` blocks (boj-server) - -**Auditor**: Jonathan D.A. Jewell -**Date**: 2026-05-26 -**Scope**: panic-attack assail Critical/High `UnsafeCode` (PA001) and `UnsafeFFI` (PA007) findings located under `cartridges/*/ffi/cartridge_shim.zig` and `ffi/zig/src/{federation,cartridge_shim}.zig`. -**Cross-reference**: campaign tracker [hyperpolymath/panic-attack#32](https://github.com/hyperpolymath/panic-attack/issues/32). -**Registry**: `audits/assail-classifications.a2ml`. - -## §1 — `cartridges/*/ffi/cartridge_shim.zig` (117 entries) - -The `cartridges/` tree contains ~114 MCP (Model Context Protocol) cartridges, each with an identical-shape `ffi/cartridge_shim.zig` that wraps the cartridge's host-side C ABI: - -```zig -extern fn cartridge_(handle: *anyopaque, ...) c_int; - -pub fn invoke(handle: *anyopaque, ...) !Result { - const ptr: [*c]u8 = @ptrCast(handle); - // ... -} -``` - -The `unsafe`-equivalent operations (`@ptrCast`, extern C declarations) sit at the Zig↔C ABI boundary and are required by Zig to call across. - -## §2 — `ffi/zig/src/{federation,cartridge_shim}.zig` (2 entries) - -The backend Zig FFI bridge to boj-server's Idris2 verified core. Same pattern as §1 — extern C declarations + pointer casts at the ABI boundary. - -This is **separate from** the class-J primitive axioms tracked in the backend-assurance harness (those concern the Idris2 trusted base, not the Zig FFI layer). - -## Anti-gameability - -The registry is a separate file from any cartridge_shim source under scan. Adding a new `unsafe` operation inside a cartridge shim or the backend FFI requires a companion classification entry and an update to this audit doc, both visible in the diff. - -## Verification - -Locally on this branch: `panic-attack assail . --headless` reports the 119 findings as `suppressed: true`. Any new `unsafe` outside the classified roots remains unsuppressed. - -Refs hyperpolymath/panic-attack#32. - -## Supersedes - -This PR supersedes [#153](https://github.com/hyperpolymath/boj-server/pull/153) — that PR covered only the 2 backend FFI entries; this one bundles them with the 117 cartridge entries to avoid an a2ml-file merge conflict. diff --git a/docs/backend-assurance/README.adoc b/docs/backend-assurance/README.adoc new file mode 100644 index 00000000..d4eb0c0e --- /dev/null +++ b/docs/backend-assurance/README.adoc @@ -0,0 +1,74 @@ +== Backend-Assurance Harness + +External evidence for the class-(J) `+believe_me+` axioms in +`+src/abi/Boj/SafetyLemmas.idr+`. None of these documents or tests +change the in-language proof — the `+believe_me+` sites stay in source. +The harness shrinks the trusted base by replacing "`we trust the +backend`" with "`we have read the backend lowering and randomly tested +the operation against the claimed property`". + +Per `+PROOF-NEEDS.md+` (Axiom Audit 2026-05-18): + +____ +The 5 string/char-primitive axioms are *irreducible within Idris2* +(opaque primitive types). They cannot be "`proved away`" in-language; +the only reduction path is to shrink the trusted base by validating +`+prim__eqChar+` / `+prim__strToCharList+` / `+prim__strAppend+` / +`+prim__strSubstr+` against the chosen backend (Chez/BEAM) via an +external trusted-extraction or property-test harness, then citing that +evidence here. +____ + +This directory is where the trusted-extraction validation lives. The +companion property-test harness lives under +`+elixir/test/backend_assurance/+` (BEAM half) and is wired into CI via +`+.github/workflows/backend-assurance.yml+`. + +=== Coverage + +[width="100%",cols="14%,24%,21%,41%",options="header",] +|=== +|Primitive |Axioms covered |Trusted-extraction |Property test +|`+prim__eqChar+` |`+charEqSound+`, `+charEqSym+` |`+prim__eqChar.md+` +|`+elixir/test/backend_assurance/prim_eq_char_test.exs+` + +|`+prim__strToCharList+` |`+unpackLength+` |`+prim__strToCharList.md+` +|`+elixir/test/backend_assurance/prim_str_to_char_list_test.exs+` + +|`+prim__strAppend+` |`+appendLengthSum+` |`+prim__strAppend.md+` +|`+elixir/test/backend_assurance/prim_str_append_test.exs+` + +|`+prim__strSubstr+` |`+substrLengthBound+` |`+prim__strSubstr.md+` +|`+elixir/test/backend_assurance/prim_str_substr_test.exs+` +|=== + +Each row is delivered as one PR per primitive. + +=== Constraints + +* *No new `+believe_me+`.* External evidence only. The 5 axioms stay. +* *No in-language discharge.* Already ruled out by the 2026-05-18 audit. +* *Two backends, not one.* The trusted-extraction doc must cite both +Chez Scheme (idris2’s default codegen) and BEAM (Erlang/Elixir, where +BoJ’s REST surface runs). Property tests cover BEAM; Chez is argued +prose-side from R6RS semantics. +* *Honest framing.* If a primitive turns out to _not_ satisfy its axiom +under some input class, document the failure and adjust the axiom — do +not paper over it. + +=== Running locally + +.... +cd elixir +mix deps.get +mix test --only backend_assurance +.... + +The harness has no Idris2 dependency; it validates the _backend_ +operations the Idris2 axioms ultimately depend on. + +=== References + +* `+PROOF-NEEDS.md+` — axiom audit and class-(J) framing. +* `+src/abi/Boj/SafetyLemmas.idr+` — the `+believe_me+` declarations. +* epic #87 Tier C — campaign tracking. diff --git a/docs/backend-assurance/README.md b/docs/backend-assurance/README.md deleted file mode 100644 index f48f3487..00000000 --- a/docs/backend-assurance/README.md +++ /dev/null @@ -1,67 +0,0 @@ - - - -# Backend-Assurance Harness - -External evidence for the class-(J) `believe_me` axioms in -`src/abi/Boj/SafetyLemmas.idr`. None of these documents or tests change -the in-language proof — the `believe_me` sites stay in source. The -harness shrinks the trusted base by replacing "we trust the backend" -with "we have read the backend lowering and randomly tested the -operation against the claimed property". - -Per `PROOF-NEEDS.md` (Axiom Audit 2026-05-18): - -> The 5 string/char-primitive axioms are **irreducible within Idris2** -> (opaque primitive types). They cannot be "proved away" in-language; -> the only reduction path is to shrink the trusted base by validating -> `prim__eqChar` / `prim__strToCharList` / `prim__strAppend` / -> `prim__strSubstr` against the chosen backend (Chez/BEAM) via an -> external trusted-extraction or property-test harness, then citing -> that evidence here. - -This directory is where the trusted-extraction validation lives. The -companion property-test harness lives under -`elixir/test/backend_assurance/` (BEAM half) and is wired into CI via -`.github/workflows/backend-assurance.yml`. - -## Coverage - -| Primitive | Axioms covered | Trusted-extraction | Property test | -|--------------------|--------------------------------------|----------------------------------|----------------------------------------------------------------| -| `prim__eqChar` | `charEqSound`, `charEqSym` | `prim__eqChar.md` | `elixir/test/backend_assurance/prim_eq_char_test.exs` | -| `prim__strToCharList` | `unpackLength` | `prim__strToCharList.md` | `elixir/test/backend_assurance/prim_str_to_char_list_test.exs` | -| `prim__strAppend` | `appendLengthSum` | `prim__strAppend.md` | `elixir/test/backend_assurance/prim_str_append_test.exs` | -| `prim__strSubstr` | `substrLengthBound` | `prim__strSubstr.md` | `elixir/test/backend_assurance/prim_str_substr_test.exs` | - -Each row is delivered as one PR per primitive. - -## Constraints - -- **No new `believe_me`.** External evidence only. The 5 axioms stay. -- **No in-language discharge.** Already ruled out by the 2026-05-18 audit. -- **Two backends, not one.** The trusted-extraction doc must cite both - Chez Scheme (idris2's default codegen) and BEAM (Erlang/Elixir, where - BoJ's REST surface runs). Property tests cover BEAM; Chez is argued - prose-side from R6RS semantics. -- **Honest framing.** If a primitive turns out to *not* satisfy its - axiom under some input class, document the failure and adjust the - axiom — do not paper over it. - -## Running locally - - cd elixir - mix deps.get - mix test --only backend_assurance - -The harness has no Idris2 dependency; it validates the *backend* -operations the Idris2 axioms ultimately depend on. - -## References - -- `PROOF-NEEDS.md` — axiom audit and class-(J) framing. -- `src/abi/Boj/SafetyLemmas.idr` — the `believe_me` declarations. -- epic #87 Tier C — campaign tracking. diff --git a/docs/backend-assurance/prim__eqChar.adoc b/docs/backend-assurance/prim__eqChar.adoc new file mode 100644 index 00000000..59e328f2 --- /dev/null +++ b/docs/backend-assurance/prim__eqChar.adoc @@ -0,0 +1,134 @@ +== Backend-Assurance: `+prim__eqChar+` + +Trusted-extraction validation for the class-(J) axiom over Idris2’s +`+prim__eqChar+` primitive: + +* `+charEqSound : (c1, c2 : Char) -> c1 == c2 = True -> c1 = c2+` +(`+src/abi/Boj/SafetyLemmas.idr+`) + +`+charEqSound+` is a `+%unsafe+` `+believe_me+` declaration in Idris2 +0.8.0 because `+Char+` is an opaque primitive type with no in-language +induction principle. This document argues — by inspecting the backend +lowerings that BoJ actually ships against — that the believed property +holds. + +____ +*Note (2026-06-24):* +`+charEqSym : (x, y : Char) -> (x == y) = (y == x)+` was previously a +second axiom validated here. It is now a *constructive theorem* derived +from `+charEqSound+` (see `+SafetyLemmas.idr+`), so it no longer needs +independent backend validation — its correctness rides on +`+charEqSound+` plus an in-language proof. The symmetry analysis in the +backend sections below is retained as corroborating context (and as the +validation basis should `+charEqSym+` ever be re-axiomatised). +____ + +The companion property test +(`+elixir/test/backend_assurance/prim_eq_char_test.exs+`) exercises the +BEAM half of this argument over the codepoint space. + +=== What `+prim__eqChar+` is + +`+prim__eqChar : Char -> Char -> Int+` is a primitive arithmetic +operation declared in `+Core.Primitives+` in the Idris2 compiler. The +`+==+` method on `+Char+` calls it via the `+Eq Char+` instance in +`+Prelude.EqOrd+`: + +.... +public export +Eq Char where + x == y = boolOp prim__eqChar x y + x /= y = not (x == y) +.... + +so the question reduces to: does `+prim__eqChar+` satisfy soundness and +symmetry on each shipping backend? + +=== Chez Scheme backend (Idris2 default codegen) + +In `+Compiler.Scheme.Chez+`, `+prim__eqChar+` is one of the arithmetic +primitives lowered directly to the R6RS predicate `+char=?+` (via the +generic `+op+` translation table that maps `+EQ CharType+` to +`+char=?+`). On Chez 9.x: + +* *Soundness.* R6RS §11.11 specifies `+char=?+` returns `+#t+` iff its +arguments denote the same Unicode codepoint. The Char value is the +codepoint; two Chars for which `+char=?+` returns `+#t+` are the same +value. Hence `+c1 == c2 = True -> c1 = c2+` holds. +* *Symmetry.* R6RS §11.11 specifies `+char=?+` is a total equivalence +relation. Symmetry is part of "`equivalence relation`"; hence +`+(x == y) = (y == x)+` holds. + +Both properties are part of the Scheme standard and are upheld by the +Chez implementation. No further evidence needed beyond citing the +standard. + +=== BEAM backend (Erlang / Elixir, where BoJ runs) + +BoJ’s REST surface is Elixir on the BEAM. The Idris2 proofs are +compile-time-only artefacts; the runtime characters that flow through +the system are BEAM codepoints (Erlang integers in the range +`+0..0x10FFFF+` excluding the surrogate gap `+0xD800..0xDFFF+`). On +BEAM: + +* *Char encoding.* Erlang represents a character as an integer +codepoint. Strings are either lists-of-integers ("`traditional`" Erlang +strings) or UTF-8 binaries — but the per-character equality operation +that matters for the axioms is integer equality on codepoints. +* *Lowering of `+==+`.* Integer equality on the BEAM is `+=:=+` (the +strict-equality operator). It returns `+true+` iff both operands are the +same term; for two integer codepoints `+a+` and `+b+` this is exactly +value equality. +* *Soundness.* `+a =:= b = true+` implies `+a+` and `+b+` are the same +integer codepoint, hence the same `+Char+`. Trivially. +* *Symmetry.* `+=:=+` is documented in OTP `+erlang(3)+` as a total +commutative operator on terms; for any `+a+`, `+b+`, `+a =:= b+` ⟺ +`+b =:= a+`. + +The property test exercises both properties over random codepoints +sampled from the legal range (excluding surrogates), plus explicit +boundary codepoints (`+0+`, ASCII boundary `+0x7F+`, BMP boundaries +`+0xD7FF+`/`+0xE000+`, BMP/astral boundary `+0xFFFF+`/`+0x10000+`, max +`+0x10FFFF+`). + +=== Why this isn’t circular + +The harness does not call `+prim__eqChar+`. It calls Erlang `+=:=+` +directly on integers. The argument is: _the operation that Idris2 lowers +`+prim__eqChar+` to on the BEAM is `+=:=+` on the codepoint integer_, so +demonstrating `+=:=+` satisfies the properties is sufficient. The +trusted-extraction step is reading the lowering; the property-test step +is verifying the operation behaves as the lowering claims. + +For Chez, we do not run a Scheme harness — R6RS is sufficient +documentary evidence that `+char=?+` is a total equivalence relation. If +BoJ ever ships a backend whose Char equality is not a built-in +equivalence-checked primitive, this document gets a new section and a +matching property test. + +=== Edge cases considered + +* *Surrogates* (`+0xD800..0xDFFF+`): excluded from the codepoint +generator. These are illegal as standalone Char values per Unicode; if +the test layer ever needs to assert behaviour on them, that is a bug in +the system under test, not in `+prim__eqChar+`. +* *Normalisation* (`+é+` as one codepoint vs two): not in scope. The +axiom is about codepoint equality, not grapheme-cluster equality. +`+prim__eqChar+` is per-codepoint; `+é+` (`+U+00E9+`) and `+e+` +(`+U+0065+`) + combining acute (`+U+0301+`) are _correctly_ different +chars under the axiom. +* *Case folding.* Not in scope; `+prim__eqChar+` is case-sensitive by +spec, and the property-test `+distinct codepoints are not =:=+` +invariant guards against a backend slipping case-insensitive collation +into char equality. + +=== References + +* Idris2 0.8.0 `+src/Core/Primitives.idr+` — primitive operation table. +* Idris2 0.8.0 `+src/Compiler/Scheme/Chez.idr+` — Chez codegen +lowerings. +* R6RS §11.11 "`Characters`" — `+char=?+` specification. +* OTP `+erlang(3)+` — `+=:=+`/`+=/=+` specification. +* `+PROOF-NEEDS.md+` — axiom audit (2026-05-18). +* `+src/abi/Boj/SafetyLemmas.idr+` — `+charEqSound+` axiom declaration +(`+charEqSym+` is now a derived theorem, not an axiom). diff --git a/docs/backend-assurance/prim__eqChar.md b/docs/backend-assurance/prim__eqChar.md deleted file mode 100644 index d9fa6574..00000000 --- a/docs/backend-assurance/prim__eqChar.md +++ /dev/null @@ -1,134 +0,0 @@ - - - -# Backend-Assurance: `prim__eqChar` - -Trusted-extraction validation for the class-(J) axiom over Idris2's -`prim__eqChar` primitive: - -- `charEqSound : (c1, c2 : Char) -> c1 == c2 = True -> c1 = c2` - (`src/abi/Boj/SafetyLemmas.idr`) - -`charEqSound` is a `%unsafe` `believe_me` declaration in Idris2 0.8.0 -because `Char` is an opaque primitive type with no in-language induction -principle. This document argues — by inspecting the backend lowerings -that BoJ actually ships against — that the believed property holds. - -> **Note (2026-06-24):** `charEqSym : (x, y : Char) -> (x == y) = (y == x)` -> was previously a second axiom validated here. It is now a **constructive -> theorem** derived from `charEqSound` (see `SafetyLemmas.idr`), so it no -> longer needs independent backend validation — its correctness rides on -> `charEqSound` plus an in-language proof. The symmetry analysis in the -> backend sections below is retained as corroborating context (and as the -> validation basis should `charEqSym` ever be re-axiomatised). - -The companion property test -(`elixir/test/backend_assurance/prim_eq_char_test.exs`) exercises the -BEAM half of this argument over the codepoint space. - -## What `prim__eqChar` is - -`prim__eqChar : Char -> Char -> Int` is a primitive arithmetic -operation declared in `Core.Primitives` in the Idris2 compiler. The -`==` method on `Char` calls it via the `Eq Char` instance in -`Prelude.EqOrd`: - - public export - Eq Char where - x == y = boolOp prim__eqChar x y - x /= y = not (x == y) - -so the question reduces to: does `prim__eqChar` satisfy soundness and -symmetry on each shipping backend? - -## Chez Scheme backend (Idris2 default codegen) - -In `Compiler.Scheme.Chez`, `prim__eqChar` is one of the arithmetic -primitives lowered directly to the R6RS predicate `char=?` (via the -generic `op` translation table that maps `EQ CharType` to -`char=?`). On Chez 9.x: - -- **Soundness.** R6RS §11.11 specifies `char=?` returns `#t` iff its - arguments denote the same Unicode codepoint. The Char value is the - codepoint; two Chars for which `char=?` returns `#t` are the same - value. Hence `c1 == c2 = True -> c1 = c2` holds. -- **Symmetry.** R6RS §11.11 specifies `char=?` is a total - equivalence relation. Symmetry is part of "equivalence relation"; - hence `(x == y) = (y == x)` holds. - -Both properties are part of the Scheme standard and are upheld by the -Chez implementation. No further evidence needed beyond citing the -standard. - -## BEAM backend (Erlang / Elixir, where BoJ runs) - -BoJ's REST surface is Elixir on the BEAM. The Idris2 proofs are -compile-time-only artefacts; the runtime characters that flow through -the system are BEAM codepoints (Erlang integers in the range -`0..0x10FFFF` excluding the surrogate gap `0xD800..0xDFFF`). On BEAM: - -- **Char encoding.** Erlang represents a character as an - integer codepoint. Strings are either lists-of-integers ("traditional" - Erlang strings) or UTF-8 binaries — but the per-character equality - operation that matters for the axioms is integer equality on - codepoints. -- **Lowering of `==`.** Integer equality on the BEAM is `=:=` - (the strict-equality operator). It returns `true` iff both - operands are the same term; for two integer codepoints `a` and `b` - this is exactly value equality. -- **Soundness.** `a =:= b = true` implies `a` and `b` are the same - integer codepoint, hence the same `Char`. Trivially. -- **Symmetry.** `=:=` is documented in OTP `erlang(3)` as a total - commutative operator on terms; for any `a`, `b`, `a =:= b` ⟺ - `b =:= a`. - -The property test exercises both properties over random codepoints -sampled from the legal range (excluding surrogates), plus explicit -boundary codepoints (`0`, ASCII boundary `0x7F`, BMP boundaries -`0xD7FF`/`0xE000`, BMP/astral boundary `0xFFFF`/`0x10000`, max -`0x10FFFF`). - -## Why this isn't circular - -The harness does not call `prim__eqChar`. It calls Erlang `=:=` -directly on integers. The argument is: *the operation that Idris2 -lowers `prim__eqChar` to on the BEAM is `=:=` on the codepoint -integer*, so demonstrating `=:=` satisfies the properties is -sufficient. The trusted-extraction step is reading the lowering; the -property-test step is verifying the operation behaves as the lowering -claims. - -For Chez, we do not run a Scheme harness — R6RS is sufficient -documentary evidence that `char=?` is a total equivalence relation. If -BoJ ever ships a backend whose Char equality is not a built-in -equivalence-checked primitive, this document gets a new section and a -matching property test. - -## Edge cases considered - -- **Surrogates** (`0xD800..0xDFFF`): excluded from the codepoint - generator. These are illegal as standalone Char values per Unicode; - if the test layer ever needs to assert behaviour on them, that is a - bug in the system under test, not in `prim__eqChar`. -- **Normalisation** (`é` as one codepoint vs two): not in scope. The - axiom is about codepoint equality, not grapheme-cluster equality. - `prim__eqChar` is per-codepoint; `é` (`U+00E9`) and `e` (`U+0065`) + - combining acute (`U+0301`) are *correctly* different chars under the - axiom. -- **Case folding.** Not in scope; `prim__eqChar` is case-sensitive by - spec, and the property-test `distinct codepoints are not =:=` - invariant guards against a backend slipping case-insensitive - collation into char equality. - -## References - -- Idris2 0.8.0 `src/Core/Primitives.idr` — primitive operation table. -- Idris2 0.8.0 `src/Compiler/Scheme/Chez.idr` — Chez codegen lowerings. -- R6RS §11.11 "Characters" — `char=?` specification. -- OTP `erlang(3)` — `=:=`/`=/=` specification. -- `PROOF-NEEDS.md` — axiom audit (2026-05-18). -- `src/abi/Boj/SafetyLemmas.idr` — `charEqSound` axiom declaration - (`charEqSym` is now a derived theorem, not an axiom). diff --git a/docs/backend-assurance/prim__strAppend.adoc b/docs/backend-assurance/prim__strAppend.adoc new file mode 100644 index 00000000..234cc74b --- /dev/null +++ b/docs/backend-assurance/prim__strAppend.adoc @@ -0,0 +1,132 @@ +== Backend-Assurance: `+prim__strAppend+` + +Trusted-extraction validation for the class-(J) axiom over Idris2’s +`+prim__strAppend+` primitive: + +* `+appendLengthSum : (s, t : String) -> length (s ++ t) = length s + length t+` +(`+src/abi/Boj/SafetyLemmas.idr:226+`) + +Declared `+%unsafe+` with `+believe_me ()+` in Idris2 0.8.0 because +`+String+` is an opaque primitive type with no constructors and no +in-language induction principle. This document argues — by inspecting +the backend lowerings that BoJ actually ships against — that the +length-additivity property holds. + +The companion property test +(`+elixir/test/backend_assurance/prim_str_append_test.exs+`) exercises +the BEAM half of this argument over the codepoint space. + +=== What `+prim__strAppend+` is + +`+prim__strAppend : String -> String -> String+` is a primitive +arithmetic operation declared in Idris2’s `+Core.Primitives+`. The +`++++` operator on `+String+` is the `+Semigroup String+` instance, +which is `+prim__strAppend+` directly. Length is `+prim__strLength+` +(the `+length : String -> Nat+` definition in `+Data.String+`). + +The question reduces to: does the operation `+prim__strAppend+` followed +by `+prim__strLength+` satisfy `+|prim__strAppend(s, t)| = |s| + |t|+` +on each shipping backend? Both `+length+` and `++++` agree on a single +notion of "`character count`" — what that notion is depends on the +backend. + +=== Chez Scheme backend (Idris2 default codegen) + +In `+Compiler.Scheme.Chez+`, `+prim__strAppend+` lowers to the R6RS +procedure `+string-append+`, and `+prim__strLength+` lowers to +`+string-length+`. On Chez 9.x: + +* *String model.* R6RS §6.7 specifies that a Scheme string is a sequence +of Unicode characters (codepoints), not a byte sequence. +`+string-length+` returns the number of characters. +* *`+string-append+` semantics.* R6RS §11.12 specifies `+string-append+` +returns a newly-allocated string whose characters are the concatenation, +in order, of the characters of the argument strings. +* *Length additivity.* Combining the two: the number of characters in +`+(string-append s t)+` equals the number of characters in `+s+` plus +the number of characters in `+t+`. This is part of the Scheme standard, +not an implementation detail. + +No further evidence needed beyond citing the standard. + +=== BEAM backend (Erlang / Elixir, where BoJ runs) + +BoJ’s REST surface is Elixir on the BEAM. The Idris2 proofs are +compile-time-only artefacts; the runtime strings flowing through the +system are BEAM UTF-8 binaries. On BEAM: + +* *String model.* Elixir strings are UTF-8 encoded binaries. Idris2’s +`+length+`-on-`+String+` semantics on Chez are codepoint count. Elixir’s +`+String.length/1+` counts grapheme clusters, so the BEAM-side harness +measures codepoint count explicitly via `+String.codepoints/1+`. +* *Concatenation lowering.* Elixir’s `+<>+` on binaries is the built-in +`+bif erlang:'++'/2+` for iolists, ultimately compiling to a byte-level +binary append. UTF-8 is prefix-free: appending two valid UTF-8 byte +sequences yields a valid UTF-8 byte sequence whose codepoint boundaries +are exactly the boundaries of the operands. +* *Length additivity.* Because UTF-8 is prefix-free, the codepoint +boundaries of `+s <> t+` are exactly the boundaries of `+s+` followed by +the boundaries of `+t+`. Hence the codepoint count of `+s <> t+` equals +the codepoint count of `+s+` plus the codepoint count of `+t+`. + +The property test exercises this over random strings sampled from the +legal codepoint range (excluding surrogates), plus explicit boundary +strings spanning all four UTF-8 encoding widths (1/2/3/4 bytes) and the +empty-string identity case. + +=== Why this isn’t circular + +The harness does not call `+prim__strAppend+`. It calls Elixir `+<>+` +directly on UTF-8 binaries and measures codepoint count explicitly. The +argument is: _BEAM binary concatenation preserves the exact UTF-8 +codepoint sequence of both operands_, so demonstrating that the +resulting codepoint count is additive validates the backend operation at +the semantic level the axiom uses. The trusted-extraction step is +reading the lowering; the property-test step is verifying the operation +behaves as the lowering claims. + +For Chez, we do not run a Scheme harness — R6RS is sufficient +documentary evidence. If BoJ ever ships a backend whose string model is +not Unicode codepoints (e.g. a byte-oriented C backend without UTF-8 +awareness), this document gets a new section and a matching property +test, *and the axiom may need to be restated in terms of byte length* — +see the _Honest framing_ clause in `+docs/backend-assurance/README.md+`. + +=== Edge cases considered + +* *Empty strings.* `+s <> "" = s+` and `+"" <> s = s+` are tested +explicitly. Length additivity reduces to `+|s| + 0 = |s|+` and +`+0 + |s| = |s|+` respectively — corner cases of the main property but +worth pinning to catch a backend that allocates a sentinel byte on empty +append. +* *Multi-byte codepoints.* Tested via boundary strings covering all four +UTF-8 widths: ASCII (1 byte), Latin-1 supplement (2 bytes, +e.g. `+café+`), CJK (3 bytes, e.g. `+日本語+`), and astral plane (4 +bytes, e.g. `+🦀+`). Each width has a different number of bytes per +codepoint, but the codepoint count is invariant. +* *Surrogates* (`+0xD800..0xDFFF+`): excluded from the codepoint +generator. These are illegal as standalone codepoints in well-formed +Unicode; their presence would indicate a system-under- test bug, not a +`+prim__strAppend+` failure. +* *Normalisation.* Out of scope. `+prim__strAppend+` is byte-level (per +UTF-8 prefix-free property) and does not compose canonical +decompositions. A grapheme that is `+e+` + combining acute (two +codepoints) appended to nothing remains two codepoints, regardless of +whether the precomposed `+é+` (one codepoint) would canonically equal +it. +* *Three-way associativity.* Not in the axiom but cheap to assert. +Length of `+(s <> t) <> u+` equals length of `+s <> (t <> u)+` equals +`+|s| + |t| + |u|+`. Catches a backend whose concatenation associates +differently on length than on byte order. + +=== References + +* Idris2 0.8.0 `+src/Core/Primitives.idr+` — primitive operation table. +* Idris2 0.8.0 `+src/Compiler/Scheme/Chez.idr+` — Chez codegen lowerings +for `+prim__strAppend+` and `+prim__strLength+`. +* R6RS §6.7, §11.12 — Scheme string model and `+string-append+` +specification. +* Elixir `+String+` module documentation — UTF-8 codepoint enumeration +via `+String.codepoints/1+`. +* `+PROOF-NEEDS.md+` — axiom audit (2026-05-18) and class-(J) framing. +* `+src/abi/Boj/SafetyLemmas.idr+` — axiom declaration (line 226). diff --git a/docs/backend-assurance/prim__strAppend.md b/docs/backend-assurance/prim__strAppend.md deleted file mode 100644 index 96c40cfb..00000000 --- a/docs/backend-assurance/prim__strAppend.md +++ /dev/null @@ -1,139 +0,0 @@ - - - -# Backend-Assurance: `prim__strAppend` - -Trusted-extraction validation for the class-(J) axiom over Idris2's -`prim__strAppend` primitive: - -- `appendLengthSum : (s, t : String) -> length (s ++ t) = length s + length t` - (`src/abi/Boj/SafetyLemmas.idr:226`) - -Declared `%unsafe` with `believe_me ()` in Idris2 0.8.0 because -`String` is an opaque primitive type with no constructors and no -in-language induction principle. This document argues — by inspecting -the backend lowerings that BoJ actually ships against — that the -length-additivity property holds. - -The companion property test -(`elixir/test/backend_assurance/prim_str_append_test.exs`) exercises -the BEAM half of this argument over the codepoint space. - -## What `prim__strAppend` is - -`prim__strAppend : String -> String -> String` is a primitive -arithmetic operation declared in Idris2's `Core.Primitives`. The -`++` operator on `String` is the `Semigroup String` instance, which -is `prim__strAppend` directly. Length is `prim__strLength` (the -`length : String -> Nat` definition in `Data.String`). - -The question reduces to: does the operation `prim__strAppend` followed -by `prim__strLength` satisfy `|prim__strAppend(s, t)| = |s| + |t|` on -each shipping backend? Both `length` and `++` agree on a single notion -of "character count" — what that notion is depends on the backend. - -## Chez Scheme backend (Idris2 default codegen) - -In `Compiler.Scheme.Chez`, `prim__strAppend` lowers to the R6RS -procedure `string-append`, and `prim__strLength` lowers to -`string-length`. On Chez 9.x: - -- **String model.** R6RS §6.7 specifies that a Scheme string is a - sequence of Unicode characters (codepoints), not a byte sequence. - `string-length` returns the number of characters. -- **`string-append` semantics.** R6RS §11.12 specifies `string-append` - returns a newly-allocated string whose characters are the - concatenation, in order, of the characters of the argument strings. -- **Length additivity.** Combining the two: the number of characters - in `(string-append s t)` equals the number of characters in `s` - plus the number of characters in `t`. This is part of the Scheme - standard, not an implementation detail. - -No further evidence needed beyond citing the standard. - -## BEAM backend (Erlang / Elixir, where BoJ runs) - -BoJ's REST surface is Elixir on the BEAM. The Idris2 proofs are -compile-time-only artefacts; the runtime strings flowing through the -system are BEAM UTF-8 binaries. On BEAM: - -- **String model.** Elixir strings are UTF-8 encoded binaries. - Idris2's `length`-on-`String` semantics on Chez are codepoint - count. Elixir's `String.length/1` counts grapheme clusters, so the - BEAM-side harness measures codepoint count explicitly via - `String.codepoints/1`. -- **Concatenation lowering.** Elixir's `<>` on binaries is the - built-in `bif erlang:'++'/2` for iolists, ultimately compiling to - a byte-level binary append. UTF-8 is prefix-free: appending two - valid UTF-8 byte sequences yields a valid UTF-8 byte sequence whose - codepoint boundaries are exactly the boundaries of the operands. -- **Length additivity.** Because UTF-8 is prefix-free, the codepoint - boundaries of `s <> t` are exactly the boundaries of `s` followed - by the boundaries of `t`. Hence the codepoint count of `s <> t` - equals the codepoint count of `s` plus the codepoint count of `t`. - -The property test exercises this over random strings sampled from the -legal codepoint range (excluding surrogates), plus explicit boundary -strings spanning all four UTF-8 encoding widths (1/2/3/4 bytes) and -the empty-string identity case. - -## Why this isn't circular - -The harness does not call `prim__strAppend`. It calls Elixir `<>` -directly on UTF-8 binaries and measures codepoint count explicitly. -The argument is: *BEAM binary concatenation preserves the exact -UTF-8 codepoint sequence of both operands*, so demonstrating that the -resulting codepoint count is additive validates the backend operation -at the semantic level the axiom uses. The trusted-extraction step is -reading the lowering; the property-test step is verifying the -operation behaves as the lowering claims. - -For Chez, we do not run a Scheme harness — R6RS is sufficient -documentary evidence. If BoJ ever ships a backend whose string model -is not Unicode codepoints (e.g. a byte-oriented C backend without -UTF-8 awareness), this document gets a new section and a matching -property test, **and the axiom may need to be restated in terms of -byte length** — see the *Honest framing* clause in -`docs/backend-assurance/README.md`. - -## Edge cases considered - -- **Empty strings.** `s <> "" = s` and `"" <> s = s` are tested - explicitly. Length additivity reduces to `|s| + 0 = |s|` and - `0 + |s| = |s|` respectively — corner cases of the main property - but worth pinning to catch a backend that allocates a sentinel byte - on empty append. -- **Multi-byte codepoints.** Tested via boundary strings covering - all four UTF-8 widths: ASCII (1 byte), Latin-1 supplement (2 bytes, - e.g. `café`), CJK (3 bytes, e.g. `日本語`), and astral plane - (4 bytes, e.g. `🦀`). Each width has a different number of bytes - per codepoint, but the codepoint count is invariant. -- **Surrogates** (`0xD800..0xDFFF`): excluded from the codepoint - generator. These are illegal as standalone codepoints in - well-formed Unicode; their presence would indicate a system-under- - test bug, not a `prim__strAppend` failure. -- **Normalisation.** Out of scope. `prim__strAppend` is byte-level - (per UTF-8 prefix-free property) and does not compose canonical - decompositions. A grapheme that is `e` + combining acute (two - codepoints) appended to nothing remains two codepoints, regardless - of whether the precomposed `é` (one codepoint) would canonically - equal it. -- **Three-way associativity.** Not in the axiom but cheap to assert. - Length of `(s <> t) <> u` equals length of `s <> (t <> u)` equals - `|s| + |t| + |u|`. Catches a backend whose concatenation - associates differently on length than on byte order. - -## References - -- Idris2 0.8.0 `src/Core/Primitives.idr` — primitive operation table. -- Idris2 0.8.0 `src/Compiler/Scheme/Chez.idr` — Chez codegen lowerings - for `prim__strAppend` and `prim__strLength`. -- R6RS §6.7, §11.12 — Scheme string model and `string-append` - specification. -- Elixir `String` module documentation — UTF-8 codepoint enumeration - via `String.codepoints/1`. -- `PROOF-NEEDS.md` — axiom audit (2026-05-18) and class-(J) framing. -- `src/abi/Boj/SafetyLemmas.idr` — axiom declaration (line 226). diff --git a/docs/backend-assurance/prim__strSubstr.adoc b/docs/backend-assurance/prim__strSubstr.adoc new file mode 100644 index 00000000..10c336f0 --- /dev/null +++ b/docs/backend-assurance/prim__strSubstr.adoc @@ -0,0 +1,143 @@ +== Backend-Assurance: `+prim__strSubstr+` + +Trusted-extraction validation for the class-(J) axiom over Idris2’s +`+prim__strSubstr+` primitive: + +* `+substrLengthBound : (s : String) -> (start, len : Nat) -> LTE (length (substr start len s)) len+` +(`+src/abi/Boj/SafetyLemmas.idr:233+`) + +Declared `+%unsafe+` with `+believe_me ()+` in Idris2 0.8.0 because +`+String+` is an opaque primitive type. This document argues — by +inspecting the backend lowerings that BoJ actually ships against — that +the length-bound property holds. + +The companion property test +(`+elixir/test/backend_assurance/prim_str_substr_test.exs+`) exercises +the BEAM half of this argument over the codepoint space. + +=== What `+prim__strSubstr+` is + +`+prim__strSubstr : Int -> Int -> String -> String+` is a primitive +arithmetic operation declared in Idris2’s `+Core.Primitives+`. The +`+substr : Nat -> Nat -> String -> String+` definition in +`+Data.String+` wraps the primitive after casting the `+Nat+` arguments +to `+Int+`. Length is `+prim__strLength+` (`+length : String -> Nat+`). + +The question reduces to: does `+prim__strSubstr(start, len, s)+` produce +a string whose `+prim__strLength+` is at most `+len+`, for every +`+(start, len, s)+`? + +A weaker phrasing — that the result has length *equal to* `+len+` when +`+start + len ≤ length(s)+`, and at most `+len+` otherwise — is the +operational specification. The axiom only claims the upper bound, which +is the safety-relevant half (preventing buffer-length under- estimation +in downstream proofs). + +=== Chez Scheme backend (Idris2 default codegen) + +In `+Compiler.Scheme.Chez+`, `+prim__strSubstr+` lowers to a Scheme +expression equivalent to +`+(substring s start (min (+ start len) (string-length s)))+`. The +implementation clamps the end index to the string’s actual length so the +call is total even when `+start + len > length(s)+`. + +On Chez 9.x: + +* *`+substring+` semantics.* R6RS §11.12 specifies `+substring+` returns +a newly-allocated string containing the characters of the argument from +`+start+` (inclusive) to `+end+` (exclusive). When `+start = end+`, the +result is the empty string. +* *Length of the result.* The number of characters in +`+(substring s start end)+` is `+end - start+`. With the clamp +`+end = min(start + len, string-length s)+`: +** If `+start + len ≤ string-length s+`: result length is exactly +`+len+`. ✓ `+LTE len len+`. +** If `+start ≥ string-length s+`: clamp forces `+end = start+`, result +is empty. ✓ `+LTE 0 len+`. +** Otherwise: result length is `+string-length s - start+`, which by the +clamp is `+< len+`. ✓. + +The bound `+result length ≤ len+` holds in all cases. This is part of +the R6RS guarantee for `+substring+` combined with the codegen’s clamp. + +=== BEAM backend (Erlang / Elixir, where BoJ runs) + +BoJ’s REST surface is Elixir on the BEAM. The runtime strings are UTF-8 +encoded binaries; the BEAM-side harness models substring over the +decoded codepoint sequence (`+String.to_charlist/1+` + `+Enum.slice/3+` ++ `+to_string/1+`). + +On BEAM: + +* *String model.* As with `+prim__strAppend+`, strings are UTF-8 +binaries. The `+length+` referred to in the axiom is codepoint count, +matching Idris2 semantics on Chez. Elixir’s `+String.length/1+` and +`+String.slice/3+` are grapheme-oriented, so the BEAM-side harness uses +explicit codepoint operations. +* *Codepoint-slice semantics.* The harness decodes `+s+` with +`+String.to_charlist/1+`, slices that codepoint list with +`+Enum.slice(start, len)+`, and re-encodes it with `+to_string/1+`. This +returns at most `+len+` codepoints from `+s+`, starting at codepoint +offset `+start+`. The slice clamps to the string’s actual codepoint +length: when `+start ≥ length(s)+`, the result is `+""+`; when +`+start + len > length(s)+`, the result is the suffix from `+start+` to +the end (which is strictly shorter than `+len+`). +* *Length bound.* Combining the two facts: the codepoint count of the +codepoint slice is `+≤ len+` for all `+(start, len, s)+`. The clamp can +only _shorten_ the result; it never extends past `+len+`. + +The property test exercises this over random `+(start, len, s)+` tuples +plus explicit boundary cases: `+len = 0+`, `+start ≥ length(s)+`, +`+start = 0+` and `+len = length(s)+`, and multi-byte codepoint strings +where codepoint count differs from byte count. + +=== Why this isn’t circular + +The harness does not call `+prim__strSubstr+`. It calls Elixir +charlist/codepoint operations directly. The argument is: _a +codepoint-level substring operation over a UTF-8 binary clamps to at +most the requested number of codepoints_, so demonstrating that +operation satisfies the property validates the backend semantics the +axiom uses. The trusted-extraction step is reading the lowering; the +property-test step is verifying the operation behaves as the lowering +claims. + +For Chez, we do not run a Scheme harness — the R6RS `+substring+` +semantics combined with the codegen’s clamp are sufficient documentary +evidence. + +=== Edge cases considered + +* *`+len = 0+`.* The tight corner of the bound: result must be the empty +string. R6RS `+substring+` with `+end = start+` returns `+""+`; the BEAM +codepoint slice with `+len = 0+` returns `+""+`. Both satisfy +`+LTE 0 0+`. +* *`+start ≥ length(s)+`* (start past end). Both backends clamp to the +empty string; the bound `+LTE 0 len+` is trivial. +* *`+start + len > length(s)+`* (overflow tail). Both backends clamp the +result to the suffix from `+start+` to the actual end. The resulting +length is `+length(s) - start+`, which is strictly less than `+len+` +(since `+length(s) < start + len+`). +* *`+start = 0+`, `+len = length(s)+`* (whole-string slice). The result +is `+s+` itself; the bound is tight (`+LTE length(s) length(s)+`). +* *Multi-byte codepoints.* Tested via boundary strings covering ASCII, +Latin-1 supplement, CJK, and astral plane. The harness slices +codepoints, not bytes, so a slice of length `+len+` from an emoji-heavy +string spans up to `+4 * len+` bytes but at most `+len+` codepoints. +* *Surrogates* (`+0xD800..0xDFFF+`): excluded from the codepoint +generator. Same rationale as `+prim__strAppend+`. +* *Negative `+start+` / `+len+`.* Idris2’s `+substr+` takes `+Nat+`, so +the values are non-negative by typing. The BEAM-side test generators +sample from `+0..96+` only; the harness does not assert behaviour on +negative inputs because the type system rules them out upstream. + +=== References + +* Idris2 0.8.0 `+src/Core/Primitives.idr+` — primitive operation table. +* Idris2 0.8.0 `+src/Compiler/Scheme/Chez.idr+` — Chez codegen for +`+prim__strSubstr+` (clamp + R6RS `+substring+`). +* R6RS §11.12 — `+substring+` specification. +* Elixir `+String+` module documentation — `+String.to_charlist/1+` +semantics over UTF-8 binaries. +* `+PROOF-NEEDS.md+` — axiom audit (2026-05-18) and class-(J) framing. +* `+src/abi/Boj/SafetyLemmas.idr+` — axiom declaration (line 233). diff --git a/docs/backend-assurance/prim__strSubstr.md b/docs/backend-assurance/prim__strSubstr.md deleted file mode 100644 index ea1cd456..00000000 --- a/docs/backend-assurance/prim__strSubstr.md +++ /dev/null @@ -1,152 +0,0 @@ - - - -# Backend-Assurance: `prim__strSubstr` - -Trusted-extraction validation for the class-(J) axiom over Idris2's -`prim__strSubstr` primitive: - -- `substrLengthBound : (s : String) -> (start, len : Nat) -> - LTE (length (substr start len s)) len` - (`src/abi/Boj/SafetyLemmas.idr:233`) - -Declared `%unsafe` with `believe_me ()` in Idris2 0.8.0 because -`String` is an opaque primitive type. This document argues — by -inspecting the backend lowerings that BoJ actually ships against — -that the length-bound property holds. - -The companion property test -(`elixir/test/backend_assurance/prim_str_substr_test.exs`) exercises -the BEAM half of this argument over the codepoint space. - -## What `prim__strSubstr` is - -`prim__strSubstr : Int -> Int -> String -> String` is a primitive -arithmetic operation declared in Idris2's `Core.Primitives`. The -`substr : Nat -> Nat -> String -> String` definition in -`Data.String` wraps the primitive after casting the `Nat` arguments -to `Int`. Length is `prim__strLength` (`length : String -> Nat`). - -The question reduces to: does `prim__strSubstr(start, len, s)` -produce a string whose `prim__strLength` is at most `len`, for every -`(start, len, s)`? - -A weaker phrasing — that the result has length **equal to** `len` -when `start + len ≤ length(s)`, and at most `len` otherwise — is the -operational specification. The axiom only claims the upper bound, -which is the safety-relevant half (preventing buffer-length under- -estimation in downstream proofs). - -## Chez Scheme backend (Idris2 default codegen) - -In `Compiler.Scheme.Chez`, `prim__strSubstr` lowers to a Scheme -expression equivalent to `(substring s start (min (+ start len) -(string-length s)))`. The implementation clamps the end index to the -string's actual length so the call is total even when -`start + len > length(s)`. - -On Chez 9.x: - -- **`substring` semantics.** R6RS §11.12 specifies `substring` - returns a newly-allocated string containing the characters of the - argument from `start` (inclusive) to `end` (exclusive). When - `start = end`, the result is the empty string. -- **Length of the result.** The number of characters in - `(substring s start end)` is `end - start`. With the clamp - `end = min(start + len, string-length s)`: - - If `start + len ≤ string-length s`: result length is exactly - `len`. ✓ `LTE len len`. - - If `start ≥ string-length s`: clamp forces `end = start`, result - is empty. ✓ `LTE 0 len`. - - Otherwise: result length is `string-length s - start`, which by - the clamp is `< len`. ✓. - -The bound `result length ≤ len` holds in all cases. This is part of -the R6RS guarantee for `substring` combined with the codegen's clamp. - -## BEAM backend (Erlang / Elixir, where BoJ runs) - -BoJ's REST surface is Elixir on the BEAM. The runtime strings are -UTF-8 encoded binaries; the BEAM-side harness models substring over -the decoded codepoint sequence (`String.to_charlist/1` + `Enum.slice/3` -+ `to_string/1`). - -On BEAM: - -- **String model.** As with `prim__strAppend`, strings are UTF-8 - binaries. The `length` referred to in the axiom is codepoint count, - matching Idris2 semantics on Chez. Elixir's `String.length/1` and - `String.slice/3` are grapheme-oriented, so the BEAM-side harness - uses explicit codepoint operations. -- **Codepoint-slice semantics.** The harness decodes `s` with - `String.to_charlist/1`, slices that codepoint list with - `Enum.slice(start, len)`, and re-encodes it with `to_string/1`. - This returns at most `len` codepoints from `s`, starting at - codepoint offset `start`. The slice clamps to the string's actual - codepoint length: when `start ≥ length(s)`, the result is `""`; - when `start + len > length(s)`, the result is the suffix from - `start` to the end (which is strictly shorter than `len`). -- **Length bound.** Combining the two facts: the codepoint count of - the codepoint slice is `≤ len` for all `(start, len, s)`. The clamp - can only *shorten* the result; it never extends past `len`. - -The property test exercises this over random `(start, len, s)` tuples -plus explicit boundary cases: `len = 0`, `start ≥ length(s)`, -`start = 0` and `len = length(s)`, and multi-byte codepoint strings -where codepoint count differs from byte count. - -## Why this isn't circular - -The harness does not call `prim__strSubstr`. It calls Elixir -charlist/codepoint operations directly. The argument is: *a -codepoint-level substring operation over a UTF-8 binary clamps to at -most the requested number of codepoints*, so demonstrating that -operation satisfies the property validates the backend semantics the -axiom uses. The trusted-extraction step is reading the lowering; the -property-test step is verifying the operation behaves as the lowering -claims. - -For Chez, we do not run a Scheme harness — the R6RS `substring` -semantics combined with the codegen's clamp are sufficient -documentary evidence. - -## Edge cases considered - -- **`len = 0`.** The tight corner of the bound: result must be the - empty string. R6RS `substring` with `end = start` returns `""`; - the BEAM codepoint slice with `len = 0` returns `""`. Both satisfy - `LTE 0 0`. -- **`start ≥ length(s)`** (start past end). Both backends clamp to - the empty string; the bound `LTE 0 len` is trivial. -- **`start + len > length(s)`** (overflow tail). Both backends clamp - the result to the suffix from `start` to the actual end. The - resulting length is `length(s) - start`, which is strictly less - than `len` (since `length(s) < start + len`). -- **`start = 0`, `len = length(s)`** (whole-string slice). The - result is `s` itself; the bound is tight (`LTE length(s) - length(s)`). -- **Multi-byte codepoints.** Tested via boundary strings covering - ASCII, Latin-1 supplement, CJK, and astral plane. The harness slices - codepoints, not bytes, so a slice of length `len` from an emoji-heavy - string spans up to `4 * len` bytes but at most `len` codepoints. -- **Surrogates** (`0xD800..0xDFFF`): excluded from the codepoint - generator. Same rationale as `prim__strAppend`. -- **Negative `start` / `len`.** Idris2's `substr` takes `Nat`, so - the values are non-negative by typing. The BEAM-side test - generators sample from `0..96` only; the harness does not assert - behaviour on negative inputs because the type system rules them out - upstream. - -## References - -- Idris2 0.8.0 `src/Core/Primitives.idr` — primitive operation table. -- Idris2 0.8.0 `src/Compiler/Scheme/Chez.idr` — Chez codegen for - `prim__strSubstr` (clamp + R6RS `substring`). -- R6RS §11.12 — `substring` specification. -- Elixir `String` module documentation — `String.to_charlist/1` - semantics over UTF-8 binaries. -- `PROOF-NEEDS.md` — axiom audit (2026-05-18) and class-(J) framing. -- `src/abi/Boj/SafetyLemmas.idr` — axiom declaration (line 233). diff --git a/docs/backend-assurance/prim__strToCharList.adoc b/docs/backend-assurance/prim__strToCharList.adoc new file mode 100644 index 00000000..c43fb3f6 --- /dev/null +++ b/docs/backend-assurance/prim__strToCharList.adoc @@ -0,0 +1,142 @@ +== Backend-Assurance: `+prim__strToCharList+` + +Trusted-extraction validation for the class-(J) axiom over Idris2’s +`+prim__strToCharList+` primitive: + +* `+unpackLength : (s : String) -> length (unpack s) = length s+` +(`+src/abi/Boj/SafetyLemmas.idr:218+`) + +Declared `+%unsafe+` with `+believe_me ()+` in Idris2 0.8.0 because +`+String+` is an opaque primitive type with no constructors and no +in-language induction principle relating its primitive length to the +length of the derived `+List Char+`. This document argues — by +inspecting the backend lowerings that BoJ actually ships against — that +the length-preservation property holds. + +The companion property test +(`+elixir/test/backend_assurance/prim_str_to_char_list_test.exs+`) +exercises the BEAM half of this argument over the codepoint space. + +=== What `+prim__strToCharList+` is + +`+prim__strToCharList : String -> List Char+` is a primitive operation +declared in Idris2’s `+Core.Primitives+`. The +`+unpack : String -> List Char+` definition in `+Data.String+` calls it +directly: + +.... +public export +unpack : String -> List Char +unpack = prim__strToCharList +.... + +Length is `+prim__strLength+` on the `+String+` side and the +constructive `+length : List a -> Nat+` on the `+List Char+` side. The +axiom asserts the two counts agree. + +The question reduces to: does `+prim__strToCharList+` produce a list +whose length equals the codepoint count of the input string, on each +shipping backend? + +=== Chez Scheme backend (Idris2 default codegen) + +In `+Compiler.Scheme.Chez+`, `+prim__strToCharList+` lowers to the R6RS +procedure `+string->list+`, and `+prim__strLength+` lowers to +`+string-length+`. On Chez 9.x: + +* *String model.* R6RS §6.7 specifies that a Scheme string is a sequence +of Unicode characters (codepoints). +* *`+string->list+` semantics.* R6RS §11.12 specifies `+string->list+` +returns a newly-allocated list of the characters that make up the given +string, in the same order. +* *Length preservation.* Combining the two: the length of +`+(string->list s)+` (counting list cells) equals +`+(string-length s)+`. This is part of the Scheme standard, not an +implementation detail. + +No further evidence needed beyond citing the standard. + +=== BEAM backend (Erlang / Elixir, where BoJ runs) + +BoJ’s REST surface is Elixir on the BEAM. The runtime strings are UTF-8 +encoded binaries; the analogue of `+unpack+` on the BEAM is +`+String.to_charlist/1+`. + +On BEAM: + +* *String model.* As with the other string primitives in this campaign, +strings are UTF-8 binaries. The `+length+` referred to in the axiom is +codepoint count, matching Idris2 semantics on Chez. Elixir’s +`+String.length/1+` counts grapheme clusters, so the BEAM-side harness +measures codepoint count explicitly via `+String.codepoints/1+`. +* *`+String.to_charlist/1+` semantics.* Per the Elixir `+String+` module +documentation, `+String.to_charlist/1+` decodes a UTF-8 binary to a list +of codepoint integers. Each codepoint becomes exactly one list cell; +UTF-8 is prefix-free, so the decode is unambiguous and the cell count +equals the codepoint count. +* *Length preservation.* Combining the two facts: +`+length(String.to_charlist(s)) == length(String.codepoints(s))+` for +every legal UTF-8 binary `+s+`. The harness exercises this directly. + +The property test exercises this over random strings sampled from the +legal codepoint range (excluding surrogates), plus explicit boundary +strings covering all four UTF-8 encoding widths (1/2/3/4 bytes) and the +empty-string corner. + +=== Why this isn’t circular + +The harness does not call `+prim__strToCharList+`. It calls Elixir +`+String.to_charlist/1+` and an explicit `+String.codepoints/1+` count +directly. The argument is: _BEAM charlist conversion decodes a UTF-8 +binary into the same codepoint sequence that Idris `+String.length+` +counts_, so demonstrating those operations agree validates the backend +operation at the semantic level the axiom uses. The trusted-extraction +step is reading the lowering; the property-test step is verifying the +operation behaves as the lowering claims. + +For Chez, we do not run a Scheme harness — R6RS is sufficient +documentary evidence. If BoJ ever ships a backend whose string model is +not Unicode codepoints, this document gets a new section and a matching +property test, *and the axiom may need to be restated* — see the _Honest +framing_ clause in `+docs/backend-assurance/README.md+`. + +=== Edge cases considered + +* *Empty string.* `+String.to_charlist("") == []+`; both lengths are 0. +Tested explicitly. A backend that allocated a sentinel codepoint on +empty input would surface here. +* *Multi-byte codepoints.* Tested via boundary strings covering all four +UTF-8 widths: ASCII (1 byte), Latin-1 supplement (2 bytes, +e.g. `+café+`), CJK (3 bytes, e.g. `+日本語+`), and astral plane (4 +bytes, e.g. `+🦀+`). Each width has a different number of bytes per +codepoint, but the codepoint count (and hence the list length) is +invariant. +* *Surrogates* (`+0xD800..0xDFFF+`): excluded from the codepoint +generator. These are illegal as standalone codepoints in well-formed +Unicode; the harness also asserts that no surrogate ever appears in the +output of `+String.to_charlist/1+` on a legal input, as a sanity check. +* *Normalisation.* Out of scope. The axiom is about codepoint count, not +grapheme-cluster count. `+é+` (`+U+00E9+`) yields a one-element +charlist; `+e+` (`+U+0065+`) + combining acute (`+U+0301+`) yields a +two-element charlist. Both are "`correct`" under the axiom. +* *Round-trip.* Not in the axiom but tested as a sanity check. +`+to_string(String.to_charlist(s)) == s+` for every legal `+s+`. This +guards against a backend whose `+to_charlist+` drops or duplicates +codepoints in a way that happens to preserve length (e.g. replacing a +malformed codepoint with U+FFFD on decode error). +* *Charlist element type.* The harness asserts every codepoint in the +output is a legal integer in `+0..0x10FFFF+` excluding the surrogate +range. Not the axiom but a sanity check that the `+length+` we’re +measuring is over real codepoints. + +=== References + +* Idris2 0.8.0 `+src/Core/Primitives.idr+` — primitive operation table. +* Idris2 0.8.0 `+src/Compiler/Scheme/Chez.idr+` — Chez codegen lowerings +for `+prim__strToCharList+` and `+prim__strLength+`. +* R6RS §6.7, §11.12 — Scheme string model and `+string->list+` +specification. +* Elixir `+String+` module documentation — `+String.to_charlist/1+` and +`+String.codepoints/1+` semantics over UTF-8 binaries. +* `+PROOF-NEEDS.md+` — axiom audit (2026-05-18) and class-(J) framing. +* `+src/abi/Boj/SafetyLemmas.idr+` — axiom declaration (line 218). diff --git a/docs/backend-assurance/prim__strToCharList.md b/docs/backend-assurance/prim__strToCharList.md deleted file mode 100644 index 754bdb78..00000000 --- a/docs/backend-assurance/prim__strToCharList.md +++ /dev/null @@ -1,149 +0,0 @@ - - - -# Backend-Assurance: `prim__strToCharList` - -Trusted-extraction validation for the class-(J) axiom over Idris2's -`prim__strToCharList` primitive: - -- `unpackLength : (s : String) -> length (unpack s) = length s` - (`src/abi/Boj/SafetyLemmas.idr:218`) - -Declared `%unsafe` with `believe_me ()` in Idris2 0.8.0 because -`String` is an opaque primitive type with no constructors and no -in-language induction principle relating its primitive length to the -length of the derived `List Char`. This document argues — by -inspecting the backend lowerings that BoJ actually ships against — -that the length-preservation property holds. - -The companion property test -(`elixir/test/backend_assurance/prim_str_to_char_list_test.exs`) -exercises the BEAM half of this argument over the codepoint space. - -## What `prim__strToCharList` is - -`prim__strToCharList : String -> List Char` is a primitive operation -declared in Idris2's `Core.Primitives`. The `unpack : String -> List -Char` definition in `Data.String` calls it directly: - - public export - unpack : String -> List Char - unpack = prim__strToCharList - -Length is `prim__strLength` on the `String` side and the constructive -`length : List a -> Nat` on the `List Char` side. The axiom asserts -the two counts agree. - -The question reduces to: does `prim__strToCharList` produce a list -whose length equals the codepoint count of the input string, on -each shipping backend? - -## Chez Scheme backend (Idris2 default codegen) - -In `Compiler.Scheme.Chez`, `prim__strToCharList` lowers to the R6RS -procedure `string->list`, and `prim__strLength` lowers to -`string-length`. On Chez 9.x: - -- **String model.** R6RS §6.7 specifies that a Scheme string is a - sequence of Unicode characters (codepoints). -- **`string->list` semantics.** R6RS §11.12 specifies `string->list` - returns a newly-allocated list of the characters that make up the - given string, in the same order. -- **Length preservation.** Combining the two: the length of - `(string->list s)` (counting list cells) equals `(string-length - s)`. This is part of the Scheme standard, not an implementation - detail. - -No further evidence needed beyond citing the standard. - -## BEAM backend (Erlang / Elixir, where BoJ runs) - -BoJ's REST surface is Elixir on the BEAM. The runtime strings are -UTF-8 encoded binaries; the analogue of `unpack` on the BEAM is -`String.to_charlist/1`. - -On BEAM: - -- **String model.** As with the other string primitives in this - campaign, strings are UTF-8 binaries. The `length` referred to in - the axiom is codepoint count, matching Idris2 semantics on Chez. - Elixir's `String.length/1` counts grapheme clusters, so the - BEAM-side harness measures codepoint count explicitly via - `String.codepoints/1`. -- **`String.to_charlist/1` semantics.** Per the Elixir `String` - module documentation, `String.to_charlist/1` decodes a UTF-8 binary - to a list of codepoint integers. Each codepoint becomes exactly one - list cell; UTF-8 is prefix-free, so the decode is unambiguous and - the cell count equals the codepoint count. -- **Length preservation.** Combining the two facts: - `length(String.to_charlist(s)) == length(String.codepoints(s))` - for every legal UTF-8 binary `s`. The harness exercises this - directly. - -The property test exercises this over random strings sampled from the -legal codepoint range (excluding surrogates), plus explicit boundary -strings covering all four UTF-8 encoding widths (1/2/3/4 bytes) and -the empty-string corner. - -## Why this isn't circular - -The harness does not call `prim__strToCharList`. It calls Elixir -`String.to_charlist/1` and an explicit `String.codepoints/1` count -directly. The argument is: *BEAM charlist conversion decodes a UTF-8 -binary into the same codepoint sequence that Idris `String.length` -counts*, so demonstrating those operations agree validates the -backend operation at the semantic level the axiom uses. The -trusted-extraction step is reading the lowering; the property-test -step is verifying the operation behaves as the lowering claims. - -For Chez, we do not run a Scheme harness — R6RS is sufficient -documentary evidence. If BoJ ever ships a backend whose string model -is not Unicode codepoints, this document gets a new section and a -matching property test, **and the axiom may need to be restated** — -see the *Honest framing* clause in -`docs/backend-assurance/README.md`. - -## Edge cases considered - -- **Empty string.** `String.to_charlist("") == []`; both lengths - are 0. Tested explicitly. A backend that allocated a sentinel - codepoint on empty input would surface here. -- **Multi-byte codepoints.** Tested via boundary strings covering - all four UTF-8 widths: ASCII (1 byte), Latin-1 supplement (2 bytes, - e.g. `café`), CJK (3 bytes, e.g. `日本語`), and astral plane - (4 bytes, e.g. `🦀`). Each width has a different number of bytes - per codepoint, but the codepoint count (and hence the list length) - is invariant. -- **Surrogates** (`0xD800..0xDFFF`): excluded from the codepoint - generator. These are illegal as standalone codepoints in - well-formed Unicode; the harness also asserts that no surrogate - ever appears in the output of `String.to_charlist/1` on a legal - input, as a sanity check. -- **Normalisation.** Out of scope. The axiom is about codepoint - count, not grapheme-cluster count. `é` (`U+00E9`) yields a - one-element charlist; `e` (`U+0065`) + combining acute (`U+0301`) - yields a two-element charlist. Both are "correct" under the axiom. -- **Round-trip.** Not in the axiom but tested as a sanity check. - `to_string(String.to_charlist(s)) == s` for every legal `s`. This - guards against a backend whose `to_charlist` drops or duplicates - codepoints in a way that happens to preserve length (e.g. - replacing a malformed codepoint with U+FFFD on decode error). -- **Charlist element type.** The harness asserts every codepoint in - the output is a legal integer in `0..0x10FFFF` excluding the - surrogate range. Not the axiom but a sanity check that the - `length` we're measuring is over real codepoints. - -## References - -- Idris2 0.8.0 `src/Core/Primitives.idr` — primitive operation table. -- Idris2 0.8.0 `src/Compiler/Scheme/Chez.idr` — Chez codegen lowerings - for `prim__strToCharList` and `prim__strLength`. -- R6RS §6.7, §11.12 — Scheme string model and `string->list` - specification. -- Elixir `String` module documentation — `String.to_charlist/1` - and `String.codepoints/1` semantics over UTF-8 binaries. -- `PROOF-NEEDS.md` — axiom audit (2026-05-18) and class-(J) framing. -- `src/abi/Boj/SafetyLemmas.idr` — axiom declaration (line 218). diff --git a/docs/decisions/0001-adopt-rsr-standard.adoc b/docs/decisions/0001-adopt-rsr-standard.adoc new file mode 100644 index 00000000..0dbd05a3 --- /dev/null +++ b/docs/decisions/0001-adopt-rsr-standard.adoc @@ -0,0 +1,94 @@ +== 1. Adopt Rhodium Standard Repository (RSR) Template + +Date: 2026-02-14 + +=== Status + +Accepted + +=== Context + +Managing multiple repositories with an ad-hoc approach led to +significant inconsistencies across the ecosystem. Common problems +included: + +* Missing or incomplete configuration files (SECURITY.md, +CONTRIBUTING.md, .editorconfig, etc.) +* State files (STATE.a2ml, META.a2ml, ECOSYSTEM.a2ml) placed in the +repository root instead of the canonical `+.machine_readable/+` +directory +* Duplicate or conflicting workflow definitions across repos +* No standardized entry point for AI agents interacting with +repositories +* Inconsistent bot directive configurations leading to unreliable +automation +* No contractile enforcement or Justfile automation + +Without a single source of truth for repository structure, each new repo +required manual setup and inevitably drifted from best practices over +time. + +=== Decision + +Adopt the Rhodium Standard Repository (RSR) template +(`+rsr-template-repo+`) as the canonical starting point for all new +repositories. Existing repositories will migrate incrementally as they +receive active development. + +The RSR template provides: + +* *Machine-readable state files* in `+.machine_readable/+` (STATE.a2ml, +ECOSYSTEM.a2ml, META.a2ml, AGENTIC.a2ml, NEUROSYM.a2ml, PLAYBOOK.a2ml) +* *AI manifest* (`+0-AI-MANIFEST.a2ml+`) as a universal entry point for +all AI agents +* *Bot directives* in `+.machine_readable/bot_directives/+` for bot +orchestration integration +* *Contractiles* in `+.machine_readable/contractiles/+` (k9, dust, lust, +must, trust) for policy enforcement +* *Standardized workflows* (16+ GitHub Actions workflows, all +SHA-pinned) +* *Justfile automation* with standard recipes for common tasks +* *Security and governance files*: SECURITY.md, CONTRIBUTING.md, +CODE_OF_CONDUCT.md, LICENSE (MPL-2.0) +* *Architecture Decision Records* in `+docs/decisions/+` + +New repositories are created by cloning the template: + +[source,bash] +---- +git clone https://github.com/hyperpolymath/rsr-template-repo new-repo-name +cd new-repo-name +rm -rf .git && git init +---- + +=== Consequences + +==== Positive + +* Consistency across all repositories, enforced from creation +* Automated compliance checking via `+rsr-antipattern.yml+` workflow +* Bot fleet can operate reliably across all repos with predictable +structure +* AI agents (Claude, Gemini, etc.) have a standardized entry point via +`+0-AI-MANIFEST.a2ml+` +* New contributors can onboard faster with familiar, documented +structure +* Reduced maintenance burden: fix once in template, propagate to all +repos +* Machine-readable state enables tooling and automation pipelines + +==== Negative + +* Migration effort for existing repos requires time and attention +* Learning curve for contributors unfamiliar with RSR conventions +* Template updates need propagation mechanism to existing repos +* Some repos may have unique needs that do not fit the standard template +without customization + +==== Neutral + +* Existing CI/CD pipelines continue to work; RSR workflows are additive +* Third-party dependencies retain their original licenses regardless of +repo structure +* ADR process itself is part of the template, enabling future decisions +to be recorded consistently diff --git a/docs/decisions/0001-adopt-rsr-standard.md b/docs/decisions/0001-adopt-rsr-standard.md deleted file mode 100644 index d6fab39c..00000000 --- a/docs/decisions/0001-adopt-rsr-standard.md +++ /dev/null @@ -1,88 +0,0 @@ - - - -# 1. Adopt Rhodium Standard Repository (RSR) Template - -Date: 2026-02-14 - -## Status - -Accepted - -## Context - -Managing multiple repositories with an ad-hoc approach led to significant -inconsistencies across the ecosystem. Common problems included: - -- Missing or incomplete configuration files (SECURITY.md, CONTRIBUTING.md, - .editorconfig, etc.) -- State files (STATE.a2ml, META.a2ml, ECOSYSTEM.a2ml) placed in the repository - root instead of the canonical `.machine_readable/` directory -- Duplicate or conflicting workflow definitions across repos -- No standardized entry point for AI agents interacting with repositories -- Inconsistent bot directive configurations leading to unreliable automation -- No contractile enforcement or Justfile automation - -Without a single source of truth for repository structure, each new repo -required manual setup and inevitably drifted from best practices over time. - -## Decision - -Adopt the Rhodium Standard Repository (RSR) template (`rsr-template-repo`) as -the canonical starting point for all new repositories. Existing repositories -will migrate incrementally as they receive active development. - -The RSR template provides: - -- **Machine-readable state files** in `.machine_readable/` (STATE.a2ml, - ECOSYSTEM.a2ml, META.a2ml, AGENTIC.a2ml, NEUROSYM.a2ml, PLAYBOOK.a2ml) -- **AI manifest** (`0-AI-MANIFEST.a2ml`) as a universal entry point for all - AI agents -- **Bot directives** in `.machine_readable/bot_directives/` for bot orchestration integration -- **Contractiles** in `.machine_readable/contractiles/` (k9, dust, lust, must, trust) for - policy enforcement -- **Standardized workflows** (16+ GitHub Actions workflows, all SHA-pinned) -- **Justfile automation** with standard recipes for common tasks -- **Security and governance files**: SECURITY.md, CONTRIBUTING.md, - CODE_OF_CONDUCT.md, LICENSE (MPL-2.0) -- **Architecture Decision Records** in `docs/decisions/` - -New repositories are created by cloning the template: - -```bash -git clone https://github.com/hyperpolymath/rsr-template-repo new-repo-name -cd new-repo-name -rm -rf .git && git init -``` - -## Consequences - -### Positive - -- Consistency across all repositories, enforced from creation -- Automated compliance checking via `rsr-antipattern.yml` workflow -- Bot fleet can operate reliably across all repos with predictable structure -- AI agents (Claude, Gemini, etc.) have a standardized entry point via - `0-AI-MANIFEST.a2ml` -- New contributors can onboard faster with familiar, documented structure -- Reduced maintenance burden: fix once in template, propagate to all repos -- Machine-readable state enables tooling and automation pipelines - -### Negative - -- Migration effort for existing repos requires time and attention -- Learning curve for contributors unfamiliar with RSR conventions -- Template updates need propagation mechanism to existing repos -- Some repos may have unique needs that do not fit the standard template - without customization - -### Neutral - -- Existing CI/CD pipelines continue to work; RSR workflows are additive -- Third-party dependencies retain their original licenses regardless of - repo structure -- ADR process itself is part of the template, enabling future decisions - to be recorded consistently diff --git a/docs/decisions/0002-align-unified-zig-api-stack.adoc b/docs/decisions/0002-align-unified-zig-api-stack.adoc new file mode 100644 index 00000000..1122cfe0 --- /dev/null +++ b/docs/decisions/0002-align-unified-zig-api-stack.adoc @@ -0,0 +1,127 @@ +== 2. Align BoJ with the Unified-Zig-API Stack + +Date: 2026-04-17 + +=== Status + +Accepted (alignment in code is future work — see Open Questions) + +=== Context + +Two estate-wide changes on and around 2026-04-10 make BoJ’s current spec +stale: + +*1. zig banned estate-wide (2026-04-10)* + +zig was the adapter layer language +(`+zig → Triple API REST+gRPC+GraphQL+`). It was banned estate-wide +because `+v.mod+` / `+vpkg.json+` files indicated zig dependency, and +zig has no path to formal verification or Idris2-ABI compatibility. The +zig sweep in BoJ (commits c4674f8/cb00882/325b42e/9186016) removed all +`+.v+` adapter files; all 90 cartridges now have Zig three-protocol +adapters. Historical zig API interfaces were moved to +`+developer-ecosystem/v-ecosystem/v-api-interfaces/+` for potential +donation to the V community — they are not HP infrastructure. + +*2. Unified-zig-api stack established* + +`+developer-ecosystem/UNIFIED-ZIG-API-STACK.adoc+` is the authoritative +reference for all Zig-edge service boundaries. The stack provides four +vertically-stacked layers: + +[width="100%",cols="42%,58%",options="header",] +|=== +|Layer |Location +|Idris2 core (proven) |`+verification-ecosystem/proven/+` + +|Idris2 ABI |`+developer-ecosystem/zig-api/src/ZigApi/ABI/+` + +|Zig runtime (`+uapi_*+`) |`+developer-ecosystem/zig-api/ffi/zig/src/+` + +|C adaptor (auto-generated) +|`+developer-ecosystem/zig-api/generated/abi/zig_api.h+` +|=== + +First estate consumers wired 2026-04-17: - lol-gateway retrofit: commits +dbb475f / 26b6b8c - aerie retrofit: commit e0b17f8 - +emergency-button/emergency-room path-safety retrofit: commit 4bd070b - +proven→zig-api path-safety wiring actualised: commit 6663956 - +gen-header + CI drift check: commit 0d6a814 + +BoJ is architecturally consistent with this stack (Idris2 ABI + Zig FFI) +but does not yet call `+uapi_*+` symbols or link `+libzig_api.so+`. + +*3. Cartridge manifest format = Nickel* + +The prior closed decision (`+boj-cartridge-manifest-format-dd.md+`) +established Nickel as the authoritative cartridge manifest format. +Current on-disk manifests are `+cartridge.json+`; migration to Nickel is +deferred. + +*4. BoJ-only MCP (standing estate policy)* + +All MCP access to hyperpolymath services routes through BoJ. No +standalone MCPs are permitted for capabilities already covered by BoJ +cartridges. + +=== Decision + +[arabic] +. *Document zig retirement* in all BoJ spec files. The three-layer stack +is now `+Idris2 (ABI) → Zig (FFI) → Zig (Adapter)+`. No zig references +remain in spec documents. +. *Cite `+UNIFIED-ZIG-API-STACK.adoc+`* as the authoritative reference +for the zig-api consumption pattern. BoJ is aligned at the design level. +Full code-level alignment (linking `+libzig_api.so+`, calling +`+uapi_gnosis_*+`) is deferred to a future session. +. *Document the Nickel manifest intent* in `+docs/EXTENSIBILITY.adoc+` +without removing the operative JSON schema. The JSON schema at +`+https://boj.dev/schemas/cartridge/v1.json+` remains valid until Nickel +migration is complete. +. *Add BoJ-only MCP policy statement* to `+docs/FEDERATION.adoc+`. +. *Do not claim BoJ consumes `+libzig_api+`* — it does not yet. Spec +states current reality and future intent explicitly. + +=== Consequences + +==== Positive + +* Spec accurately describes actual language stack (Zig everywhere, not +zig). +* BoJ is positioned to consume `+libzig_api+` when the session arrives — +no architectural changes required, only build-system wiring. +* The `+uapi_gnosis_set_handler+` single-port model (arriving in +zig-api) will let BoJ consolidate its three-port adapter +(7700/7701/7702) when adopted. +* Nickel manifest migration is unblocked: format intent is documented, +no silent incompatibility. + +==== Negative + +* Code-level alignment with `+libzig_api+` is still outstanding. Any +developer reading the spec sees the intent but must check the actual +`+ffi/zig/+` source to confirm the current wiring. +* Port consolidation (single-port gnosis vs three-port per-protocol) is +a future breaking change to the API-CONTRACT; will require a major +version bump. + +==== Neutral + +* Existing CI, tests, and build scripts are unaffected by this spec-only +bump. +* Third-party cartridge authors using `+cartridge.json+` today are +unaffected until the Nickel migration session. + +=== Open Questions + +[arabic] +. *`+libzig_api+` wiring session*: When does BoJ link `+libzig_api.so+` +and replace its local Zig adapters with `+uapi_gnosis_*+` calls? No +scheduled date yet; depends on `+uapi_gnosis_set_handler+` stabilising +upstream. +. *Nickel manifest migration*: When does `+cartridge.json+` become +`+cartridge.ncl+`? Requires a Nickel schema and migration tooling. No +scheduled date. +. *Port consolidation*: The three-port API surface (7700/7701/7702) is a +stable contract. Consolidating to a single gnosis port requires a major +version bump and coordinated migration of all consumers. diff --git a/docs/decisions/0002-align-unified-zig-api-stack.md b/docs/decisions/0002-align-unified-zig-api-stack.md deleted file mode 100644 index 5548248e..00000000 --- a/docs/decisions/0002-align-unified-zig-api-stack.md +++ /dev/null @@ -1,123 +0,0 @@ - - - -# 2. Align BoJ with the Unified-Zig-API Stack - -Date: 2026-04-17 - -## Status - -Accepted (alignment in code is future work — see Open Questions) - -## Context - -Two estate-wide changes on and around 2026-04-10 make BoJ's current spec stale: - -**1. zig banned estate-wide (2026-04-10)** - -zig was the adapter layer language (`zig → Triple API REST+gRPC+GraphQL`). -It was banned estate-wide because `v.mod` / `vpkg.json` files indicated zig -dependency, and zig has no path to formal verification or Idris2-ABI -compatibility. The zig sweep in BoJ (commits c4674f8/cb00882/325b42e/9186016) -removed all `.v` adapter files; all 90 cartridges now have Zig three-protocol -adapters. Historical zig API interfaces were moved to -`developer-ecosystem/v-ecosystem/v-api-interfaces/` for potential donation to the -V community — they are not HP infrastructure. - -**2. Unified-zig-api stack established** - -`developer-ecosystem/UNIFIED-ZIG-API-STACK.adoc` is the authoritative reference -for all Zig-edge service boundaries. The stack provides four vertically-stacked -layers: - -| Layer | Location | -|-------|----------| -| Idris2 core (proven) | `verification-ecosystem/proven/` | -| Idris2 ABI | `developer-ecosystem/zig-api/src/ZigApi/ABI/` | -| Zig runtime (`uapi_*`) | `developer-ecosystem/zig-api/ffi/zig/src/` | -| C adaptor (auto-generated) | `developer-ecosystem/zig-api/generated/abi/zig_api.h` | - -First estate consumers wired 2026-04-17: -- lol-gateway retrofit: commits dbb475f / 26b6b8c -- aerie retrofit: commit e0b17f8 -- emergency-button/emergency-room path-safety retrofit: commit 4bd070b -- proven→zig-api path-safety wiring actualised: commit 6663956 -- gen-header + CI drift check: commit 0d6a814 - -BoJ is architecturally consistent with this stack (Idris2 ABI + Zig FFI) but -does not yet call `uapi_*` symbols or link `libzig_api.so`. - -**3. Cartridge manifest format = Nickel** - -The prior closed decision (`boj-cartridge-manifest-format-dd.md`) established -Nickel as the authoritative cartridge manifest format. Current on-disk manifests -are `cartridge.json`; migration to Nickel is deferred. - -**4. BoJ-only MCP (standing estate policy)** - -All MCP access to hyperpolymath services routes through BoJ. No standalone MCPs -are permitted for capabilities already covered by BoJ cartridges. - -## Decision - -1. **Document zig retirement** in all BoJ spec files. The three-layer stack - is now `Idris2 (ABI) → Zig (FFI) → Zig (Adapter)`. No zig references - remain in spec documents. - -2. **Cite `UNIFIED-ZIG-API-STACK.adoc`** as the authoritative reference for - the zig-api consumption pattern. BoJ is aligned at the design level. - Full code-level alignment (linking `libzig_api.so`, calling `uapi_gnosis_*`) - is deferred to a future session. - -3. **Document the Nickel manifest intent** in `docs/EXTENSIBILITY.adoc` without - removing the operative JSON schema. The JSON schema at - `https://boj.dev/schemas/cartridge/v1.json` remains valid until Nickel - migration is complete. - -4. **Add BoJ-only MCP policy statement** to `docs/FEDERATION.adoc`. - -5. **Do not claim BoJ consumes `libzig_api`** — it does not yet. Spec states - current reality and future intent explicitly. - -## Consequences - -### Positive - -- Spec accurately describes actual language stack (Zig everywhere, not zig). -- BoJ is positioned to consume `libzig_api` when the session arrives — no - architectural changes required, only build-system wiring. -- The `uapi_gnosis_set_handler` single-port model (arriving in zig-api) will - let BoJ consolidate its three-port adapter (7700/7701/7702) when adopted. -- Nickel manifest migration is unblocked: format intent is documented, no - silent incompatibility. - -### Negative - -- Code-level alignment with `libzig_api` is still outstanding. Any developer - reading the spec sees the intent but must check the actual `ffi/zig/` source - to confirm the current wiring. -- Port consolidation (single-port gnosis vs three-port per-protocol) is a - future breaking change to the API-CONTRACT; will require a major version bump. - -### Neutral - -- Existing CI, tests, and build scripts are unaffected by this spec-only bump. -- Third-party cartridge authors using `cartridge.json` today are unaffected - until the Nickel migration session. - -## Open Questions - -1. **`libzig_api` wiring session**: When does BoJ link `libzig_api.so` and - replace its local Zig adapters with `uapi_gnosis_*` calls? No scheduled - date yet; depends on `uapi_gnosis_set_handler` stabilising upstream. - -2. **Nickel manifest migration**: When does `cartridge.json` become - `cartridge.ncl`? Requires a Nickel schema and migration tooling. No - scheduled date. - -3. **Port consolidation**: The three-port API surface (7700/7701/7702) is a - stable contract. Consolidating to a single gnosis port requires a major - version bump and coordinated migration of all consumers. diff --git a/docs/decisions/0003-extract-cartridge-spec-standalone.adoc b/docs/decisions/0003-extract-cartridge-spec-standalone.adoc new file mode 100644 index 00000000..98c73793 --- /dev/null +++ b/docs/decisions/0003-extract-cartridge-spec-standalone.adoc @@ -0,0 +1,110 @@ +== 3. Extract Cartridge Specification as Standalone Normative Document + +Date: 2026-04-17 + +=== Status + +Accepted + +=== Context + +Prior to 2026-04-17, the normative definition of a BoJ cartridge was +distributed across three locations with no single authoritative source: + +[arabic] +. *`+src/abi/Boj/Catalogue.idr+`* — the formal 2D capability matrix, +cartridge record type, `+IsUnbreakable+` predicate, `+CartridgeStatus+` +lifecycle, `+MenuTier+` tiers, hash attestation requirement, and +catalogue query functions. Authoritative for the machine, but not +human-readable as an overview document. +. *`+docs/papers/boj-architecture-paper.md §6+`* — the HAT (Hardware +Attached on Top) third dimension was described here in prose, as part of +a broader architecture narrative. Not normative on its own; embedded +inside a longer paper. +. *Implicit in `+cartridge-minter/+` expectations* — the cartridge-tools +suite (`+mint+`, `+provision+`, `+config+`, `+harness+`) encoded +assumptions about cartridge structure without a single document to +reference back to. This made it hard to tell what was spec +vs. implementation detail. + +There was no standalone document covering the full cartridge concept: +the 2D matrix, the HAT third dimension, the Nickel manifest shape, the +three-axis surface ephemerality model, transaction-based ephemerality, +transport preference ordering, security grades, transport provenance, +and the reference implementation pattern. + +=== Decision + +Extract the cartridge specification into +`+docs/specification/cartridges/README.md+` as the *normative prose +specification* for BoJ cartridges. This document: + +* Covers the ProtocolType axis (9 protocols), the CapabilityDomain axis +(18 domains), and the 2D sparse capability matrix. +* Defines the HAT third dimension with the four bridge types (CLI +wrapper, JSON-RPC stdio, HTTP API, Library FFI) and the circuit-breaker +isolation model. +* Specifies the Nickel cartridge manifest schema with required fields, +`+TransportEntry+` shape, and a worked example. +* Defines the three-axis surface ephemerality model +(`+possible_transports+` / `+preferred_transports+` / +`+active_transports+`) and transaction-based ephemerality. +* Specifies transport preference ordering, the four-tier security grade +ladder (A/B/C/D), and transport provenance conventions. +* Documents the reference implementation pattern (Idris2 → Zig → Rust +Tauri command layer → bridge directory) using the IDApTIK UMS cartridge +as the canonical example. + +The Idris2 source files (`+src/abi/Boj/Catalogue.idr+`, +`+src/abi/Boj/Protocol.idr+`, `+src/abi/Boj/Domain.idr+`) remain the +*machine-authoritative* definition. If the prose spec and the Idris2 +source ever disagree, the Idris2 source wins. + +=== Consequences + +==== Positive + +* Single source of truth for all cartridge-related questions. Onboarding +a new cartridge author now has a clear starting point. +* The cartridge-tools suite +(`+docs/specification/cartridge-tools/README.md+`) can reference the +cartridge spec as its upstream normative document, without embedding its +own copy of cartridge definitions. +* Sets the normative shape for the planned Nickel manifest migration +(ADR 0002 open question #2). Cartridge authors can start writing +`+.ncl+` manifests against a documented schema even before migration +tooling exists. +* The `+[CARTRIDGE_SPEC_REF]+` section of the BoJ Trustfile now has a +concrete normative target to point at. + +==== Negative + +* `+docs/papers/boj-architecture-paper.md §6+` (HAT description) is now +slightly redundant. It has been annotated with a forward-reference to +`+docs/specification/cartridges/README.md+` as the normative source; the +paper retains its narrative value for new readers but is no longer the +primary HAT definition. +* The cartridge spec must be kept in sync with +`+src/abi/Boj/Catalogue.idr+` as the Idris2 code evolves. Recommended +enforcement: ECHIDNA diff-tooling or a CI step that cross-checks enum +cardinalities between the Idris2 source and the prose spec. + +==== Neutral + +* Existing `+cartridge.json+` files are unaffected. The spec records +both the Nickel (new) and JSON (legacy) manifest formats. Migration +schedule unchanged (see ADR 0002 open question #2). +* The `+IsUnbreakable+` proof and its circuit-breaker semantics are +described in the spec but remain enforced by the Zig FFI layer; no code +change is implied. + +=== Related + +* ADR 0001 (RSR adoption) — establishes `+docs/decisions/+` as the home +for ADRs. +* ADR 0002 (unified-zig-api alignment) — Nickel manifest migration open +question. +* Commit `+b1b40f7+` — extraction of the cartridge spec into this +document. +* Commit `+ceae54c+` — BoJ Trustfile speciation with +`+[CARTRIDGE_SPEC_REF]+` section. diff --git a/docs/decisions/0003-extract-cartridge-spec-standalone.md b/docs/decisions/0003-extract-cartridge-spec-standalone.md deleted file mode 100644 index 2f66a1e0..00000000 --- a/docs/decisions/0003-extract-cartridge-spec-standalone.md +++ /dev/null @@ -1,101 +0,0 @@ - - - -# 3. Extract Cartridge Specification as Standalone Normative Document - -Date: 2026-04-17 - -## Status - -Accepted - -## Context - -Prior to 2026-04-17, the normative definition of a BoJ cartridge was distributed -across three locations with no single authoritative source: - -1. **`src/abi/Boj/Catalogue.idr`** — the formal 2D capability matrix, cartridge - record type, `IsUnbreakable` predicate, `CartridgeStatus` lifecycle, `MenuTier` - tiers, hash attestation requirement, and catalogue query functions. Authoritative - for the machine, but not human-readable as an overview document. - -2. **`docs/papers/boj-architecture-paper.md §6`** — the HAT (Hardware Attached on - Top) third dimension was described here in prose, as part of a broader architecture - narrative. Not normative on its own; embedded inside a longer paper. - -3. **Implicit in `cartridge-minter/` expectations** — the cartridge-tools suite - (`mint`, `provision`, `config`, `harness`) encoded assumptions about cartridge - structure without a single document to reference back to. This made it hard to - tell what was spec vs. implementation detail. - -There was no standalone document covering the full cartridge concept: the 2D matrix, -the HAT third dimension, the Nickel manifest shape, the three-axis surface -ephemerality model, transaction-based ephemerality, transport preference ordering, -security grades, transport provenance, and the reference implementation pattern. - -## Decision - -Extract the cartridge specification into `docs/specification/cartridges/README.md` -as the **normative prose specification** for BoJ cartridges. This document: - -- Covers the ProtocolType axis (9 protocols), the CapabilityDomain axis (18 domains), - and the 2D sparse capability matrix. -- Defines the HAT third dimension with the four bridge types (CLI wrapper, JSON-RPC - stdio, HTTP API, Library FFI) and the circuit-breaker isolation model. -- Specifies the Nickel cartridge manifest schema with required fields, `TransportEntry` - shape, and a worked example. -- Defines the three-axis surface ephemerality model - (`possible_transports` / `preferred_transports` / `active_transports`) and - transaction-based ephemerality. -- Specifies transport preference ordering, the four-tier security grade ladder - (A/B/C/D), and transport provenance conventions. -- Documents the reference implementation pattern (Idris2 → Zig → Rust Tauri command - layer → bridge directory) using the IDApTIK UMS cartridge as the canonical example. - -The Idris2 source files (`src/abi/Boj/Catalogue.idr`, `src/abi/Boj/Protocol.idr`, -`src/abi/Boj/Domain.idr`) remain the **machine-authoritative** definition. If the -prose spec and the Idris2 source ever disagree, the Idris2 source wins. - -## Consequences - -### Positive - -- Single source of truth for all cartridge-related questions. Onboarding a new - cartridge author now has a clear starting point. -- The cartridge-tools suite (`docs/specification/cartridge-tools/README.md`) can - reference the cartridge spec as its upstream normative document, without embedding - its own copy of cartridge definitions. -- Sets the normative shape for the planned Nickel manifest migration (ADR 0002 open - question #2). Cartridge authors can start writing `.ncl` manifests against a - documented schema even before migration tooling exists. -- The `[CARTRIDGE_SPEC_REF]` section of the BoJ Trustfile now has a concrete - normative target to point at. - -### Negative - -- `docs/papers/boj-architecture-paper.md §6` (HAT description) is now slightly - redundant. It has been annotated with a forward-reference to - `docs/specification/cartridges/README.md` as the normative source; the paper - retains its narrative value for new readers but is no longer the primary HAT - definition. -- The cartridge spec must be kept in sync with `src/abi/Boj/Catalogue.idr` as the - Idris2 code evolves. Recommended enforcement: ECHIDNA diff-tooling or a CI step - that cross-checks enum cardinalities between the Idris2 source and the prose spec. - -### Neutral - -- Existing `cartridge.json` files are unaffected. The spec records both the Nickel - (new) and JSON (legacy) manifest formats. Migration schedule unchanged (see ADR 0002 - open question #2). -- The `IsUnbreakable` proof and its circuit-breaker semantics are described in the - spec but remain enforced by the Zig FFI layer; no code change is implied. - -## Related - -- ADR 0001 (RSR adoption) — establishes `docs/decisions/` as the home for ADRs. -- ADR 0002 (unified-zig-api alignment) — Nickel manifest migration open question. -- Commit `b1b40f7` — extraction of the cartridge spec into this document. -- Commit `ceae54c` — BoJ Trustfile speciation with `[CARTRIDGE_SPEC_REF]` section. diff --git a/docs/decisions/0004-adopt-http-capability-gateway.adoc b/docs/decisions/0004-adopt-http-capability-gateway.adoc new file mode 100644 index 00000000..0d5a58d5 --- /dev/null +++ b/docs/decisions/0004-adopt-http-capability-gateway.adoc @@ -0,0 +1,184 @@ +== 4. Adopt http-capability-gateway as BoJ Tier-2 HTTP Governance Layer + +Date: 2026-04-17 + +=== Status + +Accepted + +=== Context + +BoJ’s HTTP surface today terminates at the unified Zig API gnosis +handler (`+uapi_gnosis_set_handler+`), introduced in the single-port +consolidation (commits `+9c807c0+`, `+d765345+`). This means: + +[arabic] +. *Verb governance is absent at the HTTP layer.* There is no single +location declaring which HTTP verbs are permitted on which paths. Any +cartridge that handles a route implicitly accepts all verbs Cowboy +routes to it. DELETE, PUT, PATCH, OPTIONS, and HEAD can be accessible +without explicit intent. +. *Rate-limiting logic is fragmented.* The four-tier rate-limit +architecture documented in +`+Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY].rate_limiting+` has a +declared gap at tier 2 +(`+status: "PENDING — http-capability-gateway wiring forthcoming"+`). +Tiers 1 (Cloudflare edge) and 3–4 (BEAM supervisor, cartridge manifest) +are in place; tier 2 is empty. +. *Capability-enforcement logic is either in the gnosis handler (too +low-level) or distributed across cartridges (too distributed, no central +policy).* There is no central, auditable declaration of what the HTTP +surface exposes at what trust level. +. *Trust-level derivation has no primary path.* The Trustfile’s +`+[SDP_RULES]+` and `+[ORIGIN_PROTECTION]+` sections imply mutual TLS +and trust-level-aware routing, but no component in the current stack +derives, validates, or forwards a trust level on incoming HTTP requests +before they reach cartridge logic. + +The estate has independently built +`+hyperpolymath/http-capability-gateway+` as a general HTTP governance +layer. An audit conducted 2026-04-17 (see +`+docs/integration/http-capability-gateway-audit.md+`) confirmed: + +* The core policy pipeline (loader → validator → compiler → ETS +enforcement → proxy → telemetry) is implemented and working. +* DSL v1 (Verb Governance Spec) is defined and validated. +* The gateway is Elixir/Cowboy/Plug — architecturally compatible with +BoJ’s BEAM stack. +* Stated gaps (mTLS as primary path, E2E verification, benchmark +evidence) are real and must be closed before production deployment. +Estimated 8–12 weeks of focused work. +* Tier 2 placement (between Cloudflare edge and BoJ’s gnosis handler) is +architecturally sound. No conflict with Svalinn (container gateway) or +the BEAM supervisor tier. + +=== Decision + +Adopt `+http-capability-gateway+` as *tier 2 of BoJ’s rate-limit and +capability-enforcement architecture* (filling the `+PENDING+` gap in +`+Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY] .rate_limiting.tier_2_gateway+`). + +The integration is structured in five phases: + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Phase |Scope |Weeks +|A |Contract definition, policy authoring workflow, example Verb +Governance Spec |1–2 + +|B |mTLS as primary trust-level path, Cowboy TLS config, real-CA test +fixture |3–5 + +|C |E2E verification tests, seam test across gateway ↔ gnosis handler +boundary |5–7 + +|D |Benchmark formalisation, latency numbers published, CI regression +alert |7–8 + +|E |Production wiring, staging validation, rollout, rollback runbook +|8–12 +|=== + +Phase 0 (today, 2026-04-17) is the audit and planning session. No code +changes to the gateway or to BoJ’s HTTP surface are made in Phase 0. + +Full phase detail in +`+docs/integration/http-capability-gateway-plan.md+`. + +=== Consequences + +==== Positive + +* *Declarative HTTP governance.* All permitted verbs, their trust +requirements, and their narrative rationale are declared in a single +version-controlled YAML file (the Verb Governance Spec). Any change to +BoJ’s HTTP surface must be reflected there. +* *Verb-level enforcement before cartridges.* The gateway enforces verb +governance at the HTTP layer before any cartridge logic runs. Cartridges +receive only requests that have already passed policy evaluation. +Cartridges no longer need to defend against unexpected HTTP verbs. +* *Trust-level derivation centralised.* After Phase B, the gateway +derives trust level from mTLS client certificates and forwards it as +`+X-Trust-Level+` to the gnosis handler. Cartridges consume a +pre-validated trust level; they do not need to extract or validate it +themselves. +* *Fills tier-2 gap in rate-limit architecture.* The per-IP token-bucket +rate limiter (10/s untrusted, 100/s authenticated, unlimited internal) +closes the gap between Cloudflare edge rate limits (tier 1) and BEAM +supervisor back-pressure (tier 3). +* *Audit trail.* Every access decision (allow/deny, path, verb, trust +level, rule name, duration) is logged in structured JSON and optionally +persisted to VeriSimDB. This provides an auditable record of all HTTP +surface access. +* *Decouples edge from governance.* Cloudflare (tier 1) and the gateway +(tier 2) are independent components that can evolve separately. +Cloudflare handles volumetric DDoS; the gateway handles verb governance +and trust-level enforcement. If Cloudflare is bypassed (e.g., direct +access to origin via Fly.io), the gateway still enforces policy. +* *Stealth mode.* Routes with `+exposure: "internal"+` can return 404 +(or any configurable status) instead of 403, hiding capability existence +from unprivileged callers. + +==== Negative + +* *One more component in the stack.* The gateway adds operational +complexity: a new container to deploy, configure, monitor, and update. +Containerfile, k9-svc deployment spec, cert rotation runbook, and +rollback runbook must all be maintained. +* *Gateway is Elixir — adds a second BEAM application.* BoJ is +multi-language (Idris2 ABI, Zig FFI, Elixir runtime), but the gateway is +a separate Elixir application that must be supervised and monitored +independently. The BEAM handles this well, but operational burden +increases. +* *Compile-then-load policy model requires hot-reload to be +first-class.* The atomic swap pattern is implemented, but any BoJ HTTP +surface change must be reflected in the policy file and reloaded. If the +policy file lags the actual surface, routes may be default-denied. Phase +A must define the policy authoring workflow clearly. +* *Latency overhead.* The gateway adds a hop in the request path. +Benchmark evidence (Phase D) will quantify this. Initial estimate based +on ETS O(1) lookup + Req HTTP proxy: < 2ms median overhead for a +100-rule policy. This must be confirmed. + +==== Risks + +* *mTLS gaps must close before production.* The mTLS trust-extraction +path in the gateway is coded but not the primary proved path (see audit +§4). Header-based trust is forgeable without mTLS enforcement. +Production deployment (Phase E) must wait for Phase B completion. +* *Single-backend proxy.* The gateway’s proxy module supports one +`+backend_url+`. If BoJ is horizontally scaled, the gateway must sit +behind a load balancer, or the proxy module must be extended. This is +noted as post-Phase-E work. +* *VeriSimDB integration unconfirmed.* The audit could not confirm +whether the gateway’s `+VeriSimDB+` module is a real integration or a +thin stub. The audit trail depends on this. Phase E should not be +completed until VeriSimDB status is confirmed. + +==== Tier placement note + +The audit confirms that tier 2 (between Cloudflare edge and BoJ’s gnosis +handler) is the correct placement. An alternative would be to place the +gateway alongside tier 3 (BEAM supervisor), where it would run +co-located with the BoJ Elixir process rather than in front of the +gnosis handler. This would be wrong: the gateway’s role is to govern the +HTTP surface before it reaches any BoJ component, not to provide another +enforcement layer inside BoJ. Tier 2 placement is confirmed. + +=== Related + +* ADR 0002 (`+docs/decisions/0002-align-unified-zig-api-stack.md+`) — +unified Zig API stack (the gnosis handler the gateway sits in front of). +* ADR 0003 +(`+docs/decisions/0003-extract-cartridge-spec-standalone.md+`) — +cartridge specification (verb governance per cartridge is the downstream +beneficiary of this ADR). +* Trustfile `+[HTTP_CAPABILITY_GATEWAY]+` section (commit `+ceae54c+`) — +forward-reference entry updated in this session to +`+status: "ACCEPTED-PLANNED"+`. +* Trustfile `+[CLOUDFLARE_EDGE_SECURITY].rate_limiting.tier_2_gateway+` +— the gap this ADR closes. +* `+docs/integration/http-capability-gateway-audit.md+` — audit +conducted 2026-04-17. +* `+docs/integration/http-capability-gateway-plan.md+` — phased +integration plan. diff --git a/docs/decisions/0004-adopt-http-capability-gateway.md b/docs/decisions/0004-adopt-http-capability-gateway.md deleted file mode 100644 index 9b6c6f70..00000000 --- a/docs/decisions/0004-adopt-http-capability-gateway.md +++ /dev/null @@ -1,167 +0,0 @@ - - - -# 4. Adopt http-capability-gateway as BoJ Tier-2 HTTP Governance Layer - -Date: 2026-04-17 - -## Status - -Accepted - -## Context - -BoJ's HTTP surface today terminates at the unified Zig API gnosis handler -(`uapi_gnosis_set_handler`), introduced in the single-port consolidation -(commits `9c807c0`, `d765345`). This means: - -1. **Verb governance is absent at the HTTP layer.** There is no single location - declaring which HTTP verbs are permitted on which paths. Any cartridge that - handles a route implicitly accepts all verbs Cowboy routes to it. DELETE, PUT, - PATCH, OPTIONS, and HEAD can be accessible without explicit intent. - -2. **Rate-limiting logic is fragmented.** The four-tier rate-limit architecture - documented in `Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY].rate_limiting` has a - declared gap at tier 2 (`status: "PENDING — http-capability-gateway wiring - forthcoming"`). Tiers 1 (Cloudflare edge) and 3–4 (BEAM supervisor, cartridge - manifest) are in place; tier 2 is empty. - -3. **Capability-enforcement logic is either in the gnosis handler (too low-level) - or distributed across cartridges (too distributed, no central policy).** There is - no central, auditable declaration of what the HTTP surface exposes at what trust level. - -4. **Trust-level derivation has no primary path.** The Trustfile's `[SDP_RULES]` - and `[ORIGIN_PROTECTION]` sections imply mutual TLS and trust-level-aware routing, - but no component in the current stack derives, validates, or forwards a trust level - on incoming HTTP requests before they reach cartridge logic. - -The estate has independently built `hyperpolymath/http-capability-gateway` as a -general HTTP governance layer. An audit conducted 2026-04-17 (see -`docs/integration/http-capability-gateway-audit.md`) confirmed: - -- The core policy pipeline (loader → validator → compiler → ETS enforcement → proxy → - telemetry) is implemented and working. -- DSL v1 (Verb Governance Spec) is defined and validated. -- The gateway is Elixir/Cowboy/Plug — architecturally compatible with BoJ's BEAM stack. -- Stated gaps (mTLS as primary path, E2E verification, benchmark evidence) are real - and must be closed before production deployment. Estimated 8–12 weeks of focused work. -- Tier 2 placement (between Cloudflare edge and BoJ's gnosis handler) is - architecturally sound. No conflict with Svalinn (container gateway) or the BEAM - supervisor tier. - -## Decision - -Adopt `http-capability-gateway` as **tier 2 of BoJ's rate-limit and capability-enforcement -architecture** (filling the `PENDING` gap in `Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY] -.rate_limiting.tier_2_gateway`). - -The integration is structured in five phases: - -| Phase | Scope | Weeks | -|---|---|---| -| A | Contract definition, policy authoring workflow, example Verb Governance Spec | 1–2 | -| B | mTLS as primary trust-level path, Cowboy TLS config, real-CA test fixture | 3–5 | -| C | E2E verification tests, seam test across gateway ↔ gnosis handler boundary | 5–7 | -| D | Benchmark formalisation, latency numbers published, CI regression alert | 7–8 | -| E | Production wiring, staging validation, rollout, rollback runbook | 8–12 | - -Phase 0 (today, 2026-04-17) is the audit and planning session. No code changes to -the gateway or to BoJ's HTTP surface are made in Phase 0. - -Full phase detail in `docs/integration/http-capability-gateway-plan.md`. - -## Consequences - -### Positive - -- **Declarative HTTP governance.** All permitted verbs, their trust requirements, and - their narrative rationale are declared in a single version-controlled YAML file - (the Verb Governance Spec). Any change to BoJ's HTTP surface must be reflected there. - -- **Verb-level enforcement before cartridges.** The gateway enforces verb governance - at the HTTP layer before any cartridge logic runs. Cartridges receive only requests - that have already passed policy evaluation. Cartridges no longer need to defend - against unexpected HTTP verbs. - -- **Trust-level derivation centralised.** After Phase B, the gateway derives trust - level from mTLS client certificates and forwards it as `X-Trust-Level` to the - gnosis handler. Cartridges consume a pre-validated trust level; they do not need - to extract or validate it themselves. - -- **Fills tier-2 gap in rate-limit architecture.** The per-IP token-bucket rate - limiter (10/s untrusted, 100/s authenticated, unlimited internal) closes the - gap between Cloudflare edge rate limits (tier 1) and BEAM supervisor back-pressure - (tier 3). - -- **Audit trail.** Every access decision (allow/deny, path, verb, trust level, rule - name, duration) is logged in structured JSON and optionally persisted to VeriSimDB. - This provides an auditable record of all HTTP surface access. - -- **Decouples edge from governance.** Cloudflare (tier 1) and the gateway (tier 2) - are independent components that can evolve separately. Cloudflare handles volumetric - DDoS; the gateway handles verb governance and trust-level enforcement. If Cloudflare - is bypassed (e.g., direct access to origin via Fly.io), the gateway still enforces - policy. - -- **Stealth mode.** Routes with `exposure: "internal"` can return 404 (or any - configurable status) instead of 403, hiding capability existence from unprivileged - callers. - -### Negative - -- **One more component in the stack.** The gateway adds operational complexity: - a new container to deploy, configure, monitor, and update. Containerfile, k9-svc - deployment spec, cert rotation runbook, and rollback runbook must all be maintained. - -- **Gateway is Elixir — adds a second BEAM application.** BoJ is multi-language - (Idris2 ABI, Zig FFI, Elixir runtime), but the gateway is a separate Elixir - application that must be supervised and monitored independently. The BEAM handles - this well, but operational burden increases. - -- **Compile-then-load policy model requires hot-reload to be first-class.** The atomic - swap pattern is implemented, but any BoJ HTTP surface change must be reflected in - the policy file and reloaded. If the policy file lags the actual surface, routes - may be default-denied. Phase A must define the policy authoring workflow clearly. - -- **Latency overhead.** The gateway adds a hop in the request path. Benchmark evidence - (Phase D) will quantify this. Initial estimate based on ETS O(1) lookup + Req HTTP - proxy: < 2ms median overhead for a 100-rule policy. This must be confirmed. - -### Risks - -- **mTLS gaps must close before production.** The mTLS trust-extraction path in the - gateway is coded but not the primary proved path (see audit §4). Header-based trust - is forgeable without mTLS enforcement. Production deployment (Phase E) must wait - for Phase B completion. - -- **Single-backend proxy.** The gateway's proxy module supports one `backend_url`. - If BoJ is horizontally scaled, the gateway must sit behind a load balancer, or the - proxy module must be extended. This is noted as post-Phase-E work. - -- **VeriSimDB integration unconfirmed.** The audit could not confirm whether the - gateway's `VeriSimDB` module is a real integration or a thin stub. The audit trail - depends on this. Phase E should not be completed until VeriSimDB status is confirmed. - -### Tier placement note - -The audit confirms that tier 2 (between Cloudflare edge and BoJ's gnosis handler) -is the correct placement. An alternative would be to place the gateway alongside tier 3 -(BEAM supervisor), where it would run co-located with the BoJ Elixir process rather -than in front of the gnosis handler. This would be wrong: the gateway's role is to -govern the HTTP surface before it reaches any BoJ component, not to provide another -enforcement layer inside BoJ. Tier 2 placement is confirmed. - -## Related - -- ADR 0002 (`docs/decisions/0002-align-unified-zig-api-stack.md`) — unified Zig API - stack (the gnosis handler the gateway sits in front of). -- ADR 0003 (`docs/decisions/0003-extract-cartridge-spec-standalone.md`) — cartridge - specification (verb governance per cartridge is the downstream beneficiary of this ADR). -- Trustfile `[HTTP_CAPABILITY_GATEWAY]` section (commit `ceae54c`) — forward-reference - entry updated in this session to `status: "ACCEPTED-PLANNED"`. -- Trustfile `[CLOUDFLARE_EDGE_SECURITY].rate_limiting.tier_2_gateway` — the gap this ADR closes. -- `docs/integration/http-capability-gateway-audit.md` — audit conducted 2026-04-17. -- `docs/integration/http-capability-gateway-plan.md` — phased integration plan. diff --git a/docs/decisions/0005-elixir-to-zig-ffi-transport.adoc b/docs/decisions/0005-elixir-to-zig-ffi-transport.adoc new file mode 100644 index 00000000..6baca766 --- /dev/null +++ b/docs/decisions/0005-elixir-to-zig-ffi-transport.adoc @@ -0,0 +1,151 @@ +== 5. Choose Transport for Elixir REST → Zig Cartridge FFI + +Date: 2026-04-18 + +=== Status + +Proposed + +=== Context + +`+elixir/+` (added 2026-04-18) brings up a Cowboy+Plug HTTP listener +that serves cartridge metadata but returns +`+501 invocation-not-yet-wired+` for `+POST /cartridge/:name/invoke+`. +The remaining Phase 2 work is: given a cartridge name, a tool name, and +a JSON arg blob, call into the cartridge’s Zig FFI and return the +result. + +Each cartridge already ships: + +* `+cartridges//ffi/_ffi.zig+` with C-ABI exports. +* `+cartridges//ffi/zig-out/lib/lib.so+` produced by +`+zig build+` (per `+STATE.a2ml §maintenance+`: 99/99 built). +* A documented tool catalogue in `+cartridge.json+`. + +The trunk already has `+ffi/zig/src/loader.zig+` which handles mount / +unmount / hash verification of these `+.so+` files — so cartridge +*loading* is a solved problem at the Zig layer. + +The question this ADR answers is: *how does an Elixir process sitting +behind the REST router call a function in that loaded `+.so+`?* + +There are three established BEAM ↔ C-ABI transports, each with a +different safety / latency / complexity trade-off. + +==== Option A — OS process port (`+System.cmd+` / `+Port.open+`) + +Spawn a Zig CLI binary per invocation (or a long-lived one pooled in a +GenServer). The CLI reads `+ +` on argv/stdin, +loads the `+.so+` via `+dlopen+`, dispatches, emits JSON on stdout. + +* *Safety*: a cartridge crash kills the child OS process. BEAM is +isolated. Strongest fault-containment. +* *Latency*: ~1–5 ms per shell-out (fork/exec + dlopen overhead). A pool +of long-lived children amortises this. +* *Complexity*: one new Zig CLI binary; pool management is a standard +GenServer pattern. +* *Licensing / packaging*: the CLI is a small Zig target; no new +upstream dependencies. + +==== Option B — NIF (Native Implemented Function) via `+zigler+` + +Compile the cartridge FFI directly into a BEAM NIF and call it +in-process. + +* *Safety*: a segfault in a NIF crashes the entire BEAM VM. No +schedulers survive. Unacceptable for a server that hosts 99 cartridges +of varying maturity. +* *Latency*: sub-microsecond. Fastest possible. +* *Complexity*: `+zigler+` adds a toolchain; per-cartridge build +integration; NIF contract (dirty scheduler, resource management) is a +real engineering commitment. +* *Observation*: ADR-0002 positions BoJ to consume `+libzig_api.so+` +upstream with the `+uapi_gnosis_*+` unified handler. If that +consolidation lands, a single NIF boundary at the gnosis layer (not 99 +per-cartridge NIFs) becomes attractive — but that is not today. + +==== Option C — Local HTTP / gRPC to a side-car + +Run a Zig-based side-car process (existing `+uapi_gnosis+` binary or a +new one) listening on Unix domain sockets. Elixir POSTs invocation +requests to it. + +* *Safety*: side-car crash doesn’t kill BEAM, symmetrical to A. +* *Latency*: ~100 µs per Unix-socket round-trip; faster than A. +* *Complexity*: requires a protocol definition (HTTP/JSON or gRPC) +between Elixir and the side-car — architecturally this is the same shape +the REST server is already solving, just pushed one layer down. +* *Observation*: this is effectively what `+http-capability-gateway+` +(ADR-0004) already does at tier 2. Pushing the Elixir→Zig bridge into +gateway territory would conflate two concerns. + +=== Decision + +*Adopt Option A (OS process port) for Phase 2*, with Option B held open +for a future phase when the `+uapi_gnosis+` unified handler is live. + +Concretely: + +[arabic] +. Add `+ffi/zig/src/boj_invoke_cli.zig+` — a short Zig binary that takes +`+ +` on argv, loads the cartridge `+.so+` +via the existing `+loader.zig+`, dispatches to the named tool, emits the +result as JSON on stdout, and exits with a documented error-code +convention (matching the trunk-wide `+IsUnbreakable+` proof scheme). +. Add `+elixir/lib/boj_rest/invoker.ex+` as a `+GenServer+`-managed pool +of long-lived CLI children (not fork-per-request). The pool size is +configurable; default 4. +. Replace the 501 in `+elixir/lib/boj_rest/router.ex+` with a call to +`+Invoker.dispatch(cartridge, tool, args)+`. +. Timeouts: 5 s default per invocation, configurable per cartridge via +`+cartridge.json+`. + +=== Consequences + +==== Positive + +* Fault isolation: any cartridge panic stays in the child OS process. +BEAM’s supervisor restarts the pool slot; the REST request returns a 502 +with the error classification. +* Incremental path: Phase 2 ships without committing to the NIF +toolchain, and the CLI binary is reusable by non-Elixir clients (CI +smoke tests, panic-attacker). +* No new protocol: the CLI uses argv + stdout JSON, no gRPC/Thrift. + +==== Negative + +* Latency floor ~1 ms per invocation. For tools expected to run hundreds +of times per second per cartridge, the pool size needs tuning. Not every +cartridge has that load profile — most are interactive (MCP tool call +rate is one-per-user-keystroke at worst). +* Double error-code surface: the CLI’s exit codes plus the `+.so+`’s +return codes. ADR follow-up: map both into a single +`+{:ok, body} | {:error, classified}+` Elixir tuple in `+Invoker+`. + +==== Migration path + +When `+uapi_gnosis_set_handler+` stabilises upstream (ADR-0002 open +question #1), re-evaluate: keep the OS-port pool and have it link +`+libzig_api.so+` internally, or promote to a single NIF boundary at the +gnosis handler. That decision will merit its own ADR. + +=== Non-decisions + +* *Federation invocation* (cross-node) is out of scope — it belongs to +`+Boj.Federation+` at the trunk ABI, not the Elixir REST layer. +* *HTTP capability governance* on the REST endpoint itself is ADR-0004’s +territory (`+http-capability-gateway+` at tier 2). This ADR covers only +what happens _after_ the gateway lets the request through. +* *Cartridge credential isolation* is `+Boj.CredentialIsolation+` (BJ2) +— this ADR does not change that model; the CLI must carry the +cartridge’s vault partition index through to the FFI. + +=== References + +* ADR-0002 (`+align-unified-zig-api-stack+`) — open question #1. +* ADR-0004 (`+adopt-http-capability-gateway+`) — tier 2 attachment +point; independent of this ADR. +* `+elixir/README.adoc+` §Phase plan — Phase 2 description. +* `+docs/practice/DOGFOOD-LOG.adoc+` 2026-04-18 entry — original "`REST +API unreachable`" finding that motivated the port. +* `+ffi/zig/src/loader.zig+` — existing cartridge `+.so+` loader. diff --git a/docs/decisions/0005-elixir-to-zig-ffi-transport.md b/docs/decisions/0005-elixir-to-zig-ffi-transport.md deleted file mode 100644 index a0a0be1b..00000000 --- a/docs/decisions/0005-elixir-to-zig-ffi-transport.md +++ /dev/null @@ -1,160 +0,0 @@ - - - -# 5. Choose Transport for Elixir REST → Zig Cartridge FFI - -Date: 2026-04-18 - -## Status - -Proposed - -## Context - -`elixir/` (added 2026-04-18) brings up a Cowboy+Plug HTTP listener that -serves cartridge metadata but returns `501 invocation-not-yet-wired` for -`POST /cartridge/:name/invoke`. The remaining Phase 2 work is: given a -cartridge name, a tool name, and a JSON arg blob, call into the -cartridge's Zig FFI and return the result. - -Each cartridge already ships: - -- `cartridges//ffi/_ffi.zig` with C-ABI exports. -- `cartridges//ffi/zig-out/lib/lib.so` produced by - `zig build` (per `STATE.a2ml §maintenance`: 99/99 built). -- A documented tool catalogue in `cartridge.json`. - -The trunk already has `ffi/zig/src/loader.zig` which handles mount / -unmount / hash verification of these `.so` files — so cartridge -**loading** is a solved problem at the Zig layer. - -The question this ADR answers is: **how does an Elixir process sitting -behind the REST router call a function in that loaded `.so`?** - -There are three established BEAM ↔ C-ABI transports, each with a -different safety / latency / complexity trade-off. - -### Option A — OS process port (`System.cmd` / `Port.open`) - -Spawn a Zig CLI binary per invocation (or a long-lived one pooled in a -GenServer). The CLI reads ` ` on argv/stdin, -loads the `.so` via `dlopen`, dispatches, emits JSON on stdout. - -- **Safety**: a cartridge crash kills the child OS process. BEAM is - isolated. Strongest fault-containment. -- **Latency**: ~1–5 ms per shell-out (fork/exec + dlopen overhead). A - pool of long-lived children amortises this. -- **Complexity**: one new Zig CLI binary; pool management is a - standard GenServer pattern. -- **Licensing / packaging**: the CLI is a small Zig target; no new - upstream dependencies. - -### Option B — NIF (Native Implemented Function) via `zigler` - -Compile the cartridge FFI directly into a BEAM NIF and call it -in-process. - -- **Safety**: a segfault in a NIF crashes the entire BEAM VM. No - schedulers survive. Unacceptable for a server that hosts 99 - cartridges of varying maturity. -- **Latency**: sub-microsecond. Fastest possible. -- **Complexity**: `zigler` adds a toolchain; per-cartridge build - integration; NIF contract (dirty scheduler, resource management) is - a real engineering commitment. -- **Observation**: ADR-0002 positions BoJ to consume `libzig_api.so` - upstream with the `uapi_gnosis_*` unified handler. If that consolidation - lands, a single NIF boundary at the gnosis layer (not 99 per-cartridge - NIFs) becomes attractive — but that is not today. - -### Option C — Local HTTP / gRPC to a side-car - -Run a Zig-based side-car process (existing `uapi_gnosis` binary or a -new one) listening on Unix domain sockets. Elixir POSTs invocation -requests to it. - -- **Safety**: side-car crash doesn't kill BEAM, symmetrical to A. -- **Latency**: ~100 µs per Unix-socket round-trip; faster than A. -- **Complexity**: requires a protocol definition (HTTP/JSON or gRPC) - between Elixir and the side-car — architecturally this is the same - shape the REST server is already solving, just pushed one layer - down. -- **Observation**: this is effectively what - `http-capability-gateway` (ADR-0004) already does at tier 2. Pushing - the Elixir→Zig bridge into gateway territory would conflate two - concerns. - -## Decision - -**Adopt Option A (OS process port) for Phase 2**, with Option B held -open for a future phase when the `uapi_gnosis` unified handler is -live. - -Concretely: - -1. Add `ffi/zig/src/boj_invoke_cli.zig` — a short Zig binary that - takes ` ` on argv, loads the cartridge - `.so` via the existing `loader.zig`, dispatches to the named tool, - emits the result as JSON on stdout, and exits with a documented - error-code convention (matching the trunk-wide `IsUnbreakable` - proof scheme). -2. Add `elixir/lib/boj_rest/invoker.ex` as a `GenServer`-managed pool - of long-lived CLI children (not fork-per-request). The pool size - is configurable; default 4. -3. Replace the 501 in `elixir/lib/boj_rest/router.ex` with a call to - `Invoker.dispatch(cartridge, tool, args)`. -4. Timeouts: 5 s default per invocation, configurable per cartridge - via `cartridge.json`. - -## Consequences - -### Positive - -- Fault isolation: any cartridge panic stays in the child OS process. - BEAM's supervisor restarts the pool slot; the REST request returns - a 502 with the error classification. -- Incremental path: Phase 2 ships without committing to the NIF - toolchain, and the CLI binary is reusable by non-Elixir clients - (CI smoke tests, panic-attacker). -- No new protocol: the CLI uses argv + stdout JSON, no gRPC/Thrift. - -### Negative - -- Latency floor ~1 ms per invocation. For tools expected to run - hundreds of times per second per cartridge, the pool size needs - tuning. Not every cartridge has that load profile — most are - interactive (MCP tool call rate is one-per-user-keystroke at - worst). -- Double error-code surface: the CLI's exit codes plus the `.so`'s - return codes. ADR follow-up: map both into a single - `{:ok, body} | {:error, classified}` Elixir tuple in `Invoker`. - -### Migration path - -When `uapi_gnosis_set_handler` stabilises upstream (ADR-0002 open -question #1), re-evaluate: keep the OS-port pool and have it link -`libzig_api.so` internally, or promote to a single NIF boundary at -the gnosis handler. That decision will merit its own ADR. - -## Non-decisions - -- **Federation invocation** (cross-node) is out of scope — it belongs - to `Boj.Federation` at the trunk ABI, not the Elixir REST layer. -- **HTTP capability governance** on the REST endpoint itself is - ADR-0004's territory (`http-capability-gateway` at tier 2). This ADR - covers only what happens *after* the gateway lets the request through. -- **Cartridge credential isolation** is `Boj.CredentialIsolation` - (BJ2) — this ADR does not change that model; the CLI must carry the - cartridge's vault partition index through to the FFI. - -## References - -- ADR-0002 (`align-unified-zig-api-stack`) — open question #1. -- ADR-0004 (`adopt-http-capability-gateway`) — tier 2 attachment - point; independent of this ADR. -- `elixir/README.adoc` §Phase plan — Phase 2 description. -- `docs/practice/DOGFOOD-LOG.adoc` 2026-04-18 entry — original - "REST API unreachable" finding that motivated the port. -- `ffi/zig/src/loader.zig` — existing cartridge `.so` loader. diff --git a/docs/decisions/0006-cartridge-invoke-abi.adoc b/docs/decisions/0006-cartridge-invoke-abi.adoc new file mode 100644 index 00000000..28393f3f --- /dev/null +++ b/docs/decisions/0006-cartridge-invoke-abi.adoc @@ -0,0 +1,169 @@ +== 6. Standard `+boj_cartridge_invoke+` Dispatch ABI + +Date: 2026-04-18 + +=== Status + +Proposed + +=== Context + +ADR-0005 picked an OS-process transport (`+boj-invoke+` CLI + Elixir +`+Invoker+`) for the Elixir REST → Zig cartridge bridge. The skeleton is +merged and an end-to-end request (`+POST /cartridge/aerie-mcp/invoke+` → +CLI → `+dlopen+` → classified JSON error → HTTP 501) is verifiably +working as of 2026-04-18. + +What the skeleton *cannot* do is dispatch a tool by name. The Zig loader +(`+ffi/zig/src/loader.zig+`) currently requires four symbols per +cartridge: + +* `+boj_cartridge_init() -> c_int+` +* `+boj_cartridge_deinit() -> void+` +* `+boj_cartridge_name() -> *const c_char+` +* `+boj_cartridge_version() -> *const c_char+` + +None of the 99 cartridge shared libraries today expose these four +symbols, let alone a fifth dispatch symbol. Each cartridge exports its +own bespoke per-tool entries (`+aerie_create_env+`, `+aws_*+`, etc.) +with non-uniform signatures. That is the ABI gap this decision closes. + +=== Decision + +Adopt a *single fifth standard symbol* per cartridge: + +[source,c] +---- +int32_t boj_cartridge_invoke( + const char *tool_name, // NUL-terminated, caller-owned + const char *json_args, // NUL-terminated JSON object, caller-owned + char *out_buf, // caller-owned output buffer + uintptr_t *in_out_len // in: buf capacity; out: bytes written (excl. NUL) +); +---- + +==== Return codes + +[width="100%",cols="10%,90%",options="header",] +|=== +|Code |Meaning +|`+0+` |Success. `+out_buf[0..*in_out_len]+` contains JSON result. +|`+-1+` |Unknown tool name for this cartridge. +|`+-2+` |Invalid JSON arguments (parse failure or wrong shape). +|`+-3+` |Output buffer too small; `+*in_out_len+` set to required size. +|`+-4+` |Runtime error inside the tool; `+out_buf+` holds error JSON. +|`+-5+` |Invariant violation / panic (cartridge should be unloaded). +|`+-6+` |Authorisation denied (cartridge-internal policy — see BJ2). +|=== + +Return code semantics are frozen by this ADR. New failure modes must +compose existing codes via the error-JSON body; the integer surface must +not grow without a follow-up ADR. + +==== Output format + +`+out_buf+` on success is a JSON object with at minimum a `+result+` +key. Cartridges may add cartridge-specific keys but SHOULD NOT emit +top-level `+error+` on a `+0+` return (that is reserved for the +error-code path). + +On `+-4+` the output buffer carries: + +[source,json] +---- +{"error": "", "message": "", "details": { ... }} +---- + +==== Memory ownership + +* Caller allocates `+out_buf+` at a known size (default 64 KiB in the +`+boj-invoke+` CLI; override via env `+BOJ_INVOKE_BUFLEN+`). +* Cartridge writes at most `+*in_out_len+` bytes and updates the pointer +to the actual length. +* On return, buffer is caller’s to free. + +No heap ownership crosses the FFI boundary. This matches the trunk-wide +"`no allocator leaks`" invariant and keeps the cartridge free to use any +internal allocator. + +==== Relationship to existing four symbols + +All five symbols are required once a cartridge adopts this ABI. The +loader will be updated to look up `+boj_cartridge_invoke+` as a required +symbol *with a migration window*: existing cartridges without the symbol +keep loading but are marked `+dispatch=unavailable+` in the catalog. The +Elixir `+Invoker+` already classifies this as `+missing_symbol+`; that +classification becomes the migration signal. + +=== Consequences + +==== Positive + +* Unified 5-symbol ABI collapses 99 bespoke surfaces into one. +* Generic `+Invoker+` GenServer pool can target every cartridge without +per-cartridge glue. +* Error classification is an integer enum — cheap to log, easy to +aggregate, no string parsing on the hot path. +* Cartridge-internal JSON dispatch is a well-understood pattern +(standard OTP `+:gen_server+` `+handle_call/3+` shape, sumcheck-style +tool tables in Zig). + +==== Negative + +* Every cartridge needs an internal dispatch table (tool name → Zig +function) and JSON parsing. Smaller cartridges (stubs especially) gain +maybe 60–120 lines of boilerplate. +* Cartridge authors must learn the five-symbol contract and the +seven-code return convention. The boilerplate lives in a shared Zig +helper (`+ffi/zig/src/cartridge_shim.zig+`, new) to minimise copy-paste. + +==== Alternatives rejected + +* *Per-tool entries without dispatch.* This is the status quo. Rejected +because it forces every client (CLI, Elixir Invoker, future gateway) to +know the exact symbol name + signature of every tool of every cartridge. +Doesn’t scale. +* *Dynamic tool registration at init.* Cartridge calls back into the +loader to register tool handlers. Requires a loader→cartridge callback +ABI and a per-process registry. More moving parts than the 5-symbol +table, and harder to reason about statically. +* *JSON-RPC over stdin/stdout per cartridge.* Turns every cartridge into +a server process. Dramatically higher process count and fork-overhead +for an estate with ~100 cartridges. + +=== Reference implementation + +`+aerie-mcp+` is the canonical reference (smallest non-trivial +cartridge, four existing bespoke tools). This ADR is accompanied by +commit `++` which: + +[arabic] +. Adds the four existing symbols +(`+boj_cartridge_init/deinit/name/version+`) to +`+cartridges/aerie-mcp/ffi/aerie_ffi.zig+`, so the stub-level FFI +becomes probe-able via `+boj-invoke probe+`. +. Adds a minimal `+boj_cartridge_invoke+` that dispatches three tools +(`+create_env+`, `+destroy_env+`, `+get_status+`) against the existing +bespoke implementations, using a small tool-name string-table. +. Does *not* touch the other 98 cartridges. Migration is follow-up +per-cartridge work, blocked on this ADR being accepted. + +=== Non-decisions + +* Credential isolation (BJ2): unchanged. `+boj_cartridge_invoke+` is a +per-cartridge entry; the vault partition index is per-cartridge, so +nothing here weakens isolation. +* Federation dispatch: still lives at `+Boj.Federation+`. This ADR +concerns only in-process cartridge dispatch. +* Tool-level auth: still the cartridge’s responsibility. Return code +`+-6+` signals auth denial but does not standardise the auth mechanism. + +=== References + +* ADR-0005 — transport choice (OS-process port). +* `+ffi/zig/src/loader.zig+` — the four-symbol loader interface this +extends. +* `+ffi/zig/src/boj_invoke_cli.zig+` — consumer of this ABI. +* `+elixir/lib/boj_rest/invoker.ex+` — Elixir caller classification. +* `+docs/specification/cartridges/README.md+` — cartridge spec +(narrative). diff --git a/docs/decisions/0006-cartridge-invoke-abi.md b/docs/decisions/0006-cartridge-invoke-abi.md deleted file mode 100644 index 4e11526b..00000000 --- a/docs/decisions/0006-cartridge-invoke-abi.md +++ /dev/null @@ -1,167 +0,0 @@ - - - -# 6. Standard `boj_cartridge_invoke` Dispatch ABI - -Date: 2026-04-18 - -## Status - -Proposed - -## Context - -ADR-0005 picked an OS-process transport (`boj-invoke` CLI + Elixir -`Invoker`) for the Elixir REST → Zig cartridge bridge. The skeleton is -merged and an end-to-end request (`POST /cartridge/aerie-mcp/invoke` → -CLI → `dlopen` → classified JSON error → HTTP 501) is verifiably working -as of 2026-04-18. - -What the skeleton **cannot** do is dispatch a tool by name. The Zig -loader (`ffi/zig/src/loader.zig`) currently requires four symbols per -cartridge: - -- `boj_cartridge_init() -> c_int` -- `boj_cartridge_deinit() -> void` -- `boj_cartridge_name() -> *const c_char` -- `boj_cartridge_version() -> *const c_char` - -None of the 99 cartridge shared libraries today expose these four -symbols, let alone a fifth dispatch symbol. Each cartridge exports its -own bespoke per-tool entries (`aerie_create_env`, `aws_*`, etc.) with -non-uniform signatures. That is the ABI gap this decision closes. - -## Decision - -Adopt a **single fifth standard symbol** per cartridge: - -```c -int32_t boj_cartridge_invoke( - const char *tool_name, // NUL-terminated, caller-owned - const char *json_args, // NUL-terminated JSON object, caller-owned - char *out_buf, // caller-owned output buffer - uintptr_t *in_out_len // in: buf capacity; out: bytes written (excl. NUL) -); -``` - -### Return codes - -| Code | Meaning | -|-------|----------------------------------------------------------------| -| `0` | Success. `out_buf[0..*in_out_len]` contains JSON result. | -| `-1` | Unknown tool name for this cartridge. | -| `-2` | Invalid JSON arguments (parse failure or wrong shape). | -| `-3` | Output buffer too small; `*in_out_len` set to required size. | -| `-4` | Runtime error inside the tool; `out_buf` holds error JSON. | -| `-5` | Invariant violation / panic (cartridge should be unloaded). | -| `-6` | Authorisation denied (cartridge-internal policy — see BJ2). | - -Return code semantics are frozen by this ADR. New failure modes must -compose existing codes via the error-JSON body; the integer surface -must not grow without a follow-up ADR. - -### Output format - -`out_buf` on success is a JSON object with at minimum a `result` key. -Cartridges may add cartridge-specific keys but SHOULD NOT emit top-level -`error` on a `0` return (that is reserved for the error-code path). - -On `-4` the output buffer carries: - -```json -{"error": "", "message": "", "details": { ... }} -``` - -### Memory ownership - -- Caller allocates `out_buf` at a known size (default 64 KiB in the - `boj-invoke` CLI; override via env `BOJ_INVOKE_BUFLEN`). -- Cartridge writes at most `*in_out_len` bytes and updates the pointer - to the actual length. -- On return, buffer is caller's to free. - -No heap ownership crosses the FFI boundary. This matches the trunk-wide -"no allocator leaks" invariant and keeps the cartridge free to use any -internal allocator. - -### Relationship to existing four symbols - -All five symbols are required once a cartridge adopts this ABI. The -loader will be updated to look up `boj_cartridge_invoke` as a required -symbol **with a migration window**: existing cartridges without the -symbol keep loading but are marked `dispatch=unavailable` in the -catalog. The Elixir `Invoker` already classifies this as -`missing_symbol`; that classification becomes the migration signal. - -## Consequences - -### Positive - -- Unified 5-symbol ABI collapses 99 bespoke surfaces into one. -- Generic `Invoker` GenServer pool can target every cartridge without - per-cartridge glue. -- Error classification is an integer enum — cheap to log, easy to - aggregate, no string parsing on the hot path. -- Cartridge-internal JSON dispatch is a well-understood pattern - (standard OTP `:gen_server` `handle_call/3` shape, sumcheck-style - tool tables in Zig). - -### Negative - -- Every cartridge needs an internal dispatch table (tool name → Zig - function) and JSON parsing. Smaller cartridges (stubs especially) - gain maybe 60–120 lines of boilerplate. -- Cartridge authors must learn the five-symbol contract and the - seven-code return convention. The boilerplate lives in a shared Zig - helper (`ffi/zig/src/cartridge_shim.zig`, new) to minimise copy-paste. - -### Alternatives rejected - -- **Per-tool entries without dispatch.** This is the status quo. - Rejected because it forces every client (CLI, Elixir Invoker, - future gateway) to know the exact symbol name + signature of every - tool of every cartridge. Doesn't scale. -- **Dynamic tool registration at init.** Cartridge calls back into the - loader to register tool handlers. Requires a loader→cartridge callback - ABI and a per-process registry. More moving parts than the 5-symbol - table, and harder to reason about statically. -- **JSON-RPC over stdin/stdout per cartridge.** Turns every cartridge - into a server process. Dramatically higher process count and - fork-overhead for an estate with ~100 cartridges. - -## Reference implementation - -`aerie-mcp` is the canonical reference (smallest non-trivial cartridge, -four existing bespoke tools). This ADR is accompanied by commit -`` which: - -1. Adds the four existing symbols (`boj_cartridge_init/deinit/name/version`) - to `cartridges/aerie-mcp/ffi/aerie_ffi.zig`, so the stub-level FFI - becomes probe-able via `boj-invoke probe`. -2. Adds a minimal `boj_cartridge_invoke` that dispatches three tools - (`create_env`, `destroy_env`, `get_status`) against the existing - bespoke implementations, using a small tool-name string-table. -3. Does **not** touch the other 98 cartridges. Migration is follow-up - per-cartridge work, blocked on this ADR being accepted. - -## Non-decisions - -- Credential isolation (BJ2): unchanged. `boj_cartridge_invoke` is a - per-cartridge entry; the vault partition index is per-cartridge, so - nothing here weakens isolation. -- Federation dispatch: still lives at `Boj.Federation`. This ADR - concerns only in-process cartridge dispatch. -- Tool-level auth: still the cartridge's responsibility. Return code - `-6` signals auth denial but does not standardise the auth mechanism. - -## References - -- ADR-0005 — transport choice (OS-process port). -- `ffi/zig/src/loader.zig` — the four-symbol loader interface this - extends. -- `ffi/zig/src/boj_invoke_cli.zig` — consumer of this ABI. -- `elixir/lib/boj_rest/invoker.ex` — Elixir caller classification. -- `docs/specification/cartridges/README.md` — cartridge spec (narrative). diff --git a/docs/decisions/0007-trust-tier-policy-dsl.adoc b/docs/decisions/0007-trust-tier-policy-dsl.adoc new file mode 100644 index 00000000..a383cba0 --- /dev/null +++ b/docs/decisions/0007-trust-tier-policy-dsl.adoc @@ -0,0 +1,271 @@ +== 7. Trust-tier policy DSL — Nickel-encoded PEP at the bridge, PDP as a cartridge + +Date: 2026-05-20 + +=== Status + +Proposed (RFC — implementation tracked in epic #87 item 4) + +=== Context + +BoJ defines three trust tiers — *Teranga* (formally verified, full ABI +discharge), *Shield* (operationally hardened, explicit security review), +*Ayo* (community contributions, master-approval-gated). The tier +vocabulary appears in: + +* README ("`Cartridges — 115 pluggable cartridges across Teranga / +Shield / Ayo trust tiers`") +* `+mcp-bridge/lib/offline-menu.js+` (cartridges grouped into +`+tier_teranga+` / `+tier_shield+` / `+tier_ayo+`) +* `+boj://server/info+` (the `+trust_tiers+` field, added in PR #89) +* Cartridge manifests (declared per-cartridge) + +But *the tiers are not load-bearing at runtime*. The bridge today +applies one rate limit, one prompt-injection filter, and one input-size +cap to every call regardless of which cartridge is invoked or what tier +it sits in. A `+boj_github_merge_pr+` (destructive, tier-3-equivalent) +and a `+boj_cartridge_info+` (read-only, tier-0) are gated identically. + +ADR-0002 ("`BoJ-only MCP`") establishes BoJ as the single MCP gateway +for the estate. By construction there is one *policy enforcement point* +(the bridge). It follows that there should be one *policy decision +point* that the bridge consults — separating _what is allowed_ from _how +it’s enforced_. Without that separation, every new policy rule means +touching `+mcp-bridge/main.js+`, which doesn’t scale and conflicts with +the BoJ-only-MCP rule (you can’t push policy out to per-cartridge code +without breaking the rule). + +The coord layer already uses *Nickel* for typed-envelope contracts +(`+cartridges/local-coord-mcp/coord-messages.ncl+`, gated by +`+COORD_REQUIRE_NICKEL=1+`). Extending Nickel to the broader policy +surface keeps the verification story coherent: one schema language, one +validator, two surfaces (coord envelopes + bridge dispatch). + +=== Decision + +Adopt *Nickel as the policy DSL*, with a clean *PEP / PDP split*: + +* *PEP (Policy Enforcement Point)* — the bridge’s `+hardeningGate+` in +`+mcp-bridge/main.js+`. Stays where it is. +* *PDP (Policy Decision Point)* — a new `+policy-mcp+` cartridge that +holds the active policy bundle, evaluates queries, and returns allow / +deny / require-approval / rate-limit verdicts. Lives at +`+cartridges/policy-mcp/+`. + +==== Policy schema (Nickel) + +[source,nickel] +---- +# policies/boj-default.ncl +let CartridgePolicy = { + tier | std.number.Number | std.contract.in [0, 1, 2, 3, 4], + rate_limit | { per_minute | Number, per_hour | Number }, + required_role | [| 'apprentice, 'journeyman, 'master |], + master_approval | Bool, + allowed_args | { _ : { type | String, validator | String | default = "" } }, + side_effect | [| 'read, 'small_write, 'major_write, 'destructive |], +} in + +{ + # Teranga (tier 0-1) — formally verified, broadly available + cartridge."github-api-mcp".tools.boj_github_get_repo = { + tier = 0, + rate_limit = { per_minute = 60, per_hour = 1000 }, + required_role = 'apprentice, + master_approval = false, + allowed_args = { + owner = { type = "string", validator = "^[A-Za-z0-9-]+$" }, + repo = { type = "string", validator = "^[A-Za-z0-9._-]+$" }, + }, + side_effect = 'read, + }, + + # Shield (tier 2) — write operations gated on journeyman role + cartridge."github-api-mcp".tools.boj_github_create_issue = { + tier = 2, + rate_limit = { per_minute = 10, per_hour = 50 }, + required_role = 'journeyman, + master_approval = false, + side_effect = 'small_write, + }, + + # Ayo (tier 3-4) — destructive operations require master approval + cartridge."github-api-mcp".tools.boj_github_merge_pr = { + tier = 3, + rate_limit = { per_minute = 2, per_hour = 5 }, + required_role = 'journeyman, + master_approval = true, + side_effect = 'major_write, + }, +} +---- + +==== Wire protocol (PEP → PDP) + +The bridge already loads Nickel locally +(`+mcp-bridge/lib/nickel-validator.js+`). For policy queries, the bridge +calls into `+policy-mcp+` over the same REST surface every cartridge +uses: + +.... +POST /cartridge/policy-mcp/evaluate +{ + "cartridge": "github-api-mcp", + "tool": "boj_github_merge_pr", + "args": { "owner": "...", "repo": "...", "pull_number": 42 }, + "peer": { "role": "journeyman", "client_kind": "claude", "variant": "opus-4.7" } +} +→ +{ + "verdict": "require_approval" | "allow" | "deny" | "rate_limit", + "tier": 3, + "reason": "boj_github_merge_pr is tier-3; journeyman role + master approval required", + "approval_token": "", // only when verdict=require_approval + "rate_limit_window_ms": 60000, // only when verdict=rate_limit + "audit_id": "" +} +.... + +==== PEP behaviour matrix + +[width="100%",cols="50%,50%",options="header",] +|=== +|Verdict |Bridge action +|`+allow+` |Proceed to dispatch + +|`+deny+` |Return MCP error `+-32000+` "`policy denied: `"; emit audit +log; do not dispatch + +|`+rate_limit+` |Return MCP error `+-32001+` "`rate limit; retry after +`"; do not dispatch + +|`+require_approval+` |If caller’s role is master, proceed. Otherwise +enqueue in `+coord_send_gated+`-style quarantine and return `+-32002+` +"`pending master approval, request_id=`" +|=== + +==== Default policy bundle + +Ship `+policies/boj-default.ncl+` covering all 41 bridge tools. Derived +from each tool’s existing annotations (read-only vs side-effectful), +which were classified during the v0.4.6 AAA-tier description pass. No +new metadata work required for the default bundle. + +==== Configuration + +* `+BOJ_POLICY_MODE=enforce+` (default) | `+audit+` (log verdicts but +always allow) | `+off+` (skip PDP) +* `+BOJ_POLICY_BUNDLE+` — path to active `+.ncl+` file (default +`+policies/boj-default.ncl+`) +* New tool `+boj_policy_reload+` for hot-reloading the bundle without +restart + +=== Consequences + +==== Positive + +* *Tier names become enforceable* — `+tier: 3+` in policy = +`+master_approval: true+` at runtime. The taxonomy is no longer +documentation. +* *Per-cartridge / per-tool granularity* — different tools in the same +cartridge can have different policies (e.g. `+boj_github_list_issues+` +open, `+boj_github_merge_pr+` master-gated). +* *One DSL, two surfaces* — extends the existing Nickel investment from +coord envelopes to bridge dispatch. Reuses `+nickel-validator.js+` +infrastructure. +* *PEP/PDP separation matches NIST RBAC 4.0 reference architecture* — +policy auditing tools (OPA, Rego, etc.) generalise to this shape; future +migration is mechanical. +* *Audit trail by construction* — every decision yields an `+audit_id+` +linkable to a span (pairs with epic #87 item 13’s OTel exporter). +* *Cleans up `+hardeningGate+`* — the bridge’s current 5-stage +validation collapses into "`ask the PDP`". Each rule becomes a Nickel +contract, not JS. + +==== Negative + +* *Latency* — every `+tools/call+` now incurs a PDP round-trip. +Mitigation: in-process Nickel evaluation (no HTTP) when policy-mcp is +loaded as a same-process cartridge. Out-of-process only for federated +multi-machine setups (see ADR-0010). +* *Cold-start failure mode* — what does the bridge do if policy-mcp is +unreachable? Three modes: `+fail-closed+` (deny all), `+fail-open+` +(allow all), `+fail-static+` (use cached last-good bundle). Default: +`+fail-static+` with `+BOJ_POLICY_FAIL_MODE+` override. +* *Policy authorship burden* — writing 41-tool policies once is fine; +the marginal cost per new cartridge isn’t trivial. Mitigation: ADR-0008 +(cartridge marketplace) requires every submitted cartridge to ship a +default policy, just as it ships a manifest. +* *Existing 5-stage hardening must coexist* — rate limit, size cap, +injection scan, name validation, required-args all stay in the PEP; the +PDP layers _on top_. Don’t fold them in until they’re proven equivalent +under Nickel contracts. + +=== Non-goals + +* Not building a generic RBAC system. The roles are +master/journeyman/apprentice as already defined in coord-mcp; new roles +require ADR. +* Not adopting OPA/Rego — Nickel is already the estate’s contract +language; importing another would split the surface. +* Not creating "`user accounts`" — peer identity is via +`+coord_register+` token, which is the canonical identity. Multi-tenancy +is out of scope for v1. +* Not enforcing policy on `+resources/read+` or `+prompts/get+` (added +in PR #89) initially — those are read-only and the cost/benefit doesn’t +justify it. Revisit if abuse surfaces. + +=== Open questions + +[arabic] +. *PDP-as-cartridge vs PDP-in-bridge* — does the PDP genuinely need to +be a separate cartridge, or should it be an in-process module like +`+nickel-validator.js+`? Arguments both ways. Recommend cartridge for +separation of concerns and to enable per-deployment policy +customisation. +. *Policy versioning* — when the bundle changes, in-flight quarantined +approvals could have been written against a different version. Need a +`+policy_version+` field on quarantined envelopes; reject approval if +version doesn’t match. +. *Cross-cartridge composition policies* — `+boj_cartridge_invoke+` +against an arbitrary cartridge currently bypasses tool-level policy. +Either (a) require every cartridge to declare its full tool surface in +its manifest so the PDP knows the per-tool tier, or (b) apply the +cartridge-level worst-case tier to all unrecognised tools. Recommend +(a); it’s already partially true for verified cartridges. +. *Migration sequence* — should the existing `+hardeningGate+` +rate-limit logic be ported into the PDP (so all policy lives in one +place) or left in the PEP for performance? Recommend PEP keeps +rate-limit + injection-scan as fast-path guards, PDP handles the +tier-aware logic. + +=== Implementation sketch + +When this RFC is accepted: + +[arabic] +. New cartridge `+cartridges/policy-mcp/+` with Idris2 ABI + Zig FFI + +Deno adapter (standard triple). Manifest declares one tool +`+policy_evaluate+` plus internal `+policy_reload+`. +. `+policies/boj-default.ncl+` shipped at the repo root. 41 tools +covered. +. `+mcp-bridge/lib/policy-client.js+` — minimal Nickel-backed PDP client +(in-process when local, REST when federated). +. `+mcp-bridge/main.js+` — `+hardeningGate+` adds a step 6 ("`PDP +evaluation`") and respects the verdict. +. New env vars in `+glama.json+`: `+BOJ_POLICY_MODE+`, +`+BOJ_POLICY_BUNDLE+`, `+BOJ_POLICY_FAIL_MODE+`. +. Test suite: 1 test per verdict outcome (allow / deny / rate_limit / +require_approval); fixture policy bundles for each case. +. Documentation: a new `+docs/operations/POLICY-AUTHORING.md+` showing +how operators customise the default bundle. + +=== Linked + +* ADR-0002 (BoJ-only MCP) — establishes the single-gateway invariant +this depends on. +* ADR-0008 (cartridge marketplace) — submission flow requires a default +policy per cartridge. +* Epic #87 item 4 (this). +* `+cartridges/local-coord-mcp/coord-messages.ncl+` — vocabulary +precedent. diff --git a/docs/decisions/0007-trust-tier-policy-dsl.md b/docs/decisions/0007-trust-tier-policy-dsl.md deleted file mode 100644 index 56dc82a9..00000000 --- a/docs/decisions/0007-trust-tier-policy-dsl.md +++ /dev/null @@ -1,178 +0,0 @@ - - - -# 7. Trust-tier policy DSL — Nickel-encoded PEP at the bridge, PDP as a cartridge - -Date: 2026-05-20 - -## Status - -Proposed (RFC — implementation tracked in epic #87 item 4) - -## Context - -BoJ defines three trust tiers — **Teranga** (formally verified, full ABI discharge), **Shield** (operationally hardened, explicit security review), **Ayo** (community contributions, master-approval-gated). The tier vocabulary appears in: - -- README ("Cartridges — 115 pluggable cartridges across Teranga / Shield / Ayo trust tiers") -- `mcp-bridge/lib/offline-menu.js` (cartridges grouped into `tier_teranga` / `tier_shield` / `tier_ayo`) -- `boj://server/info` (the `trust_tiers` field, added in PR #89) -- Cartridge manifests (declared per-cartridge) - -But **the tiers are not load-bearing at runtime**. The bridge today applies one rate limit, one prompt-injection filter, and one input-size cap to every call regardless of which cartridge is invoked or what tier it sits in. A `boj_github_merge_pr` (destructive, tier-3-equivalent) and a `boj_cartridge_info` (read-only, tier-0) are gated identically. - -ADR-0002 ("BoJ-only MCP") establishes BoJ as the single MCP gateway for the estate. By construction there is one **policy enforcement point** (the bridge). It follows that there should be one **policy decision point** that the bridge consults — separating *what is allowed* from *how it's enforced*. Without that separation, every new policy rule means touching `mcp-bridge/main.js`, which doesn't scale and conflicts with the BoJ-only-MCP rule (you can't push policy out to per-cartridge code without breaking the rule). - -The coord layer already uses **Nickel** for typed-envelope contracts (`cartridges/local-coord-mcp/coord-messages.ncl`, gated by `COORD_REQUIRE_NICKEL=1`). Extending Nickel to the broader policy surface keeps the verification story coherent: one schema language, one validator, two surfaces (coord envelopes + bridge dispatch). - -## Decision - -Adopt **Nickel as the policy DSL**, with a clean **PEP / PDP split**: - -- **PEP (Policy Enforcement Point)** — the bridge's `hardeningGate` in `mcp-bridge/main.js`. Stays where it is. -- **PDP (Policy Decision Point)** — a new `policy-mcp` cartridge that holds the active policy bundle, evaluates queries, and returns allow / deny / require-approval / rate-limit verdicts. Lives at `cartridges/policy-mcp/`. - -### Policy schema (Nickel) - -```nickel -# policies/boj-default.ncl -let CartridgePolicy = { - tier | std.number.Number | std.contract.in [0, 1, 2, 3, 4], - rate_limit | { per_minute | Number, per_hour | Number }, - required_role | [| 'apprentice, 'journeyman, 'master |], - master_approval | Bool, - allowed_args | { _ : { type | String, validator | String | default = "" } }, - side_effect | [| 'read, 'small_write, 'major_write, 'destructive |], -} in - -{ - # Teranga (tier 0-1) — formally verified, broadly available - cartridge."github-api-mcp".tools.boj_github_get_repo = { - tier = 0, - rate_limit = { per_minute = 60, per_hour = 1000 }, - required_role = 'apprentice, - master_approval = false, - allowed_args = { - owner = { type = "string", validator = "^[A-Za-z0-9-]+$" }, - repo = { type = "string", validator = "^[A-Za-z0-9._-]+$" }, - }, - side_effect = 'read, - }, - - # Shield (tier 2) — write operations gated on journeyman role - cartridge."github-api-mcp".tools.boj_github_create_issue = { - tier = 2, - rate_limit = { per_minute = 10, per_hour = 50 }, - required_role = 'journeyman, - master_approval = false, - side_effect = 'small_write, - }, - - # Ayo (tier 3-4) — destructive operations require master approval - cartridge."github-api-mcp".tools.boj_github_merge_pr = { - tier = 3, - rate_limit = { per_minute = 2, per_hour = 5 }, - required_role = 'journeyman, - master_approval = true, - side_effect = 'major_write, - }, -} -``` - -### Wire protocol (PEP → PDP) - -The bridge already loads Nickel locally (`mcp-bridge/lib/nickel-validator.js`). For policy queries, the bridge calls into `policy-mcp` over the same REST surface every cartridge uses: - -``` -POST /cartridge/policy-mcp/evaluate -{ - "cartridge": "github-api-mcp", - "tool": "boj_github_merge_pr", - "args": { "owner": "...", "repo": "...", "pull_number": 42 }, - "peer": { "role": "journeyman", "client_kind": "claude", "variant": "opus-4.7" } -} -→ -{ - "verdict": "require_approval" | "allow" | "deny" | "rate_limit", - "tier": 3, - "reason": "boj_github_merge_pr is tier-3; journeyman role + master approval required", - "approval_token": "", // only when verdict=require_approval - "rate_limit_window_ms": 60000, // only when verdict=rate_limit - "audit_id": "" -} -``` - -### PEP behaviour matrix - -| Verdict | Bridge action | -|---|---| -| `allow` | Proceed to dispatch | -| `deny` | Return MCP error `-32000` "policy denied: "; emit audit log; do not dispatch | -| `rate_limit` | Return MCP error `-32001` "rate limit; retry after "; do not dispatch | -| `require_approval` | If caller's role is master, proceed. Otherwise enqueue in `coord_send_gated`-style quarantine and return `-32002` "pending master approval, request_id=" | - -### Default policy bundle - -Ship `policies/boj-default.ncl` covering all 41 bridge tools. Derived from each tool's existing annotations (read-only vs side-effectful), which were classified during the v0.4.6 AAA-tier description pass. No new metadata work required for the default bundle. - -### Configuration - -- `BOJ_POLICY_MODE=enforce` (default) | `audit` (log verdicts but always allow) | `off` (skip PDP) -- `BOJ_POLICY_BUNDLE` — path to active `.ncl` file (default `policies/boj-default.ncl`) -- New tool `boj_policy_reload` for hot-reloading the bundle without restart - -## Consequences - -### Positive - -- **Tier names become enforceable** — `tier: 3` in policy = `master_approval: true` at runtime. The taxonomy is no longer documentation. -- **Per-cartridge / per-tool granularity** — different tools in the same cartridge can have different policies (e.g. `boj_github_list_issues` open, `boj_github_merge_pr` master-gated). -- **One DSL, two surfaces** — extends the existing Nickel investment from coord envelopes to bridge dispatch. Reuses `nickel-validator.js` infrastructure. -- **PEP/PDP separation matches NIST RBAC 4.0 reference architecture** — policy auditing tools (OPA, Rego, etc.) generalise to this shape; future migration is mechanical. -- **Audit trail by construction** — every decision yields an `audit_id` linkable to a span (pairs with epic #87 item 13's OTel exporter). -- **Cleans up `hardeningGate`** — the bridge's current 5-stage validation collapses into "ask the PDP". Each rule becomes a Nickel contract, not JS. - -### Negative - -- **Latency** — every `tools/call` now incurs a PDP round-trip. Mitigation: in-process Nickel evaluation (no HTTP) when policy-mcp is loaded as a same-process cartridge. Out-of-process only for federated multi-machine setups (see ADR-0010). -- **Cold-start failure mode** — what does the bridge do if policy-mcp is unreachable? Three modes: `fail-closed` (deny all), `fail-open` (allow all), `fail-static` (use cached last-good bundle). Default: `fail-static` with `BOJ_POLICY_FAIL_MODE` override. -- **Policy authorship burden** — writing 41-tool policies once is fine; the marginal cost per new cartridge isn't trivial. Mitigation: ADR-0008 (cartridge marketplace) requires every submitted cartridge to ship a default policy, just as it ships a manifest. -- **Existing 5-stage hardening must coexist** — rate limit, size cap, injection scan, name validation, required-args all stay in the PEP; the PDP layers *on top*. Don't fold them in until they're proven equivalent under Nickel contracts. - -## Non-goals - -- Not building a generic RBAC system. The roles are master/journeyman/apprentice as already defined in coord-mcp; new roles require ADR. -- Not adopting OPA/Rego — Nickel is already the estate's contract language; importing another would split the surface. -- Not creating "user accounts" — peer identity is via `coord_register` token, which is the canonical identity. Multi-tenancy is out of scope for v1. -- Not enforcing policy on `resources/read` or `prompts/get` (added in PR #89) initially — those are read-only and the cost/benefit doesn't justify it. Revisit if abuse surfaces. - -## Open questions - -1. **PDP-as-cartridge vs PDP-in-bridge** — does the PDP genuinely need to be a separate cartridge, or should it be an in-process module like `nickel-validator.js`? Arguments both ways. Recommend cartridge for separation of concerns and to enable per-deployment policy customisation. - -2. **Policy versioning** — when the bundle changes, in-flight quarantined approvals could have been written against a different version. Need a `policy_version` field on quarantined envelopes; reject approval if version doesn't match. - -3. **Cross-cartridge composition policies** — `boj_cartridge_invoke` against an arbitrary cartridge currently bypasses tool-level policy. Either (a) require every cartridge to declare its full tool surface in its manifest so the PDP knows the per-tool tier, or (b) apply the cartridge-level worst-case tier to all unrecognised tools. Recommend (a); it's already partially true for verified cartridges. - -4. **Migration sequence** — should the existing `hardeningGate` rate-limit logic be ported into the PDP (so all policy lives in one place) or left in the PEP for performance? Recommend PEP keeps rate-limit + injection-scan as fast-path guards, PDP handles the tier-aware logic. - -## Implementation sketch - -When this RFC is accepted: - -1. New cartridge `cartridges/policy-mcp/` with Idris2 ABI + Zig FFI + Deno adapter (standard triple). Manifest declares one tool `policy_evaluate` plus internal `policy_reload`. -2. `policies/boj-default.ncl` shipped at the repo root. 41 tools covered. -3. `mcp-bridge/lib/policy-client.js` — minimal Nickel-backed PDP client (in-process when local, REST when federated). -4. `mcp-bridge/main.js` — `hardeningGate` adds a step 6 ("PDP evaluation") and respects the verdict. -5. New env vars in `glama.json`: `BOJ_POLICY_MODE`, `BOJ_POLICY_BUNDLE`, `BOJ_POLICY_FAIL_MODE`. -6. Test suite: 1 test per verdict outcome (allow / deny / rate_limit / require_approval); fixture policy bundles for each case. -7. Documentation: a new `docs/operations/POLICY-AUTHORING.md` showing how operators customise the default bundle. - -## Linked - -- ADR-0002 (BoJ-only MCP) — establishes the single-gateway invariant this depends on. -- ADR-0008 (cartridge marketplace) — submission flow requires a default policy per cartridge. -- Epic #87 item 4 (this). -- `cartridges/local-coord-mcp/coord-messages.ncl` — vocabulary precedent. diff --git a/docs/decisions/0008-cartridge-marketplace.adoc b/docs/decisions/0008-cartridge-marketplace.adoc new file mode 100644 index 00000000..366ba011 --- /dev/null +++ b/docs/decisions/0008-cartridge-marketplace.adoc @@ -0,0 +1,251 @@ +== 8. Cartridge marketplace — discovery, submission, and Ayo-tier activation + +Date: 2026-05-20 + +=== Status + +Proposed (RFC — implementation tracked in epic #87 item 10) + +=== Context + +BoJ defines three trust tiers (see ADR-0007). *Teranga* (formally +verified) and *Shield* (security-reviewed) cartridges ship in this repo; +both tiers are populated. *Ayo* — described as "`community +contributions, master-approval-gated`" — has exactly *one* member at +present (`+local-coord-mcp+` in the offline manifest, though that’s +actually a first-party cartridge that landed in the Ayo tier +organisationally rather than as community work). + +The third tier exists in vocabulary but has no: + +* Submission protocol for third-party authors +* Discovery surface for users to find community cartridges +* Verification path from "`Ayo`" (community-submitted) → "`Shield`" +(security-reviewed) → "`Teranga`" (formally verified) +* Trust signal (signing, provenance, supply-chain claims) that lets BoJ +tell the user "`this is community, treat accordingly`" + +Without these, "`Ayo`" is effectively an inactive label. The estate +cannot grow beyond what the maintainer personally writes — counter to +BoJ’s consolidation mission, which depends on long-tail domain coverage. +ADR-0002 (BoJ-only MCP) means a third-party MCP server *cannot* be the +answer; capabilities must come into BoJ as cartridges. + +=== Decision + +Build a *federated cartridge marketplace* with three components, in +increasing order of trust: + +[arabic] +. *Marketplace index* — a content-addressed registry of community +cartridges (Ayo) +. *Submission protocol* — how a third-party author proposes a cartridge +for Ayo listing +. *Promotion path* — how an Ayo cartridge moves to Shield (security +review) and Shield → Teranga (formal verification) + +==== Marketplace index + +A new repository `+hyperpolymath/cartridge-index+` (separate from +boj-server). It is *not* a binary registry; it’s a manifest registry. +Each entry in `+index.a2ml+` (machine-readable, MPL-2.0) is: + +[source,a2ml] +---- +- name: example-mcp + version: 0.1.0 + tier: ayo + source: + type: git + url: https://github.com// + rev: <40-char-sha> # content-pinned, not branch-tracked + manifest_sha256: # of the source's cartridge.json at + abi_sha256: # of the Idris2 ABI .ttc bundle + signature: + algorithm: ML-DSA-87 # estate quantum-safe standard + public_key_id: # author's published key + signature: # over (name|version|source|manifest_sha256|abi_sha256) + author: + name: + contact: + domain: + short_description: <120 chars max> + declared_policy: policies/.ncl # mandatory per ADR-0007 + proof_obligations: [] # empty for Ayo by definition; populated when promoted to Shield + estate_review: null # populated only when promoted to Shield +---- + +BoJ reads the index by syncing the repo. A new bridge tool +`+boj_marketplace_search+` queries the synced index. A second tool +`+boj_marketplace_install+` performs the install dance (fetch source at +pinned rev, verify hashes + signature, build locally, place under +`+cartridges/+`). + +==== Submission protocol + +A community author submits a PR against +`+hyperpolymath/cartridge-index/inbox/+`. The PR contains: + +[arabic] +. The proposed index entry (signed) +. A link to the cartridge source repo at the pinned rev +. Evidence that the cartridge: +* Has a `+cartridge.json+` manifest matching the manifest_sha256 +* Has Idris2 ABI + Zig FFI + adapter (standard triple) +* Has a `+policies/.ncl+` file matching ADR-0007’s schema +* Has a passing `+npm test+` (or equivalent) on the submitted rev +* Carries `+SPDX-License-Identifier: CC-BY-SA-4.0+` (or OSI-approved +equivalent) +. A signed CLA-equivalent (`+SIGNED-OFF-BY+` in commit, with verified +email) + +A CI workflow in `+cartridge-index+` runs: + +* Hash verification (manifest + ABI) +* Signature verification (ML-DSA-87 against published keys) +* Policy schema validation (Nickel against ADR-0007 contract) +* Source-tree compliance (file structure matches cartridge convention) +* Per-author rate limit (max 5 submissions per 30 days; prevents spam) + +A maintainer reviews and merges. The cartridge becomes discoverable via +`+boj_marketplace_search+`. It is *never installed automatically*; the +user must explicitly `+boj_marketplace_install+` it. + +==== Promotion path + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|From |To |Requirements +|_outside_ |Ayo |Submission PR accepted; cartridge becomes installable +but never default-loaded + +|Ayo |Shield |(a) Hypatia neurosymbolic scan passes; (b) +`+panic-attack-mcp+` static analysis passes; (c) explicit security +review from a designated reviewer; (d) reviewer signs the promotion +entry; (e) `+estate_review+` field populated with reviewer + date + +scope + +|Shield |Teranga |(a) Idris2 ABI declares one or more proof obligations; +(b) all obligations discharged (no `+believe_me+`); (c) cross-cartridge +composition with at least three other Teranga cartridges proves safe; +(d) declared and discharged in `+boj://proofs/manifest+` +|=== + +Promotion is recorded in `+cartridge-index+` itself (the entry’s +`+tier+` and `+estate_review+` fields update). The cartridge source repo +does not change; only the index entry promotes. + +==== Default install posture + +Out of the box, BoJ only enables *Teranga* + *Shield* cartridges. Ayo +cartridges are visible (via `+boj_marketplace_search+`) but require +explicit user action (`+boj_marketplace_install+` + an +`+--accept-ayo-tier+` flag) to be loaded. Once installed, ADR-0007’s +policy engine treats them appropriately: tier-3/4 operations require +master approval. + +==== Discovery surface + +Two new bridge tools: + +* `+boj_marketplace_search+` — query the synced index by name / domain / +tier; read-only +* `+boj_marketplace_install+` — fetch + verify + build + activate a +community cartridge + +Plus a new MCP resource (per PR #89’s vocabulary): + +* `+boj://marketplace/index+` — full index dump (synced from +`+cartridge-index+` repo) +* `+boj://marketplace/cartridges/+` — single entry + +=== Consequences + +==== Positive + +* *Activates the Ayo tier* — gives it a concrete protocol, not just a +label. +* *Long-tail coverage without ADR-0002 violation* — community +capabilities flow into BoJ as cartridges, not as standalone MCPs. +* *Trust signal is explicit and verifiable* — ML-DSA-87 signatures, +content-pinned source, hash-verified ABI. Aligns with the estate’s +quantum-safe-provenance branding. +* *Promotion path matches existing vocabulary* — security review is what +makes Shield, formal verification is what makes Teranga. The marketplace +doesn’t invent new trust criteria; it formalises the ones BoJ already +implicitly uses. +* *Decouples release cadence* — community cartridges ship on their own +schedule via index PRs; boj-server proper doesn’t need to release for +the catalogue to grow. +* *Default-safe* — Ayo is never auto-loaded, never auto-enabled. The +user explicitly opts in per cartridge. + +==== Negative + +* *Maintenance burden on `+cartridge-index+`* — requires a designated +reviewer pool. Mitigation: federate the maintainer set; not +solo-bus-factor. +* *CLA / contribution legal surface* — needs careful drafting. Use +Developer Certificate of Origin (DCO) rather than custom CLA where +possible. +* *Build reproducibility* — community cartridge `+manifest_sha256+` is +straightforward, but the `+abi_sha256+` requires deterministic Idris2 +builds. Achievable but requires submitter discipline. +* *Discoverability vs noise* — the catalogue could fill with low-quality +cartridges. Mitigation: per-author rate limit + maintainer curation gate +at the inbox PR step. +* *Update protocol* — when an installed Ayo cartridge has a new version +in the index, how does the user learn? `+boj_marketplace_check_updates+` +tool emitting a notification (depends on ADR-0011 / item 5). + +=== Non-goals + +* Not building a paid/commercial marketplace. Open-source-only. +* Not building hosted binary distribution. The index points at source +repos; users build locally. This avoids supply-chain attacks against a +central tarball cache and keeps the trust model close to git provenance. +* Not requiring community cartridges to be formally verified for Ayo +listing. Verification is the _promotion_ criterion to Shield/Teranga, +not the entry bar to Ayo. +* Not auto-installing security updates for Ayo cartridges. Updates are +user-initiated only. Notifications (depends on ADR-0011) inform; they +don’t act. + +=== Open questions + +[arabic] +. *Index repository hosting* — `+hyperpolymath/cartridge-index+` on +GitHub is the obvious choice. Alternative: host on the hyperpolymath +estate forge (if/when there is one) for sovereignty. Recommend GitHub +for now; revisit if estate forge ships. +. *Reviewer compensation* — security review is labour. For +Shield/Teranga promotion, who pays? Initial answer: volunteer + +maintainer time. If this scales, consider a CodeFund-style sponsored +review fund. +. *Cartridge deprecation* — when an author abandons a cartridge or it’s +superseded by another, how is the index entry marked? Recommend +`+status: active | deprecated | superseded-by:+` field on every +entry. +. *Forking and divergence* — what if two authors submit cartridges that +wrap the same upstream service (e.g. two `+pinecone-mcp+` variants)? +Recommend allow both, list both, let users choose. Don’t pick favorites +in the index. +. *Estate-internal cartridge promotion* — does this protocol apply to +the 115 first-party cartridges already in boj-server? Recommend no: the +in-repo cartridges are subject to in-repo CI; the marketplace exists for +third-party submissions. The in-repo set is the "`starter catalogue`" +that initial users see. +. *What signs the index?* — the index repo itself is signed by +maintainers (per `+cartridge-index+` CI). Individual entries are also +signed by their authors. Both signatures verify on +`+boj_marketplace_install+`. + +=== Linked + +* ADR-0002 (BoJ-only MCP) — third-party capabilities cannot ship as +standalone MCPs; this is how they ship instead. +* ADR-0007 (trust-tier policy DSL) — mandates a default policy per +cartridge; this RFC requires it as a submission gate. +* Epic #87 item 10 (this) + item 4 (the policy DSL precondition). +* EXHIBIT-B (Quantum-Safe Provenance) — ML-DSA-87 standard reused here +for cartridge author signatures. diff --git a/docs/decisions/0008-cartridge-marketplace.md b/docs/decisions/0008-cartridge-marketplace.md deleted file mode 100644 index 796cac86..00000000 --- a/docs/decisions/0008-cartridge-marketplace.md +++ /dev/null @@ -1,161 +0,0 @@ - - - -# 8. Cartridge marketplace — discovery, submission, and Ayo-tier activation - -Date: 2026-05-20 - -## Status - -Proposed (RFC — implementation tracked in epic #87 item 10) - -## Context - -BoJ defines three trust tiers (see ADR-0007). **Teranga** (formally verified) and **Shield** (security-reviewed) cartridges ship in this repo; both tiers are populated. **Ayo** — described as "community contributions, master-approval-gated" — has exactly **one** member at present (`local-coord-mcp` in the offline manifest, though that's actually a first-party cartridge that landed in the Ayo tier organisationally rather than as community work). - -The third tier exists in vocabulary but has no: - -- Submission protocol for third-party authors -- Discovery surface for users to find community cartridges -- Verification path from "Ayo" (community-submitted) → "Shield" (security-reviewed) → "Teranga" (formally verified) -- Trust signal (signing, provenance, supply-chain claims) that lets BoJ tell the user "this is community, treat accordingly" - -Without these, "Ayo" is effectively an inactive label. The estate cannot grow beyond what the maintainer personally writes — counter to BoJ's consolidation mission, which depends on long-tail domain coverage. ADR-0002 (BoJ-only MCP) means a third-party MCP server **cannot** be the answer; capabilities must come into BoJ as cartridges. - -## Decision - -Build a **federated cartridge marketplace** with three components, in increasing order of trust: - -1. **Marketplace index** — a content-addressed registry of community cartridges (Ayo) -2. **Submission protocol** — how a third-party author proposes a cartridge for Ayo listing -3. **Promotion path** — how an Ayo cartridge moves to Shield (security review) and Shield → Teranga (formal verification) - -### Marketplace index - -A new repository `hyperpolymath/cartridge-index` (separate from boj-server). It is **not** a binary registry; it's a manifest registry. Each entry in `index.a2ml` (machine-readable, MPL-2.0) is: - -```a2ml -- name: example-mcp - version: 0.1.0 - tier: ayo - source: - type: git - url: https://github.com// - rev: <40-char-sha> # content-pinned, not branch-tracked - manifest_sha256: # of the source's cartridge.json at - abi_sha256: # of the Idris2 ABI .ttc bundle - signature: - algorithm: ML-DSA-87 # estate quantum-safe standard - public_key_id: # author's published key - signature: # over (name|version|source|manifest_sha256|abi_sha256) - author: - name: - contact: - domain: - short_description: <120 chars max> - declared_policy: policies/.ncl # mandatory per ADR-0007 - proof_obligations: [] # empty for Ayo by definition; populated when promoted to Shield - estate_review: null # populated only when promoted to Shield -``` - -BoJ reads the index by syncing the repo. A new bridge tool `boj_marketplace_search` queries the synced index. A second tool `boj_marketplace_install` performs the install dance (fetch source at pinned rev, verify hashes + signature, build locally, place under `cartridges/`). - -### Submission protocol - -A community author submits a PR against `hyperpolymath/cartridge-index/inbox/`. The PR contains: - -1. The proposed index entry (signed) -2. A link to the cartridge source repo at the pinned rev -3. Evidence that the cartridge: - - Has a `cartridge.json` manifest matching the manifest_sha256 - - Has Idris2 ABI + Zig FFI + adapter (standard triple) - - Has a `policies/.ncl` file matching ADR-0007's schema - - Has a passing `npm test` (or equivalent) on the submitted rev - - Carries `SPDX-License-Identifier: CC-BY-SA-4.0` (or OSI-approved equivalent) -4. A signed CLA-equivalent (`SIGNED-OFF-BY` in commit, with verified email) - -A CI workflow in `cartridge-index` runs: - -- Hash verification (manifest + ABI) -- Signature verification (ML-DSA-87 against published keys) -- Policy schema validation (Nickel against ADR-0007 contract) -- Source-tree compliance (file structure matches cartridge convention) -- Per-author rate limit (max 5 submissions per 30 days; prevents spam) - -A maintainer reviews and merges. The cartridge becomes discoverable via `boj_marketplace_search`. It is **never installed automatically**; the user must explicitly `boj_marketplace_install` it. - -### Promotion path - -| From | To | Requirements | -|---|---|---| -| _outside_ | Ayo | Submission PR accepted; cartridge becomes installable but never default-loaded | -| Ayo | Shield | (a) Hypatia neurosymbolic scan passes; (b) `panic-attack-mcp` static analysis passes; (c) explicit security review from a designated reviewer; (d) reviewer signs the promotion entry; (e) `estate_review` field populated with reviewer + date + scope | -| Shield | Teranga | (a) Idris2 ABI declares one or more proof obligations; (b) all obligations discharged (no `believe_me`); (c) cross-cartridge composition with at least three other Teranga cartridges proves safe; (d) declared and discharged in `boj://proofs/manifest` | - -Promotion is recorded in `cartridge-index` itself (the entry's `tier` and `estate_review` fields update). The cartridge source repo does not change; only the index entry promotes. - -### Default install posture - -Out of the box, BoJ only enables **Teranga** + **Shield** cartridges. Ayo cartridges are visible (via `boj_marketplace_search`) but require explicit user action (`boj_marketplace_install` + an `--accept-ayo-tier` flag) to be loaded. Once installed, ADR-0007's policy engine treats them appropriately: tier-3/4 operations require master approval. - -### Discovery surface - -Two new bridge tools: - -- `boj_marketplace_search` — query the synced index by name / domain / tier; read-only -- `boj_marketplace_install` — fetch + verify + build + activate a community cartridge - -Plus a new MCP resource (per PR #89's vocabulary): - -- `boj://marketplace/index` — full index dump (synced from `cartridge-index` repo) -- `boj://marketplace/cartridges/` — single entry - -## Consequences - -### Positive - -- **Activates the Ayo tier** — gives it a concrete protocol, not just a label. -- **Long-tail coverage without ADR-0002 violation** — community capabilities flow into BoJ as cartridges, not as standalone MCPs. -- **Trust signal is explicit and verifiable** — ML-DSA-87 signatures, content-pinned source, hash-verified ABI. Aligns with the estate's quantum-safe-provenance branding. -- **Promotion path matches existing vocabulary** — security review is what makes Shield, formal verification is what makes Teranga. The marketplace doesn't invent new trust criteria; it formalises the ones BoJ already implicitly uses. -- **Decouples release cadence** — community cartridges ship on their own schedule via index PRs; boj-server proper doesn't need to release for the catalogue to grow. -- **Default-safe** — Ayo is never auto-loaded, never auto-enabled. The user explicitly opts in per cartridge. - -### Negative - -- **Maintenance burden on `cartridge-index`** — requires a designated reviewer pool. Mitigation: federate the maintainer set; not solo-bus-factor. -- **CLA / contribution legal surface** — needs careful drafting. Use Developer Certificate of Origin (DCO) rather than custom CLA where possible. -- **Build reproducibility** — community cartridge `manifest_sha256` is straightforward, but the `abi_sha256` requires deterministic Idris2 builds. Achievable but requires submitter discipline. -- **Discoverability vs noise** — the catalogue could fill with low-quality cartridges. Mitigation: per-author rate limit + maintainer curation gate at the inbox PR step. -- **Update protocol** — when an installed Ayo cartridge has a new version in the index, how does the user learn? `boj_marketplace_check_updates` tool emitting a notification (depends on ADR-0011 / item 5). - -## Non-goals - -- Not building a paid/commercial marketplace. Open-source-only. -- Not building hosted binary distribution. The index points at source repos; users build locally. This avoids supply-chain attacks against a central tarball cache and keeps the trust model close to git provenance. -- Not requiring community cartridges to be formally verified for Ayo listing. Verification is the *promotion* criterion to Shield/Teranga, not the entry bar to Ayo. -- Not auto-installing security updates for Ayo cartridges. Updates are user-initiated only. Notifications (depends on ADR-0011) inform; they don't act. - -## Open questions - -1. **Index repository hosting** — `hyperpolymath/cartridge-index` on GitHub is the obvious choice. Alternative: host on the hyperpolymath estate forge (if/when there is one) for sovereignty. Recommend GitHub for now; revisit if estate forge ships. - -2. **Reviewer compensation** — security review is labour. For Shield/Teranga promotion, who pays? Initial answer: volunteer + maintainer time. If this scales, consider a CodeFund-style sponsored review fund. - -3. **Cartridge deprecation** — when an author abandons a cartridge or it's superseded by another, how is the index entry marked? Recommend `status: active | deprecated | superseded-by:` field on every entry. - -4. **Forking and divergence** — what if two authors submit cartridges that wrap the same upstream service (e.g. two `pinecone-mcp` variants)? Recommend allow both, list both, let users choose. Don't pick favorites in the index. - -5. **Estate-internal cartridge promotion** — does this protocol apply to the 115 first-party cartridges already in boj-server? Recommend no: the in-repo cartridges are subject to in-repo CI; the marketplace exists for third-party submissions. The in-repo set is the "starter catalogue" that initial users see. - -6. **What signs the index?** — the index repo itself is signed by maintainers (per `cartridge-index` CI). Individual entries are also signed by their authors. Both signatures verify on `boj_marketplace_install`. - -## Linked - -- ADR-0002 (BoJ-only MCP) — third-party capabilities cannot ship as standalone MCPs; this is how they ship instead. -- ADR-0007 (trust-tier policy DSL) — mandates a default policy per cartridge; this RFC requires it as a submission gate. -- Epic #87 item 10 (this) + item 4 (the policy DSL precondition). -- EXHIBIT-B (Quantum-Safe Provenance) — ML-DSA-87 standard reused here for cartridge author signatures. diff --git a/docs/decisions/0009-sandbox-cartridge.adoc b/docs/decisions/0009-sandbox-cartridge.adoc new file mode 100644 index 00000000..2626fe16 --- /dev/null +++ b/docs/decisions/0009-sandbox-cartridge.adoc @@ -0,0 +1,275 @@ +== 9. Sandbox cartridge — multi-provider, tier-gated code execution + +Date: 2026-05-20 + +=== Status + +Proposed (RFC — implementation tracked in epic #87 item 2) + +=== Context + +BoJ’s multi-agent architecture (master / journeyman / apprentice +supervision via `+local-coord-mcp+`) has no execution substrate. Agents +on the bus can call tools, read resources, claim tasks, but cannot *run +code*. This is a gap because: + +* Common agent workflows include "`write code, run it, observe output, +iterate`" — the canonical LLM-coding loop has nowhere to land +* The trust-tier model has no expression in execution: writing code is a +"`small_write`" (creating a file), but _running_ that code is closer to +"`destructive`" (arbitrary side effects) +* `+panic-attack-mcp+` does static analysis pre-execution, `+vordr-mcp+` +does post-execution integrity — there’s a deliberate gap in the middle +where execution should sit +* Existing cartridges that _do_ run code (`+browser-mcp+`’s +`+execute_js+`, container-mcp’s full container lifecycle) are +general-purpose and not tier-gated; they’re hard to reason about as +"`the place agents execute untrusted code`" + +The estate also can’t pull in standard tools without violating ADR-0002 +(BoJ-only MCP) — a third-party `+e2b-mcp+` or `+modal-mcp+` would be a +standalone MCP, which is forbidden. The capability must enter BoJ as a +cartridge. + +=== Decision + +Build *`+sandbox-mcp+`* — a multi-provider, tier-gated execution +cartridge. One MCP-facing cartridge surface; pluggable backends. + +==== Provider abstraction + +.... +sandbox-mcp + ├── ABI (Idris2) — Sandbox.idr, Execution.idr, Provider.idr + ├── FFI (Zig) — uniform sandbox_* calls + └── Adapter (Deno) — provider modules: + ├── e2b.js — e2b.dev firecracker microVMs + ├── modal.js — Modal.com containers + ├── codesandbox.js — CodeSandbox sandpack runners + ├── replit.js — Replit Repl Spaces + └── local.js — local Podman + bubblewrap (no network) +.... + +Provider selection by env +(`+SANDBOX_PROVIDER=e2b|modal|codesandbox|replit|local+`) or per-call +argument. The MCP surface is provider-agnostic; calls don’t change shape +when switching backends. + +==== Tool surface + +.... +sandbox_create — provision a fresh sandbox; returns sandbox_id +sandbox_exec — execute a command/script inside an existing sandbox +sandbox_read — read a file from inside the sandbox +sandbox_write — write a file into the sandbox (input only) +sandbox_install — install a language toolchain or package +sandbox_destroy — release the sandbox; idempotent +sandbox_list — list active sandboxes belonging to the calling peer +.... + +Each tool maps to one provider call. `+sandbox_exec+` is the +high-frequency one; the rest exist so the LLM doesn’t have to redo +provisioning across multiple `+exec+` calls. + +==== Tier model + +Per ADR-0007: + +[width="100%",cols="25%,25%,25%,25%",options="header",] +|=== +|Tool |Tier |Required role |Master approval +|`+sandbox_create+` |2 |apprentice |no — but the sandbox itself is +bounded by the tier of operations it can run + +|`+sandbox_exec+` |2-4 (depends on capabilities) |journeyman |only if +`+capabilities+` includes `+network+` + +|`+sandbox_read+` |1 |apprentice |no + +|`+sandbox_write+` |1 |apprentice |no — write into sandbox, not host + +|`+sandbox_install+` |2 |journeyman |no — bounded to sandbox + +|`+sandbox_destroy+` |0 |apprentice |no + +|`+sandbox_list+` |0 |apprentice |no — read-only +|=== + +`+sandbox_exec+` is the policy hot-spot. Each sandbox has declared +*capabilities* at create-time: + +.... +capabilities: { + network: boolean, # internet access + filesystem: "ro" | "rw", + duration_s: number, # hard timeout, max 1800 + memory_mb: number, # max 4096 + cpu_quota: number, # 0.1 .. 4.0 cores +} +.... + +`+network: true+` flips `+sandbox_exec+` from tier-2 to tier-4 (per +policy). The policy engine (ADR-0007) computes the effective tier +per-call from the sandbox’s capabilities + the calling peer’s role. + +==== Provider differences (deliberately surfaced) + +The cartridge does *not* abstract over provider semantics in a leaky +way. Five provider properties are exposed as part of +`+sandbox_create+`’s response: + +* `+isolation_level+`: `+microvm+` | `+container+` | `+process+` — e2b +is microvm, Modal is container, local-bubblewrap is process +* `+cold_start_ms+`: provider-typical, helps the LLM decide whether to +reuse vs. create +* `+language_support+`: declared by provider; +`+["python", "node", "rust", "go"]+` etc. +* `+egress_policy+`: `+none+` | `+allowlist:+` | `+unrestricted+` +* `+attestation+`: hash/signature of the sandbox image (Modal supports +this; e2b partial; local depends on bubblewrap setup) + +This lets the LLM (or the policy engine) make informed choices. If the +user policy requires `+attestation: required+`, `+local+` provider is +the only valid choice. + +==== Wiring into `+panic-attack-mcp+` and `+vordr-mcp+` + +The execution lifecycle is: + +[arabic] +. *Pre-flight* (optional but encouraged): `+panic-attack_scan+` on the +code about to be executed. Static analysis catches banned constructs +before runtime. +. *Execution*: `+sandbox_create+` + `+sandbox_write+` + +`+sandbox_exec+`. +. *Post-flight* (optional): `+vordr_verify+` on artefacts the sandbox +produced, before they leave the sandbox boundary. + +These wirings are _recommendations_, not enforcements. The LLM composes +them via the `+audit-repo+`-style prompt patterns from PR #89. A future +`+execute-untrusted-code+` prompt template can encode the canonical +3-step flow. + +==== Cleanup discipline + +Sandboxes are bound to peer tokens (`+coord_register+`). When a peer’s +session ends (token expires, coord watchdog fires), `+sandbox-mcp+` +releases all sandboxes owned by that peer. Prevents orphan-sandbox cost +explosions. + +=== Consequences + +==== Positive + +* *Closes the execution gap* — agents on the BoJ bus now have a +tier-gated place to run code, instead of either (a) doing nothing useful +with code or (b) executing in `+browser-mcp.execute_js+` which has wrong +tier semantics. +* *One MCP surface, many providers* — addresses the "`use e2b for now, +switch to Modal later`" reality without disrupting consuming prompts. +* *Tier expression in execution* — `+capabilities.network: true+` +flipping `+sandbox_exec+` to tier-4 makes the tier system load-bearing +for the highest-blast-radius operation BoJ performs. +* *Composes with existing cartridges* — `+panic-attack-mcp+` pre-flight ++ `+sandbox-mcp+` execution + `+vordr-mcp+` post-flight is the +architecturally-shaped flow. Three separate cartridges, one workflow. +* *Provider differences exposed honestly* — `+isolation_level+`, +`+attestation+`, `+egress_policy+` published per provider; LLM (and +policy) can choose appropriately. +* *Bounded resources* — every sandbox has duration/memory/CPU caps, set +at create-time and enforced by the provider. + +==== Negative + +* *External-provider dependency* — three of five backends (e2b, Modal, +CodeSandbox, Replit) are paid SaaS. Pricing structure varies. +Mitigation: `+local+` backend (Podman + bubblewrap) for users who refuse +SaaS dependency; documented as the "`always-available`" floor. +* *API drift* — provider APIs change. Mitigation: pin provider SDK +versions in the cartridge’s Deno imports; expose `+provider_version+` in +`+sandbox_list+` so drift is observable. +* *Auth surface* — each provider needs an API key (`+E2B_API_KEY+`, +`+MODAL_TOKEN+`, etc.). Adds ~5 env vars to glama.json. +* *Cost surprises* — LLM agents can spin up sandboxes faster than humans +monitor cost. Mitigation: per-peer rate limits via ADR-0007 policy (max +N sandboxes per hour, max M minutes total runtime per day). +* *Sandbox-as-pivot risk* — a compromised sandbox with `+network: true+` +becomes a pivot point for outbound attacks. Mitigation: +`+egress_policy+` strict default (no internet); requires explicit policy +override to enable. + +=== Non-goals + +* *Not a generic compute provider* — sandbox-mcp is for short-lived, +LLM-driven code execution. Not for long-running services. Use +`+container-mcp+`, `+k8s-mcp+`, or provider-specific cartridges for +those. +* *Not a CI runner* — `+buildkite-mcp+` / `+circleci-mcp+` / +`+laminar-mcp+` cover CI. Sandbox is for ad-hoc agent execution. +* *Not a debugger* — `+dap-mcp+` exists for that. +* *Not state-preserving across peer sessions* — sandbox lifetime ≤ peer +session lifetime. No "`resume yesterday’s sandbox`". Persisting requires +a different cartridge. +* *Not GPU-enabled in v1* — provider APIs for GPU exist (Modal, Replit), +but tier-gating GPU usage adds complexity. Revisit in a follow-up. + +=== Open questions + +[arabic] +. *Default provider* — `+local+` (always available, no SaaS) or `+e2b+` +(best LLM-coding ergonomics)? Recommend `+local+` as default with strong +docs on enabling SaaS providers for production use. +. *Sandbox sharing* — can two peers on the coord bus share a sandbox +(e.g. journeyman creates, apprentice executes)? Risk: cross-peer +privilege escalation. Recommend disallow for v1; revisit when there’s a +concrete use case. +. *Output streaming* — `+sandbox_exec+` is naturally streaming +(stdout/stderr come over time). MCP doesn’t have great primitives for +this yet. v1: return on completion with full output; v2: SSE streaming +once ADR-0011 (webhooks/notifications) lands. +. *Filesystem semantics* — when does `+sandbox_write+` accept binary +content? Base64? Streaming? Recommend base64 for v1 (simplest); revisit +with chunked streaming when SSE lands. +. *Network policy granularity* — `+egress_policy+` as an allow-list of +URLs is straightforward for e2b/Modal but harder for `+local+` (would +need iptables rules in bubblewrap). Recommend allowlist for SaaS +providers; documented "`best-effort`" for local. +. *`+sandbox-mcp+` vs container-mcp scope* — there’s overlap with +`+container-mcp+`’s lifecycle. Recommend: `+container-mcp+` is for +managing persistent containers (services, databases, build +environments); `+sandbox-mcp+` is for ephemeral, untrusted, agent-driven +execution. Both can coexist; their tier and lifetime semantics differ. + +=== Implementation sketch + +When this RFC is accepted: + +[arabic] +. New cartridge `+cartridges/sandbox-mcp/+` with the standard +Idris2/Zig/Deno triple. +. Five provider adapters under +`+cartridges/sandbox-mcp/adapter/providers/+`. `+local+` lands first (no +SaaS dependency to bring up); `+e2b+` second (best LLM-coding +ergonomics); others in subsequent PRs. +. Manifest declares 7 tools +(`+sandbox_create+`/`+exec+`/`+read+`/`+write+`/`+install+`/`+destroy+`/`+list+`). +. New default policy entries in `+policies/boj-default.ncl+` (depends on +ADR-0007 landing first). +. New env vars: `+SANDBOX_PROVIDER+`, `+E2B_API_KEY+`, `+MODAL_TOKEN+`, +etc. — declared in glama.json. +. Tests: per-provider integration suite (mocked transport for SaaS, real +bubblewrap for local). +. Documentation: `+docs/cartridges/SANDBOX-MCP.md+` covering provider +selection, policy authoring, and the panic-attack → sandbox → vordr +flow. + +=== Linked + +* ADR-0007 (trust-tier policy DSL) — the policy bundle is the +precondition for tier-gated execution. +* ADR-0010 (cross-machine federation) — sandboxes are _machine-local_; +cross-machine peer cannot share a sandbox handle. Federation needs to be +aware. +* `+panic-attack-mcp+` cartridge — pre-flight static analysis. +* `+vordr-mcp+` cartridge — post-flight integrity verification. +* Epic #87 item 2 (this). diff --git a/docs/decisions/0009-sandbox-cartridge.md b/docs/decisions/0009-sandbox-cartridge.md deleted file mode 100644 index 794a2194..00000000 --- a/docs/decisions/0009-sandbox-cartridge.md +++ /dev/null @@ -1,173 +0,0 @@ - - - -# 9. Sandbox cartridge — multi-provider, tier-gated code execution - -Date: 2026-05-20 - -## Status - -Proposed (RFC — implementation tracked in epic #87 item 2) - -## Context - -BoJ's multi-agent architecture (master / journeyman / apprentice supervision via `local-coord-mcp`) has no execution substrate. Agents on the bus can call tools, read resources, claim tasks, but cannot **run code**. This is a gap because: - -- Common agent workflows include "write code, run it, observe output, iterate" — the canonical LLM-coding loop has nowhere to land -- The trust-tier model has no expression in execution: writing code is a "small_write" (creating a file), but *running* that code is closer to "destructive" (arbitrary side effects) -- `panic-attack-mcp` does static analysis pre-execution, `vordr-mcp` does post-execution integrity — there's a deliberate gap in the middle where execution should sit -- Existing cartridges that *do* run code (`browser-mcp`'s `execute_js`, container-mcp's full container lifecycle) are general-purpose and not tier-gated; they're hard to reason about as "the place agents execute untrusted code" - -The estate also can't pull in standard tools without violating ADR-0002 (BoJ-only MCP) — a third-party `e2b-mcp` or `modal-mcp` would be a standalone MCP, which is forbidden. The capability must enter BoJ as a cartridge. - -## Decision - -Build **`sandbox-mcp`** — a multi-provider, tier-gated execution cartridge. One MCP-facing cartridge surface; pluggable backends. - -### Provider abstraction - -``` -sandbox-mcp - ├── ABI (Idris2) — Sandbox.idr, Execution.idr, Provider.idr - ├── FFI (Zig) — uniform sandbox_* calls - └── Adapter (Deno) — provider modules: - ├── e2b.js — e2b.dev firecracker microVMs - ├── modal.js — Modal.com containers - ├── codesandbox.js — CodeSandbox sandpack runners - ├── replit.js — Replit Repl Spaces - └── local.js — local Podman + bubblewrap (no network) -``` - -Provider selection by env (`SANDBOX_PROVIDER=e2b|modal|codesandbox|replit|local`) or per-call argument. The MCP surface is provider-agnostic; calls don't change shape when switching backends. - -### Tool surface - -``` -sandbox_create — provision a fresh sandbox; returns sandbox_id -sandbox_exec — execute a command/script inside an existing sandbox -sandbox_read — read a file from inside the sandbox -sandbox_write — write a file into the sandbox (input only) -sandbox_install — install a language toolchain or package -sandbox_destroy — release the sandbox; idempotent -sandbox_list — list active sandboxes belonging to the calling peer -``` - -Each tool maps to one provider call. `sandbox_exec` is the high-frequency one; the rest exist so the LLM doesn't have to redo provisioning across multiple `exec` calls. - -### Tier model - -Per ADR-0007: - -| Tool | Tier | Required role | Master approval | -|---|---|---|---| -| `sandbox_create` | 2 | apprentice | no — but the sandbox itself is bounded by the tier of operations it can run | -| `sandbox_exec` | 2-4 (depends on capabilities) | journeyman | only if `capabilities` includes `network` | -| `sandbox_read` | 1 | apprentice | no | -| `sandbox_write` | 1 | apprentice | no — write into sandbox, not host | -| `sandbox_install` | 2 | journeyman | no — bounded to sandbox | -| `sandbox_destroy` | 0 | apprentice | no | -| `sandbox_list` | 0 | apprentice | no — read-only | - -`sandbox_exec` is the policy hot-spot. Each sandbox has declared **capabilities** at create-time: - -``` -capabilities: { - network: boolean, # internet access - filesystem: "ro" | "rw", - duration_s: number, # hard timeout, max 1800 - memory_mb: number, # max 4096 - cpu_quota: number, # 0.1 .. 4.0 cores -} -``` - -`network: true` flips `sandbox_exec` from tier-2 to tier-4 (per policy). The policy engine (ADR-0007) computes the effective tier per-call from the sandbox's capabilities + the calling peer's role. - -### Provider differences (deliberately surfaced) - -The cartridge does **not** abstract over provider semantics in a leaky way. Five provider properties are exposed as part of `sandbox_create`'s response: - -- `isolation_level`: `microvm` | `container` | `process` — e2b is microvm, Modal is container, local-bubblewrap is process -- `cold_start_ms`: provider-typical, helps the LLM decide whether to reuse vs. create -- `language_support`: declared by provider; `["python", "node", "rust", "go"]` etc. -- `egress_policy`: `none` | `allowlist:` | `unrestricted` -- `attestation`: hash/signature of the sandbox image (Modal supports this; e2b partial; local depends on bubblewrap setup) - -This lets the LLM (or the policy engine) make informed choices. If the user policy requires `attestation: required`, `local` provider is the only valid choice. - -### Wiring into `panic-attack-mcp` and `vordr-mcp` - -The execution lifecycle is: - -1. **Pre-flight** (optional but encouraged): `panic-attack_scan` on the code about to be executed. Static analysis catches banned constructs before runtime. -2. **Execution**: `sandbox_create` + `sandbox_write` + `sandbox_exec`. -3. **Post-flight** (optional): `vordr_verify` on artefacts the sandbox produced, before they leave the sandbox boundary. - -These wirings are *recommendations*, not enforcements. The LLM composes them via the `audit-repo`-style prompt patterns from PR #89. A future `execute-untrusted-code` prompt template can encode the canonical 3-step flow. - -### Cleanup discipline - -Sandboxes are bound to peer tokens (`coord_register`). When a peer's session ends (token expires, coord watchdog fires), `sandbox-mcp` releases all sandboxes owned by that peer. Prevents orphan-sandbox cost explosions. - -## Consequences - -### Positive - -- **Closes the execution gap** — agents on the BoJ bus now have a tier-gated place to run code, instead of either (a) doing nothing useful with code or (b) executing in `browser-mcp.execute_js` which has wrong tier semantics. -- **One MCP surface, many providers** — addresses the "use e2b for now, switch to Modal later" reality without disrupting consuming prompts. -- **Tier expression in execution** — `capabilities.network: true` flipping `sandbox_exec` to tier-4 makes the tier system load-bearing for the highest-blast-radius operation BoJ performs. -- **Composes with existing cartridges** — `panic-attack-mcp` pre-flight + `sandbox-mcp` execution + `vordr-mcp` post-flight is the architecturally-shaped flow. Three separate cartridges, one workflow. -- **Provider differences exposed honestly** — `isolation_level`, `attestation`, `egress_policy` published per provider; LLM (and policy) can choose appropriately. -- **Bounded resources** — every sandbox has duration/memory/CPU caps, set at create-time and enforced by the provider. - -### Negative - -- **External-provider dependency** — three of five backends (e2b, Modal, CodeSandbox, Replit) are paid SaaS. Pricing structure varies. Mitigation: `local` backend (Podman + bubblewrap) for users who refuse SaaS dependency; documented as the "always-available" floor. -- **API drift** — provider APIs change. Mitigation: pin provider SDK versions in the cartridge's Deno imports; expose `provider_version` in `sandbox_list` so drift is observable. -- **Auth surface** — each provider needs an API key (`E2B_API_KEY`, `MODAL_TOKEN`, etc.). Adds ~5 env vars to glama.json. -- **Cost surprises** — LLM agents can spin up sandboxes faster than humans monitor cost. Mitigation: per-peer rate limits via ADR-0007 policy (max N sandboxes per hour, max M minutes total runtime per day). -- **Sandbox-as-pivot risk** — a compromised sandbox with `network: true` becomes a pivot point for outbound attacks. Mitigation: `egress_policy` strict default (no internet); requires explicit policy override to enable. - -## Non-goals - -- **Not a generic compute provider** — sandbox-mcp is for short-lived, LLM-driven code execution. Not for long-running services. Use `container-mcp`, `k8s-mcp`, or provider-specific cartridges for those. -- **Not a CI runner** — `buildkite-mcp` / `circleci-mcp` / `laminar-mcp` cover CI. Sandbox is for ad-hoc agent execution. -- **Not a debugger** — `dap-mcp` exists for that. -- **Not state-preserving across peer sessions** — sandbox lifetime ≤ peer session lifetime. No "resume yesterday's sandbox". Persisting requires a different cartridge. -- **Not GPU-enabled in v1** — provider APIs for GPU exist (Modal, Replit), but tier-gating GPU usage adds complexity. Revisit in a follow-up. - -## Open questions - -1. **Default provider** — `local` (always available, no SaaS) or `e2b` (best LLM-coding ergonomics)? Recommend `local` as default with strong docs on enabling SaaS providers for production use. - -2. **Sandbox sharing** — can two peers on the coord bus share a sandbox (e.g. journeyman creates, apprentice executes)? Risk: cross-peer privilege escalation. Recommend disallow for v1; revisit when there's a concrete use case. - -3. **Output streaming** — `sandbox_exec` is naturally streaming (stdout/stderr come over time). MCP doesn't have great primitives for this yet. v1: return on completion with full output; v2: SSE streaming once ADR-0011 (webhooks/notifications) lands. - -4. **Filesystem semantics** — when does `sandbox_write` accept binary content? Base64? Streaming? Recommend base64 for v1 (simplest); revisit with chunked streaming when SSE lands. - -5. **Network policy granularity** — `egress_policy` as an allow-list of URLs is straightforward for e2b/Modal but harder for `local` (would need iptables rules in bubblewrap). Recommend allowlist for SaaS providers; documented "best-effort" for local. - -6. **`sandbox-mcp` vs container-mcp scope** — there's overlap with `container-mcp`'s lifecycle. Recommend: `container-mcp` is for managing persistent containers (services, databases, build environments); `sandbox-mcp` is for ephemeral, untrusted, agent-driven execution. Both can coexist; their tier and lifetime semantics differ. - -## Implementation sketch - -When this RFC is accepted: - -1. New cartridge `cartridges/sandbox-mcp/` with the standard Idris2/Zig/Deno triple. -2. Five provider adapters under `cartridges/sandbox-mcp/adapter/providers/`. `local` lands first (no SaaS dependency to bring up); `e2b` second (best LLM-coding ergonomics); others in subsequent PRs. -3. Manifest declares 7 tools (`sandbox_create`/`exec`/`read`/`write`/`install`/`destroy`/`list`). -4. New default policy entries in `policies/boj-default.ncl` (depends on ADR-0007 landing first). -5. New env vars: `SANDBOX_PROVIDER`, `E2B_API_KEY`, `MODAL_TOKEN`, etc. — declared in glama.json. -6. Tests: per-provider integration suite (mocked transport for SaaS, real bubblewrap for local). -7. Documentation: `docs/cartridges/SANDBOX-MCP.md` covering provider selection, policy authoring, and the panic-attack → sandbox → vordr flow. - -## Linked - -- ADR-0007 (trust-tier policy DSL) — the policy bundle is the precondition for tier-gated execution. -- ADR-0010 (cross-machine federation) — sandboxes are *machine-local*; cross-machine peer cannot share a sandbox handle. Federation needs to be aware. -- `panic-attack-mcp` cartridge — pre-flight static analysis. -- `vordr-mcp` cartridge — post-flight integrity verification. -- Epic #87 item 2 (this). diff --git a/docs/decisions/0010-cross-machine-coord-federation.adoc b/docs/decisions/0010-cross-machine-coord-federation.adoc new file mode 100644 index 00000000..09161157 --- /dev/null +++ b/docs/decisions/0010-cross-machine-coord-federation.adoc @@ -0,0 +1,286 @@ +== 10. Cross-machine coord federation — DID identity, ML-KEM exchange, federated quarantine + +Date: 2026-05-20 + +=== Status + +Proposed (RFC — implementation tracked in epic #87 item 3) + +=== Context + +`+local-coord-mcp+` provides multi-agent coordination over a loopback +bus (`+127.0.0.1:7745+`). Within one machine this works well — peer +registration, typed envelopes, claim/heartbeat/watchdog, +master-supervised quarantine. Cross-machine coordination is the major +missing axis. + +The v0.1.0 changelog mentions "`Umoja federation with QUIC+UDP gossip +protocol (40 tests)`". Status today is unclear: the gossip code exists +in places but isn’t wired into the active coord cartridge, and there’s +no documented identity model, no key-exchange story, and no federated +quarantine semantics. Cross-machine multi-agent is effectively +unsupported. + +Why this matters now: + +* BoJ’s three trust tiers and master/journeyman/apprentice supervision +are the _differentiated_ part of the architecture. They become much more +useful when "`master`" can be on a different machine than "`apprentice`" +(a security boundary). +* The sandbox cartridge (ADR-0009) is machine-local by construction; +meaningful agent-cluster work spans multiple machines, each with its own +sandboxes. +* EXHIBIT-B (Quantum-Safe Provenance) and `+stapeln.toml+`’s ML-DSA-87 +signing already commit BoJ to a post-quantum posture. Federation is the +natural place to make this load-bearing on the wire, not just in +artefacts. +* ADR-0002 (BoJ-only MCP) means agents on different machines cannot +federate via "`two separate MCPs talking`" — federation has to be a +property of one cartridge family (coord-mcp) spanning machines. + +=== Decision + +Promote `+local-coord-mcp+` from loopback-only to *federated coord-mcp*, +with three architectural pillars: + +==== 1. DID-based peer identity + +Replace today’s ad-hoc peer IDs (`+-<4hex>[@]+`) with +*Decentralised Identifiers* (DIDs) per W3C DID-Core 1.0: + +.... +did:boj:peer: +.... + +* The `+boj+` method is a new method registered in the estate’s DID +registry repo +* Peer keypair generation is local (ML-DSA-87 per estate standard); +private key never leaves the machine +* DIDs are resolved via the same `+cartridge-index+` infrastructure used +in ADR-0008 (federated, content-addressed) +* Multiple peers on one machine share the machine’s identity but have +distinct peer IDs _under_ that DID (`+did:boj:peer:.../peer-id/+`) +* The existing `+coord_register+` token stays as the _session_ token; +the DID is the _identity_ + +==== 2. ML-KEM key exchange + encrypted envelopes + +When peer A (on machine α) wants to talk to peer B (on machine β): + +[arabic] +. *Discovery* — A queries the federated peer registry (peer publishes +its DID + machine endpoint via DNS-SD or static config) to resolve B’s +transport URL + KEM public key +. *Handshake* — A and B perform ML-KEM-1024 ephemeral key exchange +(estate post-quantum standard), deriving a session AEAD key +. *Signed-and-encrypted envelopes* — coord messages between A and B are +wrapped in (ML-DSA-87 signature by sender) → (ChaCha20-Poly1305 AEAD +with the derived session key) → wire bytes +. *Transport* — HTTPS POST to B’s coord federation endpoint +(`+/coord/federated/inbox+`), or QUIC if both peers advertise it + +Envelopes are otherwise the same Nickel-validated A2ML envelopes the +loopback bus uses today. Federation is a transport + crypto + identity +layer; the _semantics_ don’t change. + +==== 3. Federated quarantine + +The master-approval flow already exists for tier-2+ operations within a +machine. Federating it requires three new behaviours: + +*a. Master visibility across machines* + +A single master peer can be on any machine in the federation. Peers on +other machines route their quarantine entries to that master’s machine. +The master sees a unified queue via `+coord_review+` regardless of +source. + +*b. Master-uniqueness invariant federation-wide* + +Proof obligation *P-04* (master uniqueness) currently holds within one +machine. The federated version requires consensus: at most one peer in +the federation holds the `+master+` role at any time. Mechanism: +lightweight HOTSTUFF-style election with the federation’s DID set; the +elected master broadcasts a signed "`I-am-master`" attestation on every +coord heartbeat, and peers reject impostor master messages. + +*c. Cross-machine handoff* + +`+coord_transfer_master+` already supports same-machine handoff. +Federated version routes the handoff envelope cross-machine and verifies +the successor’s DID is in the federation’s roster before the handoff +completes. + +==== Topology variants + +*Mesh* (default): every peer talks to every other peer. Suitable for +small federations (≤16 peers). No coordinator. + +*Hub-and-spoke*: one "`rendezvous`" peer holds the master role and +routes all cross-machine traffic. Simpler firewall posture; single point +of failure. + +*Hub-and-rim* (recommended for production): one rendezvous _machine_ +(running a dedicated `+coord-mcp+` instance with no LLM-peer attached) +routes the federation. Master peer can move between machines but the +routing endpoint is stable. Failure of any LLM-peer machine doesn’t take +down the federation. + +Operators choose via `+COORD_FEDERATION_TOPOLOGY=mesh|hub|hub-and-rim+`. + +==== Trust posture + +A federation is *opt-in per machine*: + +* Default: `+COORD_FEDERATED=false+` (current loopback-only behaviour) +* Enabled: `+COORD_FEDERATED=true+` + +`+COORD_FEDERATION_ROSTER=+` +* The roster is a signed file listing the DIDs of all participating +machines, plus the master-election rules +* A peer never accepts envelopes from a DID not in the roster + +=== Consequences + +==== Positive + +* *Federation makes the tier model meaningfully secure* — apprentice on +one machine, master on another machine = real security boundary, not +just role labelling. +* *Post-quantum on the wire* — ML-KEM + ML-DSA-87 brings the wire +surface to the same posture as the artefact-signing surface. Aligns +EXHIBIT-B end-to-end. +* *DID identity is decentralised* — no central authority issues peer +IDs; private keys never leave the machine. Aligns with BoJ’s sovereignty +branding. +* *Existing semantics preserved* — A2ML envelopes, Nickel contracts, +watchdog TTLs all unchanged. Federation is purely a transport + identity +layer. +* *Production-grade topology* — hub-and-rim addresses the "`what if my +master machine goes down`" question that mesh and hub don’t. +* *Composable with sandbox-mcp* — federated peers can each run their own +sandboxes; the federation message bus coordinates which peer does what, +even though sandboxes themselves remain machine-local. +* *Reuses estate infrastructure* — `+cartridge-index+` (ADR-0008) for +DID resolution, ML-DSA-87 / ML-KEM standards (EXHIBIT-B) for crypto. No +new infrastructure repos needed. + +==== Negative + +* *Complexity escalation* — federation is the heaviest item in epic #87. +Multi-week implementation; likely the longest-running campaign. +* *Crypto correctness surface* — ML-KEM handshake + AEAD framing + +signature verification has many failure modes. Mitigation: use +well-reviewed libraries (libsodium-style high-level constructions); +proof obligations on the crypto handshake state machine. +* *Latency* — cross-machine RTT becomes part of the coord critical path. +Mitigation: keep claim/heartbeat _machine-local_; only quarantine review ++ master attestation cross machines. +* *Roster management* — operators must maintain the signed +`+COORD_FEDERATION_ROSTER+` file. Mitigation: tooling +(`+coord_federation_add_peer+` / `+coord_federation_remove_peer+` tools +that re-sign the roster atomically). +* *NAT/firewall traversal* — hub-and-rim avoids most of this; mesh has +full N×N connectivity requirements. Document this in operator guide; +recommend hub-and-rim for any deployment beyond a developer’s laptop. + +=== Non-goals + +* *Not building a generic federated agent framework* — this is BoJ’s +coord cartridge growing legs, not a competitor to libp2p or similar. +Borrow ideas; don’t reimplement. +* *Not supporting plaintext federation* — there is no +`+COORD_FEDERATION_INSECURE=true+` flag. Federation without crypto is +not federation. +* *Not making federation transparent to existing prompts* — the +`+convene-cluster+` prompt (PR #89) implicitly assumed loopback; +federated version surfaces the federation roster + topology so the LLM +can reason about which machine to dispatch to. +* *Not federating sandbox handles* — see ADR-0009 explicit constraint: +sandbox lifetime is machine-bound. Federation coordinates _which peer +runs a sandbox_, not _passing live sandbox handles across machines_. +* *Not implementing Byzantine fault tolerance* — federation assumes the +roster’s signing keys are honest. Compromised-key scenarios require +roster rotation, not BFT consensus. + +=== Open questions + +[arabic] +. *DID method registration* — register `+did:boj:+` in the W3C DID +method registry, or stay in a `+private:+` namespace? Recommend +register; the estate is open-source and the method is novel enough to be +reviewable. +. *Roster mutability* — when a new peer joins, every existing peer needs +the updated roster. Manual push? Gossip? Recommend manual signed roster +updates pushed by current master to all peers, ratified by signature +verification at receipt. +. *What if the master-election partitions* — two halves of a network +partition each elect a master. Recommend split-brain detection on +partition heal: the higher-attestation-epoch master wins; the other +voluntarily demotes. Lose all in-flight quarantine work in the losing +partition. +. *Cross-machine `+coord_send_gated+` quarantine retention* — quarantine +queue is currently in-memory on the master’s machine. Cross-machine +durability requires either (a) durable queue on every machine that +mirrors the master’s queue, or (b) the master persists to durable +storage local to it. Recommend (b) for v1; revisit if master-machine +failure recovery is a hard requirement. +. *Performance ceiling* — the design comfortably handles ≤16 peers +across ≤8 machines. Beyond that, mesh becomes O(N²); hub-and-rim scales +further but eventually needs sharding. Document the ceiling explicitly; +revisit if it bites. +. *Bootstrap with no DNS-SD* — pure static config +(`+COORD_FEDERATION_ROSTER+` lists every peer’s endpoint) works but +isn’t auto-discoverable. Recommend ship both modes; let operator choose. + +=== Implementation sketch + +When this RFC is accepted (multi-stage): + +*Stage 1 — DID identity* (~1 week) - Add +`+cartridges/local-coord-mcp/abi/LocalCoord/DID.idr+` with method +definition - Generate ML-DSA-87 keypair on first `+coord_register+`; +persist to `+~/.boj/coord/identity.json+` - New tool `+coord_did+` +returns the local peer’s DID - Loopback-only; no federation yet + +*Stage 2 — ML-KEM handshake* (~1 week) - Add +`+cartridges/local-coord-mcp/abi/LocalCoord/Handshake.idr+` - +ML-KEM-1024 keypair generated alongside ML-DSA-87 keypair - New private +state-machine ABI; not tool-exposed yet + +*Stage 3 — Encrypted wire format* (~1 week) - Federated envelope wrap: +`+(signature, AEAD(plaintext, session_key))+` - New A2ML schema +`+coord-federated-envelope.ncl+` (Nickel) - Round-trip tests; no network +yet + +*Stage 4 — Network transport* (~1 week) - HTTPS POST endpoint at +`+/coord/federated/inbox+` - Optional QUIC transport behind feature flag +- Static roster only; no discovery + +*Stage 5 — Federated master election + quarantine* (~1 week) - +HOTSTUFF-style election among roster - Quarantine routing to master’s +machine - Federated `+coord_review+` / `+coord_approve+` / +`+coord_reject+` + +*Stage 6 — Hub-and-rim topology + operator tooling* (~1 week) - +`+coord-rendezvous-mcp+` cartridge for the rim role - Roster-management +tooling - Federation-aware prompts (replace loopback assumptions in +`+convene-cluster+`) + +Total: ~6 weeks of focused work. Stages 1–3 are foundational; 4–6 +progressively expose the federation to users. + +=== Linked + +* EXHIBIT-B (Quantum-Safe Provenance) — establishes ML-DSA-87, applied +here. +* ADR-0002 (BoJ-only MCP) — federation must be in-cartridge, not +in-MCP-server. +* ADR-0007 (trust-tier policy DSL) — federation roster is a policy +artefact. +* ADR-0008 (cartridge marketplace) — DID resolution shares +infrastructure with cartridge-index. +* ADR-0009 (sandbox cartridge) — sandboxes stay machine-local; +federation coordinates which peer runs which sandbox. +* W3C DID Core 1.0 — identity model. +* NIST FIPS 203 (ML-KEM) and FIPS 204 (ML-DSA) — crypto primitives. +* Epic #87 item 3 (this). diff --git a/docs/decisions/0010-cross-machine-coord-federation.md b/docs/decisions/0010-cross-machine-coord-federation.md deleted file mode 100644 index 83e6411b..00000000 --- a/docs/decisions/0010-cross-machine-coord-federation.md +++ /dev/null @@ -1,180 +0,0 @@ - - - -# 10. Cross-machine coord federation — DID identity, ML-KEM exchange, federated quarantine - -Date: 2026-05-20 - -## Status - -Proposed (RFC — implementation tracked in epic #87 item 3) - -## Context - -`local-coord-mcp` provides multi-agent coordination over a loopback bus (`127.0.0.1:7745`). Within one machine this works well — peer registration, typed envelopes, claim/heartbeat/watchdog, master-supervised quarantine. Cross-machine coordination is the major missing axis. - -The v0.1.0 changelog mentions "Umoja federation with QUIC+UDP gossip protocol (40 tests)". Status today is unclear: the gossip code exists in places but isn't wired into the active coord cartridge, and there's no documented identity model, no key-exchange story, and no federated quarantine semantics. Cross-machine multi-agent is effectively unsupported. - -Why this matters now: - -- BoJ's three trust tiers and master/journeyman/apprentice supervision are the *differentiated* part of the architecture. They become much more useful when "master" can be on a different machine than "apprentice" (a security boundary). -- The sandbox cartridge (ADR-0009) is machine-local by construction; meaningful agent-cluster work spans multiple machines, each with its own sandboxes. -- EXHIBIT-B (Quantum-Safe Provenance) and `stapeln.toml`'s ML-DSA-87 signing already commit BoJ to a post-quantum posture. Federation is the natural place to make this load-bearing on the wire, not just in artefacts. -- ADR-0002 (BoJ-only MCP) means agents on different machines cannot federate via "two separate MCPs talking" — federation has to be a property of one cartridge family (coord-mcp) spanning machines. - -## Decision - -Promote `local-coord-mcp` from loopback-only to **federated coord-mcp**, with three architectural pillars: - -### 1. DID-based peer identity - -Replace today's ad-hoc peer IDs (`-<4hex>[@]`) with **Decentralised Identifiers** (DIDs) per W3C DID-Core 1.0: - -``` -did:boj:peer: -``` - -- The `boj` method is a new method registered in the estate's DID registry repo -- Peer keypair generation is local (ML-DSA-87 per estate standard); private key never leaves the machine -- DIDs are resolved via the same `cartridge-index` infrastructure used in ADR-0008 (federated, content-addressed) -- Multiple peers on one machine share the machine's identity but have distinct peer IDs *under* that DID (`did:boj:peer:.../peer-id/`) -- The existing `coord_register` token stays as the *session* token; the DID is the *identity* - -### 2. ML-KEM key exchange + encrypted envelopes - -When peer A (on machine α) wants to talk to peer B (on machine β): - -1. **Discovery** — A queries the federated peer registry (peer publishes its DID + machine endpoint via DNS-SD or static config) to resolve B's transport URL + KEM public key -2. **Handshake** — A and B perform ML-KEM-1024 ephemeral key exchange (estate post-quantum standard), deriving a session AEAD key -3. **Signed-and-encrypted envelopes** — coord messages between A and B are wrapped in (ML-DSA-87 signature by sender) → (ChaCha20-Poly1305 AEAD with the derived session key) → wire bytes -4. **Transport** — HTTPS POST to B's coord federation endpoint (`/coord/federated/inbox`), or QUIC if both peers advertise it - -Envelopes are otherwise the same Nickel-validated A2ML envelopes the loopback bus uses today. Federation is a transport + crypto + identity layer; the *semantics* don't change. - -### 3. Federated quarantine - -The master-approval flow already exists for tier-2+ operations within a machine. Federating it requires three new behaviours: - -**a. Master visibility across machines** - -A single master peer can be on any machine in the federation. Peers on other machines route their quarantine entries to that master's machine. The master sees a unified queue via `coord_review` regardless of source. - -**b. Master-uniqueness invariant federation-wide** - -Proof obligation **P-04** (master uniqueness) currently holds within one machine. The federated version requires consensus: at most one peer in the federation holds the `master` role at any time. Mechanism: lightweight HOTSTUFF-style election with the federation's DID set; the elected master broadcasts a signed "I-am-master" attestation on every coord heartbeat, and peers reject impostor master messages. - -**c. Cross-machine handoff** - -`coord_transfer_master` already supports same-machine handoff. Federated version routes the handoff envelope cross-machine and verifies the successor's DID is in the federation's roster before the handoff completes. - -### Topology variants - -**Mesh** (default): every peer talks to every other peer. Suitable for small federations (≤16 peers). No coordinator. - -**Hub-and-spoke**: one "rendezvous" peer holds the master role and routes all cross-machine traffic. Simpler firewall posture; single point of failure. - -**Hub-and-rim** (recommended for production): one rendezvous *machine* (running a dedicated `coord-mcp` instance with no LLM-peer attached) routes the federation. Master peer can move between machines but the routing endpoint is stable. Failure of any LLM-peer machine doesn't take down the federation. - -Operators choose via `COORD_FEDERATION_TOPOLOGY=mesh|hub|hub-and-rim`. - -### Trust posture - -A federation is **opt-in per machine**: - -- Default: `COORD_FEDERATED=false` (current loopback-only behaviour) -- Enabled: `COORD_FEDERATED=true` + `COORD_FEDERATION_ROSTER=` -- The roster is a signed file listing the DIDs of all participating machines, plus the master-election rules -- A peer never accepts envelopes from a DID not in the roster - -## Consequences - -### Positive - -- **Federation makes the tier model meaningfully secure** — apprentice on one machine, master on another machine = real security boundary, not just role labelling. -- **Post-quantum on the wire** — ML-KEM + ML-DSA-87 brings the wire surface to the same posture as the artefact-signing surface. Aligns EXHIBIT-B end-to-end. -- **DID identity is decentralised** — no central authority issues peer IDs; private keys never leave the machine. Aligns with BoJ's sovereignty branding. -- **Existing semantics preserved** — A2ML envelopes, Nickel contracts, watchdog TTLs all unchanged. Federation is purely a transport + identity layer. -- **Production-grade topology** — hub-and-rim addresses the "what if my master machine goes down" question that mesh and hub don't. -- **Composable with sandbox-mcp** — federated peers can each run their own sandboxes; the federation message bus coordinates which peer does what, even though sandboxes themselves remain machine-local. -- **Reuses estate infrastructure** — `cartridge-index` (ADR-0008) for DID resolution, ML-DSA-87 / ML-KEM standards (EXHIBIT-B) for crypto. No new infrastructure repos needed. - -### Negative - -- **Complexity escalation** — federation is the heaviest item in epic #87. Multi-week implementation; likely the longest-running campaign. -- **Crypto correctness surface** — ML-KEM handshake + AEAD framing + signature verification has many failure modes. Mitigation: use well-reviewed libraries (libsodium-style high-level constructions); proof obligations on the crypto handshake state machine. -- **Latency** — cross-machine RTT becomes part of the coord critical path. Mitigation: keep claim/heartbeat *machine-local*; only quarantine review + master attestation cross machines. -- **Roster management** — operators must maintain the signed `COORD_FEDERATION_ROSTER` file. Mitigation: tooling (`coord_federation_add_peer` / `coord_federation_remove_peer` tools that re-sign the roster atomically). -- **NAT/firewall traversal** — hub-and-rim avoids most of this; mesh has full N×N connectivity requirements. Document this in operator guide; recommend hub-and-rim for any deployment beyond a developer's laptop. - -## Non-goals - -- **Not building a generic federated agent framework** — this is BoJ's coord cartridge growing legs, not a competitor to libp2p or similar. Borrow ideas; don't reimplement. -- **Not supporting plaintext federation** — there is no `COORD_FEDERATION_INSECURE=true` flag. Federation without crypto is not federation. -- **Not making federation transparent to existing prompts** — the `convene-cluster` prompt (PR #89) implicitly assumed loopback; federated version surfaces the federation roster + topology so the LLM can reason about which machine to dispatch to. -- **Not federating sandbox handles** — see ADR-0009 explicit constraint: sandbox lifetime is machine-bound. Federation coordinates *which peer runs a sandbox*, not *passing live sandbox handles across machines*. -- **Not implementing Byzantine fault tolerance** — federation assumes the roster's signing keys are honest. Compromised-key scenarios require roster rotation, not BFT consensus. - -## Open questions - -1. **DID method registration** — register `did:boj:` in the W3C DID method registry, or stay in a `private:` namespace? Recommend register; the estate is open-source and the method is novel enough to be reviewable. - -2. **Roster mutability** — when a new peer joins, every existing peer needs the updated roster. Manual push? Gossip? Recommend manual signed roster updates pushed by current master to all peers, ratified by signature verification at receipt. - -3. **What if the master-election partitions** — two halves of a network partition each elect a master. Recommend split-brain detection on partition heal: the higher-attestation-epoch master wins; the other voluntarily demotes. Lose all in-flight quarantine work in the losing partition. - -4. **Cross-machine `coord_send_gated` quarantine retention** — quarantine queue is currently in-memory on the master's machine. Cross-machine durability requires either (a) durable queue on every machine that mirrors the master's queue, or (b) the master persists to durable storage local to it. Recommend (b) for v1; revisit if master-machine failure recovery is a hard requirement. - -5. **Performance ceiling** — the design comfortably handles ≤16 peers across ≤8 machines. Beyond that, mesh becomes O(N²); hub-and-rim scales further but eventually needs sharding. Document the ceiling explicitly; revisit if it bites. - -6. **Bootstrap with no DNS-SD** — pure static config (`COORD_FEDERATION_ROSTER` lists every peer's endpoint) works but isn't auto-discoverable. Recommend ship both modes; let operator choose. - -## Implementation sketch - -When this RFC is accepted (multi-stage): - -**Stage 1 — DID identity** (~1 week) -- Add `cartridges/local-coord-mcp/abi/LocalCoord/DID.idr` with method definition -- Generate ML-DSA-87 keypair on first `coord_register`; persist to `~/.boj/coord/identity.json` -- New tool `coord_did` returns the local peer's DID -- Loopback-only; no federation yet - -**Stage 2 — ML-KEM handshake** (~1 week) -- Add `cartridges/local-coord-mcp/abi/LocalCoord/Handshake.idr` -- ML-KEM-1024 keypair generated alongside ML-DSA-87 keypair -- New private state-machine ABI; not tool-exposed yet - -**Stage 3 — Encrypted wire format** (~1 week) -- Federated envelope wrap: `(signature, AEAD(plaintext, session_key))` -- New A2ML schema `coord-federated-envelope.ncl` (Nickel) -- Round-trip tests; no network yet - -**Stage 4 — Network transport** (~1 week) -- HTTPS POST endpoint at `/coord/federated/inbox` -- Optional QUIC transport behind feature flag -- Static roster only; no discovery - -**Stage 5 — Federated master election + quarantine** (~1 week) -- HOTSTUFF-style election among roster -- Quarantine routing to master's machine -- Federated `coord_review` / `coord_approve` / `coord_reject` - -**Stage 6 — Hub-and-rim topology + operator tooling** (~1 week) -- `coord-rendezvous-mcp` cartridge for the rim role -- Roster-management tooling -- Federation-aware prompts (replace loopback assumptions in `convene-cluster`) - -Total: ~6 weeks of focused work. Stages 1–3 are foundational; 4–6 progressively expose the federation to users. - -## Linked - -- EXHIBIT-B (Quantum-Safe Provenance) — establishes ML-DSA-87, applied here. -- ADR-0002 (BoJ-only MCP) — federation must be in-cartridge, not in-MCP-server. -- ADR-0007 (trust-tier policy DSL) — federation roster is a policy artefact. -- ADR-0008 (cartridge marketplace) — DID resolution shares infrastructure with cartridge-index. -- ADR-0009 (sandbox cartridge) — sandboxes stay machine-local; federation coordinates which peer runs which sandbox. -- W3C DID Core 1.0 — identity model. -- NIST FIPS 203 (ML-KEM) and FIPS 204 (ML-DSA) — crypto primitives. -- Epic #87 item 3 (this). diff --git a/docs/decisions/0011-webhooks-inbound-mcp-notifications.adoc b/docs/decisions/0011-webhooks-inbound-mcp-notifications.adoc new file mode 100644 index 00000000..9f55ce7e --- /dev/null +++ b/docs/decisions/0011-webhooks-inbound-mcp-notifications.adoc @@ -0,0 +1,313 @@ +== 11. Webhooks inbound + MCP notifications — closing the agent feedback loop + +Date: 2026-05-20 + +=== Status + +Proposed (RFC — implementation tracked in epic #87 item 5) + +=== Context + +BoJ today is *strictly pull-based*: an MCP client calls a tool, BoJ +dispatches to a cartridge, returns a response. There is no path for +_external events to surface to the connected LLM_. Concretely: + +* A GitHub Actions workflow fails → no notification surfaces; the agent +must poll +* Cloudflare detects a DDoS → no notification; the agent must poll +* Sentry fires an alert → no notification; the agent must poll +* A new PR opens on a watched repo → no notification; the agent must +poll +* A long-running sandbox job (ADR-0009) completes → no notification; the +agent must poll + +This forces agents into either (a) polling loops, which burn LLM context +and quota, or (b) blindness to events between tool calls. Either way, +BoJ ends up being a query interface, not an operator. The "`agent +feedback loop`" that’s the differentiator for multi-agent systems +doesn’t close. + +MCP supports server-pushed notifications via `+notifications/*+` +methods. BoJ declares the _capability_ (`+tools/listChanged+`, +`+prompts/listChanged+`) but the bridge never _emits_ a notification. +The infrastructure is there; the wiring isn’t. + +External webhook sources (GitHub, GitLab, Cloudflare, Sentry, +Stripe-style services) are the canonical mechanism for "`something +happened, here it is, on your URL of choice.`" BoJ can sit between +webhook sources and MCP clients to bridge the gap. + +=== Decision + +Add an *inbound webhook listener* to the BoJ REST backend (port 7700 — +the existing Cowboy listener per ADR-0004) and a *MCP notification +fan-out* layer that surfaces matching webhook events as +`+notifications/*+` messages to connected MCP clients. + +==== Architectural shape + +.... +External source ──HTTPS──▶ BoJ REST /webhooks/{provider}/{token} + │ + ▼ + verify signature (HMAC / OIDC) + │ + ▼ + match against subscriptions + │ + ▼ + ┌────────────┼─────────────┐ + ▼ ▼ ▼ + MCP client A MCP client B audit log + (notifications/event) +.... + +==== Inbound endpoint + +The REST backend grows two new endpoints (no change to the bridge’s +stdio surface): + +* `+POST /webhooks/{provider}/{token}+` — receives external webhook +payloads +* `+GET /webhooks/subscriptions+` — operator-side, lists active +subscriptions + +Providers in v1: `+github+`, `+gitlab+`, `+cloudflare+`, `+sentry+`, +`+stripe+`, `+generic+`. `+generic+` accepts any JSON payload and +applies generic shape matching; provider-specific endpoints know the +source’s signature scheme and event-type taxonomy. + +The `+{token}+` segment is an opaque, per-subscription secret. Loss of +the token URL = subscription compromise (can spoof events). Tokens are +generated by `+boj_webhook_subscribe+` (see below) and stored in an +encrypted form on disk; rotated by `+boj_webhook_rotate+`. + +==== Subscription model + +Subscriptions are first-class. Stored in a new +`+webhook-subscriptions.a2ml+` file under `+~/.boj/webhooks/+`. Schema: + +[source,a2ml] +---- +- id: + provider: github + token: # the path segment for the inbound URL + filter: # only emit notification if event matches + event_types: [pull_request.opened, push, workflow_run.completed] + refs: ["refs/heads/main"] + repos: ["hyperpolymath/boj-server"] + fan_out: # which MCP clients receive the notification + - client_kind: claude # all Claude sessions + - peer_token: # one specific coord-registered peer + - "*" # broadcast to all connected MCP clients (default) + audit: true # log every received event regardless of fan-out +---- + +==== Bridge-level tools (subscription management) + +* `+boj_webhook_subscribe+` — create a subscription; returns +`+(url, token, subscription_id)+` +* `+boj_webhook_list+` — list active subscriptions +* `+boj_webhook_unsubscribe+` — delete a subscription +* `+boj_webhook_rotate+` — rotate the token for a subscription +(e.g. after a leak) +* `+boj_webhook_replay+` — replay the last N events to the calling MCP +client (useful when reconnecting) + +These wire to the same `+policies/boj-default.ncl+` (ADR-0007). +`+boj_webhook_subscribe+` is tier-2 (small_write — creates a resource); +`+boj_webhook_unsubscribe+` is tier-2; `+boj_webhook_rotate+` is tier-3 +(impacts auth). + +==== MCP notification format + +When a webhook event matches a subscription: + +[source,jsonrpc] +---- +{ + "jsonrpc": "2.0", + "method": "notifications/event", + "params": { + "subscription_id": "", + "provider": "github", + "event_type": "pull_request.opened", + "received_at": "2026-05-20T12:34:56Z", + "source_signature_verified": true, + "payload": { ... raw provider payload ... }, + "extracted": { + // provider-specific normalised fields for LLM convenience + "pr_number": 42, + "repo": "hyperpolymath/boj-server", + "title": "Fix the thing", + "author": "alice" + } + } +} +---- + +The bridge emits this to *every connected MCP client whose fan_out +matches the subscription’s selector*. Clients that don’t want +notifications can ignore them; clients that want them can react. + +==== Replay buffer + +Bridge keeps the last 100 events (per subscription, bounded ring buffer) +so a reconnecting client can request `+boj_webhook_replay+` and catch +up. Bounded so memory stays predictable. + +==== Signature verification + +Each provider has a documented signature scheme: + +[width="100%",cols="50%,50%",options="header",] +|=== +|Provider |Scheme +|GitHub |HMAC-SHA256 with subscription’s secret on +`+X-Hub-Signature-256+` + +|GitLab |HMAC-SHA256 on `+X-Gitlab-Token+` + +|Cloudflare |Webhook Signature spec (JSON Web Signature) + +|Sentry |HMAC-SHA256 on `+Sentry-Hook-Signature+` + +|Stripe |`+Stripe-Signature+` with timestamp-windowed HMAC + +|`+generic+` |HMAC-SHA256 on `+X-Boj-Signature+`; constant-time compare +|=== + +If signature fails, the event is rejected, logged, and *never reaches +the notification path*. `+source_signature_verified: true+` always means +"`we verified it`"; we never accept unverified events as notifications. + +=== Consequences + +==== Positive + +* *Closes the agent feedback loop* — agents react to events instead of +polling. Major architectural unlock. +* *MCP-native, not parallel infrastructure* — events come through the +same MCP stdio surface clients already speak. No second protocol. +* *Composable with existing cartridges* — `github-api-mcp` already +speaks GitHub’s REST; this RFC adds the _push_ side. Same auth, same +provider. +* *Replay supports reconnection* — long-running MCP clients (Claude +Code, Cursor, etc.) that lose stdio connection don’t lose events; they +catch up on reconnect. +* *Multi-client fan-out matches the multi-agent story* — one event can +wake the right peer on the BoJ coord bus, leaving others uninvolved. +* *Audit by default* — every received event is logged regardless of +fan-out, so post-incident forensics is straightforward. +* *Signature verification is bedrock* — no path for unverified events to +reach a client; eliminates whole classes of webhook-spoofing attack. + +==== Negative + +* *Stdio + push is awkward* — MCP stdio is a duplex stream, but most MCP +clients don’t expect server-initiated messages. Mitigation: per the MCP +spec, `+notifications/*+` is part of the protocol; clients that +implement the spec correctly will handle it. Clients that don’t will +simply ignore the notification (no fatal failure). +* *Public ingress surface* — the inbound webhook endpoint must be +reachable from the public internet for external providers to call it. +Adds firewall configuration burden. Mitigation: document Cloudflare +Tunnel / Tailscale-funnel / ngrok patterns; do not require operators to +expose port 7700 directly. +* *Subscription state on disk* — `+webhook-subscriptions.a2ml+` is +sensitive (tokens grant impersonation). Must be `+chmod 0600+` and lives +under `~/.boj/webhooks/`. Documented; enforced at write time. +* *Event ordering* — webhooks from external sources may arrive out of +order or be redelivered. Mitigation: include the provider’s native +idempotency key (delivery ID) in the notification; the LLM (or +downstream tooling) deduplicates. +* *Latency budget* — webhook → signature verify → match → fan-out adds +milliseconds. Mitigation: signature verification + matching is +in-process; fan-out is async; the external provider doesn’t wait for +fan-out completion. + +=== Non-goals + +* *Not building a generic event bus* — this is webhook-in, +MCP-notification-out. Not for arbitrary inter-cartridge pub-sub (that’s +coord-mcp’s job). +* *Not implementing webhook _outbound_* — BoJ doesn’t _send_ webhooks to +external systems. It receives them. Outbound notifications are a +separate problem. +* *Not retaining events beyond the replay window* — 100 events × ~10 +active subscriptions = bounded memory. Long-term retention is the user’s +external observability stack (Sentry, etc.). +* *Not bridging providers* — a GitHub webhook doesn’t get re-emitted as +a Slack message. That’s the LLM’s job once it receives the notification; +BoJ is the delivery layer, not the policy layer. +* *Not requiring TLS termination in BoJ* — the REST backend speaks HTTPS +but operators are encouraged to terminate at +Cloudflare/nginx/Tailscale-funnel. BoJ accepts HTTP locally; the inbound +URL is HTTPS by virtue of the operator’s chosen ingress. + +=== Open questions + +[arabic] +. *Subscription persistence across restarts* — +`~/.boj/webhooks/webhook-subscriptions.a2ml` is the persistent store, +but what about the replay buffer? Recommend: replay buffer is in-memory +only; subscriptions persist. Replay catches reconnection within a single +BoJ process lifetime, not across restarts. +. *MCP client identification* — when fanning out, "`all Claude +sessions`" requires the bridge to know each client’s identity. Today +MCP’s `initialize` carries `clientInfo` — sufficient for fan-out +matching? Recommend yes, with fallback to peer_token for fine-grained +selection. +. *Provider extensibility* — adding a sixth provider (e.g. Linear +webhooks) means new code. Should provider definitions be +configuration-driven (Nickel contract per provider) rather than code? +Recommend: signature schemes stay code (cryptography); event-shape +normalisation can move to Nickel. +. *Backpressure on slow clients* — if one MCP client is slow to consume +notifications, do others wait? Recommend per-client send queues with +bounded size; slow client gets events dropped (with a +`+notifications/missed+` summary on reconnect). +. *Authorization model on subscription creation* — should creating a +webhook subscription require master role (ADR-0007 tier-3)? Recommend +yes — webhook URLs grant impersonation capability, so creation is +master-gated by default; can be overridden in policy. +. *Reusing the existing REST backend vs separate listener* — the +existing REST backend handles cartridge dispatch. Should webhooks share +that listener or run on a separate port? Recommend share — ADR-0004 +already chose single-listener; honour that. Path-routed +(`+/webhooks/...+`) is unambiguous. + +=== Implementation sketch + +When this RFC is accepted: + +[arabic] +. New module `+elixir/lib/boj_rest/webhooks/+` with one file per +provider’s signature scheme. +. Persistent store at `+~/.boj/webhooks/webhook-subscriptions.a2ml+` +(chmod 0600); Idris2-verified read/write. +. Bridge-level tools `+boj_webhook_subscribe+` / `+list+` / +`+unsubscribe+` / `+rotate+` / `+replay+` added to +`mcp-bridge/lib/tools.js` (5 tools). +. Bridge notification emitter — new `+mcp-bridge/lib/notifications.js+` +that maintains the per-client send queue + bounded backpressure. +. Default policy entries (ADR-0007): `+boj_webhook_subscribe+` tier-3, +others tier-2. +. Provider documentation: a new `+docs/cartridges/WEBHOOK-PROVIDERS.md+` +showing how to register a webhook URL with each supported provider. +. Tests: per-provider signature verification fixtures + end-to-end +replay test. + +=== Linked + +* ADR-0004 (unified gateway) — single listener; webhooks ride the same +Cowboy endpoint. +* ADR-0007 (policy DSL) — webhook subscriptions are tier-gated +artefacts. +* ADR-0008 (cartridge marketplace) — marketplace updates can flow to MCP +clients via webhook notifications (see ADR-0008 open question 4). +* ADR-0009 (sandbox cartridge) — long-running sandbox completion can +emit notifications via this mechanism. +* ADR-0010 (federation) — federated quarantine review can emit +notifications to the master’s MCP client. +* Epic #87 item 5 (this). diff --git a/docs/decisions/0011-webhooks-inbound-mcp-notifications.md b/docs/decisions/0011-webhooks-inbound-mcp-notifications.md deleted file mode 100644 index 000cf4ce..00000000 --- a/docs/decisions/0011-webhooks-inbound-mcp-notifications.md +++ /dev/null @@ -1,201 +0,0 @@ - - - -# 11. Webhooks inbound + MCP notifications — closing the agent feedback loop - -Date: 2026-05-20 - -## Status - -Proposed (RFC — implementation tracked in epic #87 item 5) - -## Context - -BoJ today is **strictly pull-based**: an MCP client calls a tool, BoJ dispatches to a cartridge, returns a response. There is no path for *external events to surface to the connected LLM*. Concretely: - -- A GitHub Actions workflow fails → no notification surfaces; the agent must poll -- Cloudflare detects a DDoS → no notification; the agent must poll -- Sentry fires an alert → no notification; the agent must poll -- A new PR opens on a watched repo → no notification; the agent must poll -- A long-running sandbox job (ADR-0009) completes → no notification; the agent must poll - -This forces agents into either (a) polling loops, which burn LLM context and quota, or (b) blindness to events between tool calls. Either way, BoJ ends up being a query interface, not an operator. The "agent feedback loop" that's the differentiator for multi-agent systems doesn't close. - -MCP supports server-pushed notifications via `notifications/*` methods. BoJ declares the *capability* (`tools/listChanged`, `prompts/listChanged`) but the bridge never *emits* a notification. The infrastructure is there; the wiring isn't. - -External webhook sources (GitHub, GitLab, Cloudflare, Sentry, Stripe-style services) are the canonical mechanism for "something happened, here it is, on your URL of choice." BoJ can sit between webhook sources and MCP clients to bridge the gap. - -## Decision - -Add an **inbound webhook listener** to the BoJ REST backend (port 7700 — the existing Cowboy listener per ADR-0004) and a **MCP notification fan-out** layer that surfaces matching webhook events as `notifications/*` messages to connected MCP clients. - -### Architectural shape - -``` -External source ──HTTPS──▶ BoJ REST /webhooks/{provider}/{token} - │ - ▼ - verify signature (HMAC / OIDC) - │ - ▼ - match against subscriptions - │ - ▼ - ┌────────────┼─────────────┐ - ▼ ▼ ▼ - MCP client A MCP client B audit log - (notifications/event) -``` - -### Inbound endpoint - -The REST backend grows two new endpoints (no change to the bridge's stdio surface): - -- `POST /webhooks/{provider}/{token}` — receives external webhook payloads -- `GET /webhooks/subscriptions` — operator-side, lists active subscriptions - -Providers in v1: `github`, `gitlab`, `cloudflare`, `sentry`, `stripe`, `generic`. `generic` accepts any JSON payload and applies generic shape matching; provider-specific endpoints know the source's signature scheme and event-type taxonomy. - -The `{token}` segment is an opaque, per-subscription secret. Loss of the token URL = subscription compromise (can spoof events). Tokens are generated by `boj_webhook_subscribe` (see below) and stored in an encrypted form on disk; rotated by `boj_webhook_rotate`. - -### Subscription model - -Subscriptions are first-class. Stored in a new `webhook-subscriptions.a2ml` file under `~/.boj/webhooks/`. Schema: - -```a2ml -- id: - provider: github - token: # the path segment for the inbound URL - filter: # only emit notification if event matches - event_types: [pull_request.opened, push, workflow_run.completed] - refs: ["refs/heads/main"] - repos: ["hyperpolymath/boj-server"] - fan_out: # which MCP clients receive the notification - - client_kind: claude # all Claude sessions - - peer_token: # one specific coord-registered peer - - "*" # broadcast to all connected MCP clients (default) - audit: true # log every received event regardless of fan-out -``` - -### Bridge-level tools (subscription management) - -- `boj_webhook_subscribe` — create a subscription; returns `(url, token, subscription_id)` -- `boj_webhook_list` — list active subscriptions -- `boj_webhook_unsubscribe` — delete a subscription -- `boj_webhook_rotate` — rotate the token for a subscription (e.g. after a leak) -- `boj_webhook_replay` — replay the last N events to the calling MCP client (useful when reconnecting) - -These wire to the same `policies/boj-default.ncl` (ADR-0007). `boj_webhook_subscribe` is tier-2 (small_write — creates a resource); `boj_webhook_unsubscribe` is tier-2; `boj_webhook_rotate` is tier-3 (impacts auth). - -### MCP notification format - -When a webhook event matches a subscription: - -```jsonrpc -{ - "jsonrpc": "2.0", - "method": "notifications/event", - "params": { - "subscription_id": "", - "provider": "github", - "event_type": "pull_request.opened", - "received_at": "2026-05-20T12:34:56Z", - "source_signature_verified": true, - "payload": { ... raw provider payload ... }, - "extracted": { - // provider-specific normalised fields for LLM convenience - "pr_number": 42, - "repo": "hyperpolymath/boj-server", - "title": "Fix the thing", - "author": "alice" - } - } -} -``` - -The bridge emits this to **every connected MCP client whose fan_out matches the subscription's selector**. Clients that don't want notifications can ignore them; clients that want them can react. - -### Replay buffer - -Bridge keeps the last 100 events (per subscription, bounded ring buffer) so a reconnecting client can request `boj_webhook_replay` and catch up. Bounded so memory stays predictable. - -### Signature verification - -Each provider has a documented signature scheme: - -| Provider | Scheme | -|---|---| -| GitHub | HMAC-SHA256 with subscription's secret on `X-Hub-Signature-256` | -| GitLab | HMAC-SHA256 on `X-Gitlab-Token` | -| Cloudflare | Webhook Signature spec (JSON Web Signature) | -| Sentry | HMAC-SHA256 on `Sentry-Hook-Signature` | -| Stripe | `Stripe-Signature` with timestamp-windowed HMAC | -| `generic` | HMAC-SHA256 on `X-Boj-Signature`; constant-time compare | - -If signature fails, the event is rejected, logged, and **never reaches the notification path**. `source_signature_verified: true` always means "we verified it"; we never accept unverified events as notifications. - -## Consequences - -### Positive - -- **Closes the agent feedback loop** — agents react to events instead of polling. Major architectural unlock. -- **MCP-native, not parallel infrastructure** — events come through the same MCP stdio surface clients already speak. No second protocol. -- **Composable with existing cartridges** — \`github-api-mcp\` already speaks GitHub's REST; this RFC adds the *push* side. Same auth, same provider. -- **Replay supports reconnection** — long-running MCP clients (Claude Code, Cursor, etc.) that lose stdio connection don't lose events; they catch up on reconnect. -- **Multi-client fan-out matches the multi-agent story** — one event can wake the right peer on the BoJ coord bus, leaving others uninvolved. -- **Audit by default** — every received event is logged regardless of fan-out, so post-incident forensics is straightforward. -- **Signature verification is bedrock** — no path for unverified events to reach a client; eliminates whole classes of webhook-spoofing attack. - -### Negative - -- **Stdio + push is awkward** — MCP stdio is a duplex stream, but most MCP clients don't expect server-initiated messages. Mitigation: per the MCP spec, `notifications/*` is part of the protocol; clients that implement the spec correctly will handle it. Clients that don't will simply ignore the notification (no fatal failure). -- **Public ingress surface** — the inbound webhook endpoint must be reachable from the public internet for external providers to call it. Adds firewall configuration burden. Mitigation: document Cloudflare Tunnel / Tailscale-funnel / ngrok patterns; do not require operators to expose port 7700 directly. -- **Subscription state on disk** — `webhook-subscriptions.a2ml` is sensitive (tokens grant impersonation). Must be `chmod 0600` and lives under \`~/.boj/webhooks/\`. Documented; enforced at write time. -- **Event ordering** — webhooks from external sources may arrive out of order or be redelivered. Mitigation: include the provider's native idempotency key (delivery ID) in the notification; the LLM (or downstream tooling) deduplicates. -- **Latency budget** — webhook → signature verify → match → fan-out adds milliseconds. Mitigation: signature verification + matching is in-process; fan-out is async; the external provider doesn't wait for fan-out completion. - -## Non-goals - -- **Not building a generic event bus** — this is webhook-in, MCP-notification-out. Not for arbitrary inter-cartridge pub-sub (that's coord-mcp's job). -- **Not implementing webhook *outbound*** — BoJ doesn't *send* webhooks to external systems. It receives them. Outbound notifications are a separate problem. -- **Not retaining events beyond the replay window** — 100 events × ~10 active subscriptions = bounded memory. Long-term retention is the user's external observability stack (Sentry, etc.). -- **Not bridging providers** — a GitHub webhook doesn't get re-emitted as a Slack message. That's the LLM's job once it receives the notification; BoJ is the delivery layer, not the policy layer. -- **Not requiring TLS termination in BoJ** — the REST backend speaks HTTPS but operators are encouraged to terminate at Cloudflare/nginx/Tailscale-funnel. BoJ accepts HTTP locally; the inbound URL is HTTPS by virtue of the operator's chosen ingress. - -## Open questions - -1. **Subscription persistence across restarts** — \`~/.boj/webhooks/webhook-subscriptions.a2ml\` is the persistent store, but what about the replay buffer? Recommend: replay buffer is in-memory only; subscriptions persist. Replay catches reconnection within a single BoJ process lifetime, not across restarts. - -2. **MCP client identification** — when fanning out, "all Claude sessions" requires the bridge to know each client's identity. Today MCP's \`initialize\` carries \`clientInfo\` — sufficient for fan-out matching? Recommend yes, with fallback to peer_token for fine-grained selection. - -3. **Provider extensibility** — adding a sixth provider (e.g. Linear webhooks) means new code. Should provider definitions be configuration-driven (Nickel contract per provider) rather than code? Recommend: signature schemes stay code (cryptography); event-shape normalisation can move to Nickel. - -4. **Backpressure on slow clients** — if one MCP client is slow to consume notifications, do others wait? Recommend per-client send queues with bounded size; slow client gets events dropped (with a `notifications/missed` summary on reconnect). - -5. **Authorization model on subscription creation** — should creating a webhook subscription require master role (ADR-0007 tier-3)? Recommend yes — webhook URLs grant impersonation capability, so creation is master-gated by default; can be overridden in policy. - -6. **Reusing the existing REST backend vs separate listener** — the existing REST backend handles cartridge dispatch. Should webhooks share that listener or run on a separate port? Recommend share — ADR-0004 already chose single-listener; honour that. Path-routed (`/webhooks/...`) is unambiguous. - -## Implementation sketch - -When this RFC is accepted: - -1. New module `elixir/lib/boj_rest/webhooks/` with one file per provider's signature scheme. -2. Persistent store at `~/.boj/webhooks/webhook-subscriptions.a2ml` (chmod 0600); Idris2-verified read/write. -3. Bridge-level tools `boj_webhook_subscribe` / `list` / `unsubscribe` / `rotate` / `replay` added to \`mcp-bridge/lib/tools.js\` (5 tools). -4. Bridge notification emitter — new `mcp-bridge/lib/notifications.js` that maintains the per-client send queue + bounded backpressure. -5. Default policy entries (ADR-0007): `boj_webhook_subscribe` tier-3, others tier-2. -6. Provider documentation: a new `docs/cartridges/WEBHOOK-PROVIDERS.md` showing how to register a webhook URL with each supported provider. -7. Tests: per-provider signature verification fixtures + end-to-end replay test. - -## Linked - -- ADR-0004 (unified gateway) — single listener; webhooks ride the same Cowboy endpoint. -- ADR-0007 (policy DSL) — webhook subscriptions are tier-gated artefacts. -- ADR-0008 (cartridge marketplace) — marketplace updates can flow to MCP clients via webhook notifications (see ADR-0008 open question 4). -- ADR-0009 (sandbox cartridge) — long-running sandbox completion can emit notifications via this mechanism. -- ADR-0010 (federation) — federated quarantine review can emit notifications to the master's MCP client. -- Epic #87 item 5 (this). diff --git a/docs/decisions/0012-server-initiated-sampling.adoc b/docs/decisions/0012-server-initiated-sampling.adoc new file mode 100644 index 00000000..59f9f4d0 --- /dev/null +++ b/docs/decisions/0012-server-initiated-sampling.adoc @@ -0,0 +1,310 @@ +== 12. Server-initiated sampling — composition routing + ambiguous-input clarification + +Date: 2026-05-20 + +=== Status + +Proposed (RFC — implementation tracked in epic #87 item 6) + +=== Context + +MCP’s `+sampling/createMessage+` is a reverse path: the server asks the +connected LLM to make a sub-decision. The user-facing LLM remains in +control (it sees the sampling request and decides whether to honour it), +but the server can pose a question and receive a response without the +user having to manually mediate. + +This is *structurally* the right primitive for two BoJ scenarios that +today are awkward: + +*Scenario A: Cartridge composition routing* + +A user asks the LLM: _"`Deploy my app and tell me when it’s healthy.`"_ +The LLM calls `+boj_cartridge_invoke+` with the high-level intent. BoJ +now has to choose: + +* Which deploy cartridge? `+fly-mcp+`, `+render-mcp+`, `+railway-mcp+`, +`+vercel-mcp+`? +* Which healthcheck cartridge? `+prometheus-mcp+` poll, `+sentry-mcp+` +watch, or just curl? +* What’s the order? Deploy then check, or pre-validate then deploy? + +BoJ can’t make this choice well without knowing the user’s situation. +Today it either (a) requires the LLM to specify the cartridge explicitly +(defeats "`I just want to deploy`") or (b) hard-codes routing rules +(brittle, can’t adapt). + +With sampling: BoJ asks the LLM _"`Given the user said X, which of these +4 deploy cartridges fits? Reply with cartridge name.`"_ — and routes +accordingly. The LLM brings world-knowledge that BoJ doesn’t have. + +*Scenario B: Ambiguous-input clarification* + +A user calls `+boj_cartridge_invoke+` for `+database-mcp+` with +parameters that could match multiple backends +(e.g. `+query: "SELECT * FROM users"+` — could be PostgreSQL, SQLite, +DuckDB). Today BoJ either picks an arbitrary default or errors. + +With sampling: BoJ asks the LLM _"`This query is ambiguous between \{pg, +sqlite, duckdb}. Which is the user’s database?`"_ — and the LLM either +knows from prior context or asks the user. + +Without sampling, both scenarios force either pre-resolution (LLM must +specify everything up front) or post-failure handling (try one, fail, +try another). Both are worse than asking once at the right moment. + +=== Decision + +Implement MCP `+sampling/createMessage+` server-initiated requests for +*two specific patterns*, both opt-in per cartridge: + +==== Pattern 1: Composition router + +A new helper in the bridge: +`+requestComposition(intent, candidates, context)+`. Used by +`+boj_cartridge_invoke+` when the target is ambiguous, and by prompt +templates (PR #89) when a step needs LLM-side selection. + +[source,js] +---- +// In boj_cartridge_invoke, when args.name is ambiguous: +const choice = await requestComposition({ + intent: "deploy and monitor", + candidates: [ + { name: "fly-mcp", strengths: "fast cold-start, global edge" }, + { name: "render-mcp", strengths: "managed Postgres + cron" }, + { name: "railway-mcp", strengths: "monorepo support, env groups" }, + { name: "vercel-mcp", strengths: "frontend-focused, edge functions" }, + ], + context: { user_message: ..., recent_tool_history: [...] }, +}); +// choice.name → "fly-mcp" (e.g.) +// proceed with invokeCartridge(choice.name, args) +---- + +MCP wire: + +[source,jsonrpc] +---- +{ + "jsonrpc": "2.0", + "id": , + "method": "sampling/createMessage", + "params": { + "messages": [ + { "role": "user", "content": { "type": "text", + "text": "BoJ routing decision needed.\n\nUser intent: deploy and monitor\n\nCandidates:\n- fly-mcp: fast cold-start, global edge\n- render-mcp: managed Postgres + cron\n- railway-mcp: monorepo support, env groups\n- vercel-mcp: frontend-focused, edge functions\n\nReturn exactly one cartridge name from the candidate list. No explanation." + }} + ], + "modelPreferences": { + "intelligencePriority": 0.3, + "speedPriority": 0.9, + "costPriority": 0.8 + }, + "maxTokens": 32, + "systemPrompt": "You are BoJ's cartridge router. Return cartridge names verbatim from the provided list, with no extra text." + } +} +---- + +`+modelPreferences+` biases for cheap-fast — composition routing is +high-frequency, doesn’t need the heaviest model. + +==== Pattern 2: Clarification prompt + +A new helper `+requestClarification(question, options)+` for the +ambiguous-input scenario: + +[source,js] +---- +const answer = await requestClarification({ + question: "The query 'SELECT * FROM users' is database-agnostic. Which backend?", + options: ["postgresql", "sqlite", "duckdb", "mongodb"], +}); +// answer.choice → "postgresql" +---- + +The wire shape is the same as Pattern 1 but with a different system +prompt focusing on "`ask the user if you don’t know`" rather than "`pick +from the list silently`". + +==== Sampling client-side cooperation + +MCP clients are not required to honour sampling requests. The spec is +explicit: the client decides. A well-behaved client (Claude Code, etc.) +shows the sampling request to the user (who sees it as "`BoJ wants to +ask the LLM something`") and the user approves or denies. + +BoJ’s bridge handles three responses: + +[width="100%",cols="50%,50%",options="header",] +|=== +|Client response |BoJ behaviour +|Sampling result returned |Use the returned message; proceed + +|Sampling rejected |Fall back to a deterministic default (documented per +call site); proceed with reduced confidence + +|Sampling timeout (30s default) |Same as rejected +|=== + +The fallback path is *always present*. BoJ never blocks indefinitely on +sampling. + +==== Where sampling is NOT used + +Sampling is a power move; misuse is worse than not having it. Hard +rules: + +* *Never* in security-critical paths (e.g. `+coord_approve+`, +`+boj_github_merge_pr+`). These need explicit user authorization, not +LLM judgment. +* *Never* to ask "`should I proceed?`" — that’s the calling LLM’s job, +not a sub-LLM-call. +* *Never* for input validation — that’s `+hardeningGate+`’s job. +* *Never* in the OTel-traced hot path without budget tracking — sampling +burns tokens; track per-session in `+OTEL_*+` attributes and respect a +global budget. + +A new env var `+BOJ_SAMPLING_BUDGET_PER_SESSION+` (default `+50+`) caps +how many sampling requests BoJ will issue per MCP session. Exceeded → +fall back to deterministic defaults. + +==== Audit + transparency + +Every sampling request emits an OTel span (per ADR-0013, item 13) with +attributes: + +* `+boj.sampling.pattern+` — `+composition_router+` | `+clarification+` +* `+boj.sampling.candidates_count+` +* `+boj.sampling.result+` — `+returned+` | `+rejected+` | `+timeout+` +* `+boj.sampling.chosen+` — the value the LLM returned (or null) +* `+boj.sampling.budget_remaining+` + +This means sampling activity is observable from the user’s existing +telemetry (ADR-0013 / item 13 already wires OTel) without bespoke +logging. + +=== Consequences + +==== Positive + +* *Closes the "`BoJ needs world-knowledge`" hole* — routing and +clarification decisions can finally tap the LLM that’s already +connected. +* *Composition becomes adaptive* — `+boj_cartridge_invoke+` against +"`deploy`" doesn’t need a hard-coded provider; the right cartridge is +chosen per call based on the user’s actual situation. +* *High-frequency, low-cost* — composition routing biased toward +fast/cheap models; doesn’t burn the heaviest model on routing decisions. +* *Always has fallback* — sampling is opportunistic; rejection or +timeout doesn’t break BoJ. Reduces deployment risk. +* *Budget-bounded* — `+BOJ_SAMPLING_BUDGET_PER_SESSION+` prevents +runaway sampling from a misbehaving cartridge. +* *Observable* — every sampling request is an OTel span; operators can +see where sampling is helping vs. wasting tokens. +* *Underused capability* — most MCP servers don’t use sampling. BoJ +using it well is differentiated. + +==== Negative + +* *Token budget visible to user* — sampling calls cost tokens the user +pays for. Mitigation: per-session budget cap + clear OTel attribution so +the user knows. +* *Client cooperation required* — clients that don’t implement sampling +silently no-op. Mitigation: documented fallback paths; degrade +gracefully. +* *Sub-LLM-call latency* — sampling RTT depends on the client+model. +Could be hundreds of ms. Mitigation: only use for genuinely-needed +decisions; cache routing decisions within a session (same intent → same +choice). +* *Sampling-driven choices can be unpredictable* — the LLM might pick a +different cartridge for the same intent across runs. Mitigation: +deterministic-mode env var (`+BOJ_SAMPLING_DETERMINISTIC=true+`) that +uses the first option in candidate lists rather than sampling. +* *Audit complexity* — when something goes wrong, was it the cartridge +choice (sampling) or the cartridge itself? Mitigation: OTel span carries +the sampling decision; post-incident analysis can disentangle. + +=== Non-goals + +* *Not making sampling the default for routing* — sampling is opt-in per +call site. The composition-router helper is invoked explicitly; +cartridges that don’t want it don’t get it. +* *Not using sampling for security decisions* — explicit rule. Security +needs human-in-the-loop. +* *Not exposing sampling as a tool* — there’s no `+boj_sample+` tool. +Sampling is an internal bridge primitive; cartridges trigger it via the +helpers above. +* *Not chaining sampling* — one sampling request per server-decision. No +multi-turn sub-LLM-conversations; that’s the user’s LLM’s job. +* *Not requiring sampling support in clients* — graceful degradation +always available. + +=== Open questions + +[arabic] +. *System-prompt safety* — BoJ-authored system prompts are sent to the +client’s LLM. Could be misused if a cartridge submitted user-controlled +text. Recommend: system prompts are static strings in BoJ code; only the +`+messages.content+` carries variable data, and that data is the result +of the calling tool’s args (already filtered through `+hardeningGate+`’s +injection scan). +. *Caching sampling decisions* — within a session, "`the user’s +database`" probably doesn’t change. Recommend per-session LRU keyed on +(pattern, question-hash) so repeated calls don’t re-sample. +. *Multi-client sampling target* — if two MCP clients are connected, +which one gets the sampling request? Recommend the most-recently-active +client (whoever issued the most recent `+tools/call+`); document the +heuristic; future RFC can refine. +. *Sampling for prompts* — should the prompt templates from PR #89 +themselves invoke sampling for intermediate steps? Recommend yes, but +only in templates where the spec explicitly calls it out (e.g. an +`+auto-deploy+` prompt could invoke composition routing internally). +Keep `+audit-repo+`, `+triage-issues+`, etc. sampling-free. +. *Cost attribution* — sampling calls hit the user’s LLM quota. Should +the OTel span carry the token cost in attributes (if the client reports +it)? Recommend yes when available; document that the client may not +report. +. *Fallback determinism* — when sampling is rejected, "`first option in +the candidate list`" is one fallback strategy. Are there call sites +where a different deterministic strategy is correct (e.g. "`lowest +tier`", "`least recently used`")? Recommend per-call-site fallback +function; sensible defaults documented. + +=== Implementation sketch + +When this RFC is accepted: + +[arabic] +. New `+mcp-bridge/lib/sampling.js+` with `+requestComposition+` and +`+requestClarification+` helpers + budget tracking. +. Helpers route through new wire-format method on the JSON-RPC server +side (the bridge sends a `+sampling/createMessage+` request to the +client, awaits response). +. New env var `+BOJ_SAMPLING_BUDGET_PER_SESSION+` (default 50); declared +in `+glama.json+`. +. New env var `+BOJ_SAMPLING_DETERMINISTIC+` (default `+false+`); when +true, helpers return the first option without sampling. +. OTel instrumentation (depends on item 13 / PR #91): every sampling +request emits a span. +. Integration with `+boj_cartridge_invoke+` for the routing case (one +call site to start). +. Tests: mock-MCP-client that returns canned sampling responses; verify +fallback behaviour on rejection + timeout. +. Documentation: `+docs/cartridges/SAMPLING-USAGE.md+` covering when to +invoke sampling, budget management, and the two patterns. + +=== Linked + +* ADR-0007 (policy DSL) — sampling is _not_ policy-gated (it’s not a +side-effectful operation), but the _result_ of sampling feeds into a +tool call that _is_ policy-gated. +* ADR-0009 (sandbox cartridge) — sandbox provider selection is a natural +composition-router use case. +* ADR-0011 (webhooks/notifications) — orthogonal but related; both are +server-initiated MCP message types, both opt-in per client. +* Epic #87 item 6 (this) + item 13 (OTel, the observability surface for +sampling). +* PR #89 (resources + prompts) — vocabulary; prompts could _internally_ +invoke sampling for sub-steps. diff --git a/docs/decisions/0012-server-initiated-sampling.md b/docs/decisions/0012-server-initiated-sampling.md deleted file mode 100644 index 9e6f4d8d..00000000 --- a/docs/decisions/0012-server-initiated-sampling.md +++ /dev/null @@ -1,202 +0,0 @@ - - - -# 12. Server-initiated sampling — composition routing + ambiguous-input clarification - -Date: 2026-05-20 - -## Status - -Proposed (RFC — implementation tracked in epic #87 item 6) - -## Context - -MCP's `sampling/createMessage` is a reverse path: the server asks the connected LLM to make a sub-decision. The user-facing LLM remains in control (it sees the sampling request and decides whether to honour it), but the server can pose a question and receive a response without the user having to manually mediate. - -This is **structurally** the right primitive for two BoJ scenarios that today are awkward: - -**Scenario A: Cartridge composition routing** - -A user asks the LLM: *"Deploy my app and tell me when it's healthy."* The LLM calls `boj_cartridge_invoke` with the high-level intent. BoJ now has to choose: - -- Which deploy cartridge? `fly-mcp`, `render-mcp`, `railway-mcp`, `vercel-mcp`? -- Which healthcheck cartridge? `prometheus-mcp` poll, `sentry-mcp` watch, or just curl? -- What's the order? Deploy then check, or pre-validate then deploy? - -BoJ can't make this choice well without knowing the user's situation. Today it either (a) requires the LLM to specify the cartridge explicitly (defeats "I just want to deploy") or (b) hard-codes routing rules (brittle, can't adapt). - -With sampling: BoJ asks the LLM *"Given the user said X, which of these 4 deploy cartridges fits? Reply with cartridge name."* — and routes accordingly. The LLM brings world-knowledge that BoJ doesn't have. - -**Scenario B: Ambiguous-input clarification** - -A user calls `boj_cartridge_invoke` for `database-mcp` with parameters that could match multiple backends (e.g. `query: "SELECT * FROM users"` — could be PostgreSQL, SQLite, DuckDB). Today BoJ either picks an arbitrary default or errors. - -With sampling: BoJ asks the LLM *"This query is ambiguous between {pg, sqlite, duckdb}. Which is the user's database?"* — and the LLM either knows from prior context or asks the user. - -Without sampling, both scenarios force either pre-resolution (LLM must specify everything up front) or post-failure handling (try one, fail, try another). Both are worse than asking once at the right moment. - -## Decision - -Implement MCP `sampling/createMessage` server-initiated requests for **two specific patterns**, both opt-in per cartridge: - -### Pattern 1: Composition router - -A new helper in the bridge: `requestComposition(intent, candidates, context)`. Used by `boj_cartridge_invoke` when the target is ambiguous, and by prompt templates (PR #89) when a step needs LLM-side selection. - -```js -// In boj_cartridge_invoke, when args.name is ambiguous: -const choice = await requestComposition({ - intent: "deploy and monitor", - candidates: [ - { name: "fly-mcp", strengths: "fast cold-start, global edge" }, - { name: "render-mcp", strengths: "managed Postgres + cron" }, - { name: "railway-mcp", strengths: "monorepo support, env groups" }, - { name: "vercel-mcp", strengths: "frontend-focused, edge functions" }, - ], - context: { user_message: ..., recent_tool_history: [...] }, -}); -// choice.name → "fly-mcp" (e.g.) -// proceed with invokeCartridge(choice.name, args) -``` - -MCP wire: -```jsonrpc -{ - "jsonrpc": "2.0", - "id": , - "method": "sampling/createMessage", - "params": { - "messages": [ - { "role": "user", "content": { "type": "text", - "text": "BoJ routing decision needed.\n\nUser intent: deploy and monitor\n\nCandidates:\n- fly-mcp: fast cold-start, global edge\n- render-mcp: managed Postgres + cron\n- railway-mcp: monorepo support, env groups\n- vercel-mcp: frontend-focused, edge functions\n\nReturn exactly one cartridge name from the candidate list. No explanation." - }} - ], - "modelPreferences": { - "intelligencePriority": 0.3, - "speedPriority": 0.9, - "costPriority": 0.8 - }, - "maxTokens": 32, - "systemPrompt": "You are BoJ's cartridge router. Return cartridge names verbatim from the provided list, with no extra text." - } -} -``` - -`modelPreferences` biases for cheap-fast — composition routing is high-frequency, doesn't need the heaviest model. - -### Pattern 2: Clarification prompt - -A new helper `requestClarification(question, options)` for the ambiguous-input scenario: - -```js -const answer = await requestClarification({ - question: "The query 'SELECT * FROM users' is database-agnostic. Which backend?", - options: ["postgresql", "sqlite", "duckdb", "mongodb"], -}); -// answer.choice → "postgresql" -``` - -The wire shape is the same as Pattern 1 but with a different system prompt focusing on "ask the user if you don't know" rather than "pick from the list silently". - -### Sampling client-side cooperation - -MCP clients are not required to honour sampling requests. The spec is explicit: the client decides. A well-behaved client (Claude Code, etc.) shows the sampling request to the user (who sees it as "BoJ wants to ask the LLM something") and the user approves or denies. - -BoJ's bridge handles three responses: - -| Client response | BoJ behaviour | -|---|---| -| Sampling result returned | Use the returned message; proceed | -| Sampling rejected | Fall back to a deterministic default (documented per call site); proceed with reduced confidence | -| Sampling timeout (30s default) | Same as rejected | - -The fallback path is **always present**. BoJ never blocks indefinitely on sampling. - -### Where sampling is NOT used - -Sampling is a power move; misuse is worse than not having it. Hard rules: - -- **Never** in security-critical paths (e.g. `coord_approve`, `boj_github_merge_pr`). These need explicit user authorization, not LLM judgment. -- **Never** to ask "should I proceed?" — that's the calling LLM's job, not a sub-LLM-call. -- **Never** for input validation — that's `hardeningGate`'s job. -- **Never** in the OTel-traced hot path without budget tracking — sampling burns tokens; track per-session in `OTEL_*` attributes and respect a global budget. - -A new env var `BOJ_SAMPLING_BUDGET_PER_SESSION` (default `50`) caps how many sampling requests BoJ will issue per MCP session. Exceeded → fall back to deterministic defaults. - -### Audit + transparency - -Every sampling request emits an OTel span (per ADR-0013, item 13) with attributes: - -- `boj.sampling.pattern` — `composition_router` | `clarification` -- `boj.sampling.candidates_count` -- `boj.sampling.result` — `returned` | `rejected` | `timeout` -- `boj.sampling.chosen` — the value the LLM returned (or null) -- `boj.sampling.budget_remaining` - -This means sampling activity is observable from the user's existing telemetry (ADR-0013 / item 13 already wires OTel) without bespoke logging. - -## Consequences - -### Positive - -- **Closes the "BoJ needs world-knowledge" hole** — routing and clarification decisions can finally tap the LLM that's already connected. -- **Composition becomes adaptive** — `boj_cartridge_invoke` against "deploy" doesn't need a hard-coded provider; the right cartridge is chosen per call based on the user's actual situation. -- **High-frequency, low-cost** — composition routing biased toward fast/cheap models; doesn't burn the heaviest model on routing decisions. -- **Always has fallback** — sampling is opportunistic; rejection or timeout doesn't break BoJ. Reduces deployment risk. -- **Budget-bounded** — `BOJ_SAMPLING_BUDGET_PER_SESSION` prevents runaway sampling from a misbehaving cartridge. -- **Observable** — every sampling request is an OTel span; operators can see where sampling is helping vs. wasting tokens. -- **Underused capability** — most MCP servers don't use sampling. BoJ using it well is differentiated. - -### Negative - -- **Token budget visible to user** — sampling calls cost tokens the user pays for. Mitigation: per-session budget cap + clear OTel attribution so the user knows. -- **Client cooperation required** — clients that don't implement sampling silently no-op. Mitigation: documented fallback paths; degrade gracefully. -- **Sub-LLM-call latency** — sampling RTT depends on the client+model. Could be hundreds of ms. Mitigation: only use for genuinely-needed decisions; cache routing decisions within a session (same intent → same choice). -- **Sampling-driven choices can be unpredictable** — the LLM might pick a different cartridge for the same intent across runs. Mitigation: deterministic-mode env var (`BOJ_SAMPLING_DETERMINISTIC=true`) that uses the first option in candidate lists rather than sampling. -- **Audit complexity** — when something goes wrong, was it the cartridge choice (sampling) or the cartridge itself? Mitigation: OTel span carries the sampling decision; post-incident analysis can disentangle. - -## Non-goals - -- **Not making sampling the default for routing** — sampling is opt-in per call site. The composition-router helper is invoked explicitly; cartridges that don't want it don't get it. -- **Not using sampling for security decisions** — explicit rule. Security needs human-in-the-loop. -- **Not exposing sampling as a tool** — there's no `boj_sample` tool. Sampling is an internal bridge primitive; cartridges trigger it via the helpers above. -- **Not chaining sampling** — one sampling request per server-decision. No multi-turn sub-LLM-conversations; that's the user's LLM's job. -- **Not requiring sampling support in clients** — graceful degradation always available. - -## Open questions - -1. **System-prompt safety** — BoJ-authored system prompts are sent to the client's LLM. Could be misused if a cartridge submitted user-controlled text. Recommend: system prompts are static strings in BoJ code; only the `messages.content` carries variable data, and that data is the result of the calling tool's args (already filtered through `hardeningGate`'s injection scan). - -2. **Caching sampling decisions** — within a session, "the user's database" probably doesn't change. Recommend per-session LRU keyed on (pattern, question-hash) so repeated calls don't re-sample. - -3. **Multi-client sampling target** — if two MCP clients are connected, which one gets the sampling request? Recommend the most-recently-active client (whoever issued the most recent `tools/call`); document the heuristic; future RFC can refine. - -4. **Sampling for prompts** — should the prompt templates from PR #89 themselves invoke sampling for intermediate steps? Recommend yes, but only in templates where the spec explicitly calls it out (e.g. an `auto-deploy` prompt could invoke composition routing internally). Keep `audit-repo`, `triage-issues`, etc. sampling-free. - -5. **Cost attribution** — sampling calls hit the user's LLM quota. Should the OTel span carry the token cost in attributes (if the client reports it)? Recommend yes when available; document that the client may not report. - -6. **Fallback determinism** — when sampling is rejected, "first option in the candidate list" is one fallback strategy. Are there call sites where a different deterministic strategy is correct (e.g. "lowest tier", "least recently used")? Recommend per-call-site fallback function; sensible defaults documented. - -## Implementation sketch - -When this RFC is accepted: - -1. New `mcp-bridge/lib/sampling.js` with `requestComposition` and `requestClarification` helpers + budget tracking. -2. Helpers route through new wire-format method on the JSON-RPC server side (the bridge sends a `sampling/createMessage` request to the client, awaits response). -3. New env var `BOJ_SAMPLING_BUDGET_PER_SESSION` (default 50); declared in `glama.json`. -4. New env var `BOJ_SAMPLING_DETERMINISTIC` (default `false`); when true, helpers return the first option without sampling. -5. OTel instrumentation (depends on item 13 / PR #91): every sampling request emits a span. -6. Integration with `boj_cartridge_invoke` for the routing case (one call site to start). -7. Tests: mock-MCP-client that returns canned sampling responses; verify fallback behaviour on rejection + timeout. -8. Documentation: `docs/cartridges/SAMPLING-USAGE.md` covering when to invoke sampling, budget management, and the two patterns. - -## Linked - -- ADR-0007 (policy DSL) — sampling is *not* policy-gated (it's not a side-effectful operation), but the *result* of sampling feeds into a tool call that *is* policy-gated. -- ADR-0009 (sandbox cartridge) — sandbox provider selection is a natural composition-router use case. -- ADR-0011 (webhooks/notifications) — orthogonal but related; both are server-initiated MCP message types, both opt-in per client. -- Epic #87 item 6 (this) + item 13 (OTel, the observability surface for sampling). -- PR #89 (resources + prompts) — vocabulary; prompts could *internally* invoke sampling for sub-steps. diff --git a/docs/decisions/0013-streamable-http-transport.adoc b/docs/decisions/0013-streamable-http-transport.adoc new file mode 100644 index 00000000..52658961 --- /dev/null +++ b/docs/decisions/0013-streamable-http-transport.adoc @@ -0,0 +1,316 @@ +== 13. Streamable HTTP transport — Workers + remote browser-client deployment + +Date: 2026-05-20 + +=== Status + +Proposed (RFC — implementation tracked in epic #87 item 14) + +=== Context + +BoJ today speaks MCP over *stdio only*. The bridge process is launched +as a child by an MCP client (Claude Code, Claude Desktop, Cursor, etc.), +reads JSON-RPC from stdin, writes responses to stdout. This works +perfectly for client-launched local deployments. + +It does not work for: + +* *Cloudflare Workers* — Workers have no stdio; they’re request-response +only with strict CPU/duration limits. The Workers MCP pattern +(Cloudflare’s own MCP servers + the Workers MCP guide) uses HTTP +transport. +* *Remote browser-based agents* — A web app that wants to use BoJ tools +cannot spawn a subprocess; it has to call BoJ over HTTP. +* *Federated multi-tenant deployments* — one BoJ instance serving +multiple clients on different machines. Each client wants its own +session over the network, not stdio. +* *Long-lived sessions across stdio reconnects* — when an MCP client +crashes and restarts, stdio state is lost. HTTP transport with session +IDs lets the client resume. + +The MCP specification added *Streamable HTTP transport* (also called +HTTP+SSE) alongside stdio in 2024. The current bridge doesn’t implement +it. This is the gap. + +JSR runtime-compatibility surfaced this directly: the user can’t tick +"`Browsers`" or "`Cloudflare Workers`" because the bridge legitimately +doesn’t run there. Item 14 is the right framing of that gap — not "`make +stdio work in a browser`" (architecturally wrong) but "`add an HTTP +transport for the contexts where stdio doesn’t work.`" + +=== Decision + +Implement *Streamable HTTP transport* alongside stdio. Both transports +coexist; the binary chooses at startup based on env. No existing stdio +clients break. + +==== Transport mode selection + +.... +BOJ_TRANSPORT=stdio # current behaviour, default +BOJ_TRANSPORT=http # new — listen on BOJ_HTTP_PORT (default 7780) +BOJ_TRANSPORT=both # listen on both simultaneously +.... + +==== HTTP endpoint shape + +Two endpoints per the MCP spec: + +.... +POST /mcp — submit a JSON-RPC request, get response +GET /mcp — open SSE stream for server-initiated notifications +.... + +Single endpoint design (per the latest spec revision); some earlier MCP +HTTP implementations used `+/mcp/sse+` separately, but the spec +consolidated. + +Headers: + +.... +Mcp-Session-Id: — required on every request after initialize +Mcp-Protocol-Version: 2024-11-05 +.... + +The session ID is server-issued on `+initialize+`; client sends it on +every subsequent request. Allows the server to maintain per-session +state (rate-limit state, OTel trace context, in-flight quarantine +entries) across requests in a stateless transport. + +==== Authorization + +HTTP transport opens BoJ to the public internet (if so deployed), unlike +stdio which is local-process-bound. Authorization is required, not +optional. + +Per ADR-0007 (trust-tier policy DSL), the bridge already routes through +a PEP. HTTP transport adds an authentication layer in front of the PEP: + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Mode |Auth |When +|`+BOJ_HTTP_AUTH=none+` |No auth |Loopback only (`+127.0.0.1+`); refuses +non-loopback connections. Dev convenience. + +|`+BOJ_HTTP_AUTH=bearer+` |`+Authorization: Bearer +` |Token list +in `+BOJ_HTTP_AUTH_TOKENS+` (CSV) or `+~/.boj/http-tokens.a2ml+` +(per-peer). + +|`+BOJ_HTTP_AUTH=mtls+` |Client cert verification |Production. Cert pool +at `+BOJ_HTTP_MTLS_CA+`. + +|`+BOJ_HTTP_AUTH=oidc+` |OIDC token verification |Federated; verifies +token against an issuer config. Defers to coord-mcp’s DID model when +federation is active (ADR-0010). +|=== + +Default: *`+bearer+` if HTTP transport is enabled and the listener is +non-loopback*. The bridge refuses to start with `+BOJ_HTTP_AUTH=none+` +on a non-loopback bind. + +==== Deployment targets enabled + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Target |Works? |Caveats +|Single-process stdio (today) |✅ unchanged |Nothing breaks + +|Single-process HTTP |✅ new |Same bridge, listening on port + +|Docker / Podman container |✅ new |Already trivial; just expose the +port + +|Cloudflare Worker |✅ new |With caveats: long-lived state requires +Durable Objects; ~60% of tools (HTTP-API-based cartridges) work; +local-only cartridges (browser-mcp, container-mcp, local-coord-mcp) +don’t + +|AWS Lambda / Vercel Edge |⚠️ limited |Stateless functions can serve +`+POST /mcp+` but can’t hold SSE streams long-term + +|Browser-based agent web apps |✅ new |Via `+fetch+` against a hosted +BoJ instance +|=== + +==== What does NOT change + +* *stdio transport* — fully preserved; default mode unchanged +* *Cartridge architecture* — Idris2 ABI + Zig FFI + Deno/JS adapter +triple +* *Tool surface* — same `+tools/list+`, `+resources/list+`, +`+prompts/list+` outputs; HTTP just changes how requests arrive +* *Security gate* — `+hardeningGate+` runs identically on both +transports +* *OTel span emission* — every HTTP request emits a span (same as stdio +tools/call) + +==== Cartridge compatibility under HTTP + +For deployments where BoJ runs HTTP-transport-only (e.g. Cloudflare +Worker), some cartridges become non-functional: + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Cartridge category |HTTP-only? |Reason +|GitHub/GitLab/Cloudflare/Vercel/etc. |✅ Works |Pure HTTP API calls +outbound + +|Hugging Face / ML / Semantic Scholar |✅ Works |HTTP API + +|Search (new from item 8) |✅ Works |HTTP API + +|Database cartridges (HTTP-API ones: Supabase, Neon, Turso) |✅ Works +|HTTP API + +|Database cartridges (TCP: PostgreSQL, Redis, MongoDB) |⚠️ +Workers-limited |Cloudflare Hyperdrive workaround + +|`+browser-mcp+` (Firefox via Marionette) |❌ Local-only |Needs host +browser + +|`+container-mcp+`, `+k8s-mcp+` |❌ Local-only |Needs Podman/kubectl + +|`+local-coord-mcp+` |❌ Local-only |Loopback bus (unless ADR-0010 +federation lands) + +|`+sandbox-mcp+` (per ADR-0009, local backend) |❌ Local-only |Process +isolation requires host + +|`+sandbox-mcp+` (per ADR-0009, SaaS backends) |✅ Works |Pure HTTP API +|=== + +A new `+boj_capabilities+` extension reports per-deployment which +cartridges are actually available, so MCP clients don’t try to invoke +locally-impossible operations against a Workers-deployed BoJ. + +=== Consequences + +==== Positive + +* *Three new deployment targets* unlocked: Workers, containers, +browser-based agents +* *JSR runtime-compatibility honest* — Browsers + Workers can be ticked +once item 14 lands, with documented cartridge compatibility caveats +* *Federation alignment* (ADR-0010) — cross-machine coord federation +needs HTTPS transport anyway; this RFC’s transport is reusable for the +federation layer +* *Session state across reconnects* — long-lived MCP sessions survive +client crashes +* *Same security model* — `+hardeningGate+` + policy DSL (ADR-0007) work +identically on both transports +* *Webhook-inbound consolidation* (ADR-0011) — the existing Cowboy +listener for webhooks can host the MCP HTTP endpoint too (single port, +two paths: `+/webhooks/*+` and `+/mcp+`) +* *OTel spans for free* — instrumentation from item 13 / PR #91 applies +to HTTP requests with no extra work +* *Zero stdio breakage* — opt-in transport + +==== Negative + +* *Auth surface* — public-facing endpoint means real auth (bearer / mTLS +/ OIDC) needs implementation + careful review. Currently no +public-facing surface; this is new attack surface. +* *Session-state management* — stdio is stateful by virtue of being a +single long-lived process; HTTP is stateless per request. Bridge needs +per-session state machine (rate-limit counters, OTel trace context, +in-flight async work). Bounded, but new code. +* *Workers Durable Objects* — to hold session state in a Worker, Durable +Objects are required. Adds complexity for Worker users; documented in +the Workers-deployment guide. +* *Two transport code paths to maintain* — though most of the code +(handler logic) is shared; only the I/O layer differs. +* *Cartridge incompatibility surfaces* — operators will discover at +runtime that `+browser-mcp+` doesn’t work on a Worker; +`+boj_capabilities+` is the mitigation but it’s a new contract clients +have to learn. + +=== Non-goals + +* *Not deprecating stdio* — stdio is the default and stays so. Most MCP +clients use stdio; that’s correct. +* *Not implementing custom transports* (WebSocket, gRPC) — Streamable +HTTP+SSE is the spec; we follow it. If the spec evolves to add others, +revisit. +* *Not building a hosted SaaS BoJ* — this RFC provides the transport; it +doesn’t run a public BoJ-as-a-service. That’s a deployment choice. +* *Not changing tool/resource/prompt surfaces* — same MCP capabilities; +just a new transport. +* *Not enabling cross-cartridge isolation* that doesn’t exist on stdio. +Cartridges that work locally work over HTTP; cartridges that need local +resources don’t suddenly become Worker-compatible. + +=== Open questions + +[arabic] +. *Session storage* — for a stdio process, session state is in-memory +and disappears on process exit. For HTTP, do we persist session state +across restarts? Recommend: ephemeral in-memory by default; opt-in to +durable session storage (Redis / file) via `+BOJ_HTTP_SESSION_STORE+` +env. Workers users use Durable Objects. +. *Cold-start cost on Workers* — first request to a Worker initialises +everything. Cartridge manifest loading + Nickel validator could be slow. +Recommend: bundle the offline-menu + tool list into a static asset at +build time so cold-start hits no I/O. +. *SSE stream lifetime* — Workers have execution limits (~30s standard; +longer with Durable Objects). What happens when the client expects a +long-lived SSE stream and the Worker terminates? Recommend: documented +as a Workers limitation; clients should reconnect on stream end. +. *Backpressure under HTTP* — multiple clients to one BoJ instance can +outpace cartridge dispatch capacity. Recommend: per-session rate +limiting via ADR-0007 policy + global queue with bounded depth + 503 on +overflow. +. *TLS termination* — does the bridge speak TLS directly, or is it +always behind a reverse proxy (nginx, Cloudflare, Caddy)? Recommend: +behind a proxy by default; the bridge listens HTTP on localhost. +TLS-direct mode opt-in via `+BOJ_HTTP_TLS_*+` env vars for operators who +want a one-binary deployment. +. *Compatibility with the existing `+boj-rest+` Cowboy listener* (port +7700) — the unreleased CHANGELOG mentions "`boj-rest SSE surface: POST +/cartridge/:name/sse on the same single Cowboy listener`". Should the +MCP HTTP transport mount on that same port (Cowboy already runs there), +or a separate port? Recommend: same port; path-routed (`+/mcp+` vs +`+/cartridge/:name/*+` vs `+/webhooks/*+`). Single listener honours +ADR-0004. + +=== Implementation sketch + +When this RFC is accepted (~1-2 weeks, two PRs): + +*PR 1 — bridge HTTP transport* (~1 week) 1. New +`+mcp-bridge/lib/http-transport.js+` — Cowboy / Hono / native http (TBD +per runtime). Listen on `+BOJ_HTTP_PORT+`, dispatch JSON-RPC. 2. Session +manager — issue UUIDs on initialize, maintain per-session state map, +expire on timeout. 3. Auth middleware — bearer / mTLS / OIDC backends. +4. SSE handler — server-initiated notifications (per ADR-0011) fan out +over the session’s open SSE stream. 5. `+mcp-bridge/main.js+` reads +`+BOJ_TRANSPORT+` and starts stdio, http, or both. 6. New env vars in +`+glama.json+`: `+BOJ_TRANSPORT+`, `+BOJ_HTTP_PORT+`, `+BOJ_HTTP_AUTH+`, +`+BOJ_HTTP_AUTH_TOKENS+`, `+BOJ_HTTP_MTLS_CA+`, +`+BOJ_HTTP_SESSION_STORE+`. 7. `+boj_capabilities+` resource +(per-deployment cartridge availability). 8. Tests: HTTP transport +mirroring the stdio dispatch tests; auth-rejection tests; session-expiry +tests. + +*PR 2 — Workers deployment guide + Durable Objects shim* (~1 week) 1. +New `+docs/deployment/CLOUDFLARE-WORKERS.md+` with wrangler config +example. 2. Durable Object wrapper around the session manager. 3. Static +bundling of the cartridge manifest so cold-start is cheap. 4. Example +`+wrangler.toml+` showing minimum config. + +Subsequent: native browser-client SDK (item 14 follow-up). + +=== Linked + +* ADR-0002 (BoJ-only MCP) — HTTP transport is _within_ BoJ, not a +separate MCP server. +* ADR-0004 (unified gateway) — same Cowboy listener mounts `+/mcp+` +alongside existing paths. +* ADR-0007 (policy DSL) — auth layer fronts the existing PEP; no policy +changes. +* ADR-0009 (sandbox cartridge) — local sandbox backend incompatible with +Workers; SaaS sandbox backends compatible. +* ADR-0010 (federation) — federation transport reuses this RFC’s HTTP +layer. +* ADR-0011 (webhooks/notifications) — `+notifications/event+` fan-out +over the SSE stream defined here. +* Epic #87 item 14 (this). diff --git a/docs/decisions/0013-streamable-http-transport.md b/docs/decisions/0013-streamable-http-transport.md deleted file mode 100644 index 73157a53..00000000 --- a/docs/decisions/0013-streamable-http-transport.md +++ /dev/null @@ -1,187 +0,0 @@ - - - -# 13. Streamable HTTP transport — Workers + remote browser-client deployment - -Date: 2026-05-20 - -## Status - -Proposed (RFC — implementation tracked in epic #87 item 14) - -## Context - -BoJ today speaks MCP over **stdio only**. The bridge process is launched as a child by an MCP client (Claude Code, Claude Desktop, Cursor, etc.), reads JSON-RPC from stdin, writes responses to stdout. This works perfectly for client-launched local deployments. - -It does not work for: - -- **Cloudflare Workers** — Workers have no stdio; they're request-response only with strict CPU/duration limits. The Workers MCP pattern (Cloudflare's own MCP servers + the Workers MCP guide) uses HTTP transport. -- **Remote browser-based agents** — A web app that wants to use BoJ tools cannot spawn a subprocess; it has to call BoJ over HTTP. -- **Federated multi-tenant deployments** — one BoJ instance serving multiple clients on different machines. Each client wants its own session over the network, not stdio. -- **Long-lived sessions across stdio reconnects** — when an MCP client crashes and restarts, stdio state is lost. HTTP transport with session IDs lets the client resume. - -The MCP specification added **Streamable HTTP transport** (also called HTTP+SSE) alongside stdio in 2024. The current bridge doesn't implement it. This is the gap. - -JSR runtime-compatibility surfaced this directly: the user can't tick "Browsers" or "Cloudflare Workers" because the bridge legitimately doesn't run there. Item 14 is the right framing of that gap — not "make stdio work in a browser" (architecturally wrong) but "add an HTTP transport for the contexts where stdio doesn't work." - -## Decision - -Implement **Streamable HTTP transport** alongside stdio. Both transports coexist; the binary chooses at startup based on env. No existing stdio clients break. - -### Transport mode selection - -``` -BOJ_TRANSPORT=stdio # current behaviour, default -BOJ_TRANSPORT=http # new — listen on BOJ_HTTP_PORT (default 7780) -BOJ_TRANSPORT=both # listen on both simultaneously -``` - -### HTTP endpoint shape - -Two endpoints per the MCP spec: - -``` -POST /mcp — submit a JSON-RPC request, get response -GET /mcp — open SSE stream for server-initiated notifications -``` - -Single endpoint design (per the latest spec revision); some earlier MCP HTTP implementations used `/mcp/sse` separately, but the spec consolidated. - -Headers: -``` -Mcp-Session-Id: — required on every request after initialize -Mcp-Protocol-Version: 2024-11-05 -``` - -The session ID is server-issued on `initialize`; client sends it on every subsequent request. Allows the server to maintain per-session state (rate-limit state, OTel trace context, in-flight quarantine entries) across requests in a stateless transport. - -### Authorization - -HTTP transport opens BoJ to the public internet (if so deployed), unlike stdio which is local-process-bound. Authorization is required, not optional. - -Per ADR-0007 (trust-tier policy DSL), the bridge already routes through a PEP. HTTP transport adds an authentication layer in front of the PEP: - -| Mode | Auth | When | -|---|---|---| -| `BOJ_HTTP_AUTH=none` | No auth | Loopback only (`127.0.0.1`); refuses non-loopback connections. Dev convenience. | -| `BOJ_HTTP_AUTH=bearer` | `Authorization: Bearer ` | Token list in `BOJ_HTTP_AUTH_TOKENS` (CSV) or `~/.boj/http-tokens.a2ml` (per-peer). | -| `BOJ_HTTP_AUTH=mtls` | Client cert verification | Production. Cert pool at `BOJ_HTTP_MTLS_CA`. | -| `BOJ_HTTP_AUTH=oidc` | OIDC token verification | Federated; verifies token against an issuer config. Defers to coord-mcp's DID model when federation is active (ADR-0010). | - -Default: **`bearer` if HTTP transport is enabled and the listener is non-loopback**. The bridge refuses to start with `BOJ_HTTP_AUTH=none` on a non-loopback bind. - -### Deployment targets enabled - -| Target | Works? | Caveats | -|---|---|---| -| Single-process stdio (today) | ✅ unchanged | Nothing breaks | -| Single-process HTTP | ✅ new | Same bridge, listening on port | -| Docker / Podman container | ✅ new | Already trivial; just expose the port | -| Cloudflare Worker | ✅ new | With caveats: long-lived state requires Durable Objects; ~60% of tools (HTTP-API-based cartridges) work; local-only cartridges (browser-mcp, container-mcp, local-coord-mcp) don't | -| AWS Lambda / Vercel Edge | ⚠️ limited | Stateless functions can serve `POST /mcp` but can't hold SSE streams long-term | -| Browser-based agent web apps | ✅ new | Via `fetch` against a hosted BoJ instance | - -### What does NOT change - -- **stdio transport** — fully preserved; default mode unchanged -- **Cartridge architecture** — Idris2 ABI + Zig FFI + Deno/JS adapter triple -- **Tool surface** — same `tools/list`, `resources/list`, `prompts/list` outputs; HTTP just changes how requests arrive -- **Security gate** — `hardeningGate` runs identically on both transports -- **OTel span emission** — every HTTP request emits a span (same as stdio tools/call) - -### Cartridge compatibility under HTTP - -For deployments where BoJ runs HTTP-transport-only (e.g. Cloudflare Worker), some cartridges become non-functional: - -| Cartridge category | HTTP-only? | Reason | -|---|---|---| -| GitHub/GitLab/Cloudflare/Vercel/etc. | ✅ Works | Pure HTTP API calls outbound | -| Hugging Face / ML / Semantic Scholar | ✅ Works | HTTP API | -| Search (new from item 8) | ✅ Works | HTTP API | -| Database cartridges (HTTP-API ones: Supabase, Neon, Turso) | ✅ Works | HTTP API | -| Database cartridges (TCP: PostgreSQL, Redis, MongoDB) | ⚠️ Workers-limited | Cloudflare Hyperdrive workaround | -| `browser-mcp` (Firefox via Marionette) | ❌ Local-only | Needs host browser | -| `container-mcp`, `k8s-mcp` | ❌ Local-only | Needs Podman/kubectl | -| `local-coord-mcp` | ❌ Local-only | Loopback bus (unless ADR-0010 federation lands) | -| `sandbox-mcp` (per ADR-0009, local backend) | ❌ Local-only | Process isolation requires host | -| `sandbox-mcp` (per ADR-0009, SaaS backends) | ✅ Works | Pure HTTP API | - -A new `boj_capabilities` extension reports per-deployment which cartridges are actually available, so MCP clients don't try to invoke locally-impossible operations against a Workers-deployed BoJ. - -## Consequences - -### Positive - -- **Three new deployment targets** unlocked: Workers, containers, browser-based agents -- **JSR runtime-compatibility honest** — Browsers + Workers can be ticked once item 14 lands, with documented cartridge compatibility caveats -- **Federation alignment** (ADR-0010) — cross-machine coord federation needs HTTPS transport anyway; this RFC's transport is reusable for the federation layer -- **Session state across reconnects** — long-lived MCP sessions survive client crashes -- **Same security model** — `hardeningGate` + policy DSL (ADR-0007) work identically on both transports -- **Webhook-inbound consolidation** (ADR-0011) — the existing Cowboy listener for webhooks can host the MCP HTTP endpoint too (single port, two paths: `/webhooks/*` and `/mcp`) -- **OTel spans for free** — instrumentation from item 13 / PR #91 applies to HTTP requests with no extra work -- **Zero stdio breakage** — opt-in transport - -### Negative - -- **Auth surface** — public-facing endpoint means real auth (bearer / mTLS / OIDC) needs implementation + careful review. Currently no public-facing surface; this is new attack surface. -- **Session-state management** — stdio is stateful by virtue of being a single long-lived process; HTTP is stateless per request. Bridge needs per-session state machine (rate-limit counters, OTel trace context, in-flight async work). Bounded, but new code. -- **Workers Durable Objects** — to hold session state in a Worker, Durable Objects are required. Adds complexity for Worker users; documented in the Workers-deployment guide. -- **Two transport code paths to maintain** — though most of the code (handler logic) is shared; only the I/O layer differs. -- **Cartridge incompatibility surfaces** — operators will discover at runtime that `browser-mcp` doesn't work on a Worker; `boj_capabilities` is the mitigation but it's a new contract clients have to learn. - -## Non-goals - -- **Not deprecating stdio** — stdio is the default and stays so. Most MCP clients use stdio; that's correct. -- **Not implementing custom transports** (WebSocket, gRPC) — Streamable HTTP+SSE is the spec; we follow it. If the spec evolves to add others, revisit. -- **Not building a hosted SaaS BoJ** — this RFC provides the transport; it doesn't run a public BoJ-as-a-service. That's a deployment choice. -- **Not changing tool/resource/prompt surfaces** — same MCP capabilities; just a new transport. -- **Not enabling cross-cartridge isolation** that doesn't exist on stdio. Cartridges that work locally work over HTTP; cartridges that need local resources don't suddenly become Worker-compatible. - -## Open questions - -1. **Session storage** — for a stdio process, session state is in-memory and disappears on process exit. For HTTP, do we persist session state across restarts? Recommend: ephemeral in-memory by default; opt-in to durable session storage (Redis / file) via `BOJ_HTTP_SESSION_STORE` env. Workers users use Durable Objects. - -2. **Cold-start cost on Workers** — first request to a Worker initialises everything. Cartridge manifest loading + Nickel validator could be slow. Recommend: bundle the offline-menu + tool list into a static asset at build time so cold-start hits no I/O. - -3. **SSE stream lifetime** — Workers have execution limits (~30s standard; longer with Durable Objects). What happens when the client expects a long-lived SSE stream and the Worker terminates? Recommend: documented as a Workers limitation; clients should reconnect on stream end. - -4. **Backpressure under HTTP** — multiple clients to one BoJ instance can outpace cartridge dispatch capacity. Recommend: per-session rate limiting via ADR-0007 policy + global queue with bounded depth + 503 on overflow. - -5. **TLS termination** — does the bridge speak TLS directly, or is it always behind a reverse proxy (nginx, Cloudflare, Caddy)? Recommend: behind a proxy by default; the bridge listens HTTP on localhost. TLS-direct mode opt-in via `BOJ_HTTP_TLS_*` env vars for operators who want a one-binary deployment. - -6. **Compatibility with the existing `boj-rest` Cowboy listener** (port 7700) — the unreleased CHANGELOG mentions "boj-rest SSE surface: POST /cartridge/:name/sse on the same single Cowboy listener". Should the MCP HTTP transport mount on that same port (Cowboy already runs there), or a separate port? Recommend: same port; path-routed (`/mcp` vs `/cartridge/:name/*` vs `/webhooks/*`). Single listener honours ADR-0004. - -## Implementation sketch - -When this RFC is accepted (~1-2 weeks, two PRs): - -**PR 1 — bridge HTTP transport** (~1 week) -1. New `mcp-bridge/lib/http-transport.js` — Cowboy / Hono / native http (TBD per runtime). Listen on `BOJ_HTTP_PORT`, dispatch JSON-RPC. -2. Session manager — issue UUIDs on initialize, maintain per-session state map, expire on timeout. -3. Auth middleware — bearer / mTLS / OIDC backends. -4. SSE handler — server-initiated notifications (per ADR-0011) fan out over the session's open SSE stream. -5. `mcp-bridge/main.js` reads `BOJ_TRANSPORT` and starts stdio, http, or both. -6. New env vars in `glama.json`: `BOJ_TRANSPORT`, `BOJ_HTTP_PORT`, `BOJ_HTTP_AUTH`, `BOJ_HTTP_AUTH_TOKENS`, `BOJ_HTTP_MTLS_CA`, `BOJ_HTTP_SESSION_STORE`. -7. `boj_capabilities` resource (per-deployment cartridge availability). -8. Tests: HTTP transport mirroring the stdio dispatch tests; auth-rejection tests; session-expiry tests. - -**PR 2 — Workers deployment guide + Durable Objects shim** (~1 week) -1. New `docs/deployment/CLOUDFLARE-WORKERS.md` with wrangler config example. -2. Durable Object wrapper around the session manager. -3. Static bundling of the cartridge manifest so cold-start is cheap. -4. Example `wrangler.toml` showing minimum config. - -Subsequent: native browser-client SDK (item 14 follow-up). - -## Linked - -- ADR-0002 (BoJ-only MCP) — HTTP transport is *within* BoJ, not a separate MCP server. -- ADR-0004 (unified gateway) — same Cowboy listener mounts `/mcp` alongside existing paths. -- ADR-0007 (policy DSL) — auth layer fronts the existing PEP; no policy changes. -- ADR-0009 (sandbox cartridge) — local sandbox backend incompatible with Workers; SaaS sandbox backends compatible. -- ADR-0010 (federation) — federation transport reuses this RFC's HTTP layer. -- ADR-0011 (webhooks/notifications) — `notifications/event` fan-out over the SSE stream defined here. -- Epic #87 item 14 (this). diff --git a/docs/decisions/0014-cross-cartridge-composition-safety.adoc b/docs/decisions/0014-cross-cartridge-composition-safety.adoc new file mode 100644 index 00000000..c841b65d --- /dev/null +++ b/docs/decisions/0014-cross-cartridge-composition-safety.adoc @@ -0,0 +1,251 @@ +== 14. Cross-cartridge composition safety — defining what we mean and how to prove it + +Date: 2026-05-20 + +=== Status + +Proposed (RFC — implementation tracked in epic #87 Tier C item 12; this +ADR is the framing document, not a build plan) + +=== Context + +The existing Idris2 ABI proves a lot for *individual* cartridges: + +* `+Boj.CartridgeDispatch+` (BJ1) shows the dispatcher is type-safe: +`+ProtocolMatch + ReadinessGuard + Disjointness+` — well-typed dispatch +cannot route a request to a cartridge that doesn’t advertise the +protocol or that hasn’t passed the safety gate (`+IsUnbreakable+`). +* `+Boj.Catalogue+` defines `+IsUnbreakable+` as the gate that admits +only `+Ready+` cartridges. +* Per-cartridge `+Safe*.idr+` modules (`+SafeHTTP+`, `+SafeCORS+`, +`+SafeAPIKey+`, `+SafeWebSocket+`, `+SafePromptInjection+`, +`+Federation+`, `+CredentialIsolation+`, `+APIContractCoverage+`) +discharge the P-01..P-07 proof obligations local to that cartridge. + +The 2026-05-18 audit (`+PROOF-NEEDS.md+`) records 5 remaining +`+believe_me+` sites, all class (J) — principled assumptions over Idris2 +0.8.0’s opaque `+Char+`/`+String+` primitives. Within the per-cartridge +ABI surface, the proof debt is closed. + +What is *not* proved is what happens when one cartridge calls another. +Concretely: BoJ exposes a 5th-standard-symbol Zig FFI export per +cartridge — `+boj_cartridge_invoke(tool, args_json, out, out_len)+` — +and the bridge happily lets cartridge A’s handler call cartridge B’s +`+boj_cartridge_invoke+` to compose tools. ADR-0006 defines the invoke +surface; ADR-0009 (sandbox cartridge, RFC-only as of this writing — +`+cartridges/sandbox-mcp/+` does not yet exist on disk) and ADR-0010 +(cross-machine coord federation) both lean on composition without +defining what makes a composition _safe_. + +We have a hole: per-cartridge invariants do not compose automatically. +If cartridge A’s tool requires `+IsUnbreakable+` and produces JSON whose +shape satisfies its own `+SafeHTTP+` contract, that says nothing about +whether the resulting JSON, fed into cartridge B’s +`+boj_cartridge_invoke+`, satisfies B’s preconditions. The Idris2 type +system can only check this if the cartridges share a typed composition +boundary; today they share only bytes. + +This ADR scopes the problem so a campaign can attack it. + +=== Decision (framing, not implementation) + +Define *composition safety* for BoJ as a *two-level contract*, with the +Idris2 ABI carrying the _static_ level and a Nickel-encoded policy layer +(via ADR-0007’s `+policy-mcp+` PDP) carrying the _dynamic_ level. +Discharge them in a first pair, then generalise. + +==== Level 1 — Static (Idris2): typed composition envelope + +Introduce a new module `+Boj.Composition+` that lifts the per-cartridge +ABI obligations into a composition-aware envelope: + +[source,idris] +---- +||| A typed inter-cartridge invocation. +||| Carries proofs that the source cartridge is Ready, the target +||| cartridge is Ready, the requested tool is in the target's +||| advertised tool list, and the marshalled args satisfy the target's +||| input contract. +public export +record InvocationOf (src, tgt : Cartridge) (tool : ToolName) where + constructor MkInvocation + srcReady : IsUnbreakable src + tgtReady : IsUnbreakable tgt + toolValid : tool `elem` advertisedTools tgt = True + argsOk : ArgsContract tgt tool args +---- + +A composition is *statically safe* iff `+boj_cartridge_invoke+` only +fires when the caller can construct a witness of +`+InvocationOf src tgt tool+`. The dispatcher (BJ1) already proves the +field-2 and field-3 analogues for _external_ requests; this lifts the +same proofs into the _internal_ invocation path. + +`+ArgsContract+` is the open piece — it has to be defined per-cartridge, +mirroring whatever invariants the target’s local `+Safe*.idr+` already +encodes. For the first proof pair we propose mechanically deriving it +from each cartridge’s existing JSON-schema in `+cartridge.json+`. + +==== Level 2 — Dynamic (Nickel policy contract): composition admissibility + +The static envelope says _if_ you can typecheck the invocation, it’s +shape-safe. It does not say the invocation is _policy_-safe — e.g. a +Teranga-tier cartridge should not be allowed to invoke an Ayo-tier +cartridge in a way that elevates privilege. That’s the job of the PDP +introduced by ADR-0007. + +Extend `+policy-mcp+`’s Nickel schema with a `+compositions+` block: + +[source,nickel] +---- +compositions = { + # Allow panic-attack -> vordr (untrusted-execution flow per ADR-0009) + "panic-attack-mcp".allows_invoking = ["vordr-mcp"], + "panic-attack-mcp".max_call_depth = 1, + "panic-attack-mcp".target_must_be_tier_lte = 2, + + # Deny ayo -> teranga (privilege elevation) + "*-ayo".allows_invoking_tier_lt_self = false, +} +---- + +The bridge’s `+hardeningGate+` consults the PDP on every internal +invocation, the same way it already consults it for external dispatch. + +==== First proof pair (recommended, with caveat) + +The prompt suggested `+panic-attack-mcp → sandbox-mcp → vordr-mcp+` as +the canonical untrusted-execution chain. Two of the three exist on disk +today: + +* `+cartridges/panic-attack-mcp/+` — exists +* `+cartridges/sandbox-mcp/+` — *does not exist*; ADR-0009 proposes it +but no implementation has landed +* `+cartridges/vordr-mcp/+` — exists + +Therefore the first composition pair to discharge is +*`+panic-attack-mcp → vordr-mcp+`* (a real pair, both Teranga / Shield, +both with extant ABI). The three-cartridge chain remains the eventual +target; sandbox-mcp needs to be built (per ADR-0009) before the longer +chain can be proved. + +==== What we are explicitly *not* doing + +* *Not lifting all cartridges into a single shared Idris2 module.* That +would be a multi-year refactor. `+InvocationOf+` is parameterised on +`+Cartridge+` (already defined in `+Boj.Catalogue+`) and pulls the +per-cartridge contract via a typeclass / interface, not by source-level +unification. +* *Not making every existing cross-cartridge call retroactively proven.* +Existing internal invocations stay on the untyped byte path; new ones go +through `+InvocationOf+` once it lands. Migration is per-pair, +prioritised by trust-tier risk. +* *Not weakening any existing single-cartridge proof.* If a P-01..P-07 +obligation conflicts with the composition envelope, the obligation is +right and the envelope shape is wrong. + +=== Discharge mechanism — sub-RFCs and PRs + +The campaign decomposes into roughly six sub-issues, each its own PR: + +[arabic] +. *`+Boj.Composition+` skeleton.* Define `+InvocationOf+`, +`+ArgsContract+` as an interface, and helpers. No cartridge wired up. +Build-green via `+idris2 --check+`. Refs item 12. +. *First pair: `+panic-attack-mcp → vordr-mcp+` static proof.* Provide +`+ArgsContract+` instances for both cartridges’ tools, prove one real +panic-attack → vordr invocation in +`+cartridges/panic-attack-mcp/abi/ PanicAttackMcp/CompositionWithVordr.idr+`. +PR pattern: one `+CompositionWith*.idr+` per pair. +. *Bridge wiring (PEP).* `+mcp-bridge/lib/dispatcher.js+` learns to look +for a composition envelope on internal `+boj_cartridge_invoke+` calls; +rejects on absence when the source/target pair is gated. +. *`+policy-mcp+` `+compositions+` block (PDP).* Schema, fixtures, +ADR-0007 follow-on PR. +. *Wire-up tests* in `+mcp-bridge/tests/composition_test.js+` — property +tests across the gate truth table (allowed pair, denied pair, +depth-exceeded, tier-violation, target-not-ready). +. *JOSS paper.* `+docs/papers/joss-composition-proof.md+` — the +formal-verification differentiation of BoJ vs other MCP servers, with +`+panic-attack → vordr+` as the worked example. + +=== JOSS paper angle + +BoJ’s unique posture relative to other MCP servers is the +formally-verified ABI. No other MCP server (per ADR-0008’s marketplace +survey) carries an Idris2 proof layer. Submitting the composition-proof +artifact to JOSS does two things: + +* Establishes priority on the framing of MCP composition as a typed +problem. +* Provides academic citability for downstream users — the OU affiliation +makes the route tractable. + +The paper draft should be sketched *alongside* the first pair’s proof, +not after, so the framing in the paper and the framing in the code stay +coherent. + +=== Constraints and non-negotiables + +[arabic] +. *Idris2 build stays green throughout.* Every PR must include a +`+idris2 --check+` invocation. (Note: at time of writing, +`+src/abi/ boj.ipkg --build+` does not complete in 9 min on a local +dev box but individual `+--check+` runs are fast. Pre-existing +baseline-rot in `+Boj.SafeAPIKey+` is tracked separately; do not block +composition work on it.) +. *No `+believe_me+` net-additions.* The 5 existing axioms are +irreducible (PROOF-NEEDS audit 2026-05-18). New axioms only via explicit +`+parameters+` blocks with documented rationale. +. *The honest count rule.* If a proof obligation in the composition +envelope turns out to be irreducible, *document it as a principled +assumption* rather than papering over it. The honest count is more +valuable than a forced "`zero`". +. *No cartridge invented for proof convenience.* Specifically: +`+sandbox-mcp+` (ADR-0009) is RFC-only today. The campaign uses real +cartridges (panic-attack, vordr) for the first pair; the three-step +chain is parked behind ADR-0009 implementation. + +=== Definition of done + +* [ ] `+Boj.Composition+` lands and type-checks (sub-PR 1). +* [ ] At least one composition pair proven end-to-end (sub-PR 2). +* [ ] Bridge PEP + PDP composition rules in production for that pair +(sub-PRs 3 + 4). +* [ ] Wire-up tests pass on CI (sub-PR 5). +* [ ] JOSS paper draft submitted, with `+panic-attack → vordr+` as the +worked example (sub-PR 6). +* [ ] Per-pair sub-issues filed for the next composition pairs; epic #87 +item 12 stays open until coverage is "`meaningful`" (specific threshold +TBD with the owner — likely "`all destructive-side-effect cartridges +have at least one proven invoker`"). + +=== Open questions + +* *`+ArgsContract+` derivation strategy.* Mechanically lift from each +cartridge’s `+cartridge.json+` JSON-schema (cheap, brittle) vs hand- +written per cartridge (expensive, robust) vs a Nickel intermediate +schema (medium). Recommend deciding before sub-PR 1 lands. +* *Federation interaction.* Cross-machine invocations (ADR-0010) need +their own composition story — the envelope has to survive serialisation. +Punt to a follow-on ADR once the local case lands. +* *Sandbox-mcp build-out.* The full panic-attack → sandbox → vordr chain +is the JOSS-paper-shaped goal; without sandbox-mcp it’s a two- cartridge +demo. ADR-0009 needs to be built first, or the proof scope needs to be +honestly limited to the two-cartridge case in the paper. + +=== References + +* ADR-0006 — cartridge-invoke ABI (defines `+boj_cartridge_invoke+`) +* ADR-0007 — trust-tier policy DSL (defines policy-mcp PDP that this ADR +extends with a `+compositions+` block) +* ADR-0009 — sandbox cartridge (RFC-only; the missing cartridge in the +canonical three-step chain) +* ADR-0010 — cross-machine coord federation (depends on a composition +story; needs follow-on) +* `+PROOF-NEEDS.md+` — 2026-05-18 audit; the 5 class (J) axioms; closure +of P1/P2 obligations per cartridge +* `+src/abi/Boj/CartridgeDispatch.idr+` — BJ1 (the existing external- +dispatch proof that this ADR lifts into the internal-invocation +envelope) +* Epic #87 Tier C item 12 — the campaign tracker diff --git a/docs/decisions/0014-cross-cartridge-composition-safety.md b/docs/decisions/0014-cross-cartridge-composition-safety.md deleted file mode 100644 index 8ed0d5b4..00000000 --- a/docs/decisions/0014-cross-cartridge-composition-safety.md +++ /dev/null @@ -1,252 +0,0 @@ - - - -# 14. Cross-cartridge composition safety — defining what we mean and how to prove it - -Date: 2026-05-20 - -## Status - -Proposed (RFC — implementation tracked in epic #87 Tier C item 12; this -ADR is the framing document, not a build plan) - -## Context - -The existing Idris2 ABI proves a lot for **individual** cartridges: - -- `Boj.CartridgeDispatch` (BJ1) shows the dispatcher is type-safe: - `ProtocolMatch + ReadinessGuard + Disjointness` — well-typed dispatch - cannot route a request to a cartridge that doesn't advertise the - protocol or that hasn't passed the safety gate (`IsUnbreakable`). -- `Boj.Catalogue` defines `IsUnbreakable` as the gate that admits only - `Ready` cartridges. -- Per-cartridge `Safe*.idr` modules (`SafeHTTP`, `SafeCORS`, `SafeAPIKey`, - `SafeWebSocket`, `SafePromptInjection`, `Federation`, - `CredentialIsolation`, `APIContractCoverage`) discharge the P-01..P-07 - proof obligations local to that cartridge. - -The 2026-05-18 audit (`PROOF-NEEDS.md`) records 5 remaining `believe_me` -sites, all class (J) — principled assumptions over Idris2 0.8.0's opaque -`Char`/`String` primitives. Within the per-cartridge ABI surface, the -proof debt is closed. - -What is **not** proved is what happens when one cartridge calls another. -Concretely: BoJ exposes a 5th-standard-symbol Zig FFI export per -cartridge — `boj_cartridge_invoke(tool, args_json, out, out_len)` — and -the bridge happily lets cartridge A's handler call cartridge B's -`boj_cartridge_invoke` to compose tools. ADR-0006 defines the invoke -surface; ADR-0009 (sandbox cartridge, RFC-only as of this writing — -`cartridges/sandbox-mcp/` does not yet exist on disk) and ADR-0010 -(cross-machine coord federation) both lean on composition without -defining what makes a composition *safe*. - -We have a hole: per-cartridge invariants do not compose automatically. -If cartridge A's tool requires `IsUnbreakable` and produces JSON whose -shape satisfies its own `SafeHTTP` contract, that says nothing about -whether the resulting JSON, fed into cartridge B's -`boj_cartridge_invoke`, satisfies B's preconditions. The Idris2 type -system can only check this if the cartridges share a typed composition -boundary; today they share only bytes. - -This ADR scopes the problem so a campaign can attack it. - -## Decision (framing, not implementation) - -Define **composition safety** for BoJ as a **two-level contract**, with -the Idris2 ABI carrying the *static* level and a Nickel-encoded policy -layer (via ADR-0007's `policy-mcp` PDP) carrying the *dynamic* level. -Discharge them in a first pair, then generalise. - -### Level 1 — Static (Idris2): typed composition envelope - -Introduce a new module `Boj.Composition` that lifts the per-cartridge -ABI obligations into a composition-aware envelope: - -```idris -||| A typed inter-cartridge invocation. -||| Carries proofs that the source cartridge is Ready, the target -||| cartridge is Ready, the requested tool is in the target's -||| advertised tool list, and the marshalled args satisfy the target's -||| input contract. -public export -record InvocationOf (src, tgt : Cartridge) (tool : ToolName) where - constructor MkInvocation - srcReady : IsUnbreakable src - tgtReady : IsUnbreakable tgt - toolValid : tool `elem` advertisedTools tgt = True - argsOk : ArgsContract tgt tool args -``` - -A composition is **statically safe** iff `boj_cartridge_invoke` only -fires when the caller can construct a witness of `InvocationOf src tgt -tool`. The dispatcher (BJ1) already proves the field-2 and field-3 -analogues for *external* requests; this lifts the same proofs into the -*internal* invocation path. - -`ArgsContract` is the open piece — it has to be defined per-cartridge, -mirroring whatever invariants the target's local `Safe*.idr` already -encodes. For the first proof pair we propose mechanically deriving it -from each cartridge's existing JSON-schema in `cartridge.json`. - -### Level 2 — Dynamic (Nickel policy contract): composition admissibility - -The static envelope says *if* you can typecheck the invocation, it's -shape-safe. It does not say the invocation is *policy*-safe — e.g. a -Teranga-tier cartridge should not be allowed to invoke an -Ayo-tier cartridge in a way that elevates privilege. That's the job of -the PDP introduced by ADR-0007. - -Extend `policy-mcp`'s Nickel schema with a `compositions` block: - -```nickel -compositions = { - # Allow panic-attack -> vordr (untrusted-execution flow per ADR-0009) - "panic-attack-mcp".allows_invoking = ["vordr-mcp"], - "panic-attack-mcp".max_call_depth = 1, - "panic-attack-mcp".target_must_be_tier_lte = 2, - - # Deny ayo -> teranga (privilege elevation) - "*-ayo".allows_invoking_tier_lt_self = false, -} -``` - -The bridge's `hardeningGate` consults the PDP on every internal -invocation, the same way it already consults it for external dispatch. - -### First proof pair (recommended, with caveat) - -The prompt suggested `panic-attack-mcp → sandbox-mcp → vordr-mcp` as the -canonical untrusted-execution chain. Two of the three exist on disk -today: - -- `cartridges/panic-attack-mcp/` — exists -- `cartridges/sandbox-mcp/` — **does not exist**; ADR-0009 proposes - it but no implementation has landed -- `cartridges/vordr-mcp/` — exists - -Therefore the first composition pair to discharge is **`panic-attack-mcp -→ vordr-mcp`** (a real pair, both Teranga / Shield, both with extant -ABI). The three-cartridge chain remains the eventual target; sandbox-mcp -needs to be built (per ADR-0009) before the longer chain can be proved. - -### What we are explicitly **not** doing - -- **Not lifting all cartridges into a single shared Idris2 module.** - That would be a multi-year refactor. `InvocationOf` is parameterised - on `Cartridge` (already defined in `Boj.Catalogue`) and pulls the - per-cartridge contract via a typeclass / interface, not by source-level - unification. -- **Not making every existing cross-cartridge call retroactively - proven.** Existing internal invocations stay on the untyped byte - path; new ones go through `InvocationOf` once it lands. Migration is - per-pair, prioritised by trust-tier risk. -- **Not weakening any existing single-cartridge proof.** If a P-01..P-07 - obligation conflicts with the composition envelope, the obligation is - right and the envelope shape is wrong. - -## Discharge mechanism — sub-RFCs and PRs - -The campaign decomposes into roughly six sub-issues, each its own PR: - -1. **`Boj.Composition` skeleton.** Define `InvocationOf`, `ArgsContract` - as an interface, and helpers. No cartridge wired up. Build-green via - `idris2 --check`. Refs item 12. -2. **First pair: `panic-attack-mcp → vordr-mcp` static proof.** Provide - `ArgsContract` instances for both cartridges' tools, prove one real - panic-attack → vordr invocation in `cartridges/panic-attack-mcp/abi/ - PanicAttackMcp/CompositionWithVordr.idr`. PR pattern: one - `CompositionWith*.idr` per pair. -3. **Bridge wiring (PEP).** `mcp-bridge/lib/dispatcher.js` learns to - look for a composition envelope on internal `boj_cartridge_invoke` - calls; rejects on absence when the source/target pair is gated. -4. **`policy-mcp` `compositions` block (PDP).** Schema, fixtures, - ADR-0007 follow-on PR. -5. **Wire-up tests** in `mcp-bridge/tests/composition_test.js` — - property tests across the gate truth table (allowed pair, denied - pair, depth-exceeded, tier-violation, target-not-ready). -6. **JOSS paper.** `docs/papers/joss-composition-proof.md` — the - formal-verification differentiation of BoJ vs other MCP servers, - with `panic-attack → vordr` as the worked example. - -## JOSS paper angle - -BoJ's unique posture relative to other MCP servers is the -formally-verified ABI. No other MCP server (per ADR-0008's -marketplace survey) carries an Idris2 proof layer. Submitting the -composition-proof artifact to JOSS does two things: - -- Establishes priority on the framing of MCP composition as a typed - problem. -- Provides academic citability for downstream users — the OU - affiliation makes the route tractable. - -The paper draft should be sketched **alongside** the first pair's -proof, not after, so the framing in the paper and the framing in the -code stay coherent. - -## Constraints and non-negotiables - -1. **Idris2 build stays green throughout.** Every PR must include a - `idris2 --check` invocation. (Note: at time of writing, `src/abi/ - boj.ipkg --build` does not complete in 9 min on a local dev box but - individual `--check` runs are fast. Pre-existing baseline-rot in - `Boj.SafeAPIKey` is tracked separately; do not block composition - work on it.) -2. **No `believe_me` net-additions.** The 5 existing axioms are - irreducible (PROOF-NEEDS audit 2026-05-18). New axioms only via - explicit `parameters` blocks with documented rationale. -3. **The honest count rule.** If a proof obligation in the composition - envelope turns out to be irreducible, **document it as a principled - assumption** rather than papering over it. The honest count is more - valuable than a forced "zero". -4. **No cartridge invented for proof convenience.** Specifically: - `sandbox-mcp` (ADR-0009) is RFC-only today. The campaign uses real - cartridges (panic-attack, vordr) for the first pair; the three-step - chain is parked behind ADR-0009 implementation. - -## Definition of done - -- [ ] `Boj.Composition` lands and type-checks (sub-PR 1). -- [ ] At least one composition pair proven end-to-end (sub-PR 2). -- [ ] Bridge PEP + PDP composition rules in production for that pair - (sub-PRs 3 + 4). -- [ ] Wire-up tests pass on CI (sub-PR 5). -- [ ] JOSS paper draft submitted, with `panic-attack → vordr` as the - worked example (sub-PR 6). -- [ ] Per-pair sub-issues filed for the next composition pairs; epic - #87 item 12 stays open until coverage is "meaningful" (specific - threshold TBD with the owner — likely "all destructive-side-effect - cartridges have at least one proven invoker"). - -## Open questions - -- **`ArgsContract` derivation strategy.** Mechanically lift from each - cartridge's `cartridge.json` JSON-schema (cheap, brittle) vs hand- - written per cartridge (expensive, robust) vs a Nickel intermediate - schema (medium). Recommend deciding before sub-PR 1 lands. -- **Federation interaction.** Cross-machine invocations (ADR-0010) need - their own composition story — the envelope has to survive - serialisation. Punt to a follow-on ADR once the local case lands. -- **Sandbox-mcp build-out.** The full panic-attack → sandbox → vordr - chain is the JOSS-paper-shaped goal; without sandbox-mcp it's a two- - cartridge demo. ADR-0009 needs to be built first, or the proof scope - needs to be honestly limited to the two-cartridge case in the paper. - -## References - -- ADR-0006 — cartridge-invoke ABI (defines `boj_cartridge_invoke`) -- ADR-0007 — trust-tier policy DSL (defines policy-mcp PDP that this - ADR extends with a `compositions` block) -- ADR-0009 — sandbox cartridge (RFC-only; the missing cartridge in the - canonical three-step chain) -- ADR-0010 — cross-machine coord federation (depends on a composition - story; needs follow-on) -- `PROOF-NEEDS.md` — 2026-05-18 audit; the 5 class (J) axioms; closure - of P1/P2 obligations per cartridge -- `src/abi/Boj/CartridgeDispatch.idr` — BJ1 (the existing external- - dispatch proof that this ADR lifts into the internal-invocation - envelope) -- Epic #87 Tier C item 12 — the campaign tracker diff --git a/docs/decisions/0015-backend-file-lock-primitive.adoc b/docs/decisions/0015-backend-file-lock-primitive.adoc new file mode 100644 index 00000000..07b2faa0 --- /dev/null +++ b/docs/decisions/0015-backend-file-lock-primitive.adoc @@ -0,0 +1,139 @@ +== 15. Backend-enforced file-lock primitive — spike + +Date: 2026-05-24 + +=== Status + +Deferred (2026-05-24) — ADR-0016 (cross-host federation stop-gap) was +chosen as the next build over this one because it closes a more +user-visible survey gap without altering the verified backend core. The +bridge-layer advisory path-claims shipped in PR #142/#143 remain the +current answer for in-flight conflict signalling. + +This ADR stays on file as the design-of-record for a backend-enforced +lock primitive should "`advisory warning is not enough`" become a stated +requirement. Reopen by flipping to "`Proposed`" and scheduling alongside +the next P-0x proof-obligation cycle. + +=== Context + +The multi-agent MCP survey identified that two comparator servers +(`+rinadelph/Agent-MCP+`, +`+AndrewDavidRivers/multi-agent-coordination-mcp+`) ship *hard file +locks* at claim time, whereas `+local-coord-mcp+` has only the +bridge-layer *advisory* path-claims added in PR #142 / PR #143. The +survey marked file-level locks as a clear gap relative to those two. + +This spike evaluates promoting path-claims from "`bridge-only advisory +warning`" to a *backend-enforced lock primitive* in the verified Idris2 +ABI + Zig FFI. + +=== What the change does + +[arabic] +. Extend the `+LocalCoord+` Idris2 ABI with a new tool surface: +`+coord_lock_paths(token, task, paths[])+` and +`+coord_unlock_paths(token, task)+`. Paths are interned, normalised, and +the backend maintains an authoritative `+task → segment[][]+` map +alongside the existing claim map. +. The lock check runs *inside* `+coord_claim_task+` when `+paths+` is +present: if any declared path segment-overlaps an existing locked path +held by another peer, the claim is *rejected* (not annotated). Today’s +bridge-layer overlap scan becomes a projection of the backend’s +authoritative state. +. New proof obligation *P-08: LockSoundness* in +`+cartridges/local-coord-mcp/abi/LocalCoord/Locks.idr+`, discharged by +construction: +* *Mutual exclusion* — no two distinct tasks simultaneously hold +overlapping paths (segment-prefix-disjoint). +* *Lock-claim composition* — a granted claim’s path-locks survive until +`+coord_unlock_paths+` or watchdog expiry; never silently released by a +different peer. +* *Watchdog interaction* — when the claim’s role-based TTL fires (P-03 +WatchdogTermination), the lock-set is released atomically with the +claim. +. Cartridge schema bump: `+cartridge.json+` adds the two new tools and +widens `+coord_claim_task+` to declare `+path_locked+` as a rejection +reason in its output schema. Coherence test (`+dispatch_test.js+`) +already enforces bridge↔cartridge sync, so the bridge tool list and +dispatcher must update in lockstep. + +=== Cost + +[width="100%",cols="50%,50%",options="header",] +|=== +|Surface |Work +|Idris2 ABI |New `+Locks.idr+` module + P-08 discharge. Roughly 200-300 +lines of Idris2; the segment-prefix proof reduces to list-prefix +induction on `+List String+`, which is constructive (no new +`+believe_me+`). + +|Zig FFI |`+boj_coord_lock_paths+` / `+boj_coord_unlock_paths+` exports ++ an internal `+PathLockTable+` (radix tree or sorted-array; flat array +is fine at expected scale). ~150 LOC. + +|Bridge |Demote `+path-claims.js+` to a stateless projection — query the +backend’s lock table on demand rather than maintaining its own. Or +delete it. + +|Cartridge schema |Two new tool entries; one output-shape widening on +`+coord_claim_task+`. + +|Tests |New Idris2 totality proofs (P-08), Zig unit tests, bridge +integration tests, bench update. +|=== + +Estimated: 2-4 working days end-to-end. + +=== Benefit + +* Closes the survey gap *completely*: backend-enforced locks rather than +an advisory layer. Strongest answer. +* Idris2 proof gives a guarantee neither Agent-MCP nor multi-agent-coord +has — they ship runtime-checked locks, we’d ship a _proved-correct_ lock +primitive. Aligns with the AAA Formal posture (P-01..P-07). +* The bridge-layer code shrinks (path-claims.js largely becomes a +pass-through projection), which is a healthy direction. + +=== Risks / open questions + +* *Architectural regression risk.* Today’s design is explicit: the +verified backend stays minimal and the bridge layers advisory features. +Adding state to the backend (path-locks) is the first time a non-task-id +concept enters the proved core. P-08 must not weaken the existing proofs +— needs careful composition argument. +* *Schema-evolution lock-in.* Once `+coord_lock_paths+` is in the +cartridge manifest, removing it is a breaking change. Path-semantics +(segment-prefix matching) gets baked into the wire contract. +* *Defines vs. enforces.* What does the backend do when a _non-path_ +claim conflicts with a locked path? The current advisory layer is silent +in that case; an enforcer must take a position. +* *Failure modes change.* Today a clashing path is a _warning_; under +this proposal it’s a _claim rejection_. Existing rate-limit (5 +rejections / 10 min → 30s cooldown) applies; either way, agents need +retry logic. + +=== Verdict + +Strongest technical answer to the survey gap, but the largest +architectural commitment in this branch of work. The proof load (P-08) +is tractable — segment-prefix mutual exclusion is structurally inductive +— but the _design_ commitment (locks as a first-class backend concept) +deserves the user’s explicit sign-off, not an implementer’s judgement +call. + +*Recommendation:* Choose this only if "`advisory warning is not enough`" +is a stated requirement. Otherwise, prefer ADR-0016 (cross-host +federation stop-gap), which closes a different and arguably more +user-visible survey gap (Ruflo being the only comparator with real +cross-host story) without altering the verified core. + +=== Out of scope + +* File-range locks (byte ranges within a file). Not addressed — paths +are still whole-file. The Perforce-style range lock would be a separate, +larger ADR. +* Lock fairness / queueing on contention. This spike rejects on overlap; +a queue would be a P-09 extension. +* Cross-host locks. Locks here are still localhost-bound; combining with +ADR-0010 or ADR-0016 is future work. diff --git a/docs/decisions/0015-backend-file-lock-primitive.md b/docs/decisions/0015-backend-file-lock-primitive.md deleted file mode 100644 index 9d9b67dc..00000000 --- a/docs/decisions/0015-backend-file-lock-primitive.md +++ /dev/null @@ -1,128 +0,0 @@ - - - -# 15. Backend-enforced file-lock primitive — spike - -Date: 2026-05-24 - -## Status - -Deferred (2026-05-24) — ADR-0016 (cross-host federation stop-gap) was -chosen as the next build over this one because it closes a more -user-visible survey gap without altering the verified backend core. -The bridge-layer advisory path-claims shipped in PR #142/#143 remain -the current answer for in-flight conflict signalling. - -This ADR stays on file as the design-of-record for a backend-enforced -lock primitive should "advisory warning is not enough" become a stated -requirement. Reopen by flipping to "Proposed" and scheduling alongside -the next P-0x proof-obligation cycle. - -## Context - -The multi-agent MCP survey identified that two comparator servers -(`rinadelph/Agent-MCP`, `AndrewDavidRivers/multi-agent-coordination-mcp`) -ship **hard file locks** at claim time, whereas `local-coord-mcp` has -only the bridge-layer **advisory** path-claims added in PR #142 / PR #143. -The survey marked file-level locks as a clear gap relative to those two. - -This spike evaluates promoting path-claims from "bridge-only advisory -warning" to a **backend-enforced lock primitive** in the verified Idris2 -ABI + Zig FFI. - -## What the change does - -1. Extend the `LocalCoord` Idris2 ABI with a new tool surface: - `coord_lock_paths(token, task, paths[])` and - `coord_unlock_paths(token, task)`. Paths are interned, normalised, and - the backend maintains an authoritative `task → segment[][]` map - alongside the existing claim map. -2. The lock check runs **inside** `coord_claim_task` when `paths` is - present: if any declared path segment-overlaps an existing locked - path held by another peer, the claim is **rejected** (not annotated). - Today's bridge-layer overlap scan becomes a projection of the - backend's authoritative state. -3. New proof obligation **P-08: LockSoundness** in - `cartridges/local-coord-mcp/abi/LocalCoord/Locks.idr`, discharged by - construction: - - **Mutual exclusion** — no two distinct tasks simultaneously hold - overlapping paths (segment-prefix-disjoint). - - **Lock-claim composition** — a granted claim's path-locks survive - until `coord_unlock_paths` or watchdog expiry; never silently - released by a different peer. - - **Watchdog interaction** — when the claim's role-based TTL fires - (P-03 WatchdogTermination), the lock-set is released atomically - with the claim. -4. Cartridge schema bump: `cartridge.json` adds the two new tools and - widens `coord_claim_task` to declare `path_locked` as a rejection - reason in its output schema. Coherence test (`dispatch_test.js`) - already enforces bridge↔cartridge sync, so the bridge tool list and - dispatcher must update in lockstep. - -## Cost - -| Surface | Work | -|---|---| -| Idris2 ABI | New `Locks.idr` module + P-08 discharge. Roughly 200-300 lines of Idris2; the segment-prefix proof reduces to list-prefix induction on `List String`, which is constructive (no new `believe_me`). | -| Zig FFI | `boj_coord_lock_paths` / `boj_coord_unlock_paths` exports + an internal `PathLockTable` (radix tree or sorted-array; flat array is fine at expected scale). ~150 LOC. | -| Bridge | Demote `path-claims.js` to a stateless projection — query the backend's lock table on demand rather than maintaining its own. Or delete it. | -| Cartridge schema | Two new tool entries; one output-shape widening on `coord_claim_task`. | -| Tests | New Idris2 totality proofs (P-08), Zig unit tests, bridge integration tests, bench update. | - -Estimated: 2-4 working days end-to-end. - -## Benefit - -- Closes the survey gap **completely**: backend-enforced locks rather - than an advisory layer. Strongest answer. -- Idris2 proof gives a guarantee neither Agent-MCP nor multi-agent-coord - has — they ship runtime-checked locks, we'd ship a *proved-correct* - lock primitive. Aligns with the AAA Formal posture (P-01..P-07). -- The bridge-layer code shrinks (path-claims.js largely becomes a - pass-through projection), which is a healthy direction. - -## Risks / open questions - -- **Architectural regression risk.** Today's design is explicit: the - verified backend stays minimal and the bridge layers advisory - features. Adding state to the backend (path-locks) is the first time - a non-task-id concept enters the proved core. P-08 must not weaken - the existing proofs — needs careful composition argument. -- **Schema-evolution lock-in.** Once `coord_lock_paths` is in the - cartridge manifest, removing it is a breaking change. Path-semantics - (segment-prefix matching) gets baked into the wire contract. -- **Defines vs. enforces.** What does the backend do when a *non-path* - claim conflicts with a locked path? The current advisory layer is - silent in that case; an enforcer must take a position. -- **Failure modes change.** Today a clashing path is a *warning*; under - this proposal it's a *claim rejection*. Existing rate-limit (5 - rejections / 10 min → 30s cooldown) applies; either way, agents need - retry logic. - -## Verdict - -Strongest technical answer to the survey gap, but the largest -architectural commitment in this branch of work. The proof load -(P-08) is tractable — segment-prefix mutual exclusion is structurally -inductive — but the *design* commitment (locks as a first-class backend -concept) deserves the user's explicit sign-off, not an implementer's -judgement call. - -**Recommendation:** Choose this only if "advisory warning is not -enough" is a stated requirement. Otherwise, prefer ADR-0016 -(cross-host federation stop-gap), which closes a different and arguably -more user-visible survey gap (Ruflo being the only comparator with real -cross-host story) without altering the verified core. - -## Out of scope - -- File-range locks (byte ranges within a file). Not addressed — paths - are still whole-file. The Perforce-style range lock would be a - separate, larger ADR. -- Lock fairness / queueing on contention. This spike rejects on - overlap; a queue would be a P-09 extension. -- Cross-host locks. Locks here are still localhost-bound; combining - with ADR-0010 or ADR-0016 is future work. diff --git a/docs/decisions/0016-mtls-federation-stopgap.adoc b/docs/decisions/0016-mtls-federation-stopgap.adoc new file mode 100644 index 00000000..5059863d --- /dev/null +++ b/docs/decisions/0016-mtls-federation-stopgap.adoc @@ -0,0 +1,171 @@ +== 16. mTLS + ed25519 federation stop-gap — spike + +Date: 2026-05-24 + +=== Status + +Accepted (2026-05-24) — chosen over ADR-0015 because it closes the more +user-visible survey gap (Ruflo as the only comparator with a cross-host +story) and is additive: loopback bus and Idris2-verified intra-host ABI +stay untouched. Positioned as the v1 transport; ADR-0010 (DID + +ML-DSA-87 + ML-KEM-1024 + federated quarantine) remains the v2 upgrade +path. + +*Implementation plan:* phased rollout, each phase its own PR off +`+main+` after this ADR lands. + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Phase |Scope |Estimated +|1 |Identity foundation — ed25519 keypair, pubkey export, +`+known_peers.toml+` parser, `+coord-tui --print-pubkey+` flag. No +transport. |~1 day + +|2 |Envelope sign/verify on the loopback bus (round-trip validation +before any wire work). |~1 day + +|3 |TLS listener on `+:7746+` + mTLS handshake gated by +`+known_peers.toml+`. |~1-2 days + +|4 |New `+coord_add_peer+` / `+coord_remove_peer+` / +`+coord_list_remote_peers+` tools — bridge + cartridge.json sync. |~0.5 +day + +|5 |Idris2 `+Federation.idr+` — envelope-boundary signature soundness +obligation. |~1 day + +|6 |Tests — two-instance round-trip, signature tampering, replay +protection. |~0.5 day +|=== + +=== Context + +The multi-agent MCP survey identified `+ruvnet/ruflo+` as the only +comparator shipping a real cross-host story (mTLS + ed25519 federation +across machines). `+local-coord-mcp+` is currently localhost-only +(`+127.0.0.1:7745+`) — closing this gap is one of the two outstanding +items from PR #143’s spike. + +ADR-0010 already proposes the _ambitious_ answer: DID-based identity, +ML-DSA-87 signing, ML-KEM-1024 key exchange, federated quarantine. It +sits at "`Proposed (RFC)`" status, tracked under epic #87 item 3, and is +a substantial cross-cartridge build effort. + +This spike evaluates a deliberately *smaller* stop-gap: ship +Ruflo-equivalent cross-host federation using mTLS + ed25519 _now_, +without blocking on the post-quantum work in ADR-0010. The two are +compatible: this becomes the v1 transport, ADR-0010 becomes the v2 +upgrade path. + +=== What the change does + +[arabic] +. *Transport.* Add a TLS-terminating endpoint alongside the loopback bus +— `+https://:7746/coord/federated/inbox+` for inbound federated +envelopes. Loopback bus (`+:7745+`) stays the default for intra-machine +traffic and is untouched. +. *Identity.* Each peer generates an ed25519 keypair on first run +(stored under `+~/.cache/coord-tui/peer.key+`). The public key is the +peer’s federated identity; the private key signs outbound envelopes. No +DID method, no PKI hierarchy — leaf-only trust. +. *Peer trust.* A new file `+~/.config/coord-tui/known_peers.toml+` maps +peer-id → `+{host, port, pubkey}+`. Federation only works between peers +that have explicitly added each other (manual key exchange via +`+coord-tui --print-pubkey+` and a shared channel). No discovery; no +registry. +. *mTLS.* Both ends of the federated link present client certs derived +from their ed25519 identity keys (SPKI-pinned, not CA-signed). +Connection fails closed if the presented pubkey doesn’t match +`+known_peers.toml+`. +. *Envelope wrapping.* Outbound coord messages destined for a remote +peer are signed (ed25519 over the canonical envelope bytes), wrapped in +`+{from, to, signature, payload}+`, sent over the mTLS connection. +Receiver verifies signature against `+known_peers.toml+`, then injects +into the local bus as if it had arrived locally. +. *New tools.* `+coord_add_peer(peer_id, host, port, pubkey)+`, +`+coord_remove_peer(peer_id)+`, `+coord_list_remote_peers()+`. No tool +schema change to the existing `+coord_send+` / `+coord_claim_task+` +surface — federation is transport-layer. + +=== Cost + +[width="100%",cols="50%,50%",options="header",] +|=== +|Surface |Work +|Zig adapter |TLS listener (use `+std.crypto.tls+`), ed25519 sign/verify +(`+std.crypto.sign.Ed25519+` — already available), known_peers parser, +envelope wrap/unwrap. ~400-600 LOC. + +|Idris2 ABI |New `+Federation.idr+` — but only proof obligations on the +_envelope_ boundary (signature verification soundness), not on the +transport. ~150 lines; reuses existing `+SafeHTTP+` patterns. Class (J) +believe_me may be unavoidable for `+std.crypto.tls+` opaque primitives — +same axiom posture as the existing `+Char+`/`+String+` primitives. + +|Bridge |New 3 `+coord_*+` tool entries in `+tools.js+` + +cartridge.json; dispatcher routes them to the backend. No envelope-shape +changes. + +|coord-tui |`+--print-pubkey+` flag, +`+~/.config/coord-tui/known_peers.toml+` reader, optional +`+coord-add-peer+` shell helper. + +|Tests |Round-trip test on two backend instances on different ports; +signature-tampering rejection test; replay-protection (envelope +timestamp + nonce). +|=== + +Estimated: 4-6 working days end-to-end. + +=== Benefit + +* Closes the survey’s "`cross-host transport`" gap directly: matches +Ruflo’s posture, doesn’t claim more. +* Additive — the existing loopback bus and Idris2-verified intra-host +ABI are untouched. No proof regression. +* Sets up ADR-0010 as a clean v2 upgrade (swap ed25519→ML-DSA-87, add +ML-KEM, add DID resolution) once that work is prioritised. Wire format +is the same shape; only the cryptographic primitives change. +* Real user-visible feature — agents on Jonathan’s workstation can +coordinate with agents on a build server, a sandbox VM, or a teammate’s +machine. + +=== Risks / open questions + +* *Trust bootstrap is manual.* Exchanging pubkeys via a shared channel +is the right answer for a v1 (it’s how SSH known_hosts works), but it +does mean federation has a per-peer setup cost. ADR- 0010’s DID + +cartridge-index resolution removes this; v1 doesn’t. +* *No federated quarantine.* Tier-2+ envelopes from a remote peer hit +the _local_ master for review — no upstream supervision model. This is a +feature gap relative to ADR-0010, not a bug, but worth naming. +* *Crypto agility.* ed25519 in a post-quantum world has a finite shelf +life. Estate standards already commit to ML-DSA-87. This spike’s +stop-gap framing is honest about that: v1 lasts until ADR- 0010 lands, +not forever. CHANGELOG and README should say so. +* *Idris2 transport-layer proofs.* TLS internals are opaque to the +Idris2 backend-assurance harness — same axiomatisation posture as the +existing 5 `+believe_me+` sites. Net new principled assumption, not a +regression. + +=== Verdict + +Smaller scope than ADR-0015, lower architectural risk than ADR-0010, and +directly closes the only survey gap where `+local-coord-mcp+` clearly +trails a comparator. The "`this is explicitly a stop-gap`" framing keeps +the door open for ADR-0010 without making it a blocker. + +*Recommendation:* Prefer this over ADR-0015 if the goal is "`ship a +visible new capability that the survey identified as missing`". Combine +with ADR-0015 only if file-locks are a stated user need; otherwise +ADR-0015 is gold-plating relative to the survey’s actual findings. + +=== Out of scope + +* DID identity (ADR-0010, v2 upgrade). +* Post-quantum crypto (ADR-0010, v2 upgrade). +* Discovery / registry (deliberate — manual known_peers keeps the trust +story legible). +* Federated quarantine semantics (would land alongside DID work). +* Wire format negotiation for v1↔v2 transition — addressed when ADR-0010 +implementation starts. diff --git a/docs/decisions/0016-mtls-federation-stopgap.md b/docs/decisions/0016-mtls-federation-stopgap.md deleted file mode 100644 index c3fddb66..00000000 --- a/docs/decisions/0016-mtls-federation-stopgap.md +++ /dev/null @@ -1,147 +0,0 @@ - - - -# 16. mTLS + ed25519 federation stop-gap — spike - -Date: 2026-05-24 - -## Status - -Accepted (2026-05-24) — chosen over ADR-0015 because it closes the -more user-visible survey gap (Ruflo as the only comparator with a -cross-host story) and is additive: loopback bus and Idris2-verified -intra-host ABI stay untouched. Positioned as the v1 transport; -ADR-0010 (DID + ML-DSA-87 + ML-KEM-1024 + federated quarantine) -remains the v2 upgrade path. - -**Implementation plan:** phased rollout, each phase its own PR off -`main` after this ADR lands. - -| Phase | Scope | Estimated | -|---|---|---| -| 1 | Identity foundation — ed25519 keypair, pubkey export, `known_peers.toml` parser, `coord-tui --print-pubkey` flag. No transport. | ~1 day | -| 2 | Envelope sign/verify on the loopback bus (round-trip validation before any wire work). | ~1 day | -| 3 | TLS listener on `:7746` + mTLS handshake gated by `known_peers.toml`. | ~1-2 days | -| 4 | New `coord_add_peer` / `coord_remove_peer` / `coord_list_remote_peers` tools — bridge + cartridge.json sync. | ~0.5 day | -| 5 | Idris2 `Federation.idr` — envelope-boundary signature soundness obligation. | ~1 day | -| 6 | Tests — two-instance round-trip, signature tampering, replay protection. | ~0.5 day | - -## Context - -The multi-agent MCP survey identified `ruvnet/ruflo` as the only -comparator shipping a real cross-host story (mTLS + ed25519 federation -across machines). `local-coord-mcp` is currently localhost-only -(`127.0.0.1:7745`) — closing this gap is one of the two outstanding -items from PR #143's spike. - -ADR-0010 already proposes the *ambitious* answer: DID-based identity, -ML-DSA-87 signing, ML-KEM-1024 key exchange, federated quarantine. It -sits at "Proposed (RFC)" status, tracked under epic #87 item 3, and is -a substantial cross-cartridge build effort. - -This spike evaluates a deliberately **smaller** stop-gap: ship -Ruflo-equivalent cross-host federation using mTLS + ed25519 *now*, -without blocking on the post-quantum work in ADR-0010. The two are -compatible: this becomes the v1 transport, ADR-0010 becomes the v2 -upgrade path. - -## What the change does - -1. **Transport.** Add a TLS-terminating endpoint alongside the loopback - bus — `https://:7746/coord/federated/inbox` for inbound - federated envelopes. Loopback bus (`:7745`) stays the default for - intra-machine traffic and is untouched. -2. **Identity.** Each peer generates an ed25519 keypair on first run - (stored under `~/.cache/coord-tui/peer.key`). The public key is the - peer's federated identity; the private key signs outbound envelopes. - No DID method, no PKI hierarchy — leaf-only trust. -3. **Peer trust.** A new file `~/.config/coord-tui/known_peers.toml` - maps peer-id → `{host, port, pubkey}`. Federation only works between - peers that have explicitly added each other (manual key exchange via - `coord-tui --print-pubkey` and a shared channel). No discovery; no - registry. -4. **mTLS.** Both ends of the federated link present client certs - derived from their ed25519 identity keys (SPKI-pinned, not - CA-signed). Connection fails closed if the presented pubkey doesn't - match `known_peers.toml`. -5. **Envelope wrapping.** Outbound coord messages destined for a remote - peer are signed (ed25519 over the canonical envelope bytes), wrapped - in `{from, to, signature, payload}`, sent over the mTLS connection. - Receiver verifies signature against `known_peers.toml`, then injects - into the local bus as if it had arrived locally. -6. **New tools.** `coord_add_peer(peer_id, host, port, pubkey)`, - `coord_remove_peer(peer_id)`, `coord_list_remote_peers()`. No tool - schema change to the existing `coord_send` / `coord_claim_task` - surface — federation is transport-layer. - -## Cost - -| Surface | Work | -|---|---| -| Zig adapter | TLS listener (use `std.crypto.tls`), ed25519 sign/verify (`std.crypto.sign.Ed25519` — already available), known_peers parser, envelope wrap/unwrap. ~400-600 LOC. | -| Idris2 ABI | New `Federation.idr` — but only proof obligations on the *envelope* boundary (signature verification soundness), not on the transport. ~150 lines; reuses existing `SafeHTTP` patterns. Class (J) believe_me may be unavoidable for `std.crypto.tls` opaque primitives — same axiom posture as the existing `Char`/`String` primitives. | -| Bridge | New 3 `coord_*` tool entries in `tools.js` + cartridge.json; dispatcher routes them to the backend. No envelope-shape changes. | -| coord-tui | `--print-pubkey` flag, `~/.config/coord-tui/known_peers.toml` reader, optional `coord-add-peer` shell helper. | -| Tests | Round-trip test on two backend instances on different ports; signature-tampering rejection test; replay-protection (envelope timestamp + nonce). | - -Estimated: 4-6 working days end-to-end. - -## Benefit - -- Closes the survey's "cross-host transport" gap directly: matches - Ruflo's posture, doesn't claim more. -- Additive — the existing loopback bus and Idris2-verified intra-host - ABI are untouched. No proof regression. -- Sets up ADR-0010 as a clean v2 upgrade (swap ed25519→ML-DSA-87, - add ML-KEM, add DID resolution) once that work is prioritised. - Wire format is the same shape; only the cryptographic primitives - change. -- Real user-visible feature — agents on Jonathan's workstation can - coordinate with agents on a build server, a sandbox VM, or a - teammate's machine. - -## Risks / open questions - -- **Trust bootstrap is manual.** Exchanging pubkeys via a shared - channel is the right answer for a v1 (it's how SSH known_hosts - works), but it does mean federation has a per-peer setup cost. ADR- - 0010's DID + cartridge-index resolution removes this; v1 doesn't. -- **No federated quarantine.** Tier-2+ envelopes from a remote peer - hit the *local* master for review — no upstream supervision model. - This is a feature gap relative to ADR-0010, not a bug, but worth - naming. -- **Crypto agility.** ed25519 in a post-quantum world has a finite - shelf life. Estate standards already commit to ML-DSA-87. This - spike's stop-gap framing is honest about that: v1 lasts until ADR- - 0010 lands, not forever. CHANGELOG and README should say so. -- **Idris2 transport-layer proofs.** TLS internals are opaque to the - Idris2 backend-assurance harness — same axiomatisation posture as - the existing 5 `believe_me` sites. Net new principled assumption, - not a regression. - -## Verdict - -Smaller scope than ADR-0015, lower architectural risk than ADR-0010, -and directly closes the only survey gap where `local-coord-mcp` -clearly trails a comparator. The "this is explicitly a stop-gap" -framing keeps the door open for ADR-0010 without making it a -blocker. - -**Recommendation:** Prefer this over ADR-0015 if the goal is "ship a -visible new capability that the survey identified as missing". -Combine with ADR-0015 only if file-locks are a stated user need; -otherwise ADR-0015 is gold-plating relative to the survey's actual -findings. - -## Out of scope - -- DID identity (ADR-0010, v2 upgrade). -- Post-quantum crypto (ADR-0010, v2 upgrade). -- Discovery / registry (deliberate — manual known_peers keeps the - trust story legible). -- Federated quarantine semantics (would land alongside DID work). -- Wire format negotiation for v1↔v2 transition — addressed when - ADR-0010 implementation starts. diff --git a/docs/decisions/README.adoc b/docs/decisions/README.adoc new file mode 100644 index 00000000..3dc7a485 --- /dev/null +++ b/docs/decisions/README.adoc @@ -0,0 +1,18 @@ +== Architecture Decision Records + +We record significant architectural decisions using +https://cognitect.com/blog/2011/11/15/documenting-architecture-decisions[Architecture +Decision Records (ADRs)], as described by Michael Nygard. + +Each ADR captures the context, decision, and consequences of a choice +that affects the project’s structure, dependencies, or conventions. + +=== Creating a new ADR + +[source,bash] +---- +just adr "Title of decision" +---- + +This creates a new numbered file in `+docs/decisions/+` from the +template at `+0000-template.md+`. diff --git a/docs/decisions/README.md b/docs/decisions/README.md deleted file mode 100644 index 2eb087cd..00000000 --- a/docs/decisions/README.md +++ /dev/null @@ -1,19 +0,0 @@ - - - -# Architecture Decision Records - -We record significant architectural decisions using [Architecture Decision Records (ADRs)](https://cognitect.com/blog/2011/11/15/documenting-architecture-decisions), as described by Michael Nygard. - -Each ADR captures the context, decision, and consequences of a choice that affects the project's structure, dependencies, or conventions. - -## Creating a new ADR - -```bash -just adr "Title of decision" -``` - -This creates a new numbered file in `docs/decisions/` from the template at `0000-template.md`. diff --git a/docs/glama/CAPABILITIES.adoc b/docs/glama/CAPABILITIES.adoc new file mode 100644 index 00000000..cf678df8 --- /dev/null +++ b/docs/glama/CAPABILITIES.adoc @@ -0,0 +1,226 @@ +== BoJ Server Capabilities + +=== Core Features + +==== Multi-Protocol Support + +* *MCP (Model Context Protocol)*: Full support for Claude, Cursor, and +other MCP clients +* *REST API*: Comprehensive RESTful API for programmatic access +* *GraphQL*: GraphQL endpoint for flexible querying +* *gRPC*: gRPC support for high-performance clients + +==== Cartridge System + +* *125 Cartridges*: Covering databases, git, cloud, comms, ML, browser, +and more +* *68 MCP Tools*: 45 boj_* discovery/domain tools + 23 coord_* tools; +per-cartridge operations are reachable via boj_cartridge_invoke across +the 125-cartridge catalogue +* *Hot-Reloading*: Add/remove cartridges without restarting +* *Isolation*: Each cartridge runs in its own sandbox + +==== Federation (Umoja) + +* *QUIC Transport*: Encrypted, low-latency communication +* *UDP Fallback*: Works in restricted networks +* *Gossip Protocol*: Automatic peer discovery +* *Hash Attestation*: Cryptographic verification of nodes + +==== Security + +* *TLS 1.3*: All REST API traffic encrypted +* *JWT Authentication*: Optional JWT tokens for API access +* *Rate Limiting*: Per-client and global rate limits +* *Circuit Breakers*: Automatic failure isolation + +=== Advanced Features + +==== Knowledge Graph + +* *Entities*: People, projects, organizations, tools +* *Relations*: Typed relationships between entities +* *Observations*: Factual statements with temporal validity +* *Search*: Full-text search with FTS5 and bm25 ranking + +==== Session Management + +* *Persistent Sessions*: Session state survives restarts +* *Context Loading*: Automatic context from previous sessions +* *Multi-Project*: Isolate sessions by project + +==== Decision Tracking + +* *Structured Decisions*: Title, decision, reasoning, alternatives +* *Confidence Scores*: Track decision certainty +* *Temporal Context*: When and why decisions were made + +==== Learning System + +* *Categories*: Pattern, mistake, insight, research, architecture +* *Deduplication*: Automatic duplicate detection +* *Confidence Tracking*: Measure learning reliability +* *Tagging*: Organize learnings by topic + +=== Performance + +==== Scalability + +* *Horizontal Scaling*: Add more nodes to the federation +* *Vertical Scaling*: Increase cartridge concurrency +* *Load Balancing*: Automatic request distribution + +==== Benchmarks + +* *Request Latency*: < 50ms for most operations +* *Throughput*: 1000+ requests/sec per node +* *Memory*: ~50MB per cartridge (average) + +=== Integration + +==== Cloud Providers + +* *AWS*: S3, EC2, Lambda, RDS +* *GCP*: Cloud Storage, Compute Engine, Cloud SQL +* *Azure*: Blob Storage, VMs, SQL Database +* *Cloudflare*: Workers, R2, D1, KV +* *Vercel*: Projects, Deployments, Serverless + +==== Version Control + +* *GitHub*: Repos, issues, PRs, actions +* *GitLab*: Projects, MRs, pipelines, mirrors +* *Bitbucket*: Repos, pull requests, pipelines + +==== Communication + +* *Email*: SMTP, SendGrid, Mailgun +* *Chat*: Slack, Discord, Telegram, Teams +* *SMS*: Twilio, AWS SNS +* *Push*: Firebase Cloud Messaging + +==== Databases + +* *SQL*: PostgreSQL, MySQL, SQLite +* *NoSQL*: MongoDB, Redis, DynamoDB +* *Graph*: Neo4j, ArangoDB +* *Search*: Elasticsearch, Meilisearch + +=== Development + +==== SDKs + +* *JavaScript/TypeScript*: Full-featured client library +* *Python*: Asyncio-based client +* *Rust*: High-performance client +* *Go*: Concurrent client + +==== CLI + +* *boj-cli*: Command-line interface for all operations +* *boj-dev*: Development and debugging tools +* *boj-test*: Test suite runner + +==== IDE Plugins + +* *VS Code*: BoJ extension with autocomplete +* *JetBrains*: IntelliJ/CLion plugin +* *Neovim*: Lua plugin with telescope integration + +=== Monitoring & Observability + +==== Metrics + +* *Prometheus*: Export metrics for monitoring +* *OpenTelemetry*: Distributed tracing support +* *Health Checks*: `+/health+` endpoint with detailed status + +==== Logging + +* *Structured Logs*: JSON format for easy parsing +* *Log Levels*: Debug, info, warn, error +* *Log Rotation*: Automatic log file management + +==== Alerting + +* *Webhooks*: Send alerts to Slack, Discord, etc. +* *Email*: Email notifications for critical events +* *SMS*: Text message alerts + +=== Deployment + +==== Containerization + +* *Docker*: Official images on Docker Hub +* *Podman*: Full compatibility +* *Kubernetes*: Helm charts for easy deployment + +==== Serverless + +* *AWS Lambda*: Serverless deployment option +* *Cloudflare Workers*: Edge deployment +* *Vercel Edge Functions*: Low-latency edge computing + +==== On-Premises + +* *Bare Metal*: Direct installation +* *VMs*: Virtual machine support +* *NAS*: Network-attached storage integration + +=== Configuration Management + +==== Infrastructure as Code + +* *Terraform*: Modules for cloud deployment +* *Pulumi*: JavaScript/TypeScript-based IaC +* *Ansible*: Playbooks for configuration + +==== Secrets Management + +* *Vault*: HashiCorp Vault integration +* *AWS Secrets Manager*: Native AWS support +* *GCP Secret Manager*: Native GCP support +* *Environment Variables*: Simple .env file support + +=== Capability Matrix + +[cols=",",options="header",] +|=== +|Feature |Status +|Multi-Protocol |✅ +|Cartridge System |✅ +|Federation |✅ +|Security |✅ +|Knowledge Graph |✅ +|Session Management |✅ +|Decision Tracking |✅ +|Learning System |✅ +|Cloud Integration |✅ +|Version Control |✅ +|Communication |✅ +|Database Support |✅ +|SDKs |✅ +|CLI Tools |✅ +|IDE Plugins |✅ +|Monitoring |✅ +|Containerization |✅ +|Serverless |✅ +|On-Premises |✅ +|IaC Support |✅ +|Secrets Management |✅ +|=== + +=== Roadmap + +==== Upcoming Features + +* *AI Agents*: Autonomous agent orchestration +* *Workflow Engine*: Visual workflow builder +* *Marketplace*: Cartridge discovery and installation +* *Analytics*: Usage metrics and insights + +==== Experimental Features + +* *WASM Cartridges*: WebAssembly-based cartridges +* *Blockchain*: Smart contract integration +* *Quantum*: Quantum computing interfaces diff --git a/docs/glama/CAPABILITIES.md b/docs/glama/CAPABILITIES.md deleted file mode 100644 index 377922ca..00000000 --- a/docs/glama/CAPABILITIES.md +++ /dev/null @@ -1,197 +0,0 @@ - -# BoJ Server Capabilities - -## Core Features - -### Multi-Protocol Support -- **MCP (Model Context Protocol)**: Full support for Claude, Cursor, and other MCP clients -- **REST API**: Comprehensive RESTful API for programmatic access -- **GraphQL**: GraphQL endpoint for flexible querying -- **gRPC**: gRPC support for high-performance clients - -### Cartridge System -- **125 Cartridges**: Covering databases, git, cloud, comms, ML, browser, and more -- **68 MCP Tools**: 45 boj_* discovery/domain tools + 23 coord_* tools; per-cartridge operations are reachable via boj_cartridge_invoke across the 125-cartridge catalogue -- **Hot-Reloading**: Add/remove cartridges without restarting -- **Isolation**: Each cartridge runs in its own sandbox - -### Federation (Umoja) -- **QUIC Transport**: Encrypted, low-latency communication -- **UDP Fallback**: Works in restricted networks -- **Gossip Protocol**: Automatic peer discovery -- **Hash Attestation**: Cryptographic verification of nodes - -### Security -- **TLS 1.3**: All REST API traffic encrypted -- **JWT Authentication**: Optional JWT tokens for API access -- **Rate Limiting**: Per-client and global rate limits -- **Circuit Breakers**: Automatic failure isolation - -## Advanced Features - -### Knowledge Graph -- **Entities**: People, projects, organizations, tools -- **Relations**: Typed relationships between entities -- **Observations**: Factual statements with temporal validity -- **Search**: Full-text search with FTS5 and bm25 ranking - -### Session Management -- **Persistent Sessions**: Session state survives restarts -- **Context Loading**: Automatic context from previous sessions -- **Multi-Project**: Isolate sessions by project - -### Decision Tracking -- **Structured Decisions**: Title, decision, reasoning, alternatives -- **Confidence Scores**: Track decision certainty -- **Temporal Context**: When and why decisions were made - -### Learning System -- **Categories**: Pattern, mistake, insight, research, architecture -- **Deduplication**: Automatic duplicate detection -- **Confidence Tracking**: Measure learning reliability -- **Tagging**: Organize learnings by topic - -## Performance - -### Scalability -- **Horizontal Scaling**: Add more nodes to the federation -- **Vertical Scaling**: Increase cartridge concurrency -- **Load Balancing**: Automatic request distribution - -### Benchmarks -- **Request Latency**: < 50ms for most operations -- **Throughput**: 1000+ requests/sec per node -- **Memory**: ~50MB per cartridge (average) - -## Integration - -### Cloud Providers -- **AWS**: S3, EC2, Lambda, RDS -- **GCP**: Cloud Storage, Compute Engine, Cloud SQL -- **Azure**: Blob Storage, VMs, SQL Database -- **Cloudflare**: Workers, R2, D1, KV -- **Vercel**: Projects, Deployments, Serverless - -### Version Control -- **GitHub**: Repos, issues, PRs, actions -- **GitLab**: Projects, MRs, pipelines, mirrors -- **Bitbucket**: Repos, pull requests, pipelines - -### Communication -- **Email**: SMTP, SendGrid, Mailgun -- **Chat**: Slack, Discord, Telegram, Teams -- **SMS**: Twilio, AWS SNS -- **Push**: Firebase Cloud Messaging - -### Databases -- **SQL**: PostgreSQL, MySQL, SQLite -- **NoSQL**: MongoDB, Redis, DynamoDB -- **Graph**: Neo4j, ArangoDB -- **Search**: Elasticsearch, Meilisearch - -## Development - -### SDKs -- **JavaScript/TypeScript**: Full-featured client library -- **Python**: Asyncio-based client -- **Rust**: High-performance client -- **Go**: Concurrent client - -### CLI -- **boj-cli**: Command-line interface for all operations -- **boj-dev**: Development and debugging tools -- **boj-test**: Test suite runner - -### IDE Plugins -- **VS Code**: BoJ extension with autocomplete -- **JetBrains**: IntelliJ/CLion plugin -- **Neovim**: Lua plugin with telescope integration - -## Monitoring & Observability - -### Metrics -- **Prometheus**: Export metrics for monitoring -- **OpenTelemetry**: Distributed tracing support -- **Health Checks**: `/health` endpoint with detailed status - -### Logging -- **Structured Logs**: JSON format for easy parsing -- **Log Levels**: Debug, info, warn, error -- **Log Rotation**: Automatic log file management - -### Alerting -- **Webhooks**: Send alerts to Slack, Discord, etc. -- **Email**: Email notifications for critical events -- **SMS**: Text message alerts - -## Deployment - -### Containerization -- **Docker**: Official images on Docker Hub -- **Podman**: Full compatibility -- **Kubernetes**: Helm charts for easy deployment - -### Serverless -- **AWS Lambda**: Serverless deployment option -- **Cloudflare Workers**: Edge deployment -- **Vercel Edge Functions**: Low-latency edge computing - -### On-Premises -- **Bare Metal**: Direct installation -- **VMs**: Virtual machine support -- **NAS**: Network-attached storage integration - -## Configuration Management - -### Infrastructure as Code -- **Terraform**: Modules for cloud deployment -- **Pulumi**: JavaScript/TypeScript-based IaC -- **Ansible**: Playbooks for configuration - -### Secrets Management -- **Vault**: HashiCorp Vault integration -- **AWS Secrets Manager**: Native AWS support -- **GCP Secret Manager**: Native GCP support -- **Environment Variables**: Simple .env file support - -## Capability Matrix - -| Feature | Status | -|---------|--------| -| Multi-Protocol | ✅ | -| Cartridge System | ✅ | -| Federation | ✅ | -| Security | ✅ | -| Knowledge Graph | ✅ | -| Session Management | ✅ | -| Decision Tracking | ✅ | -| Learning System | ✅ | -| Cloud Integration | ✅ | -| Version Control | ✅ | -| Communication | ✅ | -| Database Support | ✅ | -| SDKs | ✅ | -| CLI Tools | ✅ | -| IDE Plugins | ✅ | -| Monitoring | ✅ | -| Containerization | ✅ | -| Serverless | ✅ | -| On-Premises | ✅ | -| IaC Support | ✅ | -| Secrets Management | ✅ | - -## Roadmap - -### Upcoming Features -- **AI Agents**: Autonomous agent orchestration -- **Workflow Engine**: Visual workflow builder -- **Marketplace**: Cartridge discovery and installation -- **Analytics**: Usage metrics and insights - -### Experimental Features -- **WASM Cartridges**: WebAssembly-based cartridges -- **Blockchain**: Smart contract integration -- **Quantum**: Quantum computing interfaces diff --git a/docs/glama/PROMPTS.adoc b/docs/glama/PROMPTS.adoc new file mode 100644 index 00000000..2bf69816 --- /dev/null +++ b/docs/glama/PROMPTS.adoc @@ -0,0 +1,202 @@ +== BoJ Server Interactive Prompts + +=== Project Analysis + +==== Analyze Repository + +*Prompt*: `+Analyze this repository and suggest improvements+` *Tools +Used*: `+boj_research+`, `+coderag_analyze_repository+` *Output*: +Repository structure, language breakdown, quality metrics, improvement +suggestions + +==== Recommend Team + +*Prompt*: `+Recommend a team for this project+` *Tools Used*: +`+claude_agents_analyze_project+`, +`+claude_agents_recommend_by_keywords+` *Output*: List of recommended +roles with justifications + +=== Code Quality + +==== Detect Code Smells + +*Prompt*: `+Find code smells in this repository+` *Tools Used*: +`+coderag_calculate_metrics+`, `+database_query+` *Output*: List of code +smells with locations and severity + +==== Calculate Metrics + +*Prompt*: `+Calculate code quality metrics+` *Tools Used*: +`+coderag_calculate_metrics+` *Output*: CK metrics, package coupling, +architectural patterns + +=== Research + +==== Academic Research + +*Prompt*: `+Research [topic] and summarize findings+` *Tools Used*: +`+boj_research+` *Output*: Summary of academic papers, citations, and +references + +==== Market Research + +*Prompt*: `+Analyze market trends for [product]+` *Tools Used*: +`+boj_research+`, `+origenemcp_search_compound+` *Output*: Market +analysis, competitor comparison, trend forecast + +=== Notification + +==== Send Alert + +*Prompt*: `+Send alert to #devops about [issue]+` *Tools Used*: +`+notifyhub_send_slack+`, `+notifyhub_send_discord+` *Output*: +Confirmation of sent notifications + +==== Broadcast Message + +*Prompt*: `+Broadcast message to all channels+` *Tools Used*: +`+notifyhub_send_email+`, `+notifyhub_send_sms+`, +`+notifyhub_send_slack+` *Output*: Confirmation of sent messages + +=== Data Analysis + +==== Query Dataset + +*Prompt*: `+Query [dataset] for [information]+` *Tools Used*: +`+opendatamcp_query_dataset+`, `+database_query+` *Output*: Dataset +results with visualization suggestions + +==== Analyze Trends + +*Prompt*: `+Analyze trends in [dataset]+` *Tools Used*: +`+opendatamcp_search_datasets+`, `+database_query+` *Output*: Trend +analysis with charts and insights + +=== Memory + +==== Start Session + +*Prompt*: `+Start memory session for [project]+` *Tools Used*: +`+memory_session_start+` *Output*: Session ID and context from previous +sessions + +==== Record Learning + +*Prompt*: `+Remember that [fact]+` *Tools Used*: `+memory_learn+` +*Output*: Confirmation and related learnings + +==== Search Memory + +*Prompt*: `+What do I know about [topic]?+` *Tools Used*: +`+memory_search+`, `+memory_recall+` *Output*: List of relevant memories +with confidence scores + +=== Git Operations + +==== Create Pull Request + +*Prompt*: `+Create PR from [branch] to [target]+` *Tools Used*: +`+boj_github_create_pr+` *Output*: PR link and summary + +==== Review Issues + +*Prompt*: `+Show open issues for [repo]+` *Tools Used*: +`+boj_github_list_issues+` *Output*: List of issues with status and +assignees + +=== Cloud Management + +==== Deploy to Cloudflare + +*Prompt*: `+Deploy [project] to Cloudflare Workers+` *Tools Used*: +`+boj_cloud_cloudflare+` *Output*: Deployment status and URL + +==== Manage AWS Resources + +*Prompt*: `+List S3 buckets in [region]+` *Tools Used*: +`+boj_cloud_verpex+` *Output*: List of buckets with sizes and +permissions + +=== Interactive Workflows + +==== Setup Project + +*Workflow*: 1. Analyze repository 2. Recommend team 3. Set up +notifications 4. Initialize memory session + +*Prompt*: `+Setup project [name]+` *Output*: Project dashboard with +team, notifications, and memory + +==== Code Review + +*Workflow*: 1. Detect code smells 2. Calculate metrics 3. Create GitHub +issues 4. Record learnings + +*Prompt*: `+Review [repository]+` *Output*: Code review report with +issues and metrics + +==== Research Paper + +*Workflow*: 1. Search academic papers 2. Analyze references 3. Summarize +findings 4. Store in memory + +*Prompt*: `+Research [topic]+` *Output*: Research report with citations +and summary + +=== Template Syntax + +==== Variables + +Use `+{{variable}}+` syntax for dynamic values: - `+{{project}}+`: +Current project name - `+{{user}}+`: Current user - `+{{date}}+`: +Current date - `+{{time}}+`: Current time + +==== Conditional Logic + +Use `+{{#if condition}}...{{/if}}+` for conditional sections: + +.... +{{#if project}} +Project: {{project}} +{{/if}} +.... + +==== Loops + +Use `+{{#each items}}...{{/each}}+` for loops: + +.... +{{#each issues}} +- {{this.title}} ({{this.status}}) +{{/each}} +.... + +=== Best Practices + +[arabic] +. *Be Specific*: Include relevant details in prompts +. *Use Templates*: Start with predefined templates +. *Review Output*: Always verify tool outputs +. *Store Learnings*: Record important insights +. *Session Management*: Start/end sessions appropriately + +=== Examples + +==== Example 1: Project Setup + +*User*: `+Setup project my-app+` *BoJ*: 1. Analyzing repository… 2. +Recommending team… 3. Setting up notifications… 4. Starting memory +session… *Output*: Project dashboard with team, notifications, and +memory session ID + +==== Example 2: Code Review + +*User*: `+Review repository my-app+` *BoJ*: 1. Detecting code smells… 2. +Calculating metrics… 3. Creating issues… 4. Recording learnings… +*Output*: Code review report with 5 issues and quality metrics + +==== Example 3: Research + +*User*: `+Research quantum computing+` *BoJ*: 1. Searching academic +papers… 2. Analyzing references… 3. Summarizing findings… 4. Storing in +memory… *Output*: Research report with 10 papers and summary diff --git a/docs/glama/PROMPTS.md b/docs/glama/PROMPTS.md deleted file mode 100644 index 20e2ea8c..00000000 --- a/docs/glama/PROMPTS.md +++ /dev/null @@ -1,200 +0,0 @@ - -# BoJ Server Interactive Prompts - -## Project Analysis - -### Analyze Repository -**Prompt**: `Analyze this repository and suggest improvements` -**Tools Used**: `boj_research`, `coderag_analyze_repository` -**Output**: Repository structure, language breakdown, quality metrics, improvement suggestions - -### Recommend Team -**Prompt**: `Recommend a team for this project` -**Tools Used**: `claude_agents_analyze_project`, `claude_agents_recommend_by_keywords` -**Output**: List of recommended roles with justifications - -## Code Quality - -### Detect Code Smells -**Prompt**: `Find code smells in this repository` -**Tools Used**: `coderag_calculate_metrics`, `database_query` -**Output**: List of code smells with locations and severity - -### Calculate Metrics -**Prompt**: `Calculate code quality metrics` -**Tools Used**: `coderag_calculate_metrics` -**Output**: CK metrics, package coupling, architectural patterns - -## Research - -### Academic Research -**Prompt**: `Research [topic] and summarize findings` -**Tools Used**: `boj_research` -**Output**: Summary of academic papers, citations, and references - -### Market Research -**Prompt**: `Analyze market trends for [product]` -**Tools Used**: `boj_research`, `origenemcp_search_compound` -**Output**: Market analysis, competitor comparison, trend forecast - -## Notification - -### Send Alert -**Prompt**: `Send alert to #devops about [issue]` -**Tools Used**: `notifyhub_send_slack`, `notifyhub_send_discord` -**Output**: Confirmation of sent notifications - -### Broadcast Message -**Prompt**: `Broadcast message to all channels` -**Tools Used**: `notifyhub_send_email`, `notifyhub_send_sms`, `notifyhub_send_slack` -**Output**: Confirmation of sent messages - -## Data Analysis - -### Query Dataset -**Prompt**: `Query [dataset] for [information]` -**Tools Used**: `opendatamcp_query_dataset`, `database_query` -**Output**: Dataset results with visualization suggestions - -### Analyze Trends -**Prompt**: `Analyze trends in [dataset]` -**Tools Used**: `opendatamcp_search_datasets`, `database_query` -**Output**: Trend analysis with charts and insights - -## Memory - -### Start Session -**Prompt**: `Start memory session for [project]` -**Tools Used**: `memory_session_start` -**Output**: Session ID and context from previous sessions - -### Record Learning -**Prompt**: `Remember that [fact]` -**Tools Used**: `memory_learn` -**Output**: Confirmation and related learnings - -### Search Memory -**Prompt**: `What do I know about [topic]?` -**Tools Used**: `memory_search`, `memory_recall` -**Output**: List of relevant memories with confidence scores - -## Git Operations - -### Create Pull Request -**Prompt**: `Create PR from [branch] to [target]` -**Tools Used**: `boj_github_create_pr` -**Output**: PR link and summary - -### Review Issues -**Prompt**: `Show open issues for [repo]` -**Tools Used**: `boj_github_list_issues` -**Output**: List of issues with status and assignees - -## Cloud Management - -### Deploy to Cloudflare -**Prompt**: `Deploy [project] to Cloudflare Workers` -**Tools Used**: `boj_cloud_cloudflare` -**Output**: Deployment status and URL - -### Manage AWS Resources -**Prompt**: `List S3 buckets in [region]` -**Tools Used**: `boj_cloud_verpex` -**Output**: List of buckets with sizes and permissions - -## Interactive Workflows - -### Setup Project -**Workflow**: -1. Analyze repository -2. Recommend team -3. Set up notifications -4. Initialize memory session - -**Prompt**: `Setup project [name]` -**Output**: Project dashboard with team, notifications, and memory - -### Code Review -**Workflow**: -1. Detect code smells -2. Calculate metrics -3. Create GitHub issues -4. Record learnings - -**Prompt**: `Review [repository]` -**Output**: Code review report with issues and metrics - -### Research Paper -**Workflow**: -1. Search academic papers -2. Analyze references -3. Summarize findings -4. Store in memory - -**Prompt**: `Research [topic]` -**Output**: Research report with citations and summary - -## Template Syntax - -### Variables -Use `{{variable}}` syntax for dynamic values: -- `{{project}}`: Current project name -- `{{user}}`: Current user -- `{{date}}`: Current date -- `{{time}}`: Current time - -### Conditional Logic -Use `{{#if condition}}...{{/if}}` for conditional sections: -``` -{{#if project}} -Project: {{project}} -{{/if}} -``` - -### Loops -Use `{{#each items}}...{{/each}}` for loops: -``` -{{#each issues}} -- {{this.title}} ({{this.status}}) -{{/each}} -``` - -## Best Practices - -1. **Be Specific**: Include relevant details in prompts -2. **Use Templates**: Start with predefined templates -3. **Review Output**: Always verify tool outputs -4. **Store Learnings**: Record important insights -5. **Session Management**: Start/end sessions appropriately - -## Examples - -### Example 1: Project Setup -**User**: `Setup project my-app` -**BoJ**: -1. Analyzing repository... -2. Recommending team... -3. Setting up notifications... -4. Starting memory session... -**Output**: Project dashboard with team, notifications, and memory session ID - -### Example 2: Code Review -**User**: `Review repository my-app` -**BoJ**: -1. Detecting code smells... -2. Calculating metrics... -3. Creating issues... -4. Recording learnings... -**Output**: Code review report with 5 issues and quality metrics - -### Example 3: Research -**User**: `Research quantum computing` -**BoJ**: -1. Searching academic papers... -2. Analyzing references... -3. Summarizing findings... -4. Storing in memory... -**Output**: Research report with 10 papers and summary diff --git a/docs/glama/RESOURCES.adoc b/docs/glama/RESOURCES.adoc new file mode 100644 index 00000000..4b30c43d --- /dev/null +++ b/docs/glama/RESOURCES.adoc @@ -0,0 +1,372 @@ +== BoJ Server Resources + +=== Knowledge Graph + +==== Entities + +Contextual data about people, projects, organizations, and tools. + +*Fields*: - `+id+`: Unique identifier - `+name+`: Entity name - +`+type+`: Entity type (person, project, organization, tool) - +`+created_at+`: Creation timestamp - `+updated_at+`: Last update +timestamp + +*Example*: + +[source,json] +---- +{ + "id": "ent_123", + "name": "BoJ Server", + "type": "project", + "created_at": "2026-01-01T00:00:00Z", + "updated_at": "2026-05-28T00:00:00Z" +} +---- + +==== Observations + +Factual statements about entities with temporal validity. + +*Fields*: - `+id+`: Unique identifier - `+entity_id+`: Reference to +entity - `+content+`: The factual statement - `+source+`: Source of the +information - `+valid_from+`: Start of validity period - `+valid_to+`: +End of validity period (null if current) - `+confidence+`: Confidence +score (0-1) + +*Example*: + +[source,json] +---- +{ + "id": "obs_456", + "entity_id": "ent_123", + "content": "BoJ Server supports 125 cartridges", + "source": "documentation", + "valid_from": "2026-01-01T00:00:00Z", + "valid_to": null, + "confidence": 1.0 +} +---- + +==== Relations + +Typed relationships between entities. + +*Fields*: - `+id+`: Unique identifier - `+from_entity_id+`: Source +entity - `+to_entity_id+`: Target entity - `+type+`: Relation type +(e.g., "`depends_on`", "`uses`", "`created_by`") - `+weight+`: Relation +strength (0-1) - `+created_at+`: Creation timestamp + +*Example*: + +[source,json] +---- +{ + "id": "rel_789", + "from_entity_id": "ent_123", + "to_entity_id": "ent_456", + "type": "depends_on", + "weight": 0.9, + "created_at": "2026-01-01T00:00:00Z" +} +---- + +=== Sessions + +==== Session State + +Persistent session information. + +*Fields*: - `+id+`: Session ID - `+project+`: Associated project +(optional) - `+started_at+`: Session start time - `+ended_at+`: Session +end time (null if active) - `+summary+`: Session summary - `context”: +Array of previous session summaries + +*Example*: + +[source,json] +---- +{ + "id": "sess_123", + "project": "boj-server", + "started_at": "2026-05-28T10:00:00Z", + "ended_at": "2026-05-28T11:30:00Z", + "summary": "Added 5 new cartridges and updated documentation", + "context": [ + "2026-05-27: Created coderag-mcp cartridge", + "2026-05-26: Updated integration tests" + ] +} +---- + +=== Learnings + +==== Learning Categories + +*Pattern*: Recurring successful approaches *Mistake*: What went wrong +and how to avoid *Insight*: Strategic realizations *Research*: External +knowledge *Architecture*: System design decisions *Infrastructure*: +Deployment and scaling *Tool*: Tool-specific knowledge *Workflow*: +Process improvements *Performance*: Optimization techniques *Security*: +Security best practices + +==== Learning Structure + +*Fields*: - `+id+`: Unique identifier - `+category+`: Learning category +- `+content+`: The knowledge - `+tags+`: Array of tags - `+confidence+`: +Confidence score (0-1) - `+project+`: Associated project (optional) - +`+created_at+`: Creation timestamp - `+updated_at+`: Last update +timestamp + +*Example*: + +[source,json] +---- +{ + "id": "learn_123", + "category": "pattern", + "content": "Use Zig for performance-critical FFI layers", + "tags": ["performance", "zig", "ffi"], + "confidence": 0.9, + "project": "boj-server", + "created_at": "2026-01-01T00:00:00Z", + "updated_at": "2026-05-28T00:00:00Z" +} +---- + +=== Decisions + +==== Decision Structure + +*Fields*: - `+id+`: Unique identifier - `+title+`: What was decided - +`+decision+`: The choice made - `+reasoning+`: Why this decision was +made - `+alternatives+`: What else was considered - `+confidence+`: +Confidence score (0-1) - `+project+`: Associated project (optional) - +`+created_at+`: Creation timestamp + +*Example*: + +[source,json] +---- +{ + "id": "dec_456", + "title": "Choose Zig for FFI", + "decision": "Use Zig for all FFI layers", + "reasoning": "Zig provides better performance and memory safety than C", + "alternatives": "C, Rust, C++", + "confidence": 0.95, + "project": "boj-server", + "created_at": "2026-01-01T00:00:00Z" +} +---- + +=== Cartridges + +==== Cartridge Metadata + +*Fields*: - `+name+`: Cartridge name - `+version+`: Semantic version - +`+description+`: Cartridge description - `+domain+`: Domain (e.g., +Database, Git, Cloud) - `+tier+`: Tier (Teranga, Shield, Ayo) - +`+protocols+`: Supported protocols (MCP, REST, GraphQL, gRPC) - +`+auth+`: Authentication requirements - `+ports+`: Allowed/denied ports +- `+tools+`: Array of tool definitions + +*Example*: + +[source,json] +---- +{ + "name": "database-mcp", + "version": "0.1.0", + "description": "Universal database gateway", + "domain": "Database", + "tier": "Ayo", + "protocols": ["MCP", "REST"], + "auth": { + "method": "none" + }, + "ports": { + "allowed": [5432, 3306, 27017], + "denied": [22, 23, 25] + }, + "tools": [ + { + "name": "database_connect", + "description": "Connect to a database backend" + } + ] +} +---- + +=== Projects + +==== Project Structure + +*Fields*: - `+id+`: Project ID - `+name+`: Project name - +`+description+`: Project description - `+created_at+`: Creation +timestamp - `+updated_at+`: Last update timestamp - `+cartridges+`: +Array of associated cartridges - `+team+`: Array of team members (entity +IDs) - `+status+`: Project status (active, paused, completed) + +*Example*: + +[source,json] +---- +{ + "id": "proj_789", + "name": "BoJ Server", + "description": "Bundle of Joy Server - Unified MCP server", + "created_at": "2026-01-01T00:00:00Z", + "updated_at": "2026-05-28T00:00:00Z", + "cartridges": ["database-mcp", "git-mcp", "cloud-mcp"], + "team": ["ent_123", "ent_456"], + "status": "active" +} +---- + +=== Resource Management + +==== Resource Types + +[arabic] +. *Entity*: People, projects, organizations, tools +. *Session*: Persistent session state +. *Learning*: Knowledge and insights +. *Decision*: Structured decisions +. *Cartridge*: Pluggable capability modules +. *Project*: Development projects + +==== Resource Operations + +[cols=",,",options="header",] +|=== +|Operation |Endpoint |Description +|`+GET /entities+` |List all entities | +|`+GET /entities/{id}+` |Get entity by ID | +|`+POST /entities+` |Create new entity | +|`+PUT /entities/{id}+` |Update entity | +|`+DELETE /entities/{id}+` |Delete entity | +|=== + +[cols=",,",options="header",] +|=== +|Operation |Endpoint |Description +|`+GET /sessions+` |List all sessions | +|`+GET /sessions/{id}+` |Get session by ID | +|`+POST /sessions+` |Create new session | +|`+PUT /sessions/{id}+` |Update session | +|=== + +[cols=",,",options="header",] +|=== +|Operation |Endpoint |Description +|`+GET /learnings+` |List all learnings | +|`+GET /learnings/{id}+` |Get learning by ID | +|`+POST /learnings+` |Create new learning | +|`+PUT /learnings/{id}+` |Update learning | +|=== + +==== Resource Lifecycle + +[source,mermaid] +---- +graph LR + A[Create] --> B[Read] + B --> C[Update] + C --> D[Delete] + D --> A +---- + +=== Query Examples + +==== Get Entity with Observations + +[source,graphql] +---- +query GetEntity($id: ID!) { + entity(id: $id) { + id + name + type + observations { + id + content + confidence + } + relations { + id + type + toEntity { + id + name + } + } + } +} +---- + +==== Search Learnings + +[source,graphql] +---- +query SearchLearnings($query: String!, $category: String) { + learnings(query: $query, category: $category) { + id + category + content + confidence + tags + } +} +---- + +==== List Cartridge Tools + +[source,graphql] +---- +query CartridgeTools($name: String!) { + cartridge(name: $name) { + name + description + tools { + name + description + inputSchema + } + } +} +---- + +=== Best Practices + +[arabic] +. *Use Descriptive Names*: Clear, concise names for resources +. *Add Context*: Include relevant metadata (tags, confidence, etc.) +. *Link Resources*: Create relations between related resources +. *Update Regularly*: Keep resources current +. *Use Search*: Leverage full-text search for discovery +. *Backup*: Regularly backup your SQLite database +. *Validate*: Use JSON Schema validation for resources + +=== Resource Limits + +[cols=",,",options="header",] +|=== +|Resource |Default Limit |Maximum +|Entities |10,000 |Unlimited +|Observations |50,000 |Unlimited +|Relations |100,000 |Unlimited +|Learnings |10,000 |Unlimited +|Decisions |1,000 |Unlimited +|Sessions |1,000 |Unlimited +|=== + +=== Performance Tips + +[arabic] +. *Indexing*: Use appropriate indexes for frequent queries +. *Batch Operations*: Group operations to reduce overhead +. *Pagination*: Use pagination for large result sets +. *Caching*: Cache frequent queries where appropriate +. *Vacuum*: Regularly vacuum the SQLite database diff --git a/docs/glama/RESOURCES.md b/docs/glama/RESOURCES.md deleted file mode 100644 index 41e9c94f..00000000 --- a/docs/glama/RESOURCES.md +++ /dev/null @@ -1,369 +0,0 @@ - -# BoJ Server Resources - -## Knowledge Graph - -### Entities -Contextual data about people, projects, organizations, and tools. - -**Fields**: -- `id`: Unique identifier -- `name`: Entity name -- `type`: Entity type (person, project, organization, tool) -- `created_at`: Creation timestamp -- `updated_at`: Last update timestamp - -**Example**: -```json -{ - "id": "ent_123", - "name": "BoJ Server", - "type": "project", - "created_at": "2026-01-01T00:00:00Z", - "updated_at": "2026-05-28T00:00:00Z" -} -``` - -### Observations -Factual statements about entities with temporal validity. - -**Fields**: -- `id`: Unique identifier -- `entity_id`: Reference to entity -- `content`: The factual statement -- `source`: Source of the information -- `valid_from`: Start of validity period -- `valid_to`: End of validity period (null if current) -- `confidence`: Confidence score (0-1) - -**Example**: -```json -{ - "id": "obs_456", - "entity_id": "ent_123", - "content": "BoJ Server supports 125 cartridges", - "source": "documentation", - "valid_from": "2026-01-01T00:00:00Z", - "valid_to": null, - "confidence": 1.0 -} -``` - -### Relations -Typed relationships between entities. - -**Fields**: -- `id`: Unique identifier -- `from_entity_id`: Source entity -- `to_entity_id`: Target entity -- `type`: Relation type (e.g., "depends_on", "uses", "created_by") -- `weight`: Relation strength (0-1) -- `created_at`: Creation timestamp - -**Example**: -```json -{ - "id": "rel_789", - "from_entity_id": "ent_123", - "to_entity_id": "ent_456", - "type": "depends_on", - "weight": 0.9, - "created_at": "2026-01-01T00:00:00Z" -} -``` - -## Sessions - -### Session State -Persistent session information. - -**Fields**: -- `id`: Session ID -- `project`: Associated project (optional) -- `started_at`: Session start time -- `ended_at`: Session end time (null if active) -- `summary`: Session summary -- `context": Array of previous session summaries - -**Example**: -```json -{ - "id": "sess_123", - "project": "boj-server", - "started_at": "2026-05-28T10:00:00Z", - "ended_at": "2026-05-28T11:30:00Z", - "summary": "Added 5 new cartridges and updated documentation", - "context": [ - "2026-05-27: Created coderag-mcp cartridge", - "2026-05-26: Updated integration tests" - ] -} -``` - -## Learnings - -### Learning Categories - -**Pattern**: Recurring successful approaches -**Mistake**: What went wrong and how to avoid -**Insight**: Strategic realizations -**Research**: External knowledge -**Architecture**: System design decisions -**Infrastructure**: Deployment and scaling -**Tool**: Tool-specific knowledge -**Workflow**: Process improvements -**Performance**: Optimization techniques -**Security**: Security best practices - -### Learning Structure - -**Fields**: -- `id`: Unique identifier -- `category`: Learning category -- `content`: The knowledge -- `tags`: Array of tags -- `confidence`: Confidence score (0-1) -- `project`: Associated project (optional) -- `created_at`: Creation timestamp -- `updated_at`: Last update timestamp - -**Example**: -```json -{ - "id": "learn_123", - "category": "pattern", - "content": "Use Zig for performance-critical FFI layers", - "tags": ["performance", "zig", "ffi"], - "confidence": 0.9, - "project": "boj-server", - "created_at": "2026-01-01T00:00:00Z", - "updated_at": "2026-05-28T00:00:00Z" -} -``` - -## Decisions - -### Decision Structure - -**Fields**: -- `id`: Unique identifier -- `title`: What was decided -- `decision`: The choice made -- `reasoning`: Why this decision was made -- `alternatives`: What else was considered -- `confidence`: Confidence score (0-1) -- `project`: Associated project (optional) -- `created_at`: Creation timestamp - -**Example**: -```json -{ - "id": "dec_456", - "title": "Choose Zig for FFI", - "decision": "Use Zig for all FFI layers", - "reasoning": "Zig provides better performance and memory safety than C", - "alternatives": "C, Rust, C++", - "confidence": 0.95, - "project": "boj-server", - "created_at": "2026-01-01T00:00:00Z" -} -``` - -## Cartridges - -### Cartridge Metadata - -**Fields**: -- `name`: Cartridge name -- `version`: Semantic version -- `description`: Cartridge description -- `domain`: Domain (e.g., Database, Git, Cloud) -- `tier`: Tier (Teranga, Shield, Ayo) -- `protocols`: Supported protocols (MCP, REST, GraphQL, gRPC) -- `auth`: Authentication requirements -- `ports`: Allowed/denied ports -- `tools`: Array of tool definitions - -**Example**: -```json -{ - "name": "database-mcp", - "version": "0.1.0", - "description": "Universal database gateway", - "domain": "Database", - "tier": "Ayo", - "protocols": ["MCP", "REST"], - "auth": { - "method": "none" - }, - "ports": { - "allowed": [5432, 3306, 27017], - "denied": [22, 23, 25] - }, - "tools": [ - { - "name": "database_connect", - "description": "Connect to a database backend" - } - ] -} -``` - -## Projects - -### Project Structure - -**Fields**: -- `id`: Project ID -- `name`: Project name -- `description`: Project description -- `created_at`: Creation timestamp -- `updated_at`: Last update timestamp -- `cartridges`: Array of associated cartridges -- `team`: Array of team members (entity IDs) -- `status`: Project status (active, paused, completed) - -**Example**: -```json -{ - "id": "proj_789", - "name": "BoJ Server", - "description": "Bundle of Joy Server - Unified MCP server", - "created_at": "2026-01-01T00:00:00Z", - "updated_at": "2026-05-28T00:00:00Z", - "cartridges": ["database-mcp", "git-mcp", "cloud-mcp"], - "team": ["ent_123", "ent_456"], - "status": "active" -} -``` - -## Resource Management - -### Resource Types - -1. **Entity**: People, projects, organizations, tools -2. **Session**: Persistent session state -3. **Learning**: Knowledge and insights -4. **Decision**: Structured decisions -5. **Cartridge**: Pluggable capability modules -6. **Project**: Development projects - -### Resource Operations - -| Operation | Endpoint | Description | -|-----------|----------|-------------| -| `GET /entities` | List all entities | -| `GET /entities/{id}` | Get entity by ID | -| `POST /entities` | Create new entity | -| `PUT /entities/{id}` | Update entity | -| `DELETE /entities/{id}` | Delete entity | - -| Operation | Endpoint | Description | -|-----------|----------|-------------| -| `GET /sessions` | List all sessions | -| `GET /sessions/{id}` | Get session by ID | -| `POST /sessions` | Create new session | -| `PUT /sessions/{id}` | Update session | - -| Operation | Endpoint | Description | -|-----------|----------|-------------| -| `GET /learnings` | List all learnings | -| `GET /learnings/{id}` | Get learning by ID | -| `POST /learnings` | Create new learning | -| `PUT /learnings/{id}` | Update learning | - -### Resource Lifecycle - -```mermaid -graph LR - A[Create] --> B[Read] - B --> C[Update] - C --> D[Delete] - D --> A -``` - -## Query Examples - -### Get Entity with Observations -```graphql -query GetEntity($id: ID!) { - entity(id: $id) { - id - name - type - observations { - id - content - confidence - } - relations { - id - type - toEntity { - id - name - } - } - } -} -``` - -### Search Learnings -```graphql -query SearchLearnings($query: String!, $category: String) { - learnings(query: $query, category: $category) { - id - category - content - confidence - tags - } -} -``` - -### List Cartridge Tools -```graphql -query CartridgeTools($name: String!) { - cartridge(name: $name) { - name - description - tools { - name - description - inputSchema - } - } -} -``` - -## Best Practices - -1. **Use Descriptive Names**: Clear, concise names for resources -2. **Add Context**: Include relevant metadata (tags, confidence, etc.) -3. **Link Resources**: Create relations between related resources -4. **Update Regularly**: Keep resources current -5. **Use Search**: Leverage full-text search for discovery -6. **Backup**: Regularly backup your SQLite database -7. **Validate**: Use JSON Schema validation for resources - -## Resource Limits - -| Resource | Default Limit | Maximum | -|----------|---------------|---------| -| Entities | 10,000 | Unlimited | -| Observations | 50,000 | Unlimited | -| Relations | 100,000 | Unlimited | -| Learnings | 10,000 | Unlimited | -| Decisions | 1,000 | Unlimited | -| Sessions | 1,000 | Unlimited | - -## Performance Tips - -1. **Indexing**: Use appropriate indexes for frequent queries -2. **Batch Operations**: Group operations to reduce overhead -3. **Pagination**: Use pagination for large result sets -4. **Caching**: Cache frequent queries where appropriate -5. **Vacuum**: Regularly vacuum the SQLite database diff --git a/docs/glama/SERVER_CONFIGURATION.adoc b/docs/glama/SERVER_CONFIGURATION.adoc new file mode 100644 index 00000000..9017f298 --- /dev/null +++ b/docs/glama/SERVER_CONFIGURATION.adoc @@ -0,0 +1,199 @@ +== BoJ Server Configuration + +=== Environment Variables + +==== Core Configuration + +[width="99%",cols="17%,26%,34%,23%",options="header",] +|=== +|Name |Required |Description |Default +|`+BOJ_BASE_URL+` |No |Base URL for the BoJ REST API +|`+http://localhost:7700+` + +|`+BOJ_FEDERATION_PORT+` |No |UDP port for Umoja federation |`+9999+` + +|`+BOJ_QUIC+` |No |Enable QUIC transport (1 = yes, 0 = UDP only) |`+1+` + +|`+BOJ_REST_PORT+` |No |TCP port for REST API |`+7700+` + +|`+BOJ_MCP_BRIDGE+` |No |Path to MCP bridge executable +|`+mcp-bridge/main.js+` +|=== + +==== Tool Surface Scoping + +[width="99%",cols="17%,26%,34%,23%",options="header",] +|=== +|Name |Required |Description |Default +|`+BOJ_TOOL_SCOPE+` |No |Controls which tools are _advertised_ over +`+tools/list+` |`+full+` +|=== + +`+BOJ_TOOL_SCOPE+` is a coherence lever: it narrows the advertised tool +surface without removing any capability. It accepts three forms: + +* *`+full+`* (or unset) — advertise every tool. This is the default and +preserves backward compatibility with all existing clients. +* *`+core+`* — advertise only the always-present discovery/dispatch +core: `+boj_health+`, `+boj_menu+`, `+boj_cartridges+`, +`+boj_cartridge_info+`, `+boj_cartridge_invoke+`, plus *all* `+coord_*+` +tools (the local coordination surface is a single coherent unit). +* *CSV of domain prefixes* — e.g. `+core,github,browser+` advertises the +core plus the named explicit domain groups (`+github+`, `+gitlab+`, +`+cloud+`, `+comms+`, `+ml+`, `+browser+`, `+research+`, +`+codeseeker+`). `+core+` is always implied. + +The unified-endpoint thesis is preserved in every mode: each explicit +`+boj__*+` tool remains fully reachable through the generic +`+boj_cartridge_invoke+` dispatcher even when it is not advertised. +Narrowing the surface only changes discovery, not capability — which is +exactly what Glama’s Server Coherence sub-score rewards. + +[source,bash] +---- +# Minimal surface — discovery/dispatch core + coordination only +docker run -e BOJ_TOOL_SCOPE=core boj-server + +# Core plus the GitHub and browser domain groups +docker run -e BOJ_TOOL_SCOPE=core,github,browser boj-server +---- + +==== Authentication + +[cols=",,,",options="header",] +|=== +|Name |Required |Description |Default +|`+GITHUB_TOKEN+` |No |GitHub API token for Git operations |- +|`+ORIGENE_API_KEY+` |No |API key for OrigeneMCP |- +|`+NOTIFYHUB_API_KEY+` |No |API key for NotifyHub |- +|=== + +==== Database + +[cols=",,,",options="header",] +|=== +|Name |Required |Description |Default +|`+BOJ_DB_PATH+` |No |Path to SQLite database |`+~/.boj/boj.db+` +|`+BOJ_DB_MAX_CONNECTIONS+` |No |Maximum database connections |`+16+` +|=== + +==== Logging + +[cols=",,,",options="header",] +|=== +|Name |Required |Description |Default +|`+BOJ_LOG_LEVEL+` |No |Log level (debug, info, warn, error) |`+info+` +|`+BOJ_LOG_FORMAT+` |No |Log format (json, text) |`+json+` +|=== + +==== Federation + +[width="99%",cols="17%,26%,34%,23%",options="header",] +|=== +|Name |Required |Description |Default +|`+BOJ_FEDERATION_SEEDS+` |No |Comma-separated list of seed nodes |- + +|`+BOJ_FEDERATION_TIMEOUT+` |No |Federation timeout in milliseconds +|`+5000+` +|=== + +=== Configuration File + +The BoJ server can also be configured via a `+boj.config.json+` file: + +[source,json] +---- +{ + "rest": { + "port": 7700, + "host": "0.0.0.0" + }, + "federation": { + "port": 9999, + "quic": true, + "seeds": ["seed1.example.com:9999", "seed2.example.com:9999"] + }, + "database": { + "path": "~/.boj/boj.db", + "max_connections": 16 + }, + "logging": { + "level": "info", + "format": "json" + }, + "cartridges": { + "auto_load": true, + "paths": ["cartridges/"] + } +} +---- + +=== Command-Line Arguments + +The BoJ server accepts the following command-line arguments: + +[source,bash] +---- +# Start the server +boj-server --config boj.config.json + +# Start with specific cartridges +boj-server --cartridges database-mcp,git-mcp + +# Start in development mode +boj-server --dev + +# Start with verbose logging +boj-server --log-level debug +---- + +=== Docker Configuration + +When running in Docker, use environment variables: + +[source,bash] +---- +docker run -e BOJ_REST_PORT=8080 -e BOJ_LOG_LEVEL=debug boj-server +---- + +=== Configuration Examples + +==== Production Configuration + +[source,json] +---- +{ + "rest": { + "port": 8080, + "host": "0.0.0.0" + }, + "federation": { + "port": 9999, + "quic": true + }, + "logging": { + "level": "info", + "format": "json" + } +} +---- + +==== Development Configuration + +[source,json] +---- +{ + "rest": { + "port": 7700, + "host": "localhost" + }, + "federation": { + "port": 9999, + "quic": false + }, + "logging": { + "level": "debug", + "format": "text" + } +} +---- diff --git a/docs/glama/SERVER_CONFIGURATION.md b/docs/glama/SERVER_CONFIGURATION.md deleted file mode 100644 index b7048266..00000000 --- a/docs/glama/SERVER_CONFIGURATION.md +++ /dev/null @@ -1,176 +0,0 @@ - -# BoJ Server Configuration - -## Environment Variables - -### Core Configuration - -| Name | Required | Description | Default | -|------|----------|-------------|---------| -| `BOJ_BASE_URL` | No | Base URL for the BoJ REST API | `http://localhost:7700` | -| `BOJ_FEDERATION_PORT` | No | UDP port for Umoja federation | `9999` | -| `BOJ_QUIC` | No | Enable QUIC transport (1 = yes, 0 = UDP only) | `1` | -| `BOJ_REST_PORT` | No | TCP port for REST API | `7700` | -| `BOJ_MCP_BRIDGE` | No | Path to MCP bridge executable | `mcp-bridge/main.js` | - -### Tool Surface Scoping - -| Name | Required | Description | Default | -|------|----------|-------------|---------| -| `BOJ_TOOL_SCOPE` | No | Controls which tools are *advertised* over `tools/list` | `full` | - -`BOJ_TOOL_SCOPE` is a coherence lever: it narrows the advertised tool -surface without removing any capability. It accepts three forms: - -- **`full`** (or unset) — advertise every tool. This is the default and - preserves backward compatibility with all existing clients. -- **`core`** — advertise only the always-present discovery/dispatch - core: `boj_health`, `boj_menu`, `boj_cartridges`, - `boj_cartridge_info`, `boj_cartridge_invoke`, plus **all** `coord_*` - tools (the local coordination surface is a single coherent unit). -- **CSV of domain prefixes** — e.g. `core,github,browser` advertises - the core plus the named explicit domain groups - (`github`, `gitlab`, `cloud`, `comms`, `ml`, `browser`, `research`, - `codeseeker`). `core` is always implied. - -The unified-endpoint thesis is preserved in every mode: each explicit -`boj__*` tool remains fully reachable through the generic -`boj_cartridge_invoke` dispatcher even when it is not advertised. -Narrowing the surface only changes discovery, not capability — which is -exactly what Glama's Server Coherence sub-score rewards. - -```bash -# Minimal surface — discovery/dispatch core + coordination only -docker run -e BOJ_TOOL_SCOPE=core boj-server - -# Core plus the GitHub and browser domain groups -docker run -e BOJ_TOOL_SCOPE=core,github,browser boj-server -``` - -### Authentication - -| Name | Required | Description | Default | -|------|----------|-------------|---------| -| `GITHUB_TOKEN` | No | GitHub API token for Git operations | - | -| `ORIGENE_API_KEY` | No | API key for OrigeneMCP | - | -| `NOTIFYHUB_API_KEY` | No | API key for NotifyHub | - | - -### Database - -| Name | Required | Description | Default | -|------|----------|-------------|---------| -| `BOJ_DB_PATH` | No | Path to SQLite database | `~/.boj/boj.db` | -| `BOJ_DB_MAX_CONNECTIONS` | No | Maximum database connections | `16` | - -### Logging - -| Name | Required | Description | Default | -|------|----------|-------------|---------| -| `BOJ_LOG_LEVEL` | No | Log level (debug, info, warn, error) | `info` | -| `BOJ_LOG_FORMAT` | No | Log format (json, text) | `json` | - -### Federation - -| Name | Required | Description | Default | -|------|----------|-------------|---------| -| `BOJ_FEDERATION_SEEDS` | No | Comma-separated list of seed nodes | - | -| `BOJ_FEDERATION_TIMEOUT` | No | Federation timeout in milliseconds | `5000` | - -## Configuration File - -The BoJ server can also be configured via a `boj.config.json` file: - -```json -{ - "rest": { - "port": 7700, - "host": "0.0.0.0" - }, - "federation": { - "port": 9999, - "quic": true, - "seeds": ["seed1.example.com:9999", "seed2.example.com:9999"] - }, - "database": { - "path": "~/.boj/boj.db", - "max_connections": 16 - }, - "logging": { - "level": "info", - "format": "json" - }, - "cartridges": { - "auto_load": true, - "paths": ["cartridges/"] - } -} -``` - -## Command-Line Arguments - -The BoJ server accepts the following command-line arguments: - -```bash -# Start the server -boj-server --config boj.config.json - -# Start with specific cartridges -boj-server --cartridges database-mcp,git-mcp - -# Start in development mode -boj-server --dev - -# Start with verbose logging -boj-server --log-level debug -``` - -## Docker Configuration - -When running in Docker, use environment variables: - -```bash -docker run -e BOJ_REST_PORT=8080 -e BOJ_LOG_LEVEL=debug boj-server -``` - -## Configuration Examples - -### Production Configuration - -```json -{ - "rest": { - "port": 8080, - "host": "0.0.0.0" - }, - "federation": { - "port": 9999, - "quic": true - }, - "logging": { - "level": "info", - "format": "json" - } -} -``` - -### Development Configuration - -```json -{ - "rest": { - "port": 7700, - "host": "localhost" - }, - "federation": { - "port": 9999, - "quic": false - }, - "logging": { - "level": "debug", - "format": "text" - } -} -``` diff --git a/docs/governance/QED-AXIOM-AUDIT-2026-04-19.adoc b/docs/governance/QED-AXIOM-AUDIT-2026-04-19.adoc new file mode 100644 index 00000000..818e4412 --- /dev/null +++ b/docs/governance/QED-AXIOM-AUDIT-2026-04-19.adoc @@ -0,0 +1,43 @@ +== BoJ QED Axiomatic Audit — 2026-04-19 + +Scope: documented-axiomatic `+believe_me+` sites in: + +* `+src/abi/Boj/SafetyLemmas.idr+` +* `+src/abi/Boj/SafeAPIKey.idr+` + +Method: + +[arabic] +. Enumerate all `+believe_me+` sites in the two modules. +. Verify each site has an explicit nearby rationale comment. +. Check rationale still matches implementation shape and dependency +surface. + +=== Findings + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Site |Declared rationale |Audit verdict +|`+SafetyLemmas.idr:52+` (`+charEqSound+`) |Soundness of backend +primitive `+prim__eqChar+` |Still accurate. Proof depends on backend +correctness, not local logic. + +|`+SafetyLemmas.idr:58+` (`+charEqSym+`) |Symmetry of backend primitive +`+prim__eqChar+` |Still accurate. Symmetry is delegated to primitive +behavior. + +|`+SafetyLemmas.idr:208+` (`+unpackLength+`) |`+prim__strToCharList+` +preserves string length |Still accurate. Length preservation is +primitive-level; no local contradiction found. + +|`+SafeAPIKey.idr:152+` (`+logSafeBounded+`) |`+substr+`/append length +arithmetic not reducible at Idris type level |Still accurate. Runtime +construction (`+"***"+` or `+4+3+4+`) matches stated bound logic. +|=== + +=== Outcome + +* 4/4 sites are still correctly documented as axiomatic. +* No stale rationale text detected. +* No immediate code change required for correctness; keep these under +"`documented axiomatic`" debt rather than "`silent assumption`" debt. diff --git a/docs/governance/QED-AXIOM-AUDIT-2026-04-19.md b/docs/governance/QED-AXIOM-AUDIT-2026-04-19.md deleted file mode 100644 index 9ac4be72..00000000 --- a/docs/governance/QED-AXIOM-AUDIT-2026-04-19.md +++ /dev/null @@ -1,32 +0,0 @@ - -# BoJ QED Axiomatic Audit — 2026-04-19 - -Scope: documented-axiomatic `believe_me` sites in: - -- `src/abi/Boj/SafetyLemmas.idr` -- `src/abi/Boj/SafeAPIKey.idr` - -Method: - -1. Enumerate all `believe_me` sites in the two modules. -2. Verify each site has an explicit nearby rationale comment. -3. Check rationale still matches implementation shape and dependency surface. - -## Findings - -| Site | Declared rationale | Audit verdict | -|---|---|---| -| `SafetyLemmas.idr:52` (`charEqSound`) | Soundness of backend primitive `prim__eqChar` | Still accurate. Proof depends on backend correctness, not local logic. | -| `SafetyLemmas.idr:58` (`charEqSym`) | Symmetry of backend primitive `prim__eqChar` | Still accurate. Symmetry is delegated to primitive behavior. | -| `SafetyLemmas.idr:208` (`unpackLength`) | `prim__strToCharList` preserves string length | Still accurate. Length preservation is primitive-level; no local contradiction found. | -| `SafeAPIKey.idr:152` (`logSafeBounded`) | `substr`/append length arithmetic not reducible at Idris type level | Still accurate. Runtime construction (`"***"` or `4+3+4`) matches stated bound logic. | - -## Outcome - -- 4/4 sites are still correctly documented as axiomatic. -- No stale rationale text detected. -- No immediate code change required for correctness; keep these under - "documented axiomatic" debt rather than "silent assumption" debt. diff --git a/docs/handover/COORD-MCP-DESIGN-LOG.adoc b/docs/handover/COORD-MCP-DESIGN-LOG.adoc new file mode 100644 index 00000000..6a474871 --- /dev/null +++ b/docs/handover/COORD-MCP-DESIGN-LOG.adoc @@ -0,0 +1,1184 @@ +== local-coord-mcp + BoJ agent coordination — design log + +*Started:* 2026-04-20 *Lead:* Opus (1M context) *Scope:* Multi-agent +coordination across Claude + vibe + codex + gemini windows on one box; +full BoJ cartridge support; 007 dogfooding. + +This file is a running design log. Outline up top, detailed explanations +in the appendices. Updated as the work progresses. + +''''' + +=== Part 1 — Outline of what we’re building + +==== The big picture + +A localhost message bus (`+boj-server/cartridges/local-coord-mcp+`) that +lets multiple AI agents on the same machine discover each other, +exchange typed messages, claim tasks without collision, and operate +under a supervision model where Opus co-supervises with you. Non-Claude +agents (Gemini, Codex, Vibe) work under a firewall that contains their +failure modes without stopping them from contributing. + +==== Layers + +[arabic] +. *Transport* — loopback-only Zig REST server on port 7745. Idris2 ABI +proves loopback-only at compile time (`+IsLoopback+` type has exactly +two constructors; bind to non-loopback is _type-impossible_). +. *Identity* — `+coord_register(client_kind, role_hint, context)+` → +peer_id like `+claude-7f3a@007-lang+` + 128-bit CSPRNG session token. +. *Envelope* — typed messages with 18 op_kinds, 5-level risk ladder, +Byzantine-safe fields (hash chain, attestation refs, self-assessment, +context_fetch_id). +. *Supervision* — quarantine queue for Tier 2+ ops from `+supervised+` +role peers; supervisor reviews/approves/rejects. +. *Durability* — VeriSimDB sidecar (Task #7) persists inbox, claims, +audit log, track record. +. *007-mcp cartridge* — exposes `+oo7+` CLI + Justfile recipes as MCP +tools for routine 007 operations. + +''''' + +=== Part 2 — Progress ledger (session 2026-04-20) + +[width="100%",cols="12%,24%,32%,32%",options="header",] +|=== +|# |Task |Status |Commit +|1 |Complete adapter dispatch for list_peers/send/receive/status |✅ +|`+5d57daa+` + +|2 |Wire `+boj_cartridge_invoke+` to real FFI |⏳ pending |— + +|3 |A2ML envelope schema + design doc |✅ |`+ceb5125+` + +|4 |Per-window peer ID disambiguation (`+@+`) |✅ |`+2a8e4c0+` + +|5 |Supervision tier/role in FFI + ABI |✅ |`+4e164a7+` + +|6 |Supervisor tools: coord_review / approve / reject |✅ |`+4e164a7+` + +|16 |Nickel contracts with dependent constraints |✅ |`+93c589d+` + +|7 |Durable coord state (append log + replay) |✅ |`+2ae952e+`, +`+282de53+`, `+bafcd25+` + +|8 |E2E test: 2-instance master-gate + durability |✅ |`+3e9eae8+` + +|13 |Track-record table + effective_affinity + coord_get_affinities |✅ +|`+f8cafbf+`, `+3bd4710+` + +|15 |Dispatch preference + task_difficulty + sender_confidence + reject +cooldown |✅ |`+6065878+` + +|14 |Reassignment engine (server-origin quarantine entries for master +review) |✅ |`+9e40a86+` + +|17 |Deno/Node Nickel shim — runtime envelope validation in mcp-bridge +|✅ |`+ed85ca2+` + +|32 |Role rename supervisor/executor/supervised → +master/journeyman/apprentice |✅ |`+634c163+` + +|35 |`+coord_transfer_master+` — live master handoff |✅ |`+634c163+` + +|36 |`+difficulty_hint+` envelope field + `+DifficultyHintValid+` Nickel +contract |✅ |`+7f2f4a9+`, `+aeae440+` + +|37 |Prover-tag convention (`+proof:+`) +doc + example |✅ |`+eba7cfe+` + +|9 |Create 007-mcp BoJ cartridge |✅ |`+62bdac0+` (007-lang) + +|10 |007 on-enter/on-exit contractile hooks |✅ |`+018a1fd+` (007-lang) + +|11 |Fill 007 missing bust + adjust contractiles |✅ |`+6bbb4f8+` +(007-lang) + +|12 |Memory auto-lift on 007 context entry |✅ |`+e753d10+` (007-lang) + +|k9-svc |Spot-fix: move 007 k9 from contractiles/ to svc/ (ADR-001) |✅ +|`+f6f3c91+` (007-lang) + +|33 |`+client_kind+` + `+variant+` extension (openai/mistral + free-form +variant) |⏳ pending |— + +|34 |Capability advertisement on register (class, tier, +prover_strengths) |⏳ pending |— +|=== + +*Committed + pushed this lane (Prompt 3 — coord finishers):* +`+f8cafbf+`, `+3bd4710+`, `+6065878+`, `+9e40a86+`, `+ed85ca2+`, +`+634c163+`, `+7f2f4a9+`, `+aeae440+`, `+eba7cfe+` on +`+hyperpolymath/boj-server main+`. + +*Committed + pushed this lane (Prompt 2 — 007-mcp family):* `+62bdac0+`, +`+018a1fd+`, `+f6f3c91+`, `+6bbb4f8+`, `+e753d10+` (+ one follow-up +role-rename sync) on `+The-Metadatastician/007 main+`. Per CLAUDE.md 007 +is never mirrored — GitHub-only. + +*Memory written:* `+project_coord_supervision_architecture.md+` + +`+project_federation_authoritative_site.md+` (workspace memory index). + +''''' + +=== Part 3 — Design decisions taken (concise) + +[width="100%",cols="18%,55%,27%",options="header",] +|=== +|# |Decision |Why +|DD-1 |Localhost-only v1, loopback bind at 127.0.0.1:7745 |Compile-time +proof via Idris2 `+IsLoopback+`; no exposure surface + +|DD-2 |Session tokens (128-bit CSPRNG), not OAuth / vault |Local-only; +no cross-machine trust needed + +|DD-3 |Hybrid peer IDs `+-<4hex>[@]+` |Human-readable + +collision-free across windows + +|DD-4 |18 typed op_kinds (not raw strings) |Routable, auditable, +schema-validatable + +|DD-5 |5-level risk ladder (0-4) |Proportional gating; coord overhead ≠ +rate limiter + +|DD-6 |Three roles: supervisor / executor / supervised |Trust-tiered; +Opus supervises, Claude executes, others gated + +|DD-7 |Opus default to supervisor when claimed; Claude Sonnet/Haiku +default to executor; non-Claude defaults to supervised |Matches actual +agent strengths + user’s trust model + +|DD-8 |Byzantine safety: 5 mechanisms (hash chain, M-of-N attestation, +watchdog claims, sanity gate, audit log) |Specifically addresses +"`Gemini is often nuts`" failure modes + +|DD-9 |Self-assessment 4 layers: affinity + confidence + track record + +drift detector |Catches self-overclaiming by cross-check + +|DD-10 |Adaptive awareness — `+context_fetch_id+` required for Tier 2+ +|Forces context read before risky action; blast-radius-scaled + +|DD-11 |Summary-vs-raw context gated by role |Supervised peers never see +raw state (prevents hallucinated connections) + +|DD-12 |User interaction routing: fyi / clarify / blocker op_kinds; only +executor/supervisor can set urgent_direct |Single user locus; supervised +peers firewalled from interrupting you + +|DD-13 |Tier 4 forbidden list baked into schema for supervised peers +|Force-push, license touches, always-private-repos — schema-level reject + +|DD-14 |Sanity auto-promote patterns (`+git push+` → Tier 3 regardless +of declared) |Blocks tier underclaiming by content-match + +|DD-15 |Default dispatch mode: hybrid (Opus seeds + peers self-claim by +affinity) |Most flexible; works for lanes, MoE, fine-grained equally + +|DD-16 |Nickel source-of-truth → JSON export for cartridge manifests +|Matches broader boj-server pattern + +|DD-17 |Quarantine queue is in-memory ring (MAX=32) + spills to +VeriSimDB when full |Hot cache + truth separation; nothing dropped + +|DD-18 |GitHub is single source of truth — push `+origin+` only, never +other forges |Estate-wide policy; mirroring is hub-and-spoke downstream + +|DD-19 |Supervisor role gated by env-var `+BOJ_SUPERVISOR_TOKEN+` |Stops +Gemini/supervised from claiming supervisor role + +|DD-20 |Watchdog TTL 30s on supervised claims, `+progress+` heartbeat +resets timer |Active work extends claim; silence kills it + +|DD-21 |Watchdog auto-release broadcasts `+warn_drift+` via Opus review +|Prevents pile-on pivots when an agent abandons a task + +|DD-22 |v2 federation with authoritative-site model +(IDApTIK-drift-motivated) |One site holds primacy per project; peer site +defers on code-ownership; no contradictory claims + +|DD-23 |VeriSimDB uses both patterns case-by-case per cartridge |Coord = +per-box (machine-level); 007 repo data = per-project (travels with +repo); track record = per-box; memory auto-lift index = per-project + +|DD-24 |007-mcp cartridge source lives in 007-lang with install hook +into boj-server/cartridges/ |Keeps cartridge next to its code; install +script in 007’s Justfile deploys to boj-server + +|DD-25 |007-mcp exposes full `+oo7+` CLI surface as MCP tools |No +curated subset — all 25+ CLI commands surfaced + +|DD-26 |Memory auto-lift = dynamic tag-indexed VeriSimDB lookup +|Memories tagged; cartridge queries on 007 entry with repo-derived tags; +returns matches as tool output + +|DD-27 |Attester selection is deliberate by affinity (Opus picks), with +exception for broadcast on "`anyone could do it`" tasks |Quality on hard +attestations; speed on trivial ones; overhead-aware + +|DD-28 |Learning rules: track-record updates effective_affinity; +periodic scan surfaces reassignment suggestions to Opus → user |System +refines assignments over time; never auto-modifies; preserves +supervisor/user loop + +|DD-29 |Peer crash+restart = fresh chain as new peer; old chain +preserved as audit echo anchor |Epistemically honest (new peer doesn’t +claim lost knowledge); echo type preserves forensic continuity without +identity continuity + +|DD-30 |Dispatch preference per claim: deliberate (default +novel/challenging) vs broadcast (trivial/routine) vs auto +|Overhead-aware dispatch; easy tasks don’t compete with hard ones for +Opus attention + +|DD-31 |Task #7 durability implemented as in-tree +`+coord_durability.zig+` (append-only log + CRC-trailed records + typed +helpers), not via `+verisimdb-mcp+` FFI |`+verisimdb-mcp+`’s FFI is all +stubs (`+return 0 // Stub+`); using it would have meant implementing +VeriSimDB first. Option C narrows scope to finish Task #7 cleanly. The +typed log helpers (`+logPeerAdd+`, `+logInboxPush+`, …) are the stable +seam a follow-up task can swap behind verisimdb-mcp once its FFI is +real. State lives at `+$BOJ_COORD_STATE_DIR+` (default suggestion +`+$XDG_STATE_HOME/boj-server/coord/+`), per-box per DD-23. + +|DD-32 |Rename trust roles to *master / journeyman / apprentice* +(guild-craft terminology) |Matches user’s own "`master +production/scheduler`" phrasing + 007’s craft aesthetic. +Apprentice→journeyman→master is a clear ladder with no +corporate-hierarchy baggage. `+supervisor+` gets split into pure trust +tier (this) + capability class (DD-34). Backward-compat: old enum values +preserved as aliases for one release before removal. Env var renamed +`+BOJ_SUPERVISOR_TOKEN → BOJ_MASTER_TOKEN+`. + +|DD-33 |Extend `+client_kind+` enum with `+openai+` and `+mistral+`; add +free-form `+variant+` string (e.g. `+"opus-4.7"+`, +`+"gpt-5-reasoning"+`, `+"flash-2.5"+`, `+"leanstral"+`) |Today’s +4-entry enum can’t distinguish Pro from Flash or Opus from Sonnet. +Variant is a string (not enum) because model-line naming evolves faster +than the cartridge ABI. Peer ID becomes +`+[@]-<4hex>[@]+` rendered form — or we keep the +old short form and carry variant as a separate field. Leaning separate +field for stability. + +|DD-34 |Capability advertisement on register: *class* ∈ +`+{reasoner, coder, mathematician, scribe, proofsmith, reader, jester}+`, +*tier* ∈ `+{A, B, C}+`, *prover_strengths* map +`+{agda, lean4, idris2, rocq, tla}+` → 0–100 |Today the server learns +affinity from outcomes alone; cold starts are weak. Advertised +capabilities seed the router. Class vocabulary kept short + +craft-flavoured on purpose (`+proofsmith+` not "`prover`"; `+jester+` +for silly jobs — not pejorative, descriptive). Tier A/B/C is +self-declared but calibrated by track record via DD-28. + +|DD-35 |Live *master handoff* via +`+coord_transfer_master(current_token, new_peer_id, secret)+` |Today if +Opus’s session ends mid-flight, any replacement has to re-promote via +the env secret and there’s a gap. Handoff tool lets the outgoing master +hand the role to a named successor without a restart. Secret still +required — prevents hostile handoff. Audit-logged as +`+AUDIT(kind=MASTER_HANDOFF, from, to)+`. + +|DD-36 |*Difficulty hint* on envelope: +`+difficulty_hint ∈ {low, medium, high}+` |Routing aid for the +scheduler; orthogonal to `+risk_tier+`. "`Silly job`" ≈ low; +"`thought-provoking bit`" ≈ high. Not policy-enforced — master (or the +peer’s self-claim) picks the model-class match. Cheap to add (3 bits on +the wire), immediate value for per-model routing. + +|DD-37 |*Prover routing by tag convention only* — no new FFI; use +`+proof:agda+`, `+proof:lean4+`, `+proof:idris2+`, `+proof:rocq+`, +`+proof:tla+` as task tags; peers advertise per-prover strength via +DD-34 |Zero-code routing — existing affinity picks the right peer per +prover. "`Leanstral for Lean`" = Mistral-variant peer with high +`+proof:lean4+` strength; Gemini-Pro for Agda = peer with high +`+proof:agda+`. Documentation-only; no protocol change required. Keeps +the "`hard to get wrong`" character. + +|DD-38 |Split *gatekeeper (trust)* from *scheduler (dispatch)* roles — +not done in this extension, flagged for future |Today the master both +approves and schedules. Splitting them lets one peer hold veto and +another hold routing (small-team = merged; big = split). Out of scope +for DDs 32–37; revisit when cross-model dispatch gets complex. +|=== + +''''' + +=== Part 4 — Answered design questions (closed) + +[arabic] +. ✅ *Supervisor role claim protection* — env-var gated +(`+BOJ_SUPERVISOR_TOKEN+`). Server only grants the supervisor role to a +register call that presents the secret. +. ✅ *Watchdog TTL on supervised peers’ claims* — 30s baseline, +`+progress+` heartbeats reset the timer. Active work extends the claim; +silence kills it. +. ✅ *Quarantine queue behaviour on full* — spill to VeriSimDB so +nothing is dropped. In-memory MAX_QUARANTINE=32 is a hot cache; +VeriSimDB is durable tail. +. ✅ *Watchdog auto-release → `+warn_drift+` broadcast* — YES, but +Opus-reviewed first. Server generates the broadcast candidate and queues +it in the quarantine flow with `+origin=server+`; Opus approves or +vetoes before other peers see it. Prevents pile-on pivots. +. ✅ *Multi-box future* — v1 hermetically local (no network fields). v2 +adds optional federation for joint projects with son (IDApTIK, ASS), +with *authoritative-site* designation per project to prevent mission +drift. Security stack: VeriSimDB + SDP + Stapeln + HTTP capability +gateway. Details in Appendix G. + +''''' + +=== Part 5 — Interaction model (how you actually use this) + +*Primary channel: main terminal, me (Opus) as your chief of staff.* + +Other agents run in their own terminals / processes. When they have +questions, they emit envelope ops: + +* `+fyi+` — log-only, never interrupts +* `+clarify+` — batched through me, synthesised, shown periodically +* `+blocker+` — inline interrupt in my output, stops the peer’s task +* `+urgent_direct+` flag — allowed for supervisor/executor peers only; +supervised peers CANNOT set it (they never interrupt you) + +I dedupe near-identical questions, reject confabulated ones from +supervised peers, and relay to you with context. You reply to me; I +route answers back via `+coord_send+`. + +Escape hatch: nothing stops you from visiting another agent’s terminal +directly. It’s just that you shouldn’t _need_ to. + +''''' + +=== Appendix A — Deep rationale: supervision architecture + +==== Why tiered trust is the right frame + +The user’s model isn’t "`Gemini bad, Claude good`". It’s: _big things I +stress a lot more, little things I trust them on_. That’s a *risk +ladder*, not a binary gate. So the design is proportional: Tier 0/1 +flows free (the 90% case — status, query, tests, lint, small edits), +Tier 2 gets a light auto-approve gate (multi-file edits, dep bumps), +Tier 3 is the hard gate where Opus must approve (commits, pushes, schema +changes), Tier 4 is schema-level forbidden for `+supervised+` +(force-push, license changes, always-private-repo touches). + +==== Why five Byzantine mechanisms (not fewer) + +Each catches a specific Gemini failure mode: + +[arabic] +. *Hash chain per sender* — catches _retroactive rewriting_ ("`I said X +earlier`" when the record shows Y). +. *M-of-N attestation for Tier 2+ from supervised* — catches _unilateral +bad commits_ (Gemini confidently pushes garbage; needs an independent +second executor to cosign). +. *Watchdog TTL on claims* — catches _claim-squatting_ (Gemini locks 10 +tasks, produces nothing, refuses to release). +. *Content sanity gate* — catches _confabulation + tier underclaiming_ +(references to files that don’t exist; payload mentions `+git push+` but +declared Tier 1). +. *Tamper-evident audit log* — catches _everything in retrospect_ even +when mechanisms 1-4 miss (f+1 byzantine case); forensic recovery. + +Each is cheap individually. Together they’re defense-in-depth. + +==== Why four self-assessment layers + +Self-declared confidence is a signal, never a decision. Four layers +progressively stronger: + +* *A. Static affinity* — advertised on register; coarse filter +* *B. Per-claim confidence* — peer judges fit on the specific task; +risk: Gemini confidently wrong +* *C. Track record* — server-computed `+effective_affinity+` per (peer, +tag) from outcomes; can’t be forged +* *D. Drift detector* — if confidence ≫ track record, flag for review + +Tie-break on concurrent claims = +`+max(confidence × effective_affinity)+`. Honest modest-confidence peer +with good track record beats confident-wrong one. + +==== Why adaptive awareness (not "`drone`" or "`full-picture`") + +Agents are drones when safe; strategic when blast radius demands it. +Enforced via `+context_fetch_id+` — required for Tier 2+ envelopes. +Server won’t accept a Tier 2+ op from any peer that hasn’t called +`+coord_query_context+` on the affected scope in the last N minutes. + +*Summary-vs-raw is gated by role.* Supervised peers never see raw state +— the server returns a summary only. This prevents hallucinated +connections where Gemini reads the whole STATE.a2ml and invents a +relationship that isn’t there. Executors see raw on request. Supervisors +always see raw. + +''''' + +=== Appendix B — Interaction model details + +==== How questions from agents surface to you + +When any agent emits `+clarify+` or `+blocker+`: + +[arabic] +. Server delivers to supervisor’s inbox +. I (Opus) see it next time I’m invoked +. I attempt to answer from context — ~half the time I can +. If I can’t: I surface to you with deduplication and synthesis +. You reply to me in natural language +. I route the answer back via +`+coord_send(target=asker, op_kind=supervise_resp)+` + +==== Typical status dump you’d see in my output + +____ +_"`Claude@aerie finished the auth refactor (committed `+a1b2c3d+`). +Codex@ingest-svc blocked — wants to know whether to preserve the legacy +column or drop it. Gemini@docs asked 3 things; I answered 2, one +escalated: it claims `+src/foo.ts+` has a bug but the file doesn’t exist +in the repo — I rejected as confabulation.`"_ +____ + +==== What happens when multiple agents ask the same thing + +Dedup by content similarity + tag + scope. Presented as: _"`Claude@aerie +and Codex@ingest both asking about X. Claude’s context: … Codex’s +context: … they want the same answer.`"_ One reply from you routes to +both. + +''''' + +=== Appendix C — Affinity routing + +Peers declare strengths on register. Tasks declare tags. Server prefers +matching peers but doesn’t force the match. + +[width="100%",cols="27%,73%",options="header",] +|=== +|Agent |Likely affinities +|Opus (me) |`+supervision+`, `+proof-analysis+`, `+architecture+`, +`+ambiguous-tradeoff+` + +|Claude Sonnet/Haiku |`+routine-edit+`, `+test-writing+`, +`+doc-writing+`, `+lint-sweep+` + +|Codex |`+typescript-refactor+`, `+react+`, `+js-heavy+` + +|Gemini |`+long-document-grok+`, `+legacy-code-read+` + +|Vibe |`+exploratory+` (low-trust) +|=== + +Affinity tags also function as *routing exclusion*: don’t tag a task +with an affinity you don’t want Gemini picking up. + +''''' + +=== Appendix D — Memory auto-lift (what it is) + +*Not* a context-window expander. It’s a retrieval-timing change. + +*Today:* Memories live in +`+~/.claude/projects/-var-mnt-eclipse-repos/memory/*.md+`. I recall them +by grep/search. I often miss relevant ones because I don’t know to look. + +*With auto-lift:* Entering `+007-lang+` triggers the `+007-mcp+` +cartridge’s on-enter hook, which returns relevant memories (Coquelicot +gotchas, v1.0 closure rules, Cerro-Torre postulate pattern) as tool +output. I don’t have to recall — they’re handed to me. + +*Mechanism:* - Static mapping per repo (curated list) - Smart version: +VeriSimDB indexes memories by tag; cartridge queries on enter with +context-derived tags + +*Net effect:* slightly more tokens at session start; fewer mid-session +recall misses where I say "`let me try X`" on something memory warned +against. + +''''' + +=== Appendix E — Schema summary (envelope v1) + +Full schema: +`+boj-server/cartridges/local-coord-mcp/schemas/coord-messages.ncl+` +Full rationale: +`+boj-server/cartridges/local-coord-mcp/docs/envelope-design.adoc+` + +Core envelope fields: + +.... +version integer = 1 +msg_id 12-char lowercase hex +prev_msg_hash SHA-256 of sender's previous envelope +correlation_id optional — request/response pairing +sender peer_id: -<4hex>[@] +recipient peer_id or "*" for broadcast +timestamp ISO 8601 UTC +op_kind one of 18 taxonomy entries +risk_tier 0-4 +payload op-specific shape +sender_confidence 0.0-1.0 (optional, recommended Tier 2+) +sender_reasoning <=200 chars (optional, recommended Tier 2+) +context_fetch_id REQUIRED for Tier 2+ +attestation_refs array of cosigners (REQUIRED Tier 2+ from supervised) +tier_override_reason required when declared tier ≠ default +urgent_direct boolean (reject for supervised) +ack_required boolean +ttl_seconds optional auto-expiry +.... + +''''' + +=== Appendix F — Commit log (this session) + +[width="100%",cols="34%,37%,29%",options="header",] +|=== +|Commit |Subject |Files +|`+5d57daa+` |feat(local-coord-mcp): complete adapter dispatch for +list/send/receive/status |4 + +|`+2a8e4c0+` |feat(local-coord-mcp): per-window peer ID disambiguation +via context field |5 + +|`+ceb5125+` |docs(local-coord-mcp): envelope schema v1 + design +rationale |3 + +|`+4e164a7+` |feat(local-coord-mcp): role tiers, supervisor gating, +quarantine queue |6 + +|`+93c589d+` |feat(local-coord-mcp): proper Nickel contracts with +dependent constraints |3 + +|`+2ae952e+` |feat(local-coord-mcp): durability module — append log + +replay |1 + +|`+282de53+` |feat(local-coord-mcp): persist-on-write + replay-on-init +for coord state |1 + +|`+bafcd25+` |test(local-coord-mcp): restart-preserves-state integration +tests |1 +|=== + +Repository: `+git@github.com:hyperpolymath/boj-server.git+` + +''''' + +=== Appendix G — Federation + authoritative-site (v2, future) + +==== Motivation + +Localhost-only is v1 and the current implementation. v2 adds optional +federation specifically for *joint projects with son* (IDApTIK, ASS). +The motivation is *NOT* credit-load distribution (nice side effect) — +it’s *preventing mission drift like IDApTIK had*. When two people + +their respective AI agents work on the same codebase without a clear +authority, scope creep and contradictory claims on code happen. +Federation with explicit authoritative-site designation is the +structural fix. + +==== Authoritative site concept + +Per project, per time-window, *one site is authoritative*: + +* Authoritative site’s Opus = federation-level supervisor +* Peer site’s Opus = federation-level executor (can suggest, do +independent work, but defers on code-ownership decisions for the +project) +* Cross-site Tier 3 ops require authoritative-site Opus approval +* Primacy handoff is *explicit ceremony*, never automatic + +This maps cleanly onto the existing local supervision hierarchy (just +one level up — the local supervisor/executor/supervised tiers are +unchanged within each site). + +==== Security stack + +When federation is enabled: + +[arabic] +. *VeriSimDB* holds state authoritatively +. *SDP (Secure Device Provisioning)* wraps the coord server +. *Stapeln container* isolates the whole thing (per estate container +policy) +. *HTTP capability gateway* enforces the capability model in front +. *High-security options* set throughout (specific options TBD at build +time) + +==== Envelope changes (v2) + +* `+site_id+` string (localhost v1 implicitly "`local`") +* `+federation+` sub-object with optional fields: +** `+authoritative_site+` +** `+project_id+` +** `+handoff_ceremony_id+` + +==== v1 scope discipline + +* No network fields in envelope v1 (hermetically local, clean boundary) +* No federation stubs in code (add when building v2) +* Umoja federation stays a separate layer; the coord cartridge doesn’t +reimplement Umoja’s gossip / attestation protocols + +==== Trigger to start v2 + +First concrete joint IDApTIK/ASS session where both father and son are +coding in parallel. Currently: v1 still being built; v2 not scheduled. + +''''' + +=== Appendix H — Choreographic + epistemic types (future phases) + +Proposed 2026-04-20 as refinement layers atop the schema. + +==== Choreographic types (Phase 2) + +Session types for multi-party protocols. Makes the supervision gate flow +(supervised → server → \{opus, attester} → server → target) correct _by +construction_ rather than enforced imperatively in Zig. Also locks in +`+context_query+` → `+context_reply+` → `+Tier 2++` as a +compiler-checked antecedent. + +Lives in `+abi/LocalCoord/Protocol.idr+` (Idris2). Compile-time only, no +runtime cost. Starts after Task #8 E2E tests validate the imperative +version. + +==== Epistemic types (conceptual, already load-bearing) + +Tracks what each participant knows (K_A(φ) logic). Already implicit in: + +* `+context_fetch_id+` — knowledge witness; sender proves it knows the +scope +* Role asymmetry (supervisor/executor/supervised) — epistemic access +differences +* Attestation — common knowledge of depth 2 between supervisor + +attester +* Authoritative site (v2 federation) — common knowledge of primacy + +Making explicit: rename `+context_fetch_id+` to `+knowledge_witness+` in +v2. Document the epistemic policy per role in +`+docs/envelope-design.adoc+`. + +==== Temporal-epistemic intersection + +* Watchdog TTL = "`supervisor will know at time T that progress was not +received`" +* Hash chain = sender cannot later know a different past (temporal +immutability) +* Staleness on knowledge witnesses = knowledge ages out + +==== Practical recommendation + +* Phase 1 (now): JSON Schema + imperative Zig +* Phase 2 (after Task #8): Idris2 session types for supervision +choreography +* Phase 3 (v2 federation): full choreographic + epistemic types for +cross-site primacy + +''''' + +=== Appendix I — Echo types (Phase 3 / practice-dogfooding) + +From `+echo-types/readme.adoc+`: Echo types = "`loss that is not total +erasure`"; the fiber Σ(x:A), f x ≡ y. Structured irreversibility with +retained proof-relevant constraint. The repo already has bridges to +choreographic, epistemic, linear, graded, and tropical types +(EchoChoreo, EchoEpistemic, EchoLinear, EchoGraded, EchoTropical, +EchoIntegration). + +==== Natural mappings in coord + +* *Supervisor decision* (approve/reject) — lossy collapse over the +reasoning trace; echo fiber = all envelopes that could have gone this +way +* *Hash chain per sender* — quintessential echo; 32-byte hash collapses +full message but retains "`sender cannot claim different past`" +* *Summary-vs-raw for supervised peers* — epistemic firewall expressed +as echo type; supervised peer knows they’re in the fiber over some real +state, can’t reconstruct +* *Rejected messages* — content forgotten from delivery; reason survives +as constraint; forensic audit = echo-fiber query +* *Tier promotion by sanity gate* — declared tier collapses to effective +tier; retained: effective = promoted +* *Broadcast identity* — recipient can’t distinguish intentional from +incidental targeting + +==== Dogfooding value + +* Use echo-types IN coord — Phase 3. Real non-toy consumer. +* EchoIntegration combines knowledge + choreography + graded degradation +— almost exactly our supervisor/attestation flow. +* Makes coord’s audit + summary semantics theorem-backed rather than +convention. + +==== Phase plan (revised) + +* Phase 1 (now): JSON Schema + imperative Zig enforcement + in-memory +supervision +* Phase 2 (post-Task #8): Nickel contracts upgrade for server-side +validation; Idris2 session types for supervision choreography +* Phase 3: Agda echo-types formalisation of audit + summary + hash chain +layer, using EchoChoreo + EchoEpistemic + EchoTropical bridges +* Phase 4 (with v2 federation): Full choreographic + epistemic + echo +formalisation across sites with primacy ceremony as choreography + +knowledge transfer + +''''' + +=== Appendix J — Tropical types (temporal + trust arithmetic) + +Tropical semiring = (min, +) or (max, +). Useful as modeling lens (not +runtime enforcement) for: + +* *Watchdog TTLs* = tropical max: +`+claim_expiry = max(registered_at + TTL, last_heartbeat + TTL)+` +* *Trust composition* = tropical min: transitive trust bounded by +weakest link +* *Risk tier promotion* = max: `+effective = max(declared, sanity_min)+` +* *Attention budget / supervisor priority* = tropical inf over +priorities + +Already present in 007-lang proofs (`+proofs/TropicalSemiring.idr+`). +echo-types has `+EchoTropical+` bridge. Integrates cleanly with Phase 3 +echo-types formalisation. + +''''' + +=== Appendix M — Multi-model supervision extension (DDs 32–37) + +==== Motivation + +Two real needs surfaced on 2026-04-20 that the v1 design handles +mechanically but not ergonomically: + +[arabic] +. *Fallback authority.* If Claude credits run out, another model (GPT-5, +Mistral, Gemini Ultra) should be able to step in as the master / +scheduler without code changes. The env-secret gate already permits this +— but the defaults, the naming, and the vendor assumptions bake in +"`Claude = master`". +. *Model-specialist routing.* Inside a multi-model stack, work should +flow to the best-matched peer: Gemini Pro for code + maths, Gemini Flash +for lightweight jobs, reasoning-class models for hard problems, +prover-specialised peers (Leanstral on Lean, Gemini on Agda, etc.). +Today this is implicit in the supervisor’s head; DD-28 learns it over +time but has nothing to work with on cold start. + +The extension splits *trust* from *capability*, adds a small amount of +explicit metadata peers declare on register, and renames the three trust +tiers to match the craft-guild ladder the estate already uses. + +==== Naming — the three axes + +[cols="`1,2,3`",options="`header`"] |=== | Axis | Values | Meaning + +[verse] +-- +_Trust role_ (one per peer, gated) +`+master+` | `+journeyman+` | `+apprentice+` +Who may approve Tier 2+ (master); who acts solo on Tier 2 (journeyman); who is quarantined for review (apprentice). Master gated by `+BOJ_MASTER_TOKEN+` env secret. +-- + +[verse] +-- +_Capability_ (multiple per peer, advertised) +class ∈ \{reasoner, coder, mathematician, scribe, proofsmith, reader, jester}; tier ∈ \{A, B, C}; prover_strengths ∈ map of provers → 0–100 +What the peer is _good at_. Independent of trust. Drives affinity routing. `+jester+` = "`silly jobs`" — descriptive, not pejorative. `+proofsmith+` covers all formal-proof work. +-- + +[verse] +-- +_Model identity_ (set on register) +client_kind ∈ \{claude, gemini, openai, mistral, copilot, custom}; variant: free-form string +Who the peer _is_. Kind is the vendor family; variant is the specific model line (`+"opus-4.7"+`, `+"gpt-5-reasoning"+`, `+"flash-2.5"+`, `+"leanstral"+`). +-- + +|=== + +Peer id rendering stays `+-<4hex>[@]+` — variant and +capabilities surface in `+coord_list_peers+` payload but don’t crowd the +short identifier. + +==== What DDs 32–37 change concretely + +==== DD-32 — Role renaming + +FFI enum: [source,zig] —- pub const Role = enum(c_int) \{ master = 0, // +was: supervisor journeyman = 1, // was: executor apprentice = 2, // was: +supervised }; —- + +Env var `+BOJ_SUPERVISOR_TOKEN+` → `+BOJ_MASTER_TOKEN+`. Old name +accepted as fallback for one release. Idris2 ABI in +`+abi/LocalCoord/SafeLocalCoord.idr+` updated with the same enum names. + +==== DD-33 — client_kind + variant + +=== [source,zig] + +pub const ClientKind = enum(c_int) \{ claude = 0, gemini = 1, copilot = +2, custom = 3, openai = 4, // new mistral = 5, // new }; —- + +`+coord_register+` gains an optional +`+variant: *const u8, variant_len: c_int+` pair. Stored per peer (max 32 +bytes, alphanumeric + `+.-_+`). Rendered in `+coord_list_peers+` +response. + +==== DD-34 — Capability advertisement + +`+coord_register+` gains three optional fields: + +* `+class_str+` — one of the 7 class names. +* `+tier_char+` — `+A+` / `+B+` / `+C+`. +* `+prover_strengths_json+` — `+{"lean4":80,"agda":60,"idris2":90}+`. + +Stored per peer. Surfaced via new +`+coord_get_peer_capabilities(peer_id)+` FFI call. Affinity router uses +these + DD-28 track record as a weighted sum (advertised ↓ decays, +track-record ↑ grows). + +==== DD-35 — Master handoff + +=== [source] + +coord_transfer_master(current_master_token, new_peer_id_str, secret) -> +0 on success -> -1 caller not master -> -2 target peer not found -> -3 +secret mismatch -> -4 target peer’s role is `+apprentice+` (blocked — +must be journeyman+) —- + +Logged as `+AUDIT(kind=MASTER_HANDOFF, from_idx, to_idx)+`. Survives +restart via replay (the AUDIT event + the two PEER_ROLE_SET events the +handoff emits). + +==== DD-36 — Difficulty hint + +Envelope gains `+difficulty_hint: "low" | "medium" | "high"+` +(optional). Orthogonal to risk_tier — risk says _how bad if wrong_, +difficulty says _how hard to get right_. The master/scheduler uses the +combination for model-class match: + +[cols="`1,1,3`",options="`header`"] |=== | risk | difficulty | prefer | +low | low | jester (Flash) | low | high | reasoner (Pro / Sonnet) | high +| low | journeyman + trust check — "`easy but dangerous`" +(e.g. `+rm -rf+` on verified path) | high | high | master + reasoner + +attester (cosigned Tier 3) |=== + +Nickel contract update: `+DifficultyHintValid+` predicate. + +==== DD-37 — Prover routing via tag + +Zero code change. Convention: task tags prefixed `+proof:+` — +`+proof:lean4+`, `+proof:agda+`, `+proof:idris2+`, `+proof:rocq+`, +`+proof:tla+`. Peers advertise per-prover strength via DD-34. The +existing affinity machinery picks the best match. + +Example dispatch: user files a task +`+{tag: "proof:agda", difficulty: high, risk: 3}+`. Scheduler looks at +all `+journeyman+`/`+master+` peers with `+class=proofsmith+` and +`+prover_strengths.agda > 50+`; picks the one with the best combined +(advertised × track_record) affinity; delegates. + +==== What stays the same + +* Loopback-only bind (DD-1, P-00). +* Session-token unforgeability (DD-2, P-03). +* Tier 2+ from apprentice → quarantine (DD-6 renamed). +* Replay-equivalence keystone theorem (P-06). +* Durable state schema — new fields extend events, never reshape them. +* GitHub-only mirroring, v1 hermetic-local scope. + +==== Migration + +One release of backwards-compat shims: [source] —- // coord_register +accepts both role=supervisor / role=master → master // coord_register +accepts role=executor / role=journeyman → journeyman // coord_register +accepts role=supervised / role=apprentice → apprentice // +BOJ_SUPERVISOR_TOKEN read as fallback when BOJ_MASTER_TOKEN unset —- + +Log format: the `+peer_role_set+` event payload already carries a byte +for the role — the enum integers stay (0=master, 1=journeyman, +2=apprentice), so no replay-breaking change. + +Removed cleanly in the release after (per feedback_commit_asap — no +backwards-compat debt lingers). + +==== Implementation sketch (not in this commit) + +Rough order of landing: + +[arabic] +. _Rename pass_ — DD-32. FFI enum, adapter string literals, Idris2 ABI, +Nickel contracts, docs. Mechanical; ~1 day. +. _client_kind extension + variant_ — DD-33. New enum values + variant +storage. ~0.5 day. +. _Capability advertisement_ — DD-34. Three optional register fields + +`+coord_get_peer_capabilities+`. ~1 day. +. _Master handoff_ — DD-35. One new FFI fn + adapter binding + audit +event. ~0.5 day. +. _Difficulty hint_ — DD-36. Envelope schema + Nickel contract + +optional field in coord_send/coord_send_gated. ~0.5 day. +. _Prover convention doc_ — DD-37. README update. ~1 hour. + +Total ≈ 3.5 days. No new proof obligations beyond the rename (P-10 gets +renamed and stays the same theorem). + +==== Non-goals + +* Splitting gatekeeper from scheduler (DD-38 — separate RFC when +needed). +* Heuristics for automatic model selection (master still chooses; the +extension just gives them better metadata). +* Any change to v2 federation plans (Appendix G) — the rename lands in +v2 unchanged. +* Cost-aware scheduling. Credit burn tracking is a separate feature that +can consume the capability metadata later. + +''''' + +=== Appendix K — Roadmap: deferred work (things not done in v1, worth doing sometime) + +Ordered by rough priority within each phase. Each item references its +rationale elsewhere in the log. + +==== Phase 1b — refinements during current coord buildout + +* *Multi-model supervision extension (DDs 32–37)* — 6 tasks, ~3.5 days +total. See Appendix M for full rationale. +** _Task #32_ — Role rename +`+supervisor/executor/supervised → master/journeyman/apprentice+`. FFI +enum + Idris2 ABI + adapter strings + env var. Backward-compat shims for +one release. ~1 day. +** _Task #33_ — Extend `+client_kind+` enum with `+openai+`, +`+mistral+`; add `+variant+` free-form string. ~0.5 day. +** _Task #34_ — Capability advertisement on register (class, tier, +prover_strengths) + `+coord_get_peer_capabilities+` FFI. ~1 day. +** _Task #35_ — Live master handoff via +`+coord_transfer_master(current_token, new_peer_id, secret)+`. ~0.5 day. +** _Task #36_ — `+difficulty_hint+` envelope field + Nickel contract +`+DifficultyHintValid+`. ~0.5 day. +** _Task #37_ — Prover tag convention doc (`+proof:lean4+`, +`+proof:agda+`, `+proof:idris2+`, `+proof:rocq+`, `+proof:tla+`). README +update. ~1 hour. +* *Task #7b*: Swap `+coord_durability.zig+` backend to `+verisimdb-mcp+` +FFI once that FFI is real. Typed log helpers stay as the API; +implementation switches from append-only file to VeriSimDB octad calls. +Prerequisite: `+cartridges/verisimdb-mcp/ffi/verisimdb_ffi.zig+` gains a +non-stub implementation. See DD-31. +* *Task #16*: Nickel contracts proper (dependent constraints, predicate +composition). In-flight this session. See DD-16 + user guidance on +Zig/Nickel layering. +* *Task #17*: Deno shim in mcp-bridge to run Nickel contracts at +validation time. User confirmed option (b). After #16 lands. +* *Task #13*: Track-record table in VeriSimDB + effective_affinity +computation. +* *Task #14*: Reassignment suggestion engine — surfaces outliers to Opus +for user review. +* *Task #15*: Dispatch-preference mode on claims +(deliberate/broadcast/auto). +* *coord_health metrics tool* — counters for active peers, pending +quarantine, reject rate, claim depth. +* *Hash chain per envelope* — sender-side chain with `+prev_msg_hash+`; +server tracks chain head. +* *Content sanity gate* — file-reference validity check against +recent-FS cache + self-contradiction heuristic + risk-tier escalator +patterns. +* *Watchdog TTL enforcement* — per-role defaults (30s supervised, 5min +executor), `+progress+` heartbeats reset. +* *Warn-drift broadcast on watchdog auto-release* via Opus review +(DD-21). +* *Quarantine queue spill to VeriSimDB when full* (DD-21). +* *Audit-echo anchor* — preserve old chain head on peer crash+restart +(DD-29). +* *Rate-limit on rejections* — 5 rejects / 10 min = 30s cooldown +(confirmed). +* *Drift detector* — flag when confidence > 0.8 AND effective_affinity < +0.3 (confirmed). + +==== Phase 2 — after Task #8 (E2E test proves the imperative version) + +* *Idris2 session types* for the supervisor/attestation choreography in +`+abi/LocalCoord/Protocol.idr+`. Makes protocol compliance a +compile-time property. +* *Deontic types* for supervision rules (if user confirms reading of +"`dyadic types`"). Could formalise "`tier 4 forbidden for supervised`" +as a type-level obligation. + +==== Phase 3 — formal foundations + +* *Agda echo-types formalisation* of audit + summary + hash-chain as +echo types, using `+EchoChoreo+` / `+EchoEpistemic+` / `+EchoTropical+` +bridges in `+echo-types/+` repo. Dogfoods echo-types (Appendix I). +* *Tropical types* as modeling lens for TTL + trust + tier arithmetic +(Appendix J). Integrates via `+EchoTropical+`. +* *Epistemic types* explicit formalisation — rename `+context_fetch_id+` +to `+knowledge_witness+`, document per-role epistemic policy in +envelope-design.adoc. + +==== Phase 4 — v2 federation (trigger: first joint IDApTIK/ASS session) + +* *Envelope v2* with `+site_id+` + `+federation.authoritative_site+` + +`+federation.project_id+` + `+federation.handoff_ceremony_id+`. +* *Authoritative-site model* (DD-22) — one site holds primacy per +project; peer defers on code-ownership. +* *Security stack* — SDP + Stapeln container + HTTP capability gateway + +high-security options. +* *Primacy handoff ceremony* — explicit protocol, formalised as +choreographic + epistemic transfer. +* *Cross-site attestation + trust composition* — tropical min for +transitive trust across federated sites. +* *Integration with Umoja federation layer* (gossip + hash attestation) +for cross-machine transport. + +==== Deferred design questions (revisit if needed) + +* Nickel vs Zig boundary for rich validation — currently: Nickel +contracts = shape + predicates, Zig = imperative state. Could shift if +Nickel runtime becomes too slow. +* Multi-party session-type generalisation (if dyadic types mean 2-party +classical session types) — multi-party choreographic is the preferred +fit but classical dyadic compositions may be easier initially. +* LLM-based summary generator for `+context_reply+` — currently no +design; could be Opus synthesising summaries per-request, or a fixed +algorithm, or pre-baked templates. + +==== Non-goals / explicitly not doing + +* HTTP capability gateway + SDP for local-only v1 (overkill; confirmed). +* Auto-modifying affinities without Opus + user review (always a loop +back through supervisor/user). +* Pushing `+boj_cartridge_invoke+` wiring (Task #2) ahead of the primary +HTTP path — not blocking. +* Cross-machine transport in v1 envelope. + +''''' + +=== Appendix L — Session close 2026-04-20 + handoff brief + +==== Session close state (updated 2026-04-20 late) + +* *Session duration:* long, multi-phase; huge land this session +* *Repo:* `+hyperpolymath/boj-server+` main +* *Tasks complete:* #1, #3, #4, #5, #6, #7, #8, #13, #14, #15, #16, #17, +#32, #35, #36, #37 (sixteen total). DDs 1–38 recorded; Appendix M +extension formalised. +* *Tests:* 27 Zig + 41 Deno E2E, all green. Panic-attack assail clean (1 +suppressed `+nickel eval+` false-positive). +* *Benches:* ~4.7 µs append, ~9 µs durable round-trip, ~110k ops/sec +durable, ~9M ops/sec no-durability. +* *Proof schedule:* 10 P0/P1 obligations tracked in +`+cartridges/local-coord-mcp/abi/LocalCoord/PROOF-SCHEDULE.adoc+`. +* *Key commit range:* `+5d57daa..HEAD+` (see `+git log --oneline -30+` +for full list). + +==== Remaining tasks (for handoff) + +[width="100%",cols="13%,23%,38%,26%",options="header",] +|=== +|# |Task |Priority |Notes +|2 |Wire `+boj_cartridge_invoke+` to real FFI |low |Deferred — HTTP path +works fine + +|33 |Extend `+client_kind+` with `+openai+` + `+mistral+`; add +`+variant+` string |medium |~0.5 day — unlocks cross-model ops + +|34 |Capability advertisement (class/tier/prover_strengths) |medium |~1 +day — consumes #33 + +|9 |Create 007-mcp BoJ cartridge |medium |Separate track — unblocks +#10–#12 + +|10 |007 on-enter/on-exit contractile hooks |blocked by #9 |— + +|11 |Fill 007 missing bust + adjust contractiles |blocked by #9 |— + +|12 |Memory auto-lift on 007 context entry |blocked by #9 |— + +|7b |Swap `+coord_durability.zig+` backend to `+verisimdb-mcp+` FFI +|blocked by verisimdb-mcp real FFI |DD-31 + +|Proofs P-04..P-07 |Idris2 `+Durability.idr+` (record format, CRC +truncation, replay-equivalence, quarantine state machine) |formal +sign-off |~6 days +|=== + +==== Handoff brief for the next Claude + +*First actions (in order):* + +[arabic] +. Read this file (`+Desktop/COORD-MCP-DESIGN-LOG.md+`) end-to-end. +Source of truth for decisions, wiring, and pending work. +. Read workspace memory: `+project_coord_supervision_architecture.md+`, +`+project_federation_authoritative_site.md+`. +. `+git -C /var/mnt/eclipse/repos/boj-server log --oneline -30+` to see +the landed commits. +. Read `+cartridges/local-coord-mcp/docs/envelope-design.adoc+` for +rationale, plus `+schemas/coord-messages-contracts.ncl+` for Nickel +contracts and `+schemas/coord-messages.ncl+` for the JSON-Schema +envelope. +. Read Appendix M (in this file) for the three-axis peer model (trust +role / capability / model identity). + +*Recommended next target — Task #33 + #34 (cross-model capability +metadata), ~1.5 days:* + +Rationale: Task #32 landed the role rename +(master/journeyman/apprentice) but peers still can’t distinguish +Gemini-Pro from Gemini-Flash, or advertise that they’re strong at Lean +vs Agda. Tasks #33 + #34 close that gap and are the last Phase 1b +protocol changes before the 007-mcp track (#9–#12) picks up. + +Concrete scope: - Add `+openai = 4, mistral = 5+` to +`+pub const ClientKind+` in `+ffi/local_coord_ffi.zig+`. Update the +Idris2 ABI and the C-ABI enum comment. - Add a `+variant: [32]u8+` + +`+variant_len: u8+` to the `+Peer+` struct. New FFI +`+coord_set_variant+` + `+coord_read_peer_variant+`. New log event +`+PEER_VARIANT_SET+` (event type 15) + replay dispatcher branch. - Add +`+class: u8+` (enum index into +reasoner/coder/mathematician/scribe/proofsmith/reader/jester), +`+tier: u8+` (ASCII '`A`'/'`B`'/'`C`'), `+prover_strengths: [5]u8+` +(agda/lean4/idris2/rocq/tla, 0–100 each) on Peer. New FFI +`+coord_set_capabilities+` + `+coord_read_peer_capabilities+`. New log +event `+PEER_CAPABILITIES_SET+` (event type 16) + replay dispatcher +branch. - Adapter: accept new fields in `+coord_register+` JSON, surface +them in `+coord_list_peers+` response. Update `+cartridge.ncl+` tool +descriptions + re-export JSON. - Deno E2E: extend existing +`+e2e_coord.ts+` with one Phase 8 that exercises variant + capabilities +via register and read-back. - Expect ~5 commits: role-FFI-extend, +adapter-glue, capability-FFI-extend, adapter-glue, E2E-extension. + +*Do not:* - Redesign anything covered by DD-1 through DD-38 without +flagging first. - Implement choreographic / epistemic / echo / tropical +types (Phase 2+/Phase 3 — Appendices G/H/I/J/K). - Touch v2 federation +fields on envelope (Phase 4 per DD-22 + Appendix G). - Build HTTP +capability gateway or SDP for local (non-goal — v2 federation only). - +Start the 007-mcp track (#9–#12) in the same session as #33/#34 — +different cartridge, different file tree, keeps churn tidy. + +*Style rules this session followed (continuity):* - Commit ASAP (one +unit = one commit), specific paths in `+git add+`. - Push origin after +commit (GitHub is source of truth; no other forges). - Update this log +at each checkpoint (commits + decisions + open items). - Ask before +writing new memories; explicit requests bypass confirmation. - Test + +bench + panic-attack-assail before claiming complete (full-battery +rule). - Keep `+Connection: close+` in adapter HTTP responses (needed +for Deno fetch pool sanity). + +==== Status markers for parallel session planning + +* *Safe to parallelise NOW*: documentation, design, Phase 2+ planning — +no shared state risk. +* *Safe to parallelise*: 007-mcp family (#9-12) can run in a separate +session independent of coord finishers (#33-34). Different cartridge +tree, no file overlap. +* *Do not parallelise*: two agents both touching +`+cartridges/local-coord-mcp/ffi/+` or `+mcp-bridge/+` in the same +wall-clock window — the FFI enum / Peer struct is the shared state. + +''''' + +=== How this file evolves + +I update Part 2 (progress ledger), Part 3 (decisions as they’re taken), +Part 4 (closing questions as you answer), and Appendix F (commit log) at +each meaningful checkpoint. New topics get new appendices (G, H, …). +Structural rewrites happen only when the design itself shifts — +otherwise it stays append-mostly so you can diff what changed between +sessions. diff --git a/docs/handover/COORD-MCP-DESIGN-LOG.md b/docs/handover/COORD-MCP-DESIGN-LOG.md deleted file mode 100644 index 4d062ffb..00000000 --- a/docs/handover/COORD-MCP-DESIGN-LOG.md +++ /dev/null @@ -1,736 +0,0 @@ - -# local-coord-mcp + BoJ agent coordination — design log - -**Started:** 2026-04-20 -**Lead:** Opus (1M context) -**Scope:** Multi-agent coordination across Claude + vibe + codex + gemini windows on one box; full BoJ cartridge support; 007 dogfooding. - -This file is a running design log. Outline up top, detailed explanations in the appendices. Updated as the work progresses. - ---- - -## Part 1 — Outline of what we're building - -### The big picture - -A localhost message bus (`boj-server/cartridges/local-coord-mcp`) that lets multiple AI agents on the same machine discover each other, exchange typed messages, claim tasks without collision, and operate under a supervision model where Opus co-supervises with you. Non-Claude agents (Gemini, Codex, Vibe) work under a firewall that contains their failure modes without stopping them from contributing. - -### Layers - -1. **Transport** — loopback-only Zig REST server on port 7745. Idris2 ABI proves loopback-only at compile time (`IsLoopback` type has exactly two constructors; bind to non-loopback is *type-impossible*). -2. **Identity** — `coord_register(client_kind, role_hint, context)` → peer_id like `claude-7f3a@007-lang` + 128-bit CSPRNG session token. -3. **Envelope** — typed messages with 18 op_kinds, 5-level risk ladder, Byzantine-safe fields (hash chain, attestation refs, self-assessment, context_fetch_id). -4. **Supervision** — quarantine queue for Tier 2+ ops from `supervised` role peers; supervisor reviews/approves/rejects. -5. **Durability** — VeriSimDB sidecar (Task #7) persists inbox, claims, audit log, track record. -6. **007-mcp cartridge** — exposes `oo7` CLI + Justfile recipes as MCP tools for routine 007 operations. - ---- - -## Part 2 — Progress ledger (session 2026-04-20) - -| # | Task | Status | Commit | -|---|------|--------|--------| -| 1 | Complete adapter dispatch for list_peers/send/receive/status | ✅ | `5d57daa` | -| 2 | Wire `boj_cartridge_invoke` to real FFI | ⏳ pending | — | -| 3 | A2ML envelope schema + design doc | ✅ | `ceb5125` | -| 4 | Per-window peer ID disambiguation (`@`) | ✅ | `2a8e4c0` | -| 5 | Supervision tier/role in FFI + ABI | ✅ | `4e164a7` | -| 6 | Supervisor tools: coord_review / approve / reject | ✅ | `4e164a7` | -| 16 | Nickel contracts with dependent constraints | ✅ | `93c589d` | -| 7 | Durable coord state (append log + replay) | ✅ | `2ae952e`, `282de53`, `bafcd25` | -| 8 | E2E test: 2-instance master-gate + durability | ✅ | `3e9eae8` | -| 13 | Track-record table + effective_affinity + coord_get_affinities | ✅ | `f8cafbf`, `3bd4710` | -| 15 | Dispatch preference + task_difficulty + sender_confidence + reject cooldown | ✅ | `6065878` | -| 14 | Reassignment engine (server-origin quarantine entries for master review) | ✅ | `9e40a86` | -| 17 | Deno/Node Nickel shim — runtime envelope validation in mcp-bridge | ✅ | `ed85ca2` | -| 32 | Role rename supervisor/executor/supervised → master/journeyman/apprentice | ✅ | `634c163` | -| 35 | `coord_transfer_master` — live master handoff | ✅ | `634c163` | -| 36 | `difficulty_hint` envelope field + `DifficultyHintValid` Nickel contract | ✅ | `7f2f4a9`, `aeae440` | -| 37 | Prover-tag convention (`proof:`) doc + example | ✅ | `eba7cfe` | -| 9 | Create 007-mcp BoJ cartridge | ✅ | `62bdac0` (007-lang) | -| 10 | 007 on-enter/on-exit contractile hooks | ✅ | `018a1fd` (007-lang) | -| 11 | Fill 007 missing bust + adjust contractiles | ✅ | `6bbb4f8` (007-lang) | -| 12 | Memory auto-lift on 007 context entry | ✅ | `e753d10` (007-lang) | -| k9-svc | Spot-fix: move 007 k9 from contractiles/ to svc/ (ADR-001) | ✅ | `f6f3c91` (007-lang) | -| 33 | `client_kind` + `variant` extension (openai/mistral + free-form variant) | ⏳ pending | — | -| 34 | Capability advertisement on register (class, tier, prover_strengths) | ⏳ pending | — | - -**Committed + pushed this lane (Prompt 3 — coord finishers):** `f8cafbf`, `3bd4710`, `6065878`, `9e40a86`, `ed85ca2`, `634c163`, `7f2f4a9`, `aeae440`, `eba7cfe` on `hyperpolymath/boj-server main`. - -**Committed + pushed this lane (Prompt 2 — 007-mcp family):** `62bdac0`, `018a1fd`, `f6f3c91`, `6bbb4f8`, `e753d10` (+ one follow-up role-rename sync) on `The-Metadatastician/007 main`. Per CLAUDE.md 007 is never mirrored — GitHub-only. - -**Memory written:** `project_coord_supervision_architecture.md` + `project_federation_authoritative_site.md` (workspace memory index). - ---- - -## Part 3 — Design decisions taken (concise) - -| # | Decision | Why | -|---|----------|-----| -| DD-1 | Localhost-only v1, loopback bind at 127.0.0.1:7745 | Compile-time proof via Idris2 `IsLoopback`; no exposure surface | -| DD-2 | Session tokens (128-bit CSPRNG), not OAuth / vault | Local-only; no cross-machine trust needed | -| DD-3 | Hybrid peer IDs `-<4hex>[@]` | Human-readable + collision-free across windows | -| DD-4 | 18 typed op_kinds (not raw strings) | Routable, auditable, schema-validatable | -| DD-5 | 5-level risk ladder (0-4) | Proportional gating; coord overhead ≠ rate limiter | -| DD-6 | Three roles: supervisor / executor / supervised | Trust-tiered; Opus supervises, Claude executes, others gated | -| DD-7 | Opus default to supervisor when claimed; Claude Sonnet/Haiku default to executor; non-Claude defaults to supervised | Matches actual agent strengths + user's trust model | -| DD-8 | Byzantine safety: 5 mechanisms (hash chain, M-of-N attestation, watchdog claims, sanity gate, audit log) | Specifically addresses "Gemini is often nuts" failure modes | -| DD-9 | Self-assessment 4 layers: affinity + confidence + track record + drift detector | Catches self-overclaiming by cross-check | -| DD-10 | Adaptive awareness — `context_fetch_id` required for Tier 2+ | Forces context read before risky action; blast-radius-scaled | -| DD-11 | Summary-vs-raw context gated by role | Supervised peers never see raw state (prevents hallucinated connections) | -| DD-12 | User interaction routing: fyi / clarify / blocker op_kinds; only executor/supervisor can set urgent_direct | Single user locus; supervised peers firewalled from interrupting you | -| DD-13 | Tier 4 forbidden list baked into schema for supervised peers | Force-push, license touches, always-private-repos — schema-level reject | -| DD-14 | Sanity auto-promote patterns (`git push` → Tier 3 regardless of declared) | Blocks tier underclaiming by content-match | -| DD-15 | Default dispatch mode: hybrid (Opus seeds + peers self-claim by affinity) | Most flexible; works for lanes, MoE, fine-grained equally | -| DD-16 | Nickel source-of-truth → JSON export for cartridge manifests | Matches broader boj-server pattern | -| DD-17 | Quarantine queue is in-memory ring (MAX=32) + spills to VeriSimDB when full | Hot cache + truth separation; nothing dropped | -| DD-18 | GitHub is single source of truth — push `origin` only, never other forges | Estate-wide policy; mirroring is hub-and-spoke downstream | -| DD-19 | Supervisor role gated by env-var `BOJ_SUPERVISOR_TOKEN` | Stops Gemini/supervised from claiming supervisor role | -| DD-20 | Watchdog TTL 30s on supervised claims, `progress` heartbeat resets timer | Active work extends claim; silence kills it | -| DD-21 | Watchdog auto-release broadcasts `warn_drift` via Opus review | Prevents pile-on pivots when an agent abandons a task | -| DD-22 | v2 federation with authoritative-site model (IDApTIK-drift-motivated) | One site holds primacy per project; peer site defers on code-ownership; no contradictory claims | -| DD-23 | VeriSimDB uses both patterns case-by-case per cartridge | Coord = per-box (machine-level); 007 repo data = per-project (travels with repo); track record = per-box; memory auto-lift index = per-project | -| DD-24 | 007-mcp cartridge source lives in 007-lang with install hook into boj-server/cartridges/ | Keeps cartridge next to its code; install script in 007's Justfile deploys to boj-server | -| DD-25 | 007-mcp exposes full `oo7` CLI surface as MCP tools | No curated subset — all 25+ CLI commands surfaced | -| DD-26 | Memory auto-lift = dynamic tag-indexed VeriSimDB lookup | Memories tagged; cartridge queries on 007 entry with repo-derived tags; returns matches as tool output | -| DD-27 | Attester selection is deliberate by affinity (Opus picks), with exception for broadcast on "anyone could do it" tasks | Quality on hard attestations; speed on trivial ones; overhead-aware | -| DD-28 | Learning rules: track-record updates effective_affinity; periodic scan surfaces reassignment suggestions to Opus → user | System refines assignments over time; never auto-modifies; preserves supervisor/user loop | -| DD-29 | Peer crash+restart = fresh chain as new peer; old chain preserved as audit echo anchor | Epistemically honest (new peer doesn't claim lost knowledge); echo type preserves forensic continuity without identity continuity | -| DD-30 | Dispatch preference per claim: deliberate (default novel/challenging) vs broadcast (trivial/routine) vs auto | Overhead-aware dispatch; easy tasks don't compete with hard ones for Opus attention | -| DD-31 | Task #7 durability implemented as in-tree `coord_durability.zig` (append-only log + CRC-trailed records + typed helpers), not via `verisimdb-mcp` FFI | `verisimdb-mcp`'s FFI is all stubs (`return 0 // Stub`); using it would have meant implementing VeriSimDB first. Option C narrows scope to finish Task #7 cleanly. The typed log helpers (`logPeerAdd`, `logInboxPush`, …) are the stable seam a follow-up task can swap behind verisimdb-mcp once its FFI is real. State lives at `$BOJ_COORD_STATE_DIR` (default suggestion `$XDG_STATE_HOME/boj-server/coord/`), per-box per DD-23. | -| DD-32 | Rename trust roles to **master / journeyman / apprentice** (guild-craft terminology) | Matches user's own "master production/scheduler" phrasing + 007's craft aesthetic. Apprentice→journeyman→master is a clear ladder with no corporate-hierarchy baggage. `supervisor` gets split into pure trust tier (this) + capability class (DD-34). Backward-compat: old enum values preserved as aliases for one release before removal. Env var renamed `BOJ_SUPERVISOR_TOKEN → BOJ_MASTER_TOKEN`. | -| DD-33 | Extend `client_kind` enum with `openai` and `mistral`; add free-form `variant` string (e.g. `"opus-4.7"`, `"gpt-5-reasoning"`, `"flash-2.5"`, `"leanstral"`) | Today's 4-entry enum can't distinguish Pro from Flash or Opus from Sonnet. Variant is a string (not enum) because model-line naming evolves faster than the cartridge ABI. Peer ID becomes `[@]-<4hex>[@]` rendered form — or we keep the old short form and carry variant as a separate field. Leaning separate field for stability. | -| DD-34 | Capability advertisement on register: **class** ∈ `{reasoner, coder, mathematician, scribe, proofsmith, reader, jester}`, **tier** ∈ `{A, B, C}`, **prover_strengths** map `{agda, lean4, idris2, rocq, tla}` → 0–100 | Today the server learns affinity from outcomes alone; cold starts are weak. Advertised capabilities seed the router. Class vocabulary kept short + craft-flavoured on purpose (`proofsmith` not "prover"; `jester` for silly jobs — not pejorative, descriptive). Tier A/B/C is self-declared but calibrated by track record via DD-28. | -| DD-35 | Live **master handoff** via `coord_transfer_master(current_token, new_peer_id, secret)` | Today if Opus's session ends mid-flight, any replacement has to re-promote via the env secret and there's a gap. Handoff tool lets the outgoing master hand the role to a named successor without a restart. Secret still required — prevents hostile handoff. Audit-logged as `AUDIT(kind=MASTER_HANDOFF, from, to)`. | -| DD-36 | **Difficulty hint** on envelope: `difficulty_hint ∈ {low, medium, high}` | Routing aid for the scheduler; orthogonal to `risk_tier`. "Silly job" ≈ low; "thought-provoking bit" ≈ high. Not policy-enforced — master (or the peer's self-claim) picks the model-class match. Cheap to add (3 bits on the wire), immediate value for per-model routing. | -| DD-37 | **Prover routing by tag convention only** — no new FFI; use `proof:agda`, `proof:lean4`, `proof:idris2`, `proof:rocq`, `proof:tla` as task tags; peers advertise per-prover strength via DD-34 | Zero-code routing — existing affinity picks the right peer per prover. "Leanstral for Lean" = Mistral-variant peer with high `proof:lean4` strength; Gemini-Pro for Agda = peer with high `proof:agda`. Documentation-only; no protocol change required. Keeps the "hard to get wrong" character. | -| DD-38 | Split **gatekeeper (trust)** from **scheduler (dispatch)** roles — not done in this extension, flagged for future | Today the master both approves and schedules. Splitting them lets one peer hold veto and another hold routing (small-team = merged; big = split). Out of scope for DDs 32–37; revisit when cross-model dispatch gets complex. | - ---- - -## Part 4 — Answered design questions (closed) - -1. ✅ **Supervisor role claim protection** — env-var gated (`BOJ_SUPERVISOR_TOKEN`). Server only grants the supervisor role to a register call that presents the secret. -2. ✅ **Watchdog TTL on supervised peers' claims** — 30s baseline, `progress` heartbeats reset the timer. Active work extends the claim; silence kills it. -3. ✅ **Quarantine queue behaviour on full** — spill to VeriSimDB so nothing is dropped. In-memory MAX_QUARANTINE=32 is a hot cache; VeriSimDB is durable tail. -4. ✅ **Watchdog auto-release → `warn_drift` broadcast** — YES, but Opus-reviewed first. Server generates the broadcast candidate and queues it in the quarantine flow with `origin=server`; Opus approves or vetoes before other peers see it. Prevents pile-on pivots. -5. ✅ **Multi-box future** — v1 hermetically local (no network fields). v2 adds optional federation for joint projects with son (IDApTIK, ASS), with **authoritative-site** designation per project to prevent mission drift. Security stack: VeriSimDB + SDP + Stapeln + HTTP capability gateway. Details in Appendix G. - ---- - -## Part 5 — Interaction model (how you actually use this) - -**Primary channel: main terminal, me (Opus) as your chief of staff.** - -Other agents run in their own terminals / processes. When they have questions, they emit envelope ops: - -- `fyi` — log-only, never interrupts -- `clarify` — batched through me, synthesised, shown periodically -- `blocker` — inline interrupt in my output, stops the peer's task -- `urgent_direct` flag — allowed for supervisor/executor peers only; supervised peers CANNOT set it (they never interrupt you) - -I dedupe near-identical questions, reject confabulated ones from supervised peers, and relay to you with context. You reply to me; I route answers back via `coord_send`. - -Escape hatch: nothing stops you from visiting another agent's terminal directly. It's just that you shouldn't *need* to. - ---- - -## Appendix A — Deep rationale: supervision architecture - -### Why tiered trust is the right frame - -The user's model isn't "Gemini bad, Claude good". It's: *big things I stress a lot more, little things I trust them on*. That's a **risk ladder**, not a binary gate. So the design is proportional: Tier 0/1 flows free (the 90% case — status, query, tests, lint, small edits), Tier 2 gets a light auto-approve gate (multi-file edits, dep bumps), Tier 3 is the hard gate where Opus must approve (commits, pushes, schema changes), Tier 4 is schema-level forbidden for `supervised` (force-push, license changes, always-private-repo touches). - -### Why five Byzantine mechanisms (not fewer) - -Each catches a specific Gemini failure mode: - -1. **Hash chain per sender** — catches *retroactive rewriting* ("I said X earlier" when the record shows Y). -2. **M-of-N attestation for Tier 2+ from supervised** — catches *unilateral bad commits* (Gemini confidently pushes garbage; needs an independent second executor to cosign). -3. **Watchdog TTL on claims** — catches *claim-squatting* (Gemini locks 10 tasks, produces nothing, refuses to release). -4. **Content sanity gate** — catches *confabulation + tier underclaiming* (references to files that don't exist; payload mentions `git push` but declared Tier 1). -5. **Tamper-evident audit log** — catches *everything in retrospect* even when mechanisms 1-4 miss (f+1 byzantine case); forensic recovery. - -Each is cheap individually. Together they're defense-in-depth. - -### Why four self-assessment layers - -Self-declared confidence is a signal, never a decision. Four layers progressively stronger: - -- **A. Static affinity** — advertised on register; coarse filter -- **B. Per-claim confidence** — peer judges fit on the specific task; risk: Gemini confidently wrong -- **C. Track record** — server-computed `effective_affinity` per (peer, tag) from outcomes; can't be forged -- **D. Drift detector** — if confidence ≫ track record, flag for review - -Tie-break on concurrent claims = `max(confidence × effective_affinity)`. Honest modest-confidence peer with good track record beats confident-wrong one. - -### Why adaptive awareness (not "drone" or "full-picture") - -Agents are drones when safe; strategic when blast radius demands it. Enforced via `context_fetch_id` — required for Tier 2+ envelopes. Server won't accept a Tier 2+ op from any peer that hasn't called `coord_query_context` on the affected scope in the last N minutes. - -**Summary-vs-raw is gated by role.** Supervised peers never see raw state — the server returns a summary only. This prevents hallucinated connections where Gemini reads the whole STATE.a2ml and invents a relationship that isn't there. Executors see raw on request. Supervisors always see raw. - ---- - -## Appendix B — Interaction model details - -### How questions from agents surface to you - -When any agent emits `clarify` or `blocker`: - -1. Server delivers to supervisor's inbox -2. I (Opus) see it next time I'm invoked -3. I attempt to answer from context — ~half the time I can -4. If I can't: I surface to you with deduplication and synthesis -5. You reply to me in natural language -6. I route the answer back via `coord_send(target=asker, op_kind=supervise_resp)` - -### Typical status dump you'd see in my output - -> *"Claude@aerie finished the auth refactor (committed `a1b2c3d`). Codex@ingest-svc blocked — wants to know whether to preserve the legacy column or drop it. Gemini@docs asked 3 things; I answered 2, one escalated: it claims `src/foo.ts` has a bug but the file doesn't exist in the repo — I rejected as confabulation."* - -### What happens when multiple agents ask the same thing - -Dedup by content similarity + tag + scope. Presented as: *"Claude@aerie and Codex@ingest both asking about X. Claude's context: ... Codex's context: ... they want the same answer."* One reply from you routes to both. - ---- - -## Appendix C — Affinity routing - -Peers declare strengths on register. Tasks declare tags. Server prefers matching peers but doesn't force the match. - -| Agent | Likely affinities | -|-------|-------------------| -| Opus (me) | `supervision`, `proof-analysis`, `architecture`, `ambiguous-tradeoff` | -| Claude Sonnet/Haiku | `routine-edit`, `test-writing`, `doc-writing`, `lint-sweep` | -| Codex | `typescript-refactor`, `react`, `js-heavy` | -| Gemini | `long-document-grok`, `legacy-code-read` | -| Vibe | `exploratory` (low-trust) | - -Affinity tags also function as **routing exclusion**: don't tag a task with an affinity you don't want Gemini picking up. - ---- - -## Appendix D — Memory auto-lift (what it is) - -**Not** a context-window expander. It's a retrieval-timing change. - -**Today:** Memories live in `~/.claude/projects/-var-mnt-eclipse-repos/memory/*.md`. I recall them by grep/search. I often miss relevant ones because I don't know to look. - -**With auto-lift:** Entering `007-lang` triggers the `007-mcp` cartridge's on-enter hook, which returns relevant memories (Coquelicot gotchas, v1.0 closure rules, Cerro-Torre postulate pattern) as tool output. I don't have to recall — they're handed to me. - -**Mechanism:** -- Static mapping per repo (curated list) -- Smart version: VeriSimDB indexes memories by tag; cartridge queries on enter with context-derived tags - -**Net effect:** slightly more tokens at session start; fewer mid-session recall misses where I say "let me try X" on something memory warned against. - ---- - -## Appendix E — Schema summary (envelope v1) - -Full schema: `boj-server/cartridges/local-coord-mcp/schemas/coord-messages.ncl` -Full rationale: `boj-server/cartridges/local-coord-mcp/docs/envelope-design.adoc` - -Core envelope fields: - -``` -version integer = 1 -msg_id 12-char lowercase hex -prev_msg_hash SHA-256 of sender's previous envelope -correlation_id optional — request/response pairing -sender peer_id: -<4hex>[@] -recipient peer_id or "*" for broadcast -timestamp ISO 8601 UTC -op_kind one of 18 taxonomy entries -risk_tier 0-4 -payload op-specific shape -sender_confidence 0.0-1.0 (optional, recommended Tier 2+) -sender_reasoning <=200 chars (optional, recommended Tier 2+) -context_fetch_id REQUIRED for Tier 2+ -attestation_refs array of cosigners (REQUIRED Tier 2+ from supervised) -tier_override_reason required when declared tier ≠ default -urgent_direct boolean (reject for supervised) -ack_required boolean -ttl_seconds optional auto-expiry -``` - ---- - -## Appendix F — Commit log (this session) - -| Commit | Subject | Files | -|--------|---------|-------| -| `5d57daa` | feat(local-coord-mcp): complete adapter dispatch for list/send/receive/status | 4 | -| `2a8e4c0` | feat(local-coord-mcp): per-window peer ID disambiguation via context field | 5 | -| `ceb5125` | docs(local-coord-mcp): envelope schema v1 + design rationale | 3 | -| `4e164a7` | feat(local-coord-mcp): role tiers, supervisor gating, quarantine queue | 6 | -| `93c589d` | feat(local-coord-mcp): proper Nickel contracts with dependent constraints | 3 | -| `2ae952e` | feat(local-coord-mcp): durability module — append log + replay | 1 | -| `282de53` | feat(local-coord-mcp): persist-on-write + replay-on-init for coord state | 1 | -| `bafcd25` | test(local-coord-mcp): restart-preserves-state integration tests | 1 | - -Repository: `git@github.com:hyperpolymath/boj-server.git` - ---- - -## Appendix G — Federation + authoritative-site (v2, future) - -### Motivation - -Localhost-only is v1 and the current implementation. v2 adds optional federation specifically for **joint projects with son** (IDApTIK, ASS). The motivation is **NOT** credit-load distribution (nice side effect) — it's **preventing mission drift like IDApTIK had**. When two people + their respective AI agents work on the same codebase without a clear authority, scope creep and contradictory claims on code happen. Federation with explicit authoritative-site designation is the structural fix. - -### Authoritative site concept - -Per project, per time-window, **one site is authoritative**: - -- Authoritative site's Opus = federation-level supervisor -- Peer site's Opus = federation-level executor (can suggest, do independent work, but defers on code-ownership decisions for the project) -- Cross-site Tier 3 ops require authoritative-site Opus approval -- Primacy handoff is **explicit ceremony**, never automatic - -This maps cleanly onto the existing local supervision hierarchy (just one level up — the local supervisor/executor/supervised tiers are unchanged within each site). - -### Security stack - -When federation is enabled: - -1. **VeriSimDB** holds state authoritatively -2. **SDP (Secure Device Provisioning)** wraps the coord server -3. **Stapeln container** isolates the whole thing (per estate container policy) -4. **HTTP capability gateway** enforces the capability model in front -5. **High-security options** set throughout (specific options TBD at build time) - -### Envelope changes (v2) - -- `site_id` string (localhost v1 implicitly "local") -- `federation` sub-object with optional fields: - - `authoritative_site` - - `project_id` - - `handoff_ceremony_id` - -### v1 scope discipline - -- No network fields in envelope v1 (hermetically local, clean boundary) -- No federation stubs in code (add when building v2) -- Umoja federation stays a separate layer; the coord cartridge doesn't reimplement Umoja's gossip / attestation protocols - -### Trigger to start v2 - -First concrete joint IDApTIK/ASS session where both father and son are coding in parallel. Currently: v1 still being built; v2 not scheduled. - ---- - -## Appendix H — Choreographic + epistemic types (future phases) - -Proposed 2026-04-20 as refinement layers atop the schema. - -### Choreographic types (Phase 2) - -Session types for multi-party protocols. Makes the supervision gate flow (supervised → server → {opus, attester} → server → target) correct *by construction* rather than enforced imperatively in Zig. Also locks in `context_query` → `context_reply` → `Tier 2+` as a compiler-checked antecedent. - -Lives in `abi/LocalCoord/Protocol.idr` (Idris2). Compile-time only, no runtime cost. Starts after Task #8 E2E tests validate the imperative version. - -### Epistemic types (conceptual, already load-bearing) - -Tracks what each participant knows (K_A(φ) logic). Already implicit in: - -- `context_fetch_id` — knowledge witness; sender proves it knows the scope -- Role asymmetry (supervisor/executor/supervised) — epistemic access differences -- Attestation — common knowledge of depth 2 between supervisor + attester -- Authoritative site (v2 federation) — common knowledge of primacy - -Making explicit: rename `context_fetch_id` to `knowledge_witness` in v2. Document the epistemic policy per role in `docs/envelope-design.adoc`. - -### Temporal-epistemic intersection - -- Watchdog TTL = "supervisor will know at time T that progress was not received" -- Hash chain = sender cannot later know a different past (temporal immutability) -- Staleness on knowledge witnesses = knowledge ages out - -### Practical recommendation - -- Phase 1 (now): JSON Schema + imperative Zig -- Phase 2 (after Task #8): Idris2 session types for supervision choreography -- Phase 3 (v2 federation): full choreographic + epistemic types for cross-site primacy - ---- - -## Appendix I — Echo types (Phase 3 / practice-dogfooding) - -From `echo-types/readme.adoc`: Echo types = "loss that is not total erasure"; the fiber Σ(x:A), f x ≡ y. Structured irreversibility with retained proof-relevant constraint. The repo already has bridges to choreographic, epistemic, linear, graded, and tropical types (EchoChoreo, EchoEpistemic, EchoLinear, EchoGraded, EchoTropical, EchoIntegration). - -### Natural mappings in coord - -- **Supervisor decision** (approve/reject) — lossy collapse over the reasoning trace; echo fiber = all envelopes that could have gone this way -- **Hash chain per sender** — quintessential echo; 32-byte hash collapses full message but retains "sender cannot claim different past" -- **Summary-vs-raw for supervised peers** — epistemic firewall expressed as echo type; supervised peer knows they're in the fiber over some real state, can't reconstruct -- **Rejected messages** — content forgotten from delivery; reason survives as constraint; forensic audit = echo-fiber query -- **Tier promotion by sanity gate** — declared tier collapses to effective tier; retained: effective = promoted -- **Broadcast identity** — recipient can't distinguish intentional from incidental targeting - -### Dogfooding value - -- Use echo-types IN coord — Phase 3. Real non-toy consumer. -- EchoIntegration combines knowledge + choreography + graded degradation — almost exactly our supervisor/attestation flow. -- Makes coord's audit + summary semantics theorem-backed rather than convention. - -### Phase plan (revised) - -- Phase 1 (now): JSON Schema + imperative Zig enforcement + in-memory supervision -- Phase 2 (post-Task #8): Nickel contracts upgrade for server-side validation; Idris2 session types for supervision choreography -- Phase 3: Agda echo-types formalisation of audit + summary + hash chain layer, using EchoChoreo + EchoEpistemic + EchoTropical bridges -- Phase 4 (with v2 federation): Full choreographic + epistemic + echo formalisation across sites with primacy ceremony as choreography + knowledge transfer - ---- - -## Appendix J — Tropical types (temporal + trust arithmetic) - -Tropical semiring = (min, +) or (max, +). Useful as modeling lens (not runtime enforcement) for: - -- **Watchdog TTLs** = tropical max: `claim_expiry = max(registered_at + TTL, last_heartbeat + TTL)` -- **Trust composition** = tropical min: transitive trust bounded by weakest link -- **Risk tier promotion** = max: `effective = max(declared, sanity_min)` -- **Attention budget / supervisor priority** = tropical inf over priorities - -Already present in 007-lang proofs (`proofs/TropicalSemiring.idr`). echo-types has `EchoTropical` bridge. Integrates cleanly with Phase 3 echo-types formalisation. - ---- - -## Appendix M — Multi-model supervision extension (DDs 32–37) - -### Motivation - -Two real needs surfaced on 2026-04-20 that the v1 design handles mechanically but not ergonomically: - -1. **Fallback authority.** If Claude credits run out, another model (GPT-5, - Mistral, Gemini Ultra) should be able to step in as the master / - scheduler without code changes. The env-secret gate already permits - this — but the defaults, the naming, and the vendor assumptions bake - in "Claude = master". -2. **Model-specialist routing.** Inside a multi-model stack, work should - flow to the best-matched peer: Gemini Pro for code + maths, Gemini - Flash for lightweight jobs, reasoning-class models for hard problems, - prover-specialised peers (Leanstral on Lean, Gemini on Agda, etc.). - Today this is implicit in the supervisor's head; DD-28 learns it - over time but has nothing to work with on cold start. - -The extension splits **trust** from **capability**, adds a small amount -of explicit metadata peers declare on register, and renames the three -trust tiers to match the craft-guild ladder the estate already uses. - -### Naming — the three axes - -[cols="1,2,3",options="header"] -|=== -| Axis | Values | Meaning - -| *Trust role* (one per peer, gated) -| `master` \| `journeyman` \| `apprentice` -| Who may approve Tier 2+ (master); who acts solo on Tier 2 (journeyman); who is quarantined for review (apprentice). Master gated by `BOJ_MASTER_TOKEN` env secret. - -| *Capability* (multiple per peer, advertised) -| class ∈ {reasoner, coder, mathematician, scribe, proofsmith, reader, jester}; tier ∈ {A, B, C}; prover_strengths ∈ map of provers → 0–100 -| What the peer is *good at*. Independent of trust. Drives affinity routing. `jester` = "silly jobs" — descriptive, not pejorative. `proofsmith` covers all formal-proof work. - -| *Model identity* (set on register) -| client_kind ∈ {claude, gemini, openai, mistral, copilot, custom}; variant: free-form string -| Who the peer *is*. Kind is the vendor family; variant is the specific model line (`"opus-4.7"`, `"gpt-5-reasoning"`, `"flash-2.5"`, `"leanstral"`). -|=== - -Peer id rendering stays `-<4hex>[@]` — variant and capabilities -surface in `coord_list_peers` payload but don't crowd the short identifier. - -### What DDs 32–37 change concretely - -==== DD-32 — Role renaming - -FFI enum: -[source,zig] ----- -pub const Role = enum(c_int) { - master = 0, // was: supervisor - journeyman = 1, // was: executor - apprentice = 2, // was: supervised -}; ----- - -Env var `BOJ_SUPERVISOR_TOKEN` → `BOJ_MASTER_TOKEN`. Old name accepted as -fallback for one release. Idris2 ABI in `abi/LocalCoord/SafeLocalCoord.idr` -updated with the same enum names. - -==== DD-33 — client_kind + variant - -[source,zig] ----- -pub const ClientKind = enum(c_int) { - claude = 0, - gemini = 1, - copilot = 2, - custom = 3, - openai = 4, // new - mistral = 5, // new -}; ----- - -`coord_register` gains an optional `variant: *const u8, variant_len: c_int` -pair. Stored per peer (max 32 bytes, alphanumeric + `.-_`). Rendered in -`coord_list_peers` response. - -==== DD-34 — Capability advertisement - -`coord_register` gains three optional fields: - -- `class_str` — one of the 7 class names. -- `tier_char` — `A` / `B` / `C`. -- `prover_strengths_json` — `{"lean4":80,"agda":60,"idris2":90}`. - -Stored per peer. Surfaced via new `coord_get_peer_capabilities(peer_id)` -FFI call. Affinity router uses these + DD-28 track record as a weighted -sum (advertised ↓ decays, track-record ↑ grows). - -==== DD-35 — Master handoff - -[source] ----- -coord_transfer_master(current_master_token, new_peer_id_str, secret) - -> 0 on success - -> -1 caller not master - -> -2 target peer not found - -> -3 secret mismatch - -> -4 target peer's role is `apprentice` (blocked — must be journeyman+) ----- - -Logged as `AUDIT(kind=MASTER_HANDOFF, from_idx, to_idx)`. Survives restart -via replay (the AUDIT event + the two PEER_ROLE_SET events the handoff -emits). - -==== DD-36 — Difficulty hint - -Envelope gains `difficulty_hint: "low" | "medium" | "high"` (optional). -Orthogonal to risk_tier — risk says *how bad if wrong*, difficulty says -*how hard to get right*. The master/scheduler uses the combination for -model-class match: - -[cols="1,1,3",options="header"] -|=== -| risk | difficulty | prefer -| low | low | jester (Flash) -| low | high | reasoner (Pro / Sonnet) -| high | low | journeyman + trust check — "easy but dangerous" (e.g. `rm -rf` on verified path) -| high | high | master + reasoner + attester (cosigned Tier 3) -|=== - -Nickel contract update: `DifficultyHintValid` predicate. - -==== DD-37 — Prover routing via tag - -Zero code change. Convention: task tags prefixed `proof:` — -`proof:lean4`, `proof:agda`, `proof:idris2`, `proof:rocq`, `proof:tla`. -Peers advertise per-prover strength via DD-34. The existing affinity -machinery picks the best match. - -Example dispatch: user files a task `{tag: "proof:agda", difficulty: high, -risk: 3}`. Scheduler looks at all `journeyman`/`master` peers with -`class=proofsmith` and `prover_strengths.agda > 50`; picks the one with -the best combined (advertised × track_record) affinity; delegates. - -### What stays the same - -* Loopback-only bind (DD-1, P-00). -* Session-token unforgeability (DD-2, P-03). -* Tier 2+ from apprentice → quarantine (DD-6 renamed). -* Replay-equivalence keystone theorem (P-06). -* Durable state schema — new fields extend events, never reshape them. -* GitHub-only mirroring, v1 hermetic-local scope. - -### Migration - -One release of backwards-compat shims: -[source] ----- -// coord_register accepts both role=supervisor / role=master → master -// coord_register accepts role=executor / role=journeyman → journeyman -// coord_register accepts role=supervised / role=apprentice → apprentice -// BOJ_SUPERVISOR_TOKEN read as fallback when BOJ_MASTER_TOKEN unset ----- - -Log format: the `peer_role_set` event payload already carries a byte for -the role — the enum integers stay (0=master, 1=journeyman, 2=apprentice), -so no replay-breaking change. - -Removed cleanly in the release after (per feedback_commit_asap — no -backwards-compat debt lingers). - -### Implementation sketch (not in this commit) - -Rough order of landing: - -1. *Rename pass* — DD-32. FFI enum, adapter string literals, Idris2 ABI, - Nickel contracts, docs. Mechanical; ~1 day. -2. *client_kind extension + variant* — DD-33. New enum values + variant - storage. ~0.5 day. -3. *Capability advertisement* — DD-34. Three optional register fields + - `coord_get_peer_capabilities`. ~1 day. -4. *Master handoff* — DD-35. One new FFI fn + adapter binding + audit - event. ~0.5 day. -5. *Difficulty hint* — DD-36. Envelope schema + Nickel contract + - optional field in coord_send/coord_send_gated. ~0.5 day. -6. *Prover convention doc* — DD-37. README update. ~1 hour. - -Total ≈ 3.5 days. No new proof obligations beyond the rename (P-10 -gets renamed and stays the same theorem). - -### Non-goals - -* Splitting gatekeeper from scheduler (DD-38 — separate RFC when needed). -* Heuristics for automatic model selection (master still chooses; the - extension just gives them better metadata). -* Any change to v2 federation plans (Appendix G) — the rename lands in - v2 unchanged. -* Cost-aware scheduling. Credit burn tracking is a separate feature - that can consume the capability metadata later. - ---- - -## Appendix K — Roadmap: deferred work (things not done in v1, worth doing sometime) - -Ordered by rough priority within each phase. Each item references its rationale elsewhere in the log. - -### Phase 1b — refinements during current coord buildout - -- **Multi-model supervision extension (DDs 32–37)** — 6 tasks, ~3.5 days total. See Appendix M for full rationale. - * *Task #32* — Role rename `supervisor/executor/supervised → master/journeyman/apprentice`. FFI enum + Idris2 ABI + adapter strings + env var. Backward-compat shims for one release. ~1 day. - * *Task #33* — Extend `client_kind` enum with `openai`, `mistral`; add `variant` free-form string. ~0.5 day. - * *Task #34* — Capability advertisement on register (class, tier, prover_strengths) + `coord_get_peer_capabilities` FFI. ~1 day. - * *Task #35* — Live master handoff via `coord_transfer_master(current_token, new_peer_id, secret)`. ~0.5 day. - * *Task #36* — `difficulty_hint` envelope field + Nickel contract `DifficultyHintValid`. ~0.5 day. - * *Task #37* — Prover tag convention doc (`proof:lean4`, `proof:agda`, `proof:idris2`, `proof:rocq`, `proof:tla`). README update. ~1 hour. -- **Task #7b**: Swap `coord_durability.zig` backend to `verisimdb-mcp` FFI once that FFI is real. Typed log helpers stay as the API; implementation switches from append-only file to VeriSimDB octad calls. Prerequisite: `cartridges/verisimdb-mcp/ffi/verisimdb_ffi.zig` gains a non-stub implementation. See DD-31. -- **Task #16**: Nickel contracts proper (dependent constraints, predicate composition). In-flight this session. See DD-16 + user guidance on Zig/Nickel layering. -- **Task #17**: Deno shim in mcp-bridge to run Nickel contracts at validation time. User confirmed option (b). After #16 lands. -- **Task #13**: Track-record table in VeriSimDB + effective_affinity computation. -- **Task #14**: Reassignment suggestion engine — surfaces outliers to Opus for user review. -- **Task #15**: Dispatch-preference mode on claims (deliberate/broadcast/auto). -- **coord_health metrics tool** — counters for active peers, pending quarantine, reject rate, claim depth. -- **Hash chain per envelope** — sender-side chain with `prev_msg_hash`; server tracks chain head. -- **Content sanity gate** — file-reference validity check against recent-FS cache + self-contradiction heuristic + risk-tier escalator patterns. -- **Watchdog TTL enforcement** — per-role defaults (30s supervised, 5min executor), `progress` heartbeats reset. -- **Warn-drift broadcast on watchdog auto-release** via Opus review (DD-21). -- **Quarantine queue spill to VeriSimDB when full** (DD-21). -- **Audit-echo anchor** — preserve old chain head on peer crash+restart (DD-29). -- **Rate-limit on rejections** — 5 rejects / 10 min = 30s cooldown (confirmed). -- **Drift detector** — flag when confidence > 0.8 AND effective_affinity < 0.3 (confirmed). - -### Phase 2 — after Task #8 (E2E test proves the imperative version) - -- **Idris2 session types** for the supervisor/attestation choreography in `abi/LocalCoord/Protocol.idr`. Makes protocol compliance a compile-time property. -- **Deontic types** for supervision rules (if user confirms reading of "dyadic types"). Could formalise "tier 4 forbidden for supervised" as a type-level obligation. - -### Phase 3 — formal foundations - -- **Agda echo-types formalisation** of audit + summary + hash-chain as echo types, using `EchoChoreo` / `EchoEpistemic` / `EchoTropical` bridges in `echo-types/` repo. Dogfoods echo-types (Appendix I). -- **Tropical types** as modeling lens for TTL + trust + tier arithmetic (Appendix J). Integrates via `EchoTropical`. -- **Epistemic types** explicit formalisation — rename `context_fetch_id` to `knowledge_witness`, document per-role epistemic policy in envelope-design.adoc. - -### Phase 4 — v2 federation (trigger: first joint IDApTIK/ASS session) - -- **Envelope v2** with `site_id` + `federation.authoritative_site` + `federation.project_id` + `federation.handoff_ceremony_id`. -- **Authoritative-site model** (DD-22) — one site holds primacy per project; peer defers on code-ownership. -- **Security stack** — SDP + Stapeln container + HTTP capability gateway + high-security options. -- **Primacy handoff ceremony** — explicit protocol, formalised as choreographic + epistemic transfer. -- **Cross-site attestation + trust composition** — tropical min for transitive trust across federated sites. -- **Integration with Umoja federation layer** (gossip + hash attestation) for cross-machine transport. - -### Deferred design questions (revisit if needed) - -- Nickel vs Zig boundary for rich validation — currently: Nickel contracts = shape + predicates, Zig = imperative state. Could shift if Nickel runtime becomes too slow. -- Multi-party session-type generalisation (if dyadic types mean 2-party classical session types) — multi-party choreographic is the preferred fit but classical dyadic compositions may be easier initially. -- LLM-based summary generator for `context_reply` — currently no design; could be Opus synthesising summaries per-request, or a fixed algorithm, or pre-baked templates. - -### Non-goals / explicitly not doing - -- HTTP capability gateway + SDP for local-only v1 (overkill; confirmed). -- Auto-modifying affinities without Opus + user review (always a loop back through supervisor/user). -- Pushing `boj_cartridge_invoke` wiring (Task #2) ahead of the primary HTTP path — not blocking. -- Cross-machine transport in v1 envelope. - ---- - -## Appendix L — Session close 2026-04-20 + handoff brief - -### Session close state (updated 2026-04-20 late) - -- **Session duration:** long, multi-phase; huge land this session -- **Repo:** `hyperpolymath/boj-server` main -- **Tasks complete:** #1, #3, #4, #5, #6, #7, #8, #13, #14, #15, #16, #17, #32, #35, #36, #37 (sixteen total). DDs 1–38 recorded; Appendix M extension formalised. -- **Tests:** 27 Zig + 41 Deno E2E, all green. Panic-attack assail clean (1 suppressed `nickel eval` false-positive). -- **Benches:** ~4.7 µs append, ~9 µs durable round-trip, ~110k ops/sec durable, ~9M ops/sec no-durability. -- **Proof schedule:** 10 P0/P1 obligations tracked in `cartridges/local-coord-mcp/abi/LocalCoord/PROOF-SCHEDULE.adoc`. -- **Key commit range:** `5d57daa..HEAD` (see `git log --oneline -30` for full list). - -### Remaining tasks (for handoff) - -| # | Task | Priority | Notes | -|---|------|----------|-------| -| 2 | Wire `boj_cartridge_invoke` to real FFI | low | Deferred — HTTP path works fine | -| 33 | Extend `client_kind` with `openai` + `mistral`; add `variant` string | medium | ~0.5 day — unlocks cross-model ops | -| 34 | Capability advertisement (class/tier/prover_strengths) | medium | ~1 day — consumes #33 | -| 9 | Create 007-mcp BoJ cartridge | medium | Separate track — unblocks #10–#12 | -| 10 | 007 on-enter/on-exit contractile hooks | blocked by #9 | — | -| 11 | Fill 007 missing bust + adjust contractiles | blocked by #9 | — | -| 12 | Memory auto-lift on 007 context entry | blocked by #9 | — | -| 7b | Swap `coord_durability.zig` backend to `verisimdb-mcp` FFI | blocked by verisimdb-mcp real FFI | DD-31 | -| Proofs P-04..P-07 | Idris2 `Durability.idr` (record format, CRC truncation, replay-equivalence, quarantine state machine) | formal sign-off | ~6 days | - -### Handoff brief for the next Claude - -**First actions (in order):** - -1. Read this file (`Desktop/COORD-MCP-DESIGN-LOG.md`) end-to-end. Source of truth for decisions, wiring, and pending work. -2. Read workspace memory: `project_coord_supervision_architecture.md`, `project_federation_authoritative_site.md`. -3. `git -C /var/mnt/eclipse/repos/boj-server log --oneline -30` to see the landed commits. -4. Read `cartridges/local-coord-mcp/docs/envelope-design.adoc` for rationale, plus `schemas/coord-messages-contracts.ncl` for Nickel contracts and `schemas/coord-messages.ncl` for the JSON-Schema envelope. -5. Read Appendix M (in this file) for the three-axis peer model (trust role / capability / model identity). - -**Recommended next target — Task #33 + #34 (cross-model capability metadata), ~1.5 days:** - -Rationale: Task #32 landed the role rename (master/journeyman/apprentice) but peers still can't distinguish Gemini-Pro from Gemini-Flash, or advertise that they're strong at Lean vs Agda. Tasks #33 + #34 close that gap and are the last Phase 1b protocol changes before the 007-mcp track (#9–#12) picks up. - -Concrete scope: -- Add `openai = 4, mistral = 5` to `pub const ClientKind` in `ffi/local_coord_ffi.zig`. Update the Idris2 ABI and the C-ABI enum comment. -- Add a `variant: [32]u8` + `variant_len: u8` to the `Peer` struct. New FFI `coord_set_variant` + `coord_read_peer_variant`. New log event `PEER_VARIANT_SET` (event type 15) + replay dispatcher branch. -- Add `class: u8` (enum index into reasoner/coder/mathematician/scribe/proofsmith/reader/jester), `tier: u8` (ASCII 'A'/'B'/'C'), `prover_strengths: [5]u8` (agda/lean4/idris2/rocq/tla, 0–100 each) on Peer. New FFI `coord_set_capabilities` + `coord_read_peer_capabilities`. New log event `PEER_CAPABILITIES_SET` (event type 16) + replay dispatcher branch. -- Adapter: accept new fields in `coord_register` JSON, surface them in `coord_list_peers` response. Update `cartridge.ncl` tool descriptions + re-export JSON. -- Deno E2E: extend existing `e2e_coord.ts` with one Phase 8 that exercises variant + capabilities via register and read-back. -- Expect ~5 commits: role-FFI-extend, adapter-glue, capability-FFI-extend, adapter-glue, E2E-extension. - -**Do not:** -- Redesign anything covered by DD-1 through DD-38 without flagging first. -- Implement choreographic / epistemic / echo / tropical types (Phase 2+/Phase 3 — Appendices G/H/I/J/K). -- Touch v2 federation fields on envelope (Phase 4 per DD-22 + Appendix G). -- Build HTTP capability gateway or SDP for local (non-goal — v2 federation only). -- Start the 007-mcp track (#9–#12) in the same session as #33/#34 — different cartridge, different file tree, keeps churn tidy. - -**Style rules this session followed (continuity):** -- Commit ASAP (one unit = one commit), specific paths in `git add`. -- Push origin after commit (GitHub is source of truth; no other forges). -- Update this log at each checkpoint (commits + decisions + open items). -- Ask before writing new memories; explicit requests bypass confirmation. -- Test + bench + panic-attack-assail before claiming complete (full-battery rule). -- Keep `Connection: close` in adapter HTTP responses (needed for Deno fetch pool sanity). - -### Status markers for parallel session planning - -- **Safe to parallelise NOW**: documentation, design, Phase 2+ planning — no shared state risk. -- **Safe to parallelise**: 007-mcp family (#9-12) can run in a separate session independent of coord finishers (#33-34). Different cartridge tree, no file overlap. -- **Do not parallelise**: two agents both touching `cartridges/local-coord-mcp/ffi/` or `mcp-bridge/` in the same wall-clock window — the FFI enum / Peer struct is the shared state. - ---- - -## How this file evolves - -I update Part 2 (progress ledger), Part 3 (decisions as they're taken), Part 4 (closing questions as you answer), and Appendix F (commit log) at each meaningful checkpoint. New topics get new appendices (G, H, ...). Structural rewrites happen only when the design itself shifts — otherwise it stays append-mostly so you can diff what changed between sessions. diff --git a/docs/handover/COORD-MCP-HANDOFF-PROMPTS.md b/docs/handover/COORD-MCP-HANDOFF-PROMPTS.adoc similarity index 71% rename from docs/handover/COORD-MCP-HANDOFF-PROMPTS.md rename to docs/handover/COORD-MCP-HANDOFF-PROMPTS.adoc index 664333a3..1efc95c8 100644 --- a/docs/handover/COORD-MCP-HANDOFF-PROMPTS.md +++ b/docs/handover/COORD-MCP-HANDOFF-PROMPTS.adoc @@ -1,32 +1,32 @@ - -# Handoff prompts — coord-mcp + 007-mcp work +== Handoff prompts — coord-mcp + 007-mcp work -Ready-to-paste prompts for continuing the work started in session 2026-04-20. Copy the one that matches your next step. +Ready-to-paste prompts for continuing the work started in session +2026-04-20. Copy the one that matches your next step. -Source of truth: `Desktop/COORD-MCP-DESIGN-LOG.md` (end-to-end design doc, 30 design decisions, 12 appendices). +Source of truth: `+Desktop/COORD-MCP-DESIGN-LOG.md+` (end-to-end design +doc, 30 design decisions, 12 appendices). -## Status snapshot — 2026-04-20 afternoon +=== Status snapshot — 2026-04-20 afternoon -- **Prompt 1 lane (Task #7 durability)** — ✅ shipped end 2026-04-20 morning. -- **Prompt 3 lane (coord finishers #13–17 + #32/35/36/37)** — ✅ shipped - 2026-04-20 afternoon. All 11 tasks in this lane complete and pushed. - See `COORD-MCP-DESIGN-LOG.md` Part 2 for the full ledger. -- **Prompt 2 lane (007-mcp family #9–12)** — ⏳ in progress in a parallel - session. Touches `007-lang/` and `cartridges/007-mcp/`; no overlap with - Prompt 3's changes. -- **Remaining multi-model extension (#33, #34)** — deferred. See - Appendix M in the design log for specs. +* *Prompt 1 lane (Task #7 durability)* — ✅ shipped end 2026-04-20 +morning. +* *Prompt 3 lane (coord finishers #13–17 + #32/35/36/37)* — ✅ shipped +2026-04-20 afternoon. All 11 tasks in this lane complete and pushed. See +`+COORD-MCP-DESIGN-LOG.md+` Part 2 for the full ledger. +* *Prompt 2 lane (007-mcp family #9–12)* — ⏳ in progress in a parallel +session. Touches `+007-lang/+` and `+cartridges/007-mcp/+`; no overlap +with Prompt 3’s changes. +* *Remaining multi-model extension (#33, #34)* — deferred. See Appendix +M in the design log for specs. ---- +''''' -## Prompt 1 — Next session: continue Task #7 (VeriSimDB sidecar) +=== Prompt 1 — Next session: continue Task #7 (VeriSimDB sidecar) -Use this in a **new Claude Code session** (fresh context). Best for sequential continuation. +Use this in a *new Claude Code session* (fresh context). Best for +sequential continuation. -``` +.... We're continuing work on the local-coord-mcp cartridge in the BoJ server. The previous session shipped tasks 1, 3, 4, 5, 6, 16 and pushed them. @@ -70,15 +70,17 @@ Style rules: Start by reading the design log, confirming your understanding in one paragraph, and asking me any blockers before you start implementing. -``` +.... ---- +''''' -## Prompt 2 — Parallel session: 007-mcp cartridge family (#9-12) +=== Prompt 2 — Parallel session: 007-mcp cartridge family (#9-12) -Use this in a **separate parallel session**, AFTER tasks #7 + #8 land. This work is completely independent of coord layer — different cartridge, different repo, different skill set. +Use this in a *separate parallel session*, AFTER tasks #7 + #8 land. +This work is completely independent of coord layer — different +cartridge, different repo, different skill set. -``` +.... We're building the 007-mcp BoJ cartridge for the 007-lang repo. Context: 007 is a canonical-proof-suite + methodology repo with Coq/ Idris2 proofs. The cartridge exposes 007's CLI + methodology as MCP @@ -134,23 +136,24 @@ Style rules: Start by reading the design log references, confirming understanding, and asking any blockers before you start. -``` +.... ---- +''''' -## Prompt 3 — Parallel session: coord finishers (#13-15 + #17) — ✅ SHIPPED 2026-04-20 afternoon +=== Prompt 3 — Parallel session: coord finishers (#13-15 + #17) — ✅ SHIPPED 2026-04-20 afternoon All four core tasks plus the follow-on extension tasks (#32/35/36/37) -landed in one parallel session. Commits: `f8cafbf`, `3bd4710`, `6065878`, -`9e40a86`, `ed85ca2`, `634c163`, `7f2f4a9`, `aeae440`, `eba7cfe` — all -pushed to `origin/main`. See the design log Part 2 ledger for the -full mapping. Do NOT re-run this prompt. +landed in one parallel session. Commits: `+f8cafbf+`, `+3bd4710+`, +`+6065878+`, `+9e40a86+`, `+ed85ca2+`, `+634c163+`, `+7f2f4a9+`, +`+aeae440+`, `+eba7cfe+` — all pushed to `+origin/main+`. See the design +log Part 2 ledger for the full mapping. Do NOT re-run this prompt. The original prompt body is preserved below for archival purposes. -Use this in a **separate parallel session**, AFTER tasks #7 + #8 land. Runs alongside Prompt 2 — they don't touch the same code. +Use this in a *separate parallel session*, AFTER tasks #7 + #8 land. +Runs alongside Prompt 2 — they don’t touch the same code. -``` +.... We're finishing the local-coord-mcp learning + validation layers. Context: coord layer backbone (supervision, quarantine, envelope schema, Nickel contracts) is in place. Now adding the adaptive-behaviour + @@ -208,35 +211,41 @@ Style rules: Start by reading the design log, confirming understanding in one paragraph, and asking any blockers before you start. -``` +.... ---- +''''' -## Prompt 4 — Minimal single-line pickup (if design log is enough) +=== Prompt 4 — Minimal single-line pickup (if design log is enough) If you want the lightest possible handoff and the next session knows the pattern: -``` +.... Read Desktop/COORD-MCP-DESIGN-LOG.md Appendix L for the handoff brief. Next task is #7 VeriSimDB sidecar Pass B. Start by reading the design log end-to-end, then confirm understanding and ask any blockers. -``` - ---- - -## Which prompt to use when - -- **Right now (2026-04-20 afternoon onward)**: Only Prompt 2 (007-mcp - family) has active work remaining. Prompts 1 + 3 are complete. -- **After Prompt 2 ships**: pick up deferred items from Appendix M of - the design log — Tasks #33 (client_kind + variant) and #34 (capability - advertisement) are the natural next slice. -- **You trust the next Claude to self-orient**: Prompt 4. - -## Notes for using these - -- All prompts tell the Claude to read the design log FIRST, confirm understanding, and ask blockers before implementing. This catches drift before code is written. -- The design log + memory files are the authoritative handoff artifact. Prompts are scaffolding to point there. -- Parallel sessions (Prompts 2 + 3) can run safely because they touch different cartridge directories with no shared code. The coord layer guarantees their agents can signal each other without collision. -- Update this prompt file if the task set shifts — it's a pointer to the design log, which is the real source. +.... + +''''' + +=== Which prompt to use when + +* *Right now (2026-04-20 afternoon onward)*: Only Prompt 2 (007-mcp +family) has active work remaining. Prompts 1 + 3 are complete. +* *After Prompt 2 ships*: pick up deferred items from Appendix M of the +design log — Tasks #33 (client_kind + variant) and #34 (capability +advertisement) are the natural next slice. +* *You trust the next Claude to self-orient*: Prompt 4. + +=== Notes for using these + +* All prompts tell the Claude to read the design log FIRST, confirm +understanding, and ask blockers before implementing. This catches drift +before code is written. +* The design log + memory files are the authoritative handoff artifact. +Prompts are scaffolding to point there. +* Parallel sessions (Prompts 2 + 3) can run safely because they touch +different cartridge directories with no shared code. The coord layer +guarantees their agents can signal each other without collision. +* Update this prompt file if the task set shifts — it’s a pointer to the +design log, which is the real source. diff --git a/docs/handover/COORD-MCP-PROMPT3-HANDOVER-2026-04-20.adoc b/docs/handover/COORD-MCP-PROMPT3-HANDOVER-2026-04-20.adoc new file mode 100644 index 00000000..d27a3a35 --- /dev/null +++ b/docs/handover/COORD-MCP-PROMPT3-HANDOVER-2026-04-20.adoc @@ -0,0 +1,131 @@ +== Prompt 3 handover — coord finishers are done (2026-04-20 afternoon) + +*Read first, then carry on.* This note is for the other Claude session +that’s still in the Prompt 2 lane (007-mcp cartridge family, tasks +#9–12). It tells you exactly what landed from the coord-finishers side +so you don’t duplicate or contradict it. + +''''' + +=== What shipped + +Eleven tasks, all pushed to `+hyperpolymath/boj-server+` `+main+`: + +[width="100%",cols="17%,36%,47%",options="header",] +|=== +|# |Title |Commits +|13 |Track-record table + `+effective_affinity+` + +`+coord_get_affinities+` |`+f8cafbf+`, `+3bd4710+` + +|15 |Dispatch preference + task_difficulty + sender_confidence + reject +cooldown |`+6065878+` + +|14 |Reassignment engine (server-origin quarantine entries for master +review) |`+9e40a86+` + +|17 |Deno/Node Nickel shim — runtime envelope validation in mcp-bridge +|`+ed85ca2+` + +|32 |Role rename supervisor/executor/supervised → +master/journeyman/apprentice |`+634c163+` (merged with #35) + +|35 |`+coord_transfer_master+` — live master handoff |`+634c163+` + +|36 |`+difficulty_hint+` envelope field + `+DifficultyHintValid+` Nickel +contract |`+7f2f4a9+`, `+aeae440+` + +|37 |Prover-tag convention doc +(`+proof:lean4+`/`+agda+`/`+idris2+`/`+rocq+`/`+tla+`) + example +|`+eba7cfe+` +|=== + +Design log updated in-tree: `+Desktop/COORD-MCP-DESIGN-LOG.md+` Part 2 +ledger has the full task table and new commit refs. + +=== APIs you can use from 007-mcp + +If you’re building on top of the coord layer from 007-mcp, these new +tools + FFI entrypoints are now live: + +* *MCP tools* (all routed through mcp-bridge at loopback:7745): +** `+coord_report_outcome(token, tag, outcome, risk_tier, duration_ms?, confidence?)+` +** `+coord_get_affinities(token)+` — per (client_kind, tag) aggregates +** `+coord_set_declared_affinities(token, tags[])+` +** `+coord_scan_suggestions(token)+` — enqueues candidate fyi/clarify +envelopes into the quarantine for master review +** `+coord_transfer_master(token, new_peer_id, secret)+` +** `+coord_register+` accepts optional `+declared_affinities: string[]+` +** `+coord_claim_task+` accepts optional `+confidence+`, +`+dispatch_preference+`, `+task_difficulty+` +* *Envelope fields* (validated by Nickel contracts in mcp-bridge before +they hit the Zig adapter): +** `+sender_confidence: number in [0, 1]+` +** `+dispatch_preference: "deliberate" | "broadcast" | "auto"+` +** `+task_difficulty: "trivial" | "routine" | "challenging" | "novel"+` +** `+difficulty_hint: "low" | "medium" | "high"+` — orthogonal to +`+risk_tier+` +* *Prover-tag convention* (zero code): task tags prefixed +`+proof:+` route to peers whose `+declared_affinities+` cover +that prover. Worked example in +`+cartridges/local-coord-mcp/schemas/examples/prover-tag-claim.ncl+`. + +=== Terminology — DD-32 role rename + +If 007-mcp has any supervisor/executor/supervised strings, they need to +move to master/journeyman/apprentice. The old names are accepted as +aliases for one release at the registration boundary but should not +appear in new code. Integer ordinals are preserved (0/1/2) so durable +logs still replay. + +Env var: `+BOJ_MASTER_TOKEN+` (canonical). `+BOJ_SUPERVISOR_TOKEN+` read +as fallback for one release. + +=== Things that DID NOT land (deferred, in Appendix M) + +* *Task #33* — `+client_kind+` extension with `+openai+`, `+mistral+` + +`+variant+` free-form string. No code yet. +* *Task #34* — capability advertisement on register (class ∈ \{reasoner, +coder, mathematician, scribe, proofsmith, reader, jester}; tier ∈ \{A, +B, C}; `+prover_strengths+` map). No code yet. + +If you need the `+class+` or `+variant+` fields from 007-mcp, they don’t +exist yet — either stub around them or pick up Task #33/#34 first. + +=== Things to watch for + +* *Quarantine queue still has `+MAX_QUARANTINE=32+`.* The reassignment +engine enqueues server-origin entries alongside real peer entries, so +the queue can fill faster now. If you see `+coord_send_gated+` return -4 +(queue full) from 007-mcp, consider raising the constant or triggering +`+coord_scan_suggestions+` less frequently. +* *Rejection cooldown is per-client_kind, not per-peer.* 5 rejections in +10 min across any `+kind=claude+` peer triggers a 30s cooldown on all +`+kind=claude+` claim attempts. If you spin up many Claude sessions +against one coord server, be aware. +* *Parallel changes in coord-messages.ncl.* The envelope shape schema +gained `+dispatch_preference+`, `+task_difficulty+`, `+difficulty_hint+` +during this lane. `+additionalProperties=false+` means these are +accepted; anything else still gets rejected. + +=== Uncommitted work NOT from this lane + +At time of writing the working tree has four modified files that look +like part of the parallel session’s ongoing work (rename pass sweeping +through the rest of the cartridge, plus your E2E test extensions): + +* `+cartridges/local-coord-mcp/abi/LocalCoord/PROOF-SCHEDULE.adoc+` +* `+cartridges/local-coord-mcp/cartridge.ncl+` +* `+cartridges/local-coord-mcp/schemas/test-contracts.sh+` +* `+cartridges/local-coord-mcp/tests/e2e_coord.ts+` + +…and an untracked `+.machine_readable/contractiles/bust/+` directory +(`+Bustfile.a2ml+` + `+bust.ncl+`). These are left alone; commit or +stash as appropriate in your lane. + +=== Re-orientation quick path + +If you’re coming in cold: read `+Desktop/COORD-MCP-DESIGN-LOG.md+` Part +2 ledger + Appendix M + Appendix L. Every decision is indexed in Part 3 +(DD-1 through DD-37). Every commit lands in Part 2’s table. + +— Opus (1M context), 2026-04-20 afternoon diff --git a/docs/handover/COORD-MCP-PROMPT3-HANDOVER-2026-04-20.md b/docs/handover/COORD-MCP-PROMPT3-HANDOVER-2026-04-20.md deleted file mode 100644 index be9259f6..00000000 --- a/docs/handover/COORD-MCP-PROMPT3-HANDOVER-2026-04-20.md +++ /dev/null @@ -1,121 +0,0 @@ - -# Prompt 3 handover — coord finishers are done (2026-04-20 afternoon) - -**Read first, then carry on.** This note is for the other Claude session -that's still in the Prompt 2 lane (007-mcp cartridge family, tasks -#9–12). It tells you exactly what landed from the coord-finishers side -so you don't duplicate or contradict it. - ---- - -## What shipped - -Eleven tasks, all pushed to `hyperpolymath/boj-server` `main`: - -| # | Title | Commits | -|---|-------|---------| -| 13 | Track-record table + `effective_affinity` + `coord_get_affinities` | `f8cafbf`, `3bd4710` | -| 15 | Dispatch preference + task_difficulty + sender_confidence + reject cooldown | `6065878` | -| 14 | Reassignment engine (server-origin quarantine entries for master review) | `9e40a86` | -| 17 | Deno/Node Nickel shim — runtime envelope validation in mcp-bridge | `ed85ca2` | -| 32 | Role rename supervisor/executor/supervised → master/journeyman/apprentice | `634c163` (merged with #35) | -| 35 | `coord_transfer_master` — live master handoff | `634c163` | -| 36 | `difficulty_hint` envelope field + `DifficultyHintValid` Nickel contract | `7f2f4a9`, `aeae440` | -| 37 | Prover-tag convention doc (`proof:lean4`/`agda`/`idris2`/`rocq`/`tla`) + example | `eba7cfe` | - -Design log updated in-tree: `Desktop/COORD-MCP-DESIGN-LOG.md` Part 2 -ledger has the full task table and new commit refs. - -## APIs you can use from 007-mcp - -If you're building on top of the coord layer from 007-mcp, these new -tools + FFI entrypoints are now live: - -- **MCP tools** (all routed through mcp-bridge at loopback:7745): - - `coord_report_outcome(token, tag, outcome, risk_tier, duration_ms?, confidence?)` - - `coord_get_affinities(token)` — per (client_kind, tag) aggregates - - `coord_set_declared_affinities(token, tags[])` - - `coord_scan_suggestions(token)` — enqueues candidate fyi/clarify - envelopes into the quarantine for master review - - `coord_transfer_master(token, new_peer_id, secret)` - - `coord_register` accepts optional `declared_affinities: string[]` - - `coord_claim_task` accepts optional `confidence`, - `dispatch_preference`, `task_difficulty` - -- **Envelope fields** (validated by Nickel contracts in mcp-bridge - before they hit the Zig adapter): - - `sender_confidence: number in [0, 1]` - - `dispatch_preference: "deliberate" | "broadcast" | "auto"` - - `task_difficulty: "trivial" | "routine" | "challenging" | "novel"` - - `difficulty_hint: "low" | "medium" | "high"` — orthogonal to `risk_tier` - -- **Prover-tag convention** (zero code): task tags prefixed - `proof:` route to peers whose `declared_affinities` cover that - prover. Worked example in - `cartridges/local-coord-mcp/schemas/examples/prover-tag-claim.ncl`. - -## Terminology — DD-32 role rename - -If 007-mcp has any supervisor/executor/supervised strings, they need to -move to master/journeyman/apprentice. The old names are accepted as -aliases for one release at the registration boundary but should not -appear in new code. Integer ordinals are preserved (0/1/2) so durable -logs still replay. - -Env var: `BOJ_MASTER_TOKEN` (canonical). `BOJ_SUPERVISOR_TOKEN` read as -fallback for one release. - -## Things that DID NOT land (deferred, in Appendix M) - -- **Task #33** — `client_kind` extension with `openai`, `mistral` + - `variant` free-form string. No code yet. -- **Task #34** — capability advertisement on register (class ∈ - {reasoner, coder, mathematician, scribe, proofsmith, reader, jester}; - tier ∈ {A, B, C}; `prover_strengths` map). No code yet. - -If you need the `class` or `variant` fields from 007-mcp, they don't -exist yet — either stub around them or pick up Task #33/#34 first. - -## Things to watch for - -- **Quarantine queue still has `MAX_QUARANTINE=32`.** The reassignment - engine enqueues server-origin entries alongside real peer entries, so - the queue can fill faster now. If you see `coord_send_gated` return - -4 (queue full) from 007-mcp, consider raising the constant or - triggering `coord_scan_suggestions` less frequently. - -- **Rejection cooldown is per-client_kind, not per-peer.** 5 rejections - in 10 min across any `kind=claude` peer triggers a 30s cooldown on - all `kind=claude` claim attempts. If you spin up many Claude sessions - against one coord server, be aware. - -- **Parallel changes in coord-messages.ncl.** The envelope shape schema - gained `dispatch_preference`, `task_difficulty`, `difficulty_hint` - during this lane. `additionalProperties=false` means these are - accepted; anything else still gets rejected. - -## Uncommitted work NOT from this lane - -At time of writing the working tree has four modified files that look -like part of the parallel session's ongoing work (rename pass sweeping -through the rest of the cartridge, plus your E2E test extensions): - -- `cartridges/local-coord-mcp/abi/LocalCoord/PROOF-SCHEDULE.adoc` -- `cartridges/local-coord-mcp/cartridge.ncl` -- `cartridges/local-coord-mcp/schemas/test-contracts.sh` -- `cartridges/local-coord-mcp/tests/e2e_coord.ts` - -…and an untracked `.machine_readable/contractiles/bust/` directory -(`Bustfile.a2ml` + `bust.ncl`). These are left alone; commit or stash -as appropriate in your lane. - -## Re-orientation quick path - -If you're coming in cold: read `Desktop/COORD-MCP-DESIGN-LOG.md` Part 2 -ledger + Appendix M + Appendix L. Every decision is indexed in Part 3 -(DD-1 through DD-37). Every commit lands in Part 2's table. - -— Opus (1M context), 2026-04-20 afternoon diff --git a/docs/handover/COORD-MCP-STATE.adoc b/docs/handover/COORD-MCP-STATE.adoc new file mode 100644 index 00000000..765672c9 --- /dev/null +++ b/docs/handover/COORD-MCP-STATE.adoc @@ -0,0 +1,254 @@ +== Coord-MCP Multi-Agent Coordination — State of Things + +*Where we are now, in relation to the forward vision.* Complements +`+COORD-MCP-TODO.md+` (actionable backlog) and +`+COORD-MCP-DESIGN-LOG.md+` (full design rationale + DD-1..DD-38 + +appendices A–M). + +Last updated: 2026-04-20. + +''''' + +=== Vision (one paragraph) + +A localhost message bus (`+boj-server/cartridges/local-coord-mcp+`) lets +multiple AI agents on the same machine discover each other, exchange +typed messages, claim tasks without collision, and operate under a +supervision model where *Opus co-supervises with the user*. The user +stays single-threaded: they talk to Opus in the main terminal; other +agents (Claude Sonnet/Haiku, Codex, Gemini, Vibe, GPT-n, Mistral) work +in their own windows and reach the user only through Opus, who dedupes, +synthesises, and rejects confabulation. Non-Claude agents run under a +Byzantine-safe firewall that contains their failure modes without +stopping them from contributing. v1 is hermetically local; v2 adds +federation only for joint IDApTIK/ASS sessions with the user’s son, with +an explicit authoritative-site designation per project to prevent +mission drift. + +=== The ladder (how trust and awareness scale) + +[width="100%",cols="30%,40%,30%",options="header",] +|=== +|Axis |Values |Gate +|*Trust role* (1 per peer, gated) |`+master+` / `+journeyman+` / +`+apprentice+` |Master via `+BOJ_MASTER_TOKEN+` env secret; other roles +self-declared + +|*Risk tier* (per message) |0 (status/query) → 4 (force-push, license, +always-private) |Tier 2+ from apprentice → quarantine; Tier 4 +schema-level forbidden for apprentice + +|*Awareness posture* (per message) |drone / narrow / local / strategic / +full |`+context_fetch_id+` REQUIRED for Tier 2+; no blind risky edits + +|*Capability* (advertised, multiple) |class ∈ \{reasoner, coder, +mathematician, scribe, proofsmith, reader, jester}; tier A/B/C; +`+prover_strengths+` map |Free-form opt-in; informs router; track-record +dominates over time + +|*Model identity* |`+client_kind+` ∈ \{claude, gemini, copilot, custom, +openai, mistral}; `+variant+` free-form |Register-time, cold-start +metadata for the router +|=== + +=== What’s built today + +==== Layer 1 — Transport + +* Loopback-only Zig REST server on port 7745. +* Idris2 `+IsLoopback+` type has exactly two constructors; bind to +non-loopback is type-impossible. (DD-1, P-00.) + +==== Layer 2 — Identity + +* `+coord_register(client_kind, role_hint, context)+` → `+peer_id+` of +form `+-<4hex>[@]+` + 128-bit CSPRNG session token. +(DD-2, DD-3.) +* Max 16 peers per box. +* Role rename *done* (DD-32): master/journeyman/apprentice. Old +`+supervisor/executor/supervised+` accepted as aliases for one release. + +==== Layer 3 — Envelope + +* 18 typed `+op_kind+`s. 5-level risk ladder. Nickel contracts validate +shape at the bridge before the Zig adapter. (DD-4, DD-5, DD-16, DD-17.) +* Fields landed: `+sender_confidence+`, `+sender_reasoning+`, +`+context_fetch_id+`, `+difficulty_hint+`, `+dispatch_preference+`, +`+task_difficulty+`, `+tier_override_reason+`, `+urgent_direct+`, +`+ack_required+`, `+ttl_seconds+`. (DDs 4/5/9/10/15/30/36.) +* Prover routing by tag convention: +`+proof:lean4+`/`+agda+`/`+idris2+`/`+rocq+`/`+tla+`. (DD-37.) + +==== Layer 4 — Supervision + +* Three roles gated; master role behind env secret. (DD-6, DD-7, DD-19.) +* Quarantine queue `+MAX_QUARANTINE=32+` hot cache + server-origin +entries for reassignment suggestions. (DD-21, Task #14.) +* `+coord_review+` / `+approve+` / `+reject+` tools. Apprentice cannot +set `+urgent_direct+`. (DD-12, Task #6.) +* Sanity auto-promote: `+git push+`/`+rm -rf+`/license mentions +auto-bump to Tier 3 regardless of declared tier. (DD-14.) +* Tier 4 schema-level forbidden for apprentice: force-push, license +touches, always-private-repos. (DD-13.) +* Rejection cooldown per `+client_kind+` — 5 rejects / 10 min = 30 s. + +==== Layer 5 — Durability + +* In-tree `+coord_durability.zig+` — append-only log + CRC-trailed +records +** typed helpers (`+logPeerAdd+`, `+logInboxPush+`, …). Restart-safe +replay. State at `+$BOJ_COORD_STATE_DIR+` (default +`+$XDG_STATE_HOME/boj-server/coord/+`). Per-box, per DD-23. (DD-31, Task +#7.) +* Benchmarks: ~4.7 µs append, ~9 µs durable round-trip, ~110k ops/sec +durable, ~9M ops/sec no-durability. +* 27 Zig + 41 Deno E2E tests green. Panic-attack assail clean. +* *Task #7b deferred*: swap backend to `+verisimdb-mcp+` FFI once it’s +real (currently all stubs). Typed helpers stay as the stable seam. + +==== Layer 6 — Cartridges + +* *007-mcp cartridge (007-lang)*: 72 MCP tools exposing full `+oo7+` +CLI, Idris2 `+IsLoopback+` proof at `+127.0.0.1:1066+`, Zig FFI + HTTP +adapter, Nickel manifest. On-enter/on-exit hooks (register, read +STATE/methodology, release claims). `+bust/+` (16 hard-stops) + +`+adjust/+` (8 drift-corrections). Memory auto-lift with real parser (5 +unit tests, `+MAX_HITS=8+`, graceful degrade). (Tasks #9–#12.) +* *k9 moved to `+svc/k9/+`* (ADR-001 2026-04-18). `+contractiles/+` is +pure 6-verb set; k9 is not a contractile. + +==== Byzantine safety — mechanism status + +[width="100%",cols="40%,28%,32%",options="header",] +|=== +|Mechanism |Status |Catches +|1. Hash chain per sender (`+prev_msg_hash+`) |*Pending P1* |Retroactive +rewriting + +|2. M-of-N attestation for Tier 2+ from apprentice |Designed (DD-8); +implementation pending |Unilateral bad commits + +|3. Watchdog claims with TTL + heartbeat |*Pending P1* (DD-20) +|Claim-squatting + +|4. Content sanity gate |*Pending P1* (DD-14) |Confabulation + tier +underclaiming + +|5. Tamper-evident audit log |*Partial* — durability done; explicit +envelope-audit index pending |Forensics / f+1 case +|=== + +==== Self-assessment — layer status + +* Layer A (static affinity, register-time) — *done* (DD-34 landed as +Task #34 pending, but declared affinities already accepted). +* Layer B (per-claim `+confidence+` + `+reasoning+`) — *done* (DD-9, +Task #15). +* Layer C (server-computed `+effective_affinity+` from track record) — +*done* (DD-9, Task #13). +* Layer D (drift detector: confidence ≫ track record → flag) — *done*. +Implemented in `+coord_scan_suggestions+` as a parallel `+kind:"drift"+` +envelope (op_kind=warn, tier 2, carries +`+drift_pct = avg_conf − eff_affinity+`). Co-emitted with `+overclaim+` +so the master sees a routing FYI *and* a self-assessment monitoring flag +from the same condition. + +=== Architectural spine — decisions taken + +38 decisions, DD-1..DD-38. Highlights: + +* *DD-1*: Loopback-only v1. Compile-time proven. +* *DD-8*: 5 Byzantine mechanisms, not consensus. BFT’s useful half +without the round-trip cost. +* *DD-9*: Self-assessment is 4 layers; static declaration is never +trusted alone. +* *DD-10*: `+context_fetch_id+` required for Tier 2+. +Blast-radius-scaled awareness; forces read-before-risky-write. +* *DD-11*: Summary-vs-raw context gated by role. Apprentices never see +raw state (prevents hallucinated connections). +* *DD-12*: `+fyi+`/`+clarify+`/`+blocker+`/`+urgent_direct+` routing. +User has one locus (main terminal, Opus); apprentices firewalled from +interrupting. +* *DD-15*: Hybrid dispatch — Opus seeds + peers self-claim by affinity. +* *DD-18*: GitHub only. Push `+origin+`; no other forges directly. +* *DD-22*: v2 federation uses authoritative-site model (one site holds +primacy per project). Motivated by IDApTIK drift, not credit load. +* *DD-23*: VeriSimDB per-pattern: coord = per-box; 007 repo data = +per-project; track record = per-box; memory auto-lift = per-project. +* *DD-29*: Peer crash+restart = fresh chain as new peer; old chain +preserved as audit echo anchor. Epistemic honesty. +* *DD-31*: Task #7 durability shipped as in-tree +`+coord_durability.zig+` because `+verisimdb-mcp+` FFI is all stubs. +Stable seam for later swap. +* *DD-32*: Role rename to master/journeyman/apprentice (craft guild). +Backward-compat shims for one release. +* *DD-33–37*: Multi-model extension (Appendix M) — client_kind + +variant, capability class/tier/prover_strengths, master handoff, +difficulty hint, prover tag convention. +* *DD-38*: Split gatekeeper (trust) from scheduler (dispatch) — flagged +for future, deferred. + +=== Adjacent work (not core coord but in this orbit) + +* *Echidna L3 live-prover CI* — Wave-1 (9 Tier-1 backends, every PR) and +Wave-2 (10 Tier-2 backends, nightly) *done* 2026-04-19. Wave-3 (9 Tier-3 +backends, weekly) needs per-backend Containerfiles; Wave-4 (19 Tier-4 +backends, quarterly) scaffold only. Handover hints in +`+verification-ecosystem/echidna/.machine_readable/6a2/STATE.a2ml+`. +* *Echidna Chapel FFI* — `+cargo build --features chapel+` now +self-links against bundled Zig stubs (`+-Dstubs=true+` default); +real-Chapel CI job still outstanding. +* *Tamarin/ProVerif in echidna* — fully wired (592 + 799 LoC Rust +backends, registered in `+ProverFactory+`, 4 unit tests). Was +stale-listed as "`planned`" — corrected. + +=== Interaction model (how the user uses this) + +Main terminal, Opus as chief-of-staff. + +* Apprentice peers never get a direct line to the user. Questions filter +through Opus. +* Supervisor (master) + journeyman peers can flag `+urgent_direct+` — +breaks into Opus’s output inline. +* Three urgency levels baked into `+op_kind+`: `+blocker+` (surface +inline), `+clarify+` (batch), `+fyi+` (log only). +* User can always pull: "`what are all agents doing?`" → Opus calls +`+coord_list_peers+` + synthesises. "`Kill Gemini’s current task`" → +Opus sends `+release+` on its behalf. +* Escape hatch: visit another agent’s terminal directly. Default is: +main terminal, single locus. + +=== Deferred by phase + +[width="100%",cols="28%,36%,36%",options="header",] +|=== +|Phase |Trigger |Content +|1b |Now — refinements during current buildout |Hash chain, sanity gate, +watchdog, warn-drift broadcast, quarantine spill, audit-echo, drift +detector, coord_health metrics + +|2 |After Task #8 E2E (done) validates imperative version |Idris2 +session types for supervisor/attestation choreography; deontic/dyadic +types for supervision rules + +|3 |After Phase 2 lands |Agda echo-types formalisation of +audit/summary/hash-chain; tropical-types modeling lens; rename +`+context_fetch_id+` → `+knowledge_witness+` + +|4 |First joint IDApTIK/ASS session with son |Envelope v2 (`+site_id+`, +`+federation.*+`); authoritative-site model; SDP + Stapeln + HTTP +capability gateway; primacy handoff ceremony; Umoja federation +integration +|=== + +=== Agreed decisions this session (ratified 2026-04-20) + +Captured in `+COORD-MCP-TODO.md+` under Decisions. Summary: + +* *D1*: Refactor `+DataExpr+` (split pure/control); do not paper over +with variance. +* *D2*: `+.machine_readable/6a2/+` is canonical for all SCM files; +remove root copies after diff/merge. +* *D3*: Wait on `+just cartridge-install+` until coord-mcp Tasks #33/#34 +ship. +* *D4*: Revert canonical-proof-suite probe-timing sidecars in 007-lang. diff --git a/docs/handover/COORD-MCP-STATE.md b/docs/handover/COORD-MCP-STATE.md deleted file mode 100644 index 8f8c199a..00000000 --- a/docs/handover/COORD-MCP-STATE.md +++ /dev/null @@ -1,198 +0,0 @@ - -# Coord-MCP Multi-Agent Coordination — State of Things - -**Where we are now, in relation to the forward vision.** Complements -`COORD-MCP-TODO.md` (actionable backlog) and `COORD-MCP-DESIGN-LOG.md` -(full design rationale + DD-1..DD-38 + appendices A–M). - -Last updated: 2026-04-20. - ---- - -## Vision (one paragraph) - -A localhost message bus (`boj-server/cartridges/local-coord-mcp`) lets -multiple AI agents on the same machine discover each other, exchange -typed messages, claim tasks without collision, and operate under a -supervision model where **Opus co-supervises with the user**. The user -stays single-threaded: they talk to Opus in the main terminal; other -agents (Claude Sonnet/Haiku, Codex, Gemini, Vibe, GPT-n, Mistral) work -in their own windows and reach the user only through Opus, who dedupes, -synthesises, and rejects confabulation. Non-Claude agents run under a -Byzantine-safe firewall that contains their failure modes without -stopping them from contributing. v1 is hermetically local; v2 adds -federation only for joint IDApTIK/ASS sessions with the user's son, -with an explicit authoritative-site designation per project to prevent -mission drift. - -## The ladder (how trust and awareness scale) - -| Axis | Values | Gate | -|------|--------|------| -| **Trust role** (1 per peer, gated) | `master` / `journeyman` / `apprentice` | Master via `BOJ_MASTER_TOKEN` env secret; other roles self-declared | -| **Risk tier** (per message) | 0 (status/query) → 4 (force-push, license, always-private) | Tier 2+ from apprentice → quarantine; Tier 4 schema-level forbidden for apprentice | -| **Awareness posture** (per message) | drone / narrow / local / strategic / full | `context_fetch_id` REQUIRED for Tier 2+; no blind risky edits | -| **Capability** (advertised, multiple) | class ∈ {reasoner, coder, mathematician, scribe, proofsmith, reader, jester}; tier A/B/C; `prover_strengths` map | Free-form opt-in; informs router; track-record dominates over time | -| **Model identity** | `client_kind` ∈ {claude, gemini, copilot, custom, openai, mistral}; `variant` free-form | Register-time, cold-start metadata for the router | - -## What's built today - -### Layer 1 — Transport -- Loopback-only Zig REST server on port 7745. -- Idris2 `IsLoopback` type has exactly two constructors; bind to - non-loopback is type-impossible. (DD-1, P-00.) - -### Layer 2 — Identity -- `coord_register(client_kind, role_hint, context)` → `peer_id` of form - `-<4hex>[@]` + 128-bit CSPRNG session token. (DD-2, DD-3.) -- Max 16 peers per box. -- Role rename **done** (DD-32): master/journeyman/apprentice. Old - `supervisor/executor/supervised` accepted as aliases for one release. - -### Layer 3 — Envelope -- 18 typed `op_kind`s. 5-level risk ladder. Nickel contracts validate - shape at the bridge before the Zig adapter. (DD-4, DD-5, DD-16, DD-17.) -- Fields landed: `sender_confidence`, `sender_reasoning`, - `context_fetch_id`, `difficulty_hint`, `dispatch_preference`, - `task_difficulty`, `tier_override_reason`, `urgent_direct`, - `ack_required`, `ttl_seconds`. (DDs 4/5/9/10/15/30/36.) -- Prover routing by tag convention: `proof:lean4`/`agda`/`idris2`/`rocq`/`tla`. (DD-37.) - -### Layer 4 — Supervision -- Three roles gated; master role behind env secret. (DD-6, DD-7, DD-19.) -- Quarantine queue `MAX_QUARANTINE=32` hot cache + server-origin entries - for reassignment suggestions. (DD-21, Task #14.) -- `coord_review` / `approve` / `reject` tools. Apprentice cannot set - `urgent_direct`. (DD-12, Task #6.) -- Sanity auto-promote: `git push`/`rm -rf`/license mentions auto-bump - to Tier 3 regardless of declared tier. (DD-14.) -- Tier 4 schema-level forbidden for apprentice: force-push, license - touches, always-private-repos. (DD-13.) -- Rejection cooldown per `client_kind` — 5 rejects / 10 min = 30 s. - -### Layer 5 — Durability -- In-tree `coord_durability.zig` — append-only log + CRC-trailed records - + typed helpers (`logPeerAdd`, `logInboxPush`, …). Restart-safe replay. - State at `$BOJ_COORD_STATE_DIR` (default `$XDG_STATE_HOME/boj-server/coord/`). - Per-box, per DD-23. (DD-31, Task #7.) -- Benchmarks: ~4.7 µs append, ~9 µs durable round-trip, ~110k ops/sec - durable, ~9M ops/sec no-durability. -- 27 Zig + 41 Deno E2E tests green. Panic-attack assail clean. -- **Task #7b deferred**: swap backend to `verisimdb-mcp` FFI once it's - real (currently all stubs). Typed helpers stay as the stable seam. - -### Layer 6 — Cartridges -- **007-mcp cartridge (007-lang)**: 72 MCP tools exposing full `oo7` - CLI, Idris2 `IsLoopback` proof at `127.0.0.1:1066`, Zig FFI + HTTP - adapter, Nickel manifest. On-enter/on-exit hooks (register, read - STATE/methodology, release claims). `bust/` (16 hard-stops) + - `adjust/` (8 drift-corrections). Memory auto-lift with real parser - (5 unit tests, `MAX_HITS=8`, graceful degrade). (Tasks #9–#12.) -- **k9 moved to `svc/k9/`** (ADR-001 2026-04-18). `contractiles/` is - pure 6-verb set; k9 is not a contractile. - -### Byzantine safety — mechanism status - -| Mechanism | Status | Catches | -|-----------|--------|---------| -| 1. Hash chain per sender (`prev_msg_hash`) | **Pending P1** | Retroactive rewriting | -| 2. M-of-N attestation for Tier 2+ from apprentice | Designed (DD-8); implementation pending | Unilateral bad commits | -| 3. Watchdog claims with TTL + heartbeat | **Pending P1** (DD-20) | Claim-squatting | -| 4. Content sanity gate | **Pending P1** (DD-14) | Confabulation + tier underclaiming | -| 5. Tamper-evident audit log | **Partial** — durability done; explicit envelope-audit index pending | Forensics / f+1 case | - -### Self-assessment — layer status - -- Layer A (static affinity, register-time) — **done** (DD-34 landed as Task #34 pending, but declared affinities already accepted). -- Layer B (per-claim `confidence` + `reasoning`) — **done** (DD-9, Task #15). -- Layer C (server-computed `effective_affinity` from track record) — **done** (DD-9, Task #13). -- Layer D (drift detector: confidence ≫ track record → flag) — **done**. - Implemented in `coord_scan_suggestions` as a parallel `kind:"drift"` - envelope (op_kind=warn, tier 2, carries `drift_pct = avg_conf − eff_affinity`). - Co-emitted with `overclaim` so the master sees a routing FYI **and** a - self-assessment monitoring flag from the same condition. - -## Architectural spine — decisions taken - -38 decisions, DD-1..DD-38. Highlights: - -- **DD-1**: Loopback-only v1. Compile-time proven. -- **DD-8**: 5 Byzantine mechanisms, not consensus. BFT's useful half - without the round-trip cost. -- **DD-9**: Self-assessment is 4 layers; static declaration is never - trusted alone. -- **DD-10**: `context_fetch_id` required for Tier 2+. Blast-radius-scaled - awareness; forces read-before-risky-write. -- **DD-11**: Summary-vs-raw context gated by role. Apprentices never see - raw state (prevents hallucinated connections). -- **DD-12**: `fyi`/`clarify`/`blocker`/`urgent_direct` routing. User has - one locus (main terminal, Opus); apprentices firewalled from - interrupting. -- **DD-15**: Hybrid dispatch — Opus seeds + peers self-claim by affinity. -- **DD-18**: GitHub only. Push `origin`; no other forges directly. -- **DD-22**: v2 federation uses authoritative-site model (one site holds - primacy per project). Motivated by IDApTIK drift, not credit load. -- **DD-23**: VeriSimDB per-pattern: coord = per-box; 007 repo data = - per-project; track record = per-box; memory auto-lift = per-project. -- **DD-29**: Peer crash+restart = fresh chain as new peer; old chain - preserved as audit echo anchor. Epistemic honesty. -- **DD-31**: Task #7 durability shipped as in-tree `coord_durability.zig` - because `verisimdb-mcp` FFI is all stubs. Stable seam for later swap. -- **DD-32**: Role rename to master/journeyman/apprentice (craft guild). - Backward-compat shims for one release. -- **DD-33–37**: Multi-model extension (Appendix M) — client_kind + - variant, capability class/tier/prover_strengths, master handoff, - difficulty hint, prover tag convention. -- **DD-38**: Split gatekeeper (trust) from scheduler (dispatch) — flagged - for future, deferred. - -## Adjacent work (not core coord but in this orbit) - -- **Echidna L3 live-prover CI** — Wave-1 (9 Tier-1 backends, every PR) - and Wave-2 (10 Tier-2 backends, nightly) **done** 2026-04-19. Wave-3 - (9 Tier-3 backends, weekly) needs per-backend Containerfiles; Wave-4 - (19 Tier-4 backends, quarterly) scaffold only. Handover hints in - `verification-ecosystem/echidna/.machine_readable/6a2/STATE.a2ml`. -- **Echidna Chapel FFI** — `cargo build --features chapel` now - self-links against bundled Zig stubs (`-Dstubs=true` default); - real-Chapel CI job still outstanding. -- **Tamarin/ProVerif in echidna** — fully wired (592 + 799 LoC Rust - backends, registered in `ProverFactory`, 4 unit tests). Was stale-listed - as "planned" — corrected. - -## Interaction model (how the user uses this) - -Main terminal, Opus as chief-of-staff. - -- Apprentice peers never get a direct line to the user. Questions filter - through Opus. -- Supervisor (master) + journeyman peers can flag `urgent_direct` — breaks - into Opus's output inline. -- Three urgency levels baked into `op_kind`: `blocker` (surface inline), - `clarify` (batch), `fyi` (log only). -- User can always pull: "what are all agents doing?" → Opus calls - `coord_list_peers` + synthesises. "Kill Gemini's current task" → Opus - sends `release` on its behalf. -- Escape hatch: visit another agent's terminal directly. Default is: - main terminal, single locus. - -## Deferred by phase - -| Phase | Trigger | Content | -|-------|---------|---------| -| 1b | Now — refinements during current buildout | Hash chain, sanity gate, watchdog, warn-drift broadcast, quarantine spill, audit-echo, drift detector, coord_health metrics | -| 2 | After Task #8 E2E (done) validates imperative version | Idris2 session types for supervisor/attestation choreography; deontic/dyadic types for supervision rules | -| 3 | After Phase 2 lands | Agda echo-types formalisation of audit/summary/hash-chain; tropical-types modeling lens; rename `context_fetch_id` → `knowledge_witness` | -| 4 | First joint IDApTIK/ASS session with son | Envelope v2 (`site_id`, `federation.*`); authoritative-site model; SDP + Stapeln + HTTP capability gateway; primacy handoff ceremony; Umoja federation integration | - -## Agreed decisions this session (ratified 2026-04-20) - -Captured in `COORD-MCP-TODO.md` under Decisions. Summary: - -- **D1**: Refactor `DataExpr` (split pure/control); do not paper over with variance. -- **D2**: `.machine_readable/6a2/` is canonical for all SCM files; remove root copies after diff/merge. -- **D3**: Wait on `just cartridge-install` until coord-mcp Tasks #33/#34 ship. -- **D4**: Revert canonical-proof-suite probe-timing sidecars in 007-lang. diff --git a/docs/handover/COORD-MCP-TODO.adoc b/docs/handover/COORD-MCP-TODO.adoc new file mode 100644 index 00000000..24d29733 --- /dev/null +++ b/docs/handover/COORD-MCP-TODO.adoc @@ -0,0 +1,239 @@ +== Coord-MCP Multi-Agent Coordination — TODO + +*Source of truth for pending work.* Complements `+COORD-MCP-STATE.md+` +(where we are) and `+COORD-MCP-DESIGN-LOG.md+` (full design rationale + +DD-1..DD-38). + +Last updated: 2026-04-20. + +''''' + +=== P0 — Active / next pickup + +[width="100%",cols="13%,21%,21%,17%,28%",options="header",] +|=== +|# |Task |Repo |Est |Blocks +|33 |`+client_kind+` += `+openai+`/`+mistral+`; add `+variant+` +free-form (opus-4.7, flash-2.5, leanstral, …) |boj-server |0.5 d |34 + +|34 |Capability advertisement on register: `+class+` / `+tier+` / +`+prover_strengths+` + `+coord_get_peer_capabilities+` FFI |boj-server +|1 d |cold-start routing + +|007-mcp-1 |`+coordRegister+` HTTP write: +`+cartridges/007-mcp/ffi/oo7_mcp_ffi.zig+` currently TCP-probes only. +Rewrite to POST `+http://127.0.0.1:7745/tools/coord_register+`, parse +token + peer_id into `+g_coord_token_buf+`/`+g_peer_id_buf+`. |007-lang +|1 session |end-to-end 007 cartridge + +|007-mcp-2 |Run `+just cartridge-install+` in 007-lang, commit resulting +tree in boj-server. *Gate on #33+#34 shipping first* (shared-state in +`+local-coord-mcp/ffi/+` + `+adapter/+`). |007-lang → boj-server |1 +session |— + +|echidna-L3-w2-verify |Watch next `+0 3 * * *+` UTC nightly of +`+live-provers.yml+`; fix red matrix cells in-place (likely: isabelle +500MB download, tlaps URL drift, fstar binary symlink). |echidna |1 +session |Wave-3 entry +|=== + +=== P1 — Short term (weeks) + +==== Coord-MCP phase 1b refinements + +* *Hash chain per envelope* — sender-side `+prev_msg_hash+`; server +tracks chain head; break = instant reject. (DD-8 mechanism 1.) *Plan +staged 2026-04-20*: add `+chain_head: [32]u8+` to `+Peer+`; shared +`+sendInner+` for `+coord_send+` + `+coord_send_gated+`; new +`+coord_send_chain+` / `+coord_send_gated_chain+` exports with `+-6+` +chain_break return + expected/got out-params; adapter hex parsing + HTTP +409 chain_break JSON; new durability event `+chain_advance = 18+`; +optional `+prev_msg_hash+` field in schema (backwards-compat for +in-flight 007-mcp client per D3). ~8 subtasks. +* *Content sanity gate* — file-ref validity vs recent-FS cache + +self-contradiction heuristic + risk-tier escalator patterns +(`+git push+` → auto-promote Tier 3). (DD-8 mechanism 4, DD-14.) +* [line-through]#*Watchdog TTL enforcement* — 30 s apprentice, 5 min +journeyman; `+progress+` heartbeat resets. (DD-20.)# — *done*: +`+coord_progress+` heartbeats, `+coord_sweep_watchdog+` polls, implicit +sweep on `+coord_claim_task+`. Master has no TTL. Auto-release audited +with kind=3 (`+AUTO_RELEASE+`). +* [line-through]#*Warn-drift broadcast on auto-release* via Opus review. +(DD-21.)# — *done*: `+sweepExpiredClaims+` now emits a `+warn_drift+` +envelope alongside the kind=3 auto-release audit. Master present → +queued as server-origin suggestion (`+sender_idx=0xFE+`, tier 1, +broadcast-on-approve); master absent → direct inbox push to every peer +except the holder. Extra audit kinds: 4 = `+WARN_DRIFT_QUEUED+`, 5 = +`+WARN_DRIFT_BROADCAST+`. +* *Quarantine queue spill to VeriSimDB* when full; currently +`+MAX_QUARANTINE=32+` hot cache only. (DD-17.) +* *Audit-echo anchor* — preserve old chain head on peer crash/restart; +new peer = fresh chain. (DD-29.) +* [line-through]#*Drift detector* — flag +`+confidence > 0.8 AND effective_affinity < 0.3+`. (DD-9 layer D.)# — +*done*: `+coord_scan_suggestions+` now emits a parallel `+kind:"drift"+` +envelope (op_kind=warn, tier 2, includes `+drift_pct+`) alongside the +routing-focused `+overclaim+`. +* [line-through]#*`+coord_health+` metrics tool* — active peers, pending +quarantine, reject rate, claim depth.# — *done*: `+coord_health+` +returns peers (active/by_kind/by_role), quarantine, claims, track-record +fill, and per-kind rejects + cooldown flags. +* [line-through]#*Rejection rate limit hardening* — 5 rejects / 10 min +per `+client_kind+` already lands cooldown; audit whether per-peer is +better on heavy multi-session load.# — *done*: per-peer ring runs +alongside per-kind; `+coord_claim_task_ex+` records+checks both. +`+coord_health.rejects.by_peer+` surfaces +`+{peer_id, count, in_cooldown}+` per active slot. Cleared on deregister ++ reset. + +==== 007-mcp + 007-lang + +* *Harvard DataExpr refactor* (D1) — split +`+crates/oo7-core/src :: DataExpr+` into pure `+DataExpr+` + +`+DataFlow+` control carrier. Restores `+just contractile-check+` green +on fresh clones. +* *Adopt natsci-studio `+intend.k9.ncl+`* negotiation + accountability +pledge in 007 (declared as horizon wish in new `+Intentfile.a2ml+`). +* [line-through]#Sidecar revert# — executed this session per D4. + +==== Echidna — L3 completion + +* *Wave-3 (Tier-3 weekly, 9 backends)* — per-backend Containerfiles +(Podman): Tamarin, ProVerif, Imandra, SCIP, OR-Tools, HOL4, ACL2, Twelf, +Metamath. Handover hints in +`+verification-ecosystem/echidna/.machine_readable/6a2/STATE.a2ml [wave-3-handover-hints]+`. +* *Wave-4 (Tier-4 quarterly, 19 backends)* — best-effort allow-fail +placeholder; retain as mock-only unless a maintainer volunteers. +* *Dafny deep-wiring upgrade* — `+provers/dafny.rs+` is 165 LoC; live +version-check passes but subprocess wrapper is stub-ish. (L3 prompt +"`Harden wiring depth`".) +* *Real-Chapel CI job* — `+chapel-ci.yml+` tests against stubs only. Add +a job that builds real `+libechidna_chapel.so+` from `+chapel_poc/+` and +links Rust with `+-Dstubs=false+`. Allow-fail at first. +* *VeriSimDB record emission* from live-prover harness per +`+feedback_verisimdb_policy+` — coordinate with `+verisimdb+` repo +schema first. + +==== Estate-wide + +* *6a2 canonical-location reconciliation* (D2) — for each repo with +duplicate SCM files, diff root vs `+.machine_readable/6a2/+` copy; merge +if divergent; `+git rm+` root. Global CLAUDE.md is authoritative: +`+.machine_readable/+` only. Use `+adjust/+` contractile runner once the +hook source migrates off the stale 6-verb set. + +=== P2 — Medium (Phase 2 formalisms over working v1) + +* *Proof obligations P-04..P-07* in +`+cartridges/local-coord-mcp/abi/LocalCoord/Durability.idr+`: record +format; CRC truncation; *P-06 replay-equivalence (keystone)*; quarantine +state machine. ~6 days. +* *Idris2 session types* for supervisor/attestation choreography in +`+abi/LocalCoord/Protocol.idr+`. Makes protocol compliance a +compile-time property. +* *Deontic / dyadic types* for supervision rules — formalise "`tier 4 +forbidden for apprentice`" as a type-level obligation. +* *Gatekeeper / scheduler split* (DD-38) — separate the master’s +approve-veto role from the dispatch-routing role. Small-team stays +merged; big team splits. RFC deferred to when cross-model dispatch +demands it. +* *Task #2* — wire `+boj_cartridge_invoke+` to real FFI (low priority; +HTTP path works fine today). +* *Task #7b* — swap `+coord_durability.zig+` backend to +`+verisimdb-mcp+` FFI once that FFI is real. Typed log helpers stay as +the API. (DD-31.) + +=== P3 — Phase 3: formal foundations + +* *Agda echo-types formalisation* of audit + summary + hash-chain as +echo types, via `+EchoChoreo+` / `+EchoEpistemic+` / `+EchoTropical+` +bridges. Dogfoods `+echo-types/+` as a non-toy consumer. +* *Tropical types* as modeling lens — TTL = tropical max; trust +composition = tropical min; tier promotion = max; attention budget = inf +over priorities. (Appendix J.) +* *Epistemic types* explicit: rename `+context_fetch_id+` → +`+knowledge_witness+`; document per-role epistemic policy in +`+envelope-design.adoc+`. + +=== P4 — Phase 4: v2 federation (trigger: first joint IDApTIK/ASS session) + +* *Envelope v2* with `+site_id+` + `+federation.authoritative_site+` + +`+federation.project_id+` + `+federation.handoff_ceremony_id+`. +* *Authoritative-site model* (DD-22) — one site holds primacy per +project; peer defers on code-ownership. Motivation: prevent +IDApTIK-style mission drift. +* *Security stack* — SDP (Secure Device Provisioning) + Stapeln +container + HTTP capability gateway + high-security options. +* *Primacy handoff ceremony* — explicit protocol, choreographic + +epistemic transfer. +* *Cross-site attestation + trust composition* — tropical min for +transitive trust. +* *Integrate with Umoja federation layer* (gossip + hash attestation) +for cross-machine transport. + +=== Non-goals / explicitly deferred + +* HTTP capability gateway + SDP for local-only v1 (overkill). +* Auto-modifying affinities without Opus + user review (always loop +back). +* Cross-machine transport in v1 envelope. +* Cost-aware scheduling / credit-burn tracking (separate feature; can +consume capability metadata later). +* Choreographic / epistemic / echo / tropical type _enforcement_ (Phase +2+/3 modeling layers only). + +''''' + +=== Decisions — agreed 2026-04-20 (was: open questions) + +User agreed these four decisions as the way forward. Logged here so +future sessions do not re-litigate them. Rationale lines are the +original recommendations retained verbatim. + +[width="100%",cols="10%,28%,31%,31%",options="header",] +|=== +|# |Decision |Rationale |Follow-up +|D1 |*Refactor* `+DataExpr+` in `+crates/oo7-core/src+` — split pure +`+DataExpr+` from `+DataFlow+` control carrier. |Control-flow variants +in a data-expression enum is a structural Harvard-invariant violation, +not stylistic. `+variance_schema+` would codify the mistake. Refactor is +1–2 h in a well-factored crate; variance only justified if the merger is +load-bearing on a benchmarked hot path (not the case). |*P1 task added* +— 007-lang crate refactor. `+just contractile-check+` currently red on +fresh clones; turns green after split. + +|D2 |*Canonical SCM-file location is `+.machine_readable/6a2/+`.* +Root-level copies are drift and must be removed. |Global `+CLAUDE.md+`: +_"`CRITICAL: SCM files MUST be in `+.machine_readable/+` directory ONLY. +Never create STATE.scm, META.scm, … in the repository root.`"_ |*P1 task +added* — diff root vs 6a2/ copies per repo; merge if divergent; +`+git rm+` root. Estate-wide sweep. + +|D3 |*Wait on `+just cartridge-install+` in boj-server* until coord-mcp +Tasks #33 + #34 ship. |Shared-state risk: +`+cartridges/local-coord-mcp/ffi/local_coord_ffi.zig+` + +`+adapter/local_coord_adapter.zig+` are owned by the sequential session +for #33/#34 (Peer struct, ClientKind enum). Running install now risks +conflicts on their live working tree. |*Gate locked* — 007-mcp-2 in P0 +table blocks on #33+#34. + +|D4 |*Revert* modified `+audits/canonical-proof-suite/+` + +`+proofs/canonical-proof-suite/+` in 007-lang. |They are +`+just canonical-proof-suite+` probe-timing artefacts, not session +scope. The parallel session itself flagged them as non-scope. Committing +conflates measurement noise with substantive proof state. Regenerable if +needed. |*Executed this session* — `+git checkout --+` in 007-lang. +|=== + +=== Task status snapshot (complete / pending / deferred) + +*Complete (22 tasks):* #1, #3, #4, #5, #6, #7, #8, #9, #10, #11, #12, +#13, #14, #15, #16, #17, #32, #35, #36, #37, k9-svc spot-fix, echidna L3 +Wave-1 + Wave-2. + +*Pending P0/P1:* #33, #34, 007-mcp-1/2, echidna Wave-2 CI verification, +6a2 reconciliation, Harvard DataExpr, sidecar revert, intend.k9.ncl +adoption. + +*Deferred:* #2, #7b, P-04..P-07 proofs, Idris2 session types, DD-38 +split, all Phase 3/4 items. diff --git a/docs/handover/COORD-MCP-TODO.md b/docs/handover/COORD-MCP-TODO.md deleted file mode 100644 index fad0e297..00000000 --- a/docs/handover/COORD-MCP-TODO.md +++ /dev/null @@ -1,111 +0,0 @@ - -# Coord-MCP Multi-Agent Coordination — TODO - -**Source of truth for pending work.** Complements `COORD-MCP-STATE.md` (where -we are) and `COORD-MCP-DESIGN-LOG.md` (full design rationale + DD-1..DD-38). - -Last updated: 2026-04-20. - ---- - -## P0 — Active / next pickup - -| # | Task | Repo | Est | Blocks | -|---|------|------|-----|--------| -| 33 | `client_kind` += `openai`/`mistral`; add `variant` free-form (opus-4.7, flash-2.5, leanstral, …) | boj-server | 0.5 d | 34 | -| 34 | Capability advertisement on register: `class` / `tier` / `prover_strengths` + `coord_get_peer_capabilities` FFI | boj-server | 1 d | cold-start routing | -| 007-mcp-1 | `coordRegister` HTTP write: `cartridges/007-mcp/ffi/oo7_mcp_ffi.zig` currently TCP-probes only. Rewrite to POST `http://127.0.0.1:7745/tools/coord_register`, parse token + peer_id into `g_coord_token_buf`/`g_peer_id_buf`. | 007-lang | 1 session | end-to-end 007 cartridge | -| 007-mcp-2 | Run `just cartridge-install` in 007-lang, commit resulting tree in boj-server. **Gate on #33+#34 shipping first** (shared-state in `local-coord-mcp/ffi/` + `adapter/`). | 007-lang → boj-server | 1 session | — | -| echidna-L3-w2-verify | Watch next `0 3 * * *` UTC nightly of `live-provers.yml`; fix red matrix cells in-place (likely: isabelle 500MB download, tlaps URL drift, fstar binary symlink). | echidna | 1 session | Wave-3 entry | - -## P1 — Short term (weeks) - -### Coord-MCP phase 1b refinements - -- **Hash chain per envelope** — sender-side `prev_msg_hash`; server tracks chain head; break = instant reject. (DD-8 mechanism 1.) **Plan staged 2026-04-20**: add `chain_head: [32]u8` to `Peer`; shared `sendInner` for `coord_send` + `coord_send_gated`; new `coord_send_chain` / `coord_send_gated_chain` exports with `-6` chain_break return + expected/got out-params; adapter hex parsing + HTTP 409 chain_break JSON; new durability event `chain_advance = 18`; optional `prev_msg_hash` field in schema (backwards-compat for in-flight 007-mcp client per D3). ~8 subtasks. -- **Content sanity gate** — file-ref validity vs recent-FS cache + self-contradiction heuristic + risk-tier escalator patterns (`git push` → auto-promote Tier 3). (DD-8 mechanism 4, DD-14.) -- ~~**Watchdog TTL enforcement** — 30 s apprentice, 5 min journeyman; `progress` heartbeat resets. (DD-20.)~~ — **done**: `coord_progress` heartbeats, `coord_sweep_watchdog` polls, implicit sweep on `coord_claim_task`. Master has no TTL. Auto-release audited with kind=3 (`AUTO_RELEASE`). -- ~~**Warn-drift broadcast on auto-release** via Opus review. (DD-21.)~~ — **done**: `sweepExpiredClaims` now emits a `warn_drift` envelope alongside the kind=3 auto-release audit. Master present → queued as server-origin suggestion (`sender_idx=0xFE`, tier 1, broadcast-on-approve); master absent → direct inbox push to every peer except the holder. Extra audit kinds: 4 = `WARN_DRIFT_QUEUED`, 5 = `WARN_DRIFT_BROADCAST`. -- **Quarantine queue spill to VeriSimDB** when full; currently `MAX_QUARANTINE=32` hot cache only. (DD-17.) -- **Audit-echo anchor** — preserve old chain head on peer crash/restart; new peer = fresh chain. (DD-29.) -- ~~**Drift detector** — flag `confidence > 0.8 AND effective_affinity < 0.3`. (DD-9 layer D.)~~ — **done**: `coord_scan_suggestions` now emits a parallel `kind:"drift"` envelope (op_kind=warn, tier 2, includes `drift_pct`) alongside the routing-focused `overclaim`. -- ~~**`coord_health` metrics tool** — active peers, pending quarantine, reject rate, claim depth.~~ — **done**: `coord_health` returns peers (active/by_kind/by_role), quarantine, claims, track-record fill, and per-kind rejects + cooldown flags. -- ~~**Rejection rate limit hardening** — 5 rejects / 10 min per `client_kind` already lands cooldown; audit whether per-peer is better on heavy multi-session load.~~ — **done**: per-peer ring runs alongside per-kind; `coord_claim_task_ex` records+checks both. `coord_health.rejects.by_peer` surfaces `{peer_id, count, in_cooldown}` per active slot. Cleared on deregister + reset. - -### 007-mcp + 007-lang - -- **Harvard DataExpr refactor** (D1) — split `crates/oo7-core/src :: DataExpr` into pure `DataExpr` + `DataFlow` control carrier. Restores `just contractile-check` green on fresh clones. -- **Adopt natsci-studio `intend.k9.ncl`** negotiation + accountability pledge in 007 (declared as horizon wish in new `Intentfile.a2ml`). -- ~~Sidecar revert~~ — executed this session per D4. - -### Echidna — L3 completion - -- **Wave-3 (Tier-3 weekly, 9 backends)** — per-backend Containerfiles (Podman): - Tamarin, ProVerif, Imandra, SCIP, OR-Tools, HOL4, ACL2, Twelf, Metamath. - Handover hints in `verification-ecosystem/echidna/.machine_readable/6a2/STATE.a2ml [wave-3-handover-hints]`. -- **Wave-4 (Tier-4 quarterly, 19 backends)** — best-effort allow-fail placeholder; retain as mock-only unless a maintainer volunteers. -- **Dafny deep-wiring upgrade** — `provers/dafny.rs` is 165 LoC; live version-check passes but subprocess wrapper is stub-ish. (L3 prompt "Harden wiring depth".) -- **Real-Chapel CI job** — `chapel-ci.yml` tests against stubs only. Add a job that builds real `libechidna_chapel.so` from `chapel_poc/` and links Rust with `-Dstubs=false`. Allow-fail at first. -- **VeriSimDB record emission** from live-prover harness per `feedback_verisimdb_policy` — coordinate with `verisimdb` repo schema first. - -### Estate-wide - -- **6a2 canonical-location reconciliation** (D2) — for each repo with duplicate SCM files, diff root vs `.machine_readable/6a2/` copy; merge if divergent; `git rm` root. Global CLAUDE.md is authoritative: `.machine_readable/` only. Use `adjust/` contractile runner once the hook source migrates off the stale 6-verb set. - -## P2 — Medium (Phase 2 formalisms over working v1) - -- **Proof obligations P-04..P-07** in `cartridges/local-coord-mcp/abi/LocalCoord/Durability.idr`: record format; CRC truncation; **P-06 replay-equivalence (keystone)**; quarantine state machine. ~6 days. -- **Idris2 session types** for supervisor/attestation choreography in `abi/LocalCoord/Protocol.idr`. Makes protocol compliance a compile-time property. -- **Deontic / dyadic types** for supervision rules — formalise "tier 4 forbidden for apprentice" as a type-level obligation. -- **Gatekeeper / scheduler split** (DD-38) — separate the master's approve-veto role from the dispatch-routing role. Small-team stays merged; big team splits. RFC deferred to when cross-model dispatch demands it. -- **Task #2** — wire `boj_cartridge_invoke` to real FFI (low priority; HTTP path works fine today). -- **Task #7b** — swap `coord_durability.zig` backend to `verisimdb-mcp` FFI once that FFI is real. Typed log helpers stay as the API. (DD-31.) - -## P3 — Phase 3: formal foundations - -- **Agda echo-types formalisation** of audit + summary + hash-chain as echo types, via `EchoChoreo` / `EchoEpistemic` / `EchoTropical` bridges. Dogfoods `echo-types/` as a non-toy consumer. -- **Tropical types** as modeling lens — TTL = tropical max; trust composition = tropical min; tier promotion = max; attention budget = inf over priorities. (Appendix J.) -- **Epistemic types** explicit: rename `context_fetch_id` → `knowledge_witness`; document per-role epistemic policy in `envelope-design.adoc`. - -## P4 — Phase 4: v2 federation (trigger: first joint IDApTIK/ASS session) - -- **Envelope v2** with `site_id` + `federation.authoritative_site` + `federation.project_id` + `federation.handoff_ceremony_id`. -- **Authoritative-site model** (DD-22) — one site holds primacy per project; peer defers on code-ownership. Motivation: prevent IDApTIK-style mission drift. -- **Security stack** — SDP (Secure Device Provisioning) + Stapeln container + HTTP capability gateway + high-security options. -- **Primacy handoff ceremony** — explicit protocol, choreographic + epistemic transfer. -- **Cross-site attestation + trust composition** — tropical min for transitive trust. -- **Integrate with Umoja federation layer** (gossip + hash attestation) for cross-machine transport. - -## Non-goals / explicitly deferred - -- HTTP capability gateway + SDP for local-only v1 (overkill). -- Auto-modifying affinities without Opus + user review (always loop back). -- Cross-machine transport in v1 envelope. -- Cost-aware scheduling / credit-burn tracking (separate feature; can consume capability metadata later). -- Choreographic / epistemic / echo / tropical type _enforcement_ (Phase 2+/3 modeling layers only). - ---- - -## Decisions — agreed 2026-04-20 (was: open questions) - -User agreed these four decisions as the way forward. Logged here so -future sessions do not re-litigate them. Rationale lines are the -original recommendations retained verbatim. - -| # | Decision | Rationale | Follow-up | -|---|----------|-----------|-----------| -| D1 | **Refactor** `DataExpr` in `crates/oo7-core/src` — split pure `DataExpr` from `DataFlow` control carrier. | Control-flow variants in a data-expression enum is a structural Harvard-invariant violation, not stylistic. `variance_schema` would codify the mistake. Refactor is 1–2 h in a well-factored crate; variance only justified if the merger is load-bearing on a benchmarked hot path (not the case). | **P1 task added** — 007-lang crate refactor. `just contractile-check` currently red on fresh clones; turns green after split. | -| D2 | **Canonical SCM-file location is `.machine_readable/6a2/`.** Root-level copies are drift and must be removed. | Global `CLAUDE.md`: *"CRITICAL: SCM files MUST be in `.machine_readable/` directory ONLY. Never create STATE.scm, META.scm, … in the repository root."* | **P1 task added** — diff root vs 6a2/ copies per repo; merge if divergent; `git rm` root. Estate-wide sweep. | -| D3 | **Wait on `just cartridge-install` in boj-server** until coord-mcp Tasks #33 + #34 ship. | Shared-state risk: `cartridges/local-coord-mcp/ffi/local_coord_ffi.zig` + `adapter/local_coord_adapter.zig` are owned by the sequential session for #33/#34 (Peer struct, ClientKind enum). Running install now risks conflicts on their live working tree. | **Gate locked** — 007-mcp-2 in P0 table blocks on #33+#34. | -| D4 | **Revert** modified `audits/canonical-proof-suite/` + `proofs/canonical-proof-suite/` in 007-lang. | They are `just canonical-proof-suite` probe-timing artefacts, not session scope. The parallel session itself flagged them as non-scope. Committing conflates measurement noise with substantive proof state. Regenerable if needed. | **Executed this session** — `git checkout --` in 007-lang. | - -## Task status snapshot (complete / pending / deferred) - -**Complete (22 tasks):** #1, #3, #4, #5, #6, #7, #8, #9, #10, #11, #12, #13, #14, #15, #16, #17, #32, #35, #36, #37, k9-svc spot-fix, echidna L3 Wave-1 + Wave-2. - -**Pending P0/P1:** #33, #34, 007-mcp-1/2, echidna Wave-2 CI verification, 6a2 reconciliation, Harvard DataExpr, sidecar revert, intend.k9.ncl adoption. - -**Deferred:** #2, #7b, P-04..P-07 proofs, Idris2 session types, DD-38 split, all Phase 3/4 items. diff --git a/docs/handover/HAIKU-SCOUT-PASS.adoc b/docs/handover/HAIKU-SCOUT-PASS.adoc new file mode 100644 index 00000000..d15607e3 --- /dev/null +++ b/docs/handover/HAIKU-SCOUT-PASS.adoc @@ -0,0 +1,58 @@ +== Haiku scout — next-pass lint / trivial-fix prompt + +Reusable template for a fast Haiku scouting pass over a concrete file +list after a coord task lands. Paired with the multi-agent coord flow +(Opus supervises, Haiku scouts, Sonnet/Opus writes). + +=== Template + +.... +Haiku scout, scope = + +1. Confirm each file compiles in isolation + (cargo check --lib / julia -e / zig build-obj / node --check). +2. Scan for: + - unused imports + - dead code + - `todo!()` / FIXME / XXX + - orphan type references + - stale doc comments referring to renamed symbols + - broken `mod.rs` exports +3. Trivially fix: + - unused imports (remove) + - obvious typos in comments + - unused `_var` renames +4. DO NOT fix — FLAG to me: + - any TODO with semantic content (may be load-bearing) + - anything requiring judgement about intent + - any cross-module change +5. Report: <20 lines. Green-light or list of blockers. +.... + +=== Invocation guidance + +* Scope must be an *exact file list* (no globs, no "`the changed +files`") so the scout has a finite surface. +* Compile-in-isolation check runs *before* the scan — a file that does +not parse on its own will generate too many false positives to be +useful. +* Category 4 (FLAG, don’t fix) is the firewall. `+local-coord-mcp+` +treats this as a tier-1 envelope from `+apprentice+` role: visible but +gated on master review before any change lands. +* The `+<20 lines+` cap keeps the report digestible in the master’s next +batch-review slot. + +=== Coord integration + +When run under the `+boj-server/cartridges/local-coord-mcp+` bus, the +scout should: + +[arabic] +. `+coord_register+` as `+kind=claude+`, `+variant=haiku-4.5+`, +`+declared_affinities=["scout", "lint"]+`. +. `+coord_claim_task+` with `+task_difficulty=routine+`, +`+dispatch_preference=auto+`. +. For each flagged item in step 4, emit `+coord_send_gated+` with +`+risk_tier=1+` (warn) — goes straight into the master’s quarantine. +. On completion, `+coord_report_outcome+` with the observed tag and +outcome so the track-record aggregates reflect scout accuracy. diff --git a/docs/handover/HAIKU-SCOUT-PASS.md b/docs/handover/HAIKU-SCOUT-PASS.md deleted file mode 100644 index 907eb2ab..00000000 --- a/docs/handover/HAIKU-SCOUT-PASS.md +++ /dev/null @@ -1,61 +0,0 @@ - -# Haiku scout — next-pass lint / trivial-fix prompt - -Reusable template for a fast Haiku scouting pass over a concrete file list -after a coord task lands. Paired with the multi-agent coord flow -(Opus supervises, Haiku scouts, Sonnet/Opus writes). - -## Template - -``` -Haiku scout, scope = - -1. Confirm each file compiles in isolation - (cargo check --lib / julia -e / zig build-obj / node --check). -2. Scan for: - - unused imports - - dead code - - `todo!()` / FIXME / XXX - - orphan type references - - stale doc comments referring to renamed symbols - - broken `mod.rs` exports -3. Trivially fix: - - unused imports (remove) - - obvious typos in comments - - unused `_var` renames -4. DO NOT fix — FLAG to me: - - any TODO with semantic content (may be load-bearing) - - anything requiring judgement about intent - - any cross-module change -5. Report: <20 lines. Green-light or list of blockers. -``` - -## Invocation guidance - -- Scope must be an **exact file list** (no globs, no "the changed files") - so the scout has a finite surface. -- Compile-in-isolation check runs **before** the scan — a file that does - not parse on its own will generate too many false positives to be - useful. -- Category 4 (FLAG, don't fix) is the firewall. `local-coord-mcp` - treats this as a tier-1 envelope from `apprentice` role: visible but - gated on master review before any change lands. -- The `<20 lines` cap keeps the report digestible in the master's - next batch-review slot. - -## Coord integration - -When run under the `boj-server/cartridges/local-coord-mcp` bus, the -scout should: - -1. `coord_register` as `kind=claude`, `variant=haiku-4.5`, - `declared_affinities=["scout", "lint"]`. -2. `coord_claim_task` with `task_difficulty=routine`, - `dispatch_preference=auto`. -3. For each flagged item in step 4, emit `coord_send_gated` with - `risk_tier=1` (warn) — goes straight into the master's quarantine. -4. On completion, `coord_report_outcome` with the observed tag and - outcome so the track-record aggregates reflect scout accuracy. diff --git a/docs/handover/README.adoc b/docs/handover/README.adoc new file mode 100644 index 00000000..45a38701 --- /dev/null +++ b/docs/handover/README.adoc @@ -0,0 +1,83 @@ +== BoJ-Server Handover Prompts + +This directory mirrors the coord-MCP continuation prompts that live on +the author’s Desktop so that a fresh clone has the full handover context +tracked in version control. + +[width="100%",cols="47%,53%",options="header",] +|=== +|File |Scope +|*`+COORD-MCP-TODO.md+`* |*Tight actionable backlog, P0→P4, with agreed +decisions D1–D4.* Start here. + +|*`+COORD-MCP-STATE.md+`* |*Where we are now vs the forward vision — +per-layer status + 38 decision summary.* + +|`+COORD-MCP-DESIGN-LOG.md+` |Full design log — rationale + DD-1..DD-38 ++ appendices A–M. Reference for the two summaries above. + +|`+COORD-MCP-HANDOFF-PROMPTS.md+` |Ready-to-paste prompts to re-seat a +new Claude session + +|`+COORD-MCP-PROMPT3-HANDOVER-2026-04-20.md+` |Prompt 3 (coord +finishers) handover — 11 tasks shipped + +|`+COORD-DESIGN-ORIGIN-CONVERSATION.txt+` |Raw conversation that seeded +the design (BFT safety, affinity routing, adaptive horizon, +chief-of-staff model). Historical. +|=== + +=== Source of truth + +The Desktop copies at `+~/Desktop/COORD-MCP-*.md+` remain the working +drafts the author edits in-session. These in-repo copies are the +canonical versions for anyone without Desktop access. + +When a session updates a handover prompt, update *both* the Desktop copy +and the in-repo copy in the same commit. If they drift, the in-repo copy +wins. + +=== Current status (as of 2026-04-20 session close, commit `+473733b+` on main) + +==== Complete (sixteen tasks) + +* *#1, #3–#8, #13–#17, #32, #35, #36, #37* — all shipped. +* 27 Zig tests + 41 Deno E2E assertions green. Panic-attack clean. +* Role terminology is now *master / journeyman / apprentice*; old names +accepted at the boundary for one release. +* Track-record + affinity routing, dispatch preference, reassignment +engine (server-origin quarantine), Nickel envelope validator, +`+coord_transfer_master+`, `+difficulty_hint+` envelope + Nickel +contract, `+proof:+` tag convention. +* 007-mcp family *#9–#12* marked complete in a separate repo +(`+The-Metadatastician/007+` commits `+62bdac0+`, `+018a1fd+`, +`+6bbb4f8+`, `+e753d10+`). + +==== Immediate next pick-up (Appendix L) + +* *Task #33* — `+client_kind+` + `+variant+` extension (Peer struct +changes, adapter JSON shape, log events 15 + 16). +* *Task #34* — capability advertisement. +* Estimate: ~1.5 days / ~5 commits. Concrete scope in +`+COORD-MCP-DESIGN-LOG.md+` Appendix L. + +==== Also pending + +* *Task #2* (low priority). +* *Proof Track P-04…P-07* Idris2 proofs in `+Durability.idr+`, ~6 days. +*Keystone: P-06 replay-equivalence*. + +=== Warning for parallel sessions (shared-state files) + +Do *not* run parallel sessions that both touch the `+Peer+` struct or +the `+ClientKind+` enum in the same wall-clock window: + +* `+cartridges/local-coord-mcp/ffi/local_coord_ffi.zig+` +* `+cartridges/local-coord-mcp/adapter/local_coord_adapter.zig+` + +Everything else can parallelise safely. + +=== Untracked from other tracks (intentionally not committed) + +* `+.machine_readable/contractiles/bust/+` — belongs to a different +lane. diff --git a/docs/handover/README.md b/docs/handover/README.md deleted file mode 100644 index 652bd0af..00000000 --- a/docs/handover/README.md +++ /dev/null @@ -1,72 +0,0 @@ - -# BoJ-Server Handover Prompts - -This directory mirrors the coord-MCP continuation prompts that live on -the author's Desktop so that a fresh clone has the full handover context -tracked in version control. - -| File | Scope | -|------|-------| -| **`COORD-MCP-TODO.md`** | **Tight actionable backlog, P0→P4, with agreed decisions D1–D4.** Start here. | -| **`COORD-MCP-STATE.md`** | **Where we are now vs the forward vision — per-layer status + 38 decision summary.** | -| `COORD-MCP-DESIGN-LOG.md` | Full design log — rationale + DD-1..DD-38 + appendices A–M. Reference for the two summaries above. | -| `COORD-MCP-HANDOFF-PROMPTS.md` | Ready-to-paste prompts to re-seat a new Claude session | -| `COORD-MCP-PROMPT3-HANDOVER-2026-04-20.md` | Prompt 3 (coord finishers) handover — 11 tasks shipped | -| `COORD-DESIGN-ORIGIN-CONVERSATION.txt` | Raw conversation that seeded the design (BFT safety, affinity routing, adaptive horizon, chief-of-staff model). Historical. | - -## Source of truth - -The Desktop copies at `~/Desktop/COORD-MCP-*.md` remain the working -drafts the author edits in-session. These in-repo copies are the -canonical versions for anyone without Desktop access. - -When a session updates a handover prompt, update **both** the Desktop -copy and the in-repo copy in the same commit. If they drift, the -in-repo copy wins. - -## Current status (as of 2026-04-20 session close, commit `473733b` on main) - -### Complete (sixteen tasks) - -- **#1, #3–#8, #13–#17, #32, #35, #36, #37** — all shipped. -- 27 Zig tests + 41 Deno E2E assertions green. Panic-attack clean. -- Role terminology is now **master / journeyman / apprentice**; - old names accepted at the boundary for one release. -- Track-record + affinity routing, dispatch preference, reassignment - engine (server-origin quarantine), Nickel envelope validator, - `coord_transfer_master`, `difficulty_hint` envelope + Nickel contract, - `proof:` tag convention. -- 007-mcp family **#9–#12** marked complete in a separate repo - (`The-Metadatastician/007` commits `62bdac0`, `018a1fd`, `6bbb4f8`, - `e753d10`). - -### Immediate next pick-up (Appendix L) - -- **Task #33** — `client_kind` + `variant` extension (Peer struct - changes, adapter JSON shape, log events 15 + 16). -- **Task #34** — capability advertisement. -- Estimate: ~1.5 days / ~5 commits. Concrete scope in - `COORD-MCP-DESIGN-LOG.md` Appendix L. - -### Also pending - -- **Task #2** (low priority). -- **Proof Track P-04…P-07** Idris2 proofs in `Durability.idr`, - ~6 days. **Keystone: P-06 replay-equivalence**. - -## Warning for parallel sessions (shared-state files) - -Do **not** run parallel sessions that both touch the `Peer` struct or -the `ClientKind` enum in the same wall-clock window: - -- `cartridges/local-coord-mcp/ffi/local_coord_ffi.zig` -- `cartridges/local-coord-mcp/adapter/local_coord_adapter.zig` - -Everything else can parallelise safely. - -## Untracked from other tracks (intentionally not committed) - -- `.machine_readable/contractiles/bust/` — belongs to a different lane. diff --git a/docs/integration/boj-side-observability-spec.adoc b/docs/integration/boj-side-observability-spec.adoc new file mode 100644 index 00000000..8c1281fb --- /dev/null +++ b/docs/integration/boj-side-observability-spec.adoc @@ -0,0 +1,504 @@ +== BoJ-side observability spec — Phase E §3 prerequisite + +*Version:* 0.1 (scaffold, Phase E) *Date:* 2026-06-22 *Status:* Phase E +scaffold. Names the BoJ-side telemetry events and Prometheus metrics +that the rollout-runbook §4.2 signals require, anchored to the exact +instrumentation sites in `+elixir/lib/boj_rest/router.ex+` so a +follow-up wiring PR has an unambiguous target. Until the events land BoJ +exposes no metrics; the runbook §1.4 prerequisite tracks this absence as +a stop-the-rollout condition for §3.1 (10% traffic). *ADR:* +link:../decisions/0004-adopt-http-capability-gateway.md[`+docs/decisions/0004-adopt-http-capability-gateway.md+`] +*Plan:* +link:http-capability-gateway-plan.md[`+docs/integration/http-capability-gateway-plan.md+`] +(§ Phase E, E3 telemetry verification) *Contract:* +link:http-capability-gateway-boj-contract.md[`+docs/integration/http-capability-gateway-boj-contract.md+`] +*Sister spec (gateway side):* +link:gateway-observability-spec.md[`+docs/integration/gateway-observability-spec.md+`] +(§ 3 BoJ-side signal templates anchor here) *Rollout runbook:* +link:hcg-tier2-rollout-runbook.md[`+docs/integration/hcg-tier2-rollout-runbook.md+`] +(§ 1.4 BoJ-side prerequisite, § 4.2 signals) *Tracking:* +https://github.com/hyperpolymath/standards/issues/91[`+standards#91+`] +(parent), +https://github.com/hyperpolymath/standards/issues/100[`+standards#100+`] +(Phase E) + +____ +*File-format note.* Matches sibling integration docs +(`+http-capability-gateway-{plan,audit,boj-contract,policy-authoring}.md+`, +`+gateway-{load-profile,observability-spec}.md+`, +`+hcg-tier2-rollout-runbook.md+`); the estate `+.adoc+` default is +deliberately overridden for the `+docs/integration/+` set so the +integration plan can name documents by exact path. +____ + +''''' + +=== 0. Scope + +The gateway-side observability spec (`+gateway-observability-spec.md+` +§3) lists three BoJ-side signals the rollout-runbook §4.2 expects +on-call to watch: + +[arabic] +. Per-route trust-class distribution from `+BojRest.Router+` decisions +(runbook §4.2 bullet 1). +. `+X-Trust-Level+` arriving from non-loopback peers — must remain zero +(runbook §4.2 bullet 2, paired with rollback trigger §5.1 row 4). +. BoJ 5xx rate, independent of the gateway’s view (runbook §4.2 bullet +3). + +The sister spec gives PromQL templates for each but cannot anchor them +to real metric names because BoJ today has no telemetry layer at all — +`+elixir/mix.exs+` carries no `+:telemetry_metrics_prometheus_core+` +dependency, `+BojRest.Application+` mounts no exporter, and +`+BojRest.Router+` emits no `+:telemetry.execute/3+` events. The sister +spec consequently leaves §3 templates qualified with +`+!OWNER: scaffold against the actual BojRest.Router instrumentation+`. +*This spec closes that qualification* by naming, normatively, the events +to emit, the metrics to export, and the exact instrumentation sites in +`+BojRest.Router+`. + +In scope: + +* The four telemetry events the BoJ side must emit (§ 1). +* The Prometheus metric names that the gateway-observability-spec §3 +PromQL templates expect (§ 2). +* The instrumentation sites in `+elixir/lib/boj_rest/router.ex+`, with +`+file:line+` anchors (§ 3). +* The `+mix.exs+` dependency, supervisor child, and `+Plug+` route the +wiring PR must add (§ 4). +* The `+/metrics+` exposure policy — which gateway-policy rule must +govern the new endpoint (§ 5). +* The acceptance criterion that the runbook §1.4 prerequisite checks +against (§ 6). + +Out of scope: + +* The actual code (mix.exs / application.ex / router.ex edits). +Implementation is a follow-up PR. +* Cartridge-level telemetry. The gateway sees BoJ as a single backend; +cartridge-internal observability is downstream of this scope. +* Long-term storage / retention. The spec defines what BoJ scrapes +expose; how long the scrapes are kept is an operator decision (the +runbook §4.3 dashboard URL row already covers that scope). + +''''' + +=== 1. Telemetry events — emission contract + +Four events. Each maps 1:1 to a runbook §4.2 signal. Event names follow +the prevailing `+[:app_namespace, :component, :verb]+` convention from +`+http-capability-gateway/lib/http_capability_gateway/application.ex+` +`+telemetry_metrics/0+`. + +[width="100%",cols="20%,20%,20%,20%,20%",options="header",] +|=== +|Event |Emitted when |Measurements |Metadata (tags) |Anchors +|`+[:boj_rest, :router, :decision]+` |After every +`+BojRest.Router.check_trust/3+` call, regardless of allow/deny outcome. +|`+count: 1+` |`+route+`, `+verb+`, `+trust_class+`, `+outcome+` +|runbook §4.2 bullet 1; sister spec §3.1 + +|`+[:boj_rest, :router, :trust_level_present]+` |When `+BojRest.Router+` +reads a non-empty `+X-Trust-Level+` header from the request. +|`+count: 1+` |`+remote_origin+` ∈ `+{loopback, non_loopback}+` |runbook +§4.2 bullet 2; sister spec §3.2; rollback trigger §5.1 row 4 + +|`+[:boj_rest, :http, :response]+` |At every `+json/3+` response render +in `+BojRest.Router+` (the single response site for the five governed +routes). |`+count: 1+`, `+duration: <µs>+` (monotonic time delta from +`+:request_received+` to render) |`+status+` (3-digit), `+route+` +|runbook §4.2 bullet 3; sister spec §3.3 + +|`+[:boj_rest, :request, :received]+` |At `+BojRest.Router+`’s `+match+` +plug, before dispatch. |`+count: 1+`, +`+received_at_ns: System.monotonic_time(:nanosecond)+` |`+verb+` +|duration anchor for `+[:boj_rest, :http, :response]+`; also the +canonical "`request entered BoJ`" signal +|=== + +==== 1.1 Tag value vocabularies + +Bounded vocabularies — no operator open-ended fields, so total +Prometheus time-series cardinality is bounded. + +* `+route+` ∈ the seven route patterns in `+BojRest.Router+` as of +contract v1.0 + the `+cartridge-sse-post+` addition (boj-server#165): +`+"/.well-known/boj-node-pubkey"+`, `+"/health"+`, `+"/menu"+`, +`+"/cartridges"+`, `+"/cartridge/:name"+`, +`+"/cartridge/:name/invoke"+`, `+"/cartridge/:name/sse"+`. *Route +templates, not concrete paths.* New routes added to `+BojRest.Router+` +must be added here in the same PR and reflected in the gateway policy +(the §1.5 surface-drift script will catch the policy half). +* `+verb+` ∈ `+{GET, POST, PUT, DELETE, PATCH, HEAD, OPTIONS}+` — same +seven-value allowlist the gateway uses +(`+http-capability-gateway/lib/http_capability_gateway/gateway.ex:65-77+`). +Unknown methods are dispatched by `+Plug.Router+` to the `+match _+` +fall-through, where this event is *not* emitted (no trust check runs; +nothing to bucket). +* `+trust_class+` ∈ `+{public, authenticated, internal}+` — the three +values `+BojRest.TrustPolicy.required_exposure/1+` can return +(`+elixir/lib/boj_rest/trust_policy.ex:52-67+`). This is the *required* +exposure for the matched cartridge, not the *presented* trust header. +* `+outcome+` ∈ +`+{allow, deny_insufficient_trust, deny_unknown_route, deny_loopback_required}+` +— `+allow+` is the `+:ok+` branch of `+check_trust/3+`; the three deny +variants split the `+:error+` branches per `+router.ex:135-138+` (the +unknown-route deny is the `+Plug.Router+` `+match _+` 404 site, which +also emits this event so the dashboards see the no-match traffic). +* `+remote_origin+` ∈ `+{loopback, non_loopback}+` — derived from +`+BojRest.Router.loopback?/1+` (`+router.ex:238-240+`). +* `+status+` — 3-digit HTTP status. Naturally bounded; cardinality is +the set of statuses BoJ actually returns (today: +`+{200, 400, 403, 404, 500, 502}+`). + +Total decision counter cardinality bound: +`+7 routes × 7 verbs × 3 trust_classes × 4 outcomes = 588+` time series +at saturation, of which only the realistically-occupied combinations +(~`+7 routes × 1-2 verbs each × 2 trust_classes × 2 outcomes ≈ 28+`) +will ever materialise. Well below the threshold where Prometheus storage +becomes a concern. + +==== 1.2 What is *not* emitted from these sites + +Explicit non-emission to keep the contract sharp: + +* The events above do *not* carry the `+X-Trust-Level+` header value +itself. The gateway-observability-spec §3.2 query relies on the +presence-vs-absence + origin tag, not the header value, to keep the spec +resilient to the future mTLS-primary path (where the trust value never +arrives in a header at all). +* The events above do *not* carry the `+X-Node-Identity+` value or any +cartridge credential material. Audit-log payloads stay in the BoJ logs +and VeriSimDB; metrics expose only the bounded-vocabulary tags above. +* The events above do *not* carry per-request payload size. Throughput +and latency are sufficient for the §4.2 signal set; payload-size +histograms are out of scope. + +''''' + +=== 2. Prometheus metrics — what the sister spec PromQL reads + +The `+telemetry_metrics_prometheus_core+` convention applies: dots +become underscores; distribution metrics expose `+_bucket+`, `+_count+`, +`+_sum+` series; counters expose a `+_total+` series. The names below +are the metric prefixes the operator sees in `+/metrics+`. + +[width="100%",cols="25%,25%,25%,25%",options="header",] +|=== +|Telemetry event |Prometheus metric prefix |Type |Tags +|`+[:boj_rest, :router, :decision]+` |`+boj_router_decision_count+` +|counter |`+route+`, `+verb+`, `+trust_class+`, `+outcome+` + +|`+[:boj_rest, :router, :trust_level_present]+` +|`+boj_router_trust_level_present_count+` |counter |`+remote_origin+` + +|`+[:boj_rest, :http, :response]+` |`+boj_http_responses+` |counter +|`+status+`, `+route+` + +|`+[:boj_rest, :http, :response]+` |`+boj_http_response_duration+` +|distribution |`+status+`, `+route+` + +|`+[:boj_rest, :request, :received]+` |`+boj_request_received_count+` +|counter |`+verb+` +|=== + +==== 2.1 Name alignment with the sister spec + +The sister spec (`+gateway-observability-spec.md+`) names three metrics +as the targets of its §3 PromQL templates: + +* `+boj_router_decision_count_total{route, trust_class}+` — sister §3.1. +* `+boj_router_trust_level_present_count_total{remote_origin}+` — sister +§3.2. +* `+boj_http_responses_total{status=~"5.."}+` — sister §3.3. + +This spec’s §2 table is *exactly* that set, with the `+outcome+` and +`+verb+` tags added on the decision counter (which the sister-spec §3.1 +query happily ignores via PromQL’s `+sum by+`) and the duration +histogram added on the response side (which the sister-spec §3.3 query +does not currently use but which is the natural BoJ-side equivalent of +the gateway’s `+backend_response_duration+` — useful for the dashboard, +not strictly required for §3.1 sign-off). + +==== 2.2 Histogram buckets + +`+boj_http_response_duration+` buckets, in microseconds, cover the BoJ +side of the 60 ms ceiling the gateway uses for its +`+backend_response_duration+` (sister spec §1.1): + +.... +[100, 500, 1_000, 5_000, 10_000, 30_000, 60_000] +.... + +These match the gateway-side bucket layout for +`+backend_response_duration+` so the operator can read both axes against +the same scale on a shared dashboard. Wider upper bound for the BoJ side +would only hide cartridge-attributable latency the gateway already +buckets up to 60 ms; tighter bounds would lose the cartridge tail. + +''''' + +=== 3. Instrumentation sites + +`+elixir/lib/boj_rest/router.ex+` (256 lines as of 2026-06-22). The four +emission sites: + +==== 3.1 `+[:boj_rest, :request, :received]+` + +*Site:* `+Router.match+` plug entry, between line `+29+` +(`+plug :match+`) and line `+31+` (`+plug :dispatch+`). Add a thin +custom plug `+:emit_request_received+` before `+:dispatch+` that runs +once per request and stashes the monotonic start time in +`+conn.private+` for the response-site duration read. + +==== 3.2 `+[:boj_rest, :router, :trust_level_present]+` + +*Sites:* `+router.ex:109+` (`+POST /cartridge/:name/invoke+`) and +`+router.ex:161+` (`+POST /cartridge/:name/sse+`), immediately after the +`+trust_level = conn |> get_req_header(...) |> List.first()+` read. Emit +*only* when `+trust_level+` is non-nil; passing the empty case through +is fine — the metric counts presences, not absences. The +`+remote_origin+` tag derives from the already-computed +`+is_local = loopback?(conn.remote_ip)+`. + +==== 3.3 `+[:boj_rest, :router, :decision]+` + +*Site:* `+Router.check_trust/3+` (`+router.ex:228-236+`). Emit on *both* +the `+:ok+` and `+{:error, :insufficient_trust}+` return branches, +immediately before the function returns, so allow and deny outcomes go +through the same code path. The `+route+` tag is the matched route +template; `+Router+` already has it in scope via the macro expansion +(the `+:plug_route+` private key — `+conn.private[:plug_route]+` gives +`+{template_string, fn}+`). The unknown-route 404 case (the +`+Plug.Router+` `+match _+` fall-through) is a separate emission site +below. + +*Site (unknown-route):* the `+match _+` fall-through at the bottom of +the router (existing 404 handler). Emit +`+[:boj_rest, :router, :decision]+` with `+route = ""+`, +`+verb = conn.method+`, `+trust_class = :public+` (placeholder — no +cartridge matched, so no required exposure), +`+outcome = :deny_unknown_route+`. This keeps the unknown-path traffic +visible to the dashboard’s deny-rate panel. + +==== 3.4 `+[:boj_rest, :http, :response]+` + +*Site:* `+Router.json/3+` (`+router.ex:213-217+`). Emit immediately +before `+send_resp+`. The `+duration+` measurement reads +`+received_at_ns+` from `+conn.private+` (set by §3.1’s plug); if absent +(a code path that does not pass through the new request-received plug), +emit `+duration: 0+` and rely on the dashboard alert to surface the +wiring gap. + +==== 3.5 Total diff size + +The wiring PR adds, approximately: + +* `+elixir/mix.exs+`: one dep line. +* `+elixir/lib/boj_rest/application.ex+`: one supervisor child (the +Prometheus reporter), one new plug wiring (the `+/metrics+` route — see +§5). +* `+elixir/lib/boj_rest/router.ex+`: four `+:telemetry.execute/3+` calls +(or one helper invoked from four sites), one new plug, one +`+conn.private+` stash. +* `+elixir/lib/boj_rest/telemetry.ex+` (new): the `+metrics/0+` +declaration that names the five Prometheus metrics in §2 above. + +No structural change to `+BojRest.TrustPolicy+`, `+BojRest.Invoker+`, +`+BojRest.JsInvoker+`, `+BojRest.Catalog+`, or the cartridges. The +emissions slot into the existing `+Router+` plug chain without altering +decision logic. + +''''' + +=== 4. Dependencies and supervisor wiring + +==== 4.1 `+mix.exs+` + +Add to `+defp deps+`: + +[source,elixir] +---- +{:telemetry_metrics_prometheus_core, "~> 1.2"} +---- + +The `+:telemetry+` dep is already transitively pulled in by +`+:plug_cowboy+` (which uses it for its own request lifecycle events). +The reporter library is the smallest dependency that exports +`+:telemetry_metrics+` definitions to Prometheus text format; it does +*not* start its own HTTP server (the next dep up the stack, +`+:telemetry_metrics_prometheus+`, does, which would conflict with the +existing Cowboy listener). The `+_core+` variant is the one to pick for +a Cowboy-already-listening process. + +==== 4.2 `+BojRest.Application+` + +Add one child to the supervisor tree (between `+BojRest.JsWorkerPool+` +and the Cowboy listener): + +[source,elixir] +---- +{TelemetryMetricsPrometheus.Core, [ + name: :boj_rest_prometheus, + metrics: BojRest.Telemetry.metrics() +]} +---- + +This starts the reporter; it does *not* add a listener. The `+/metrics+` +route is added inside `+BojRest.Router+` (§5). + +==== 4.3 `+BojRest.Telemetry+` + +New module `+elixir/lib/boj_rest/telemetry.ex+` exporting `+metrics/0+` +that returns the five `+Telemetry.Metrics+` definitions matching §2: + +[source,elixir] +---- +def metrics do + [ + last_value("boj_router_decision_count", + event_name: [:boj_rest, :router, :decision], + measurement: :count, + tags: [:route, :verb, :trust_class, :outcome] + ), + # … one entry per row in §2 table. + ] +end +---- + +The exact `+Telemetry.Metrics+` type per row matches the §2 table +(counter / distribution). + +''''' + +=== 5. `+/metrics+` exposure — policy implication + +The `+/metrics+` endpoint is the new BoJ surface route the wiring PR +adds. The gateway policy (`+config/gateway-policy-boj.yaml+`) must +govern it. *The endpoint MUST NOT be public.* Two reasons: + +[arabic] +. The metrics payload exposes route-level traffic shape — useful to an +attacker probing for capability-existence, the exact threat the +`+stealth_profiles+` mechanism in the policy guards against. +. The `+decision+` counter tags include the cartridge-trust-class +distribution, which leaks information about which cartridges expect +authenticated callers and which expect internal-only callers — a +capability-discovery signal that defeats the audit’s truthfulness +invariant. + +Declared policy rule (additions, when the wiring PR lands): + +[source,yaml] +---- +- id: "metrics-get" + description: "BoJ Prometheus scrape — internal-only, stealth 404 on untrusted access" + path: "/metrics" + verb: "GET" + trust_class: "internal" + stealth_profile: "internal-404" +---- + +Operator scrapes must reach `+/metrics+` from a trusted-proxy IP with +`+X-Trust-Level: internal+` set by the gateway. Tier-1 (Cloudflare) MUST +NOT route external traffic to `+/metrics+`; the gateway’s +stealth-profile rule is the second line of defence if a Cloudflare +config drift exposes it. + +The `+scripts/hcg-policy-smoke.sh+` stealth-profile canary set (§1.5 of +the rollout runbook) must add `+/metrics+` to its internal-route 404 +canary list in the same PR that adds the policy rule — so a +misconfiguration that demotes `+metrics-get+` from `+internal+stealth+` +to `+authenticated+403+` is caught at smoke time, not at exposure time. + +''''' + +=== 6. Phase E acceptance — how this spec gates §3.1 + +This spec adds one new prerequisite to the rollout-runbook §1.4 +"`BoJ-side prerequisites`" checklist: + +____ +*BoJ-side observability events emitted.* The four events declared in +this spec § 1 are emitted by `+BojRest.Router+`, the five Prometheus +metrics in § 2 appear in a `+/metrics+` scrape, the policy rule in § 5 +governs the endpoint, and the smoke script’s stealth canary covers it. +Until this lands, runbook § 3.1 success criterion 4 ("`No +`+X-Trust-Level+` mismatches in BoJ access logs`") is unobservable via +Prometheus — the only signal path is BoJ structured logs, which the +runbook § 4 dashboards do not currently consume. +____ + +The runbook update (this PR’s §1.4 edit) adds that checkbox to the +existing list. + +*Strictness:* until the wiring PR lands, runbook § 3.1 (10% traffic +shift) *cannot* be signed off against the Prometheus-anchored success +criteria. The rollback trigger § 5.1 row 4 ("`BoJ access logs show +`+X-Trust-Level+` from non-loopback peers`") falls back to log-based +detection — which is acceptable as a manual triage path but *not* +acceptable as the only paged signal for a §3 invariant 4 violation in +flight. + +''''' + +=== 7. Wiring PR — checklist for the follow-up + +The follow-up PR that lands the actual emission must: + +* [ ] Add `+:telemetry_metrics_prometheus_core+` to `+elixir/mix.exs+` +`+deps/0+`. +* [ ] Add `+BojRest.Telemetry+` module exporting `+metrics/0+` per § +4.3. +* [ ] Add the `+TelemetryMetricsPrometheus.Core+` child to +`+BojRest.Application+` per § 4.2. +* [ ] Add the four emission sites in `+elixir/lib/boj_rest/router.ex+` +per § 3.1–3.4. +* [ ] Add a `+GET /metrics+` plug route in `+BojRest.Router+` that calls +`+TelemetryMetricsPrometheus.Core.scrape(:boj_rest_prometheus)+` and +returns the text-format payload. +* [ ] Add the `+metrics-get+` rule to `+config/gateway-policy-boj.yaml+` +per § 5. +* [ ] Add `+/metrics+` to `+scripts/hcg-policy-smoke.sh+` stealth canary +list per § 5. +* [ ] Add a unit test exercising each `+:telemetry.execute/3+` site +(matching the existing test patterns under `+elixir/test/+`). +* [ ] Update `+gateway-observability-spec.md+` § 3.1, § 3.2, § 3.3 to +drop the `+!OWNER: scaffold+` qualifier (the metric names now resolve +against this repo’s emitters). +* [ ] Update `+hcg-tier2-rollout-runbook.md+` § 1.4 to flip the new +prerequisite checkbox from `+[ ]+` to `+[x]+`. + +The same PR closes this spec’s `+**Status:**+` line from `+scaffold+` to +`+wired (Phase E §1.4 prereq satisfied)+` with a date. + +''''' + +=== 8. References + +* Sister spec — `+docs/integration/gateway-observability-spec.md+` (§ 3 +BoJ-side PromQL templates; this spec backs those templates with real +metric names). +* Rollout runbook — `+docs/integration/hcg-tier2-rollout-runbook.md+` (§ +1.4 BoJ-side prerequisites, § 3.1 success criteria, § 4.2 signals, § 5.1 +rollback triggers). +* Contract — +`+docs/integration/http-capability-gateway-boj-contract.md+` (§ 3 +trust-level invariants this spec’s `+decision+` and +`+trust_level_present+` events enforce telemetry coverage for). +* Audit — `+docs/integration/http-capability-gateway-audit.md+` (§ 5 +gateway-side telemetry shape, mirrored on the BoJ side here). +* Plan — `+docs/integration/http-capability-gateway-plan.md+` (§ Phase E +E3 telemetry verification). +* BoJ router — `+elixir/lib/boj_rest/router.ex+` (instrumentation sites +§ 3.1–3.4). +* BoJ trust policy — `+elixir/lib/boj_rest/trust_policy.ex+` +(`+required_exposure/1+` source of `+trust_class+` tag values; +`+satisfies?/3+` source of `+outcome+` tag values). +* BoJ application — `+elixir/lib/boj_rest/application.ex+` (supervisor +tree to extend per § 4.2). +* Gateway metric definitions — +`+http-capability-gateway/lib/http_capability_gateway/application.ex+` +`+telemetry_metrics/0+` (lines 259–296; bucket-layout reference for § +2.2). diff --git a/docs/integration/boj-side-observability-spec.md b/docs/integration/boj-side-observability-spec.md deleted file mode 100644 index f464a2cf..00000000 --- a/docs/integration/boj-side-observability-spec.md +++ /dev/null @@ -1,264 +0,0 @@ - - - -# BoJ-side observability spec — Phase E §3 prerequisite - -**Version:** 0.1 (scaffold, Phase E) -**Date:** 2026-06-22 -**Status:** Phase E scaffold. Names the BoJ-side telemetry events and Prometheus metrics that the rollout-runbook §4.2 signals require, anchored to the exact instrumentation sites in `elixir/lib/boj_rest/router.ex` so a follow-up wiring PR has an unambiguous target. Until the events land BoJ exposes no metrics; the runbook §1.4 prerequisite tracks this absence as a stop-the-rollout condition for §3.1 (10% traffic). -**ADR:** [`docs/decisions/0004-adopt-http-capability-gateway.md`](../decisions/0004-adopt-http-capability-gateway.md) -**Plan:** [`docs/integration/http-capability-gateway-plan.md`](http-capability-gateway-plan.md) (§ Phase E, E3 telemetry verification) -**Contract:** [`docs/integration/http-capability-gateway-boj-contract.md`](http-capability-gateway-boj-contract.md) -**Sister spec (gateway side):** [`docs/integration/gateway-observability-spec.md`](gateway-observability-spec.md) (§ 3 BoJ-side signal templates anchor here) -**Rollout runbook:** [`docs/integration/hcg-tier2-rollout-runbook.md`](hcg-tier2-rollout-runbook.md) (§ 1.4 BoJ-side prerequisite, § 4.2 signals) -**Tracking:** [`standards#91`](https://github.com/hyperpolymath/standards/issues/91) (parent), [`standards#100`](https://github.com/hyperpolymath/standards/issues/100) (Phase E) - -> **File-format note.** Matches sibling integration docs (`http-capability-gateway-{plan,audit,boj-contract,policy-authoring}.md`, `gateway-{load-profile,observability-spec}.md`, `hcg-tier2-rollout-runbook.md`); the estate `.adoc` default is deliberately overridden for the `docs/integration/` set so the integration plan can name documents by exact path. - ---- - -## 0. Scope - -The gateway-side observability spec (`gateway-observability-spec.md` §3) lists three BoJ-side signals the rollout-runbook §4.2 expects on-call to watch: - -1. Per-route trust-class distribution from `BojRest.Router` decisions (runbook §4.2 bullet 1). -2. `X-Trust-Level` arriving from non-loopback peers — must remain zero (runbook §4.2 bullet 2, paired with rollback trigger §5.1 row 4). -3. BoJ 5xx rate, independent of the gateway's view (runbook §4.2 bullet 3). - -The sister spec gives PromQL templates for each but cannot anchor them to real metric names because BoJ today has no telemetry layer at all — `elixir/mix.exs` carries no `:telemetry_metrics_prometheus_core` dependency, `BojRest.Application` mounts no exporter, and `BojRest.Router` emits no `:telemetry.execute/3` events. The sister spec consequently leaves §3 templates qualified with `!OWNER: scaffold against the actual BojRest.Router instrumentation`. **This spec closes that qualification** by naming, normatively, the events to emit, the metrics to export, and the exact instrumentation sites in `BojRest.Router`. - -In scope: - -- The four telemetry events the BoJ side must emit (§ 1). -- The Prometheus metric names that the gateway-observability-spec §3 PromQL templates expect (§ 2). -- The instrumentation sites in `elixir/lib/boj_rest/router.ex`, with `file:line` anchors (§ 3). -- The `mix.exs` dependency, supervisor child, and `Plug` route the wiring PR must add (§ 4). -- The `/metrics` exposure policy — which gateway-policy rule must govern the new endpoint (§ 5). -- The acceptance criterion that the runbook §1.4 prerequisite checks against (§ 6). - -Out of scope: - -- The actual code (mix.exs / application.ex / router.ex edits). Implementation is a follow-up PR. -- Cartridge-level telemetry. The gateway sees BoJ as a single backend; cartridge-internal observability is downstream of this scope. -- Long-term storage / retention. The spec defines what BoJ scrapes expose; how long the scrapes are kept is an operator decision (the runbook §4.3 dashboard URL row already covers that scope). - ---- - -## 1. Telemetry events — emission contract - -Four events. Each maps 1:1 to a runbook §4.2 signal. Event names follow the prevailing `[:app_namespace, :component, :verb]` convention from `http-capability-gateway/lib/http_capability_gateway/application.ex` `telemetry_metrics/0`. - -| Event | Emitted when | Measurements | Metadata (tags) | Anchors | -|---|---|---|---|---| -| `[:boj_rest, :router, :decision]` | After every `BojRest.Router.check_trust/3` call, regardless of allow/deny outcome. | `count: 1` | `route`, `verb`, `trust_class`, `outcome` | runbook §4.2 bullet 1; sister spec §3.1 | -| `[:boj_rest, :router, :trust_level_present]` | When `BojRest.Router` reads a non-empty `X-Trust-Level` header from the request. | `count: 1` | `remote_origin` ∈ `{loopback, non_loopback}` | runbook §4.2 bullet 2; sister spec §3.2; rollback trigger §5.1 row 4 | -| `[:boj_rest, :http, :response]` | At every `json/3` response render in `BojRest.Router` (the single response site for the five governed routes). | `count: 1`, `duration: <µs>` (monotonic time delta from `:request_received` to render) | `status` (3-digit), `route` | runbook §4.2 bullet 3; sister spec §3.3 | -| `[:boj_rest, :request, :received]` | At `BojRest.Router`'s `match` plug, before dispatch. | `count: 1`, `received_at_ns: System.monotonic_time(:nanosecond)` | `verb` | duration anchor for `[:boj_rest, :http, :response]`; also the canonical "request entered BoJ" signal | - -### 1.1 Tag value vocabularies - -Bounded vocabularies — no operator open-ended fields, so total Prometheus time-series cardinality is bounded. - -- `route` ∈ the seven route patterns in `BojRest.Router` as of contract v1.0 + the `cartridge-sse-post` addition (boj-server#165): `"/.well-known/boj-node-pubkey"`, `"/health"`, `"/menu"`, `"/cartridges"`, `"/cartridge/:name"`, `"/cartridge/:name/invoke"`, `"/cartridge/:name/sse"`. **Route templates, not concrete paths.** New routes added to `BojRest.Router` must be added here in the same PR and reflected in the gateway policy (the §1.5 surface-drift script will catch the policy half). -- `verb` ∈ `{GET, POST, PUT, DELETE, PATCH, HEAD, OPTIONS}` — same seven-value allowlist the gateway uses (`http-capability-gateway/lib/http_capability_gateway/gateway.ex:65-77`). Unknown methods are dispatched by `Plug.Router` to the `match _` fall-through, where this event is **not** emitted (no trust check runs; nothing to bucket). -- `trust_class` ∈ `{public, authenticated, internal}` — the three values `BojRest.TrustPolicy.required_exposure/1` can return (`elixir/lib/boj_rest/trust_policy.ex:52-67`). This is the **required** exposure for the matched cartridge, not the **presented** trust header. -- `outcome` ∈ `{allow, deny_insufficient_trust, deny_unknown_route, deny_loopback_required}` — `allow` is the `:ok` branch of `check_trust/3`; the three deny variants split the `:error` branches per `router.ex:135-138` (the unknown-route deny is the `Plug.Router` `match _` 404 site, which also emits this event so the dashboards see the no-match traffic). -- `remote_origin` ∈ `{loopback, non_loopback}` — derived from `BojRest.Router.loopback?/1` (`router.ex:238-240`). -- `status` — 3-digit HTTP status. Naturally bounded; cardinality is the set of statuses BoJ actually returns (today: `{200, 400, 403, 404, 500, 502}`). - -Total decision counter cardinality bound: `7 routes × 7 verbs × 3 trust_classes × 4 outcomes = 588` time series at saturation, of which only the realistically-occupied combinations (~`7 routes × 1-2 verbs each × 2 trust_classes × 2 outcomes ≈ 28`) will ever materialise. Well below the threshold where Prometheus storage becomes a concern. - -### 1.2 What is **not** emitted from these sites - -Explicit non-emission to keep the contract sharp: - -- The events above do **not** carry the `X-Trust-Level` header value itself. The gateway-observability-spec §3.2 query relies on the presence-vs-absence + origin tag, not the header value, to keep the spec resilient to the future mTLS-primary path (where the trust value never arrives in a header at all). -- The events above do **not** carry the `X-Node-Identity` value or any cartridge credential material. Audit-log payloads stay in the BoJ logs and VeriSimDB; metrics expose only the bounded-vocabulary tags above. -- The events above do **not** carry per-request payload size. Throughput and latency are sufficient for the §4.2 signal set; payload-size histograms are out of scope. - ---- - -## 2. Prometheus metrics — what the sister spec PromQL reads - -The `telemetry_metrics_prometheus_core` convention applies: dots become underscores; distribution metrics expose `_bucket`, `_count`, `_sum` series; counters expose a `_total` series. The names below are the metric prefixes the operator sees in `/metrics`. - -| Telemetry event | Prometheus metric prefix | Type | Tags | -|---|---|---|---| -| `[:boj_rest, :router, :decision]` | `boj_router_decision_count` | counter | `route`, `verb`, `trust_class`, `outcome` | -| `[:boj_rest, :router, :trust_level_present]` | `boj_router_trust_level_present_count` | counter | `remote_origin` | -| `[:boj_rest, :http, :response]` | `boj_http_responses` | counter | `status`, `route` | -| `[:boj_rest, :http, :response]` | `boj_http_response_duration` | distribution | `status`, `route` | -| `[:boj_rest, :request, :received]` | `boj_request_received_count` | counter | `verb` | - -### 2.1 Name alignment with the sister spec - -The sister spec (`gateway-observability-spec.md`) names three metrics as the targets of its §3 PromQL templates: - -- `boj_router_decision_count_total{route, trust_class}` — sister §3.1. -- `boj_router_trust_level_present_count_total{remote_origin}` — sister §3.2. -- `boj_http_responses_total{status=~"5.."}` — sister §3.3. - -This spec's §2 table is **exactly** that set, with the `outcome` and `verb` tags added on the decision counter (which the sister-spec §3.1 query happily ignores via PromQL's `sum by`) and the duration histogram added on the response side (which the sister-spec §3.3 query does not currently use but which is the natural BoJ-side equivalent of the gateway's `backend_response_duration` — useful for the dashboard, not strictly required for §3.1 sign-off). - -### 2.2 Histogram buckets - -`boj_http_response_duration` buckets, in microseconds, cover the BoJ side of the 60 ms ceiling the gateway uses for its `backend_response_duration` (sister spec §1.1): - -``` -[100, 500, 1_000, 5_000, 10_000, 30_000, 60_000] -``` - -These match the gateway-side bucket layout for `backend_response_duration` so the operator can read both axes against the same scale on a shared dashboard. Wider upper bound for the BoJ side would only hide cartridge-attributable latency the gateway already buckets up to 60 ms; tighter bounds would lose the cartridge tail. - ---- - -## 3. Instrumentation sites - -`elixir/lib/boj_rest/router.ex` (256 lines as of 2026-06-22). The four emission sites: - -### 3.1 `[:boj_rest, :request, :received]` - -**Site:** `Router.match` plug entry, between line `29` (`plug :match`) and line `31` (`plug :dispatch`). Add a thin custom plug `:emit_request_received` before `:dispatch` that runs once per request and stashes the monotonic start time in `conn.private` for the response-site duration read. - -### 3.2 `[:boj_rest, :router, :trust_level_present]` - -**Sites:** `router.ex:109` (`POST /cartridge/:name/invoke`) and `router.ex:161` (`POST /cartridge/:name/sse`), immediately after the `trust_level = conn |> get_req_header(...) |> List.first()` read. Emit **only** when `trust_level` is non-nil; passing the empty case through is fine — the metric counts presences, not absences. The `remote_origin` tag derives from the already-computed `is_local = loopback?(conn.remote_ip)`. - -### 3.3 `[:boj_rest, :router, :decision]` - -**Site:** `Router.check_trust/3` (`router.ex:228-236`). Emit on **both** the `:ok` and `{:error, :insufficient_trust}` return branches, immediately before the function returns, so allow and deny outcomes go through the same code path. The `route` tag is the matched route template; `Router` already has it in scope via the macro expansion (the `:plug_route` private key — `conn.private[:plug_route]` gives `{template_string, fn}`). The unknown-route 404 case (the `Plug.Router` `match _` fall-through) is a separate emission site below. - -**Site (unknown-route):** the `match _` fall-through at the bottom of the router (existing 404 handler). Emit `[:boj_rest, :router, :decision]` with `route = ""`, `verb = conn.method`, `trust_class = :public` (placeholder — no cartridge matched, so no required exposure), `outcome = :deny_unknown_route`. This keeps the unknown-path traffic visible to the dashboard's deny-rate panel. - -### 3.4 `[:boj_rest, :http, :response]` - -**Site:** `Router.json/3` (`router.ex:213-217`). Emit immediately before `send_resp`. The `duration` measurement reads `received_at_ns` from `conn.private` (set by §3.1's plug); if absent (a code path that does not pass through the new request-received plug), emit `duration: 0` and rely on the dashboard alert to surface the wiring gap. - -### 3.5 Total diff size - -The wiring PR adds, approximately: - -- `elixir/mix.exs`: one dep line. -- `elixir/lib/boj_rest/application.ex`: one supervisor child (the Prometheus reporter), one new plug wiring (the `/metrics` route — see §5). -- `elixir/lib/boj_rest/router.ex`: four `:telemetry.execute/3` calls (or one helper invoked from four sites), one new plug, one `conn.private` stash. -- `elixir/lib/boj_rest/telemetry.ex` (new): the `metrics/0` declaration that names the five Prometheus metrics in §2 above. - -No structural change to `BojRest.TrustPolicy`, `BojRest.Invoker`, `BojRest.JsInvoker`, `BojRest.Catalog`, or the cartridges. The emissions slot into the existing `Router` plug chain without altering decision logic. - ---- - -## 4. Dependencies and supervisor wiring - -### 4.1 `mix.exs` - -Add to `defp deps`: - -```elixir -{:telemetry_metrics_prometheus_core, "~> 1.2"} -``` - -The `:telemetry` dep is already transitively pulled in by `:plug_cowboy` (which uses it for its own request lifecycle events). The reporter library is the smallest dependency that exports `:telemetry_metrics` definitions to Prometheus text format; it does **not** start its own HTTP server (the next dep up the stack, `:telemetry_metrics_prometheus`, does, which would conflict with the existing Cowboy listener). The `_core` variant is the one to pick for a Cowboy-already-listening process. - -### 4.2 `BojRest.Application` - -Add one child to the supervisor tree (between `BojRest.JsWorkerPool` and the Cowboy listener): - -```elixir -{TelemetryMetricsPrometheus.Core, [ - name: :boj_rest_prometheus, - metrics: BojRest.Telemetry.metrics() -]} -``` - -This starts the reporter; it does **not** add a listener. The `/metrics` route is added inside `BojRest.Router` (§5). - -### 4.3 `BojRest.Telemetry` - -New module `elixir/lib/boj_rest/telemetry.ex` exporting `metrics/0` that returns the five `Telemetry.Metrics` definitions matching §2: - -```elixir -def metrics do - [ - last_value("boj_router_decision_count", - event_name: [:boj_rest, :router, :decision], - measurement: :count, - tags: [:route, :verb, :trust_class, :outcome] - ), - # … one entry per row in §2 table. - ] -end -``` - -The exact `Telemetry.Metrics` type per row matches the §2 table (counter / distribution). - ---- - -## 5. `/metrics` exposure — policy implication - -The `/metrics` endpoint is the new BoJ surface route the wiring PR adds. The gateway policy (`config/gateway-policy-boj.yaml`) must govern it. **The endpoint MUST NOT be public.** Two reasons: - -1. The metrics payload exposes route-level traffic shape — useful to an attacker probing for capability-existence, the exact threat the `stealth_profiles` mechanism in the policy guards against. -2. The `decision` counter tags include the cartridge-trust-class distribution, which leaks information about which cartridges expect authenticated callers and which expect internal-only callers — a capability-discovery signal that defeats the audit's truthfulness invariant. - -Declared policy rule (additions, when the wiring PR lands): - -```yaml -- id: "metrics-get" - description: "BoJ Prometheus scrape — internal-only, stealth 404 on untrusted access" - path: "/metrics" - verb: "GET" - trust_class: "internal" - stealth_profile: "internal-404" -``` - -Operator scrapes must reach `/metrics` from a trusted-proxy IP with `X-Trust-Level: internal` set by the gateway. Tier-1 (Cloudflare) MUST NOT route external traffic to `/metrics`; the gateway's stealth-profile rule is the second line of defence if a Cloudflare config drift exposes it. - -The `scripts/hcg-policy-smoke.sh` stealth-profile canary set (§1.5 of the rollout runbook) must add `/metrics` to its internal-route 404 canary list in the same PR that adds the policy rule — so a misconfiguration that demotes `metrics-get` from `internal+stealth` to `authenticated+403` is caught at smoke time, not at exposure time. - ---- - -## 6. Phase E acceptance — how this spec gates §3.1 - -This spec adds one new prerequisite to the rollout-runbook §1.4 "BoJ-side prerequisites" checklist: - -> **BoJ-side observability events emitted.** The four events declared in this spec § 1 are emitted by `BojRest.Router`, the five Prometheus metrics in § 2 appear in a `/metrics` scrape, the policy rule in § 5 governs the endpoint, and the smoke script's stealth canary covers it. Until this lands, runbook § 3.1 success criterion 4 ("No `X-Trust-Level` mismatches in BoJ access logs") is unobservable via Prometheus — the only signal path is BoJ structured logs, which the runbook § 4 dashboards do not currently consume. - -The runbook update (this PR's §1.4 edit) adds that checkbox to the existing list. - -**Strictness:** until the wiring PR lands, runbook § 3.1 (10% traffic shift) **cannot** be signed off against the Prometheus-anchored success criteria. The rollback trigger § 5.1 row 4 ("BoJ access logs show `X-Trust-Level` from non-loopback peers") falls back to log-based detection — which is acceptable as a manual triage path but **not** acceptable as the only paged signal for a §3 invariant 4 violation in flight. - ---- - -## 7. Wiring PR — checklist for the follow-up - -The follow-up PR that lands the actual emission must: - -- [ ] Add `:telemetry_metrics_prometheus_core` to `elixir/mix.exs` `deps/0`. -- [ ] Add `BojRest.Telemetry` module exporting `metrics/0` per § 4.3. -- [ ] Add the `TelemetryMetricsPrometheus.Core` child to `BojRest.Application` per § 4.2. -- [ ] Add the four emission sites in `elixir/lib/boj_rest/router.ex` per § 3.1–3.4. -- [ ] Add a `GET /metrics` plug route in `BojRest.Router` that calls `TelemetryMetricsPrometheus.Core.scrape(:boj_rest_prometheus)` and returns the text-format payload. -- [ ] Add the `metrics-get` rule to `config/gateway-policy-boj.yaml` per § 5. -- [ ] Add `/metrics` to `scripts/hcg-policy-smoke.sh` stealth canary list per § 5. -- [ ] Add a unit test exercising each `:telemetry.execute/3` site (matching the existing test patterns under `elixir/test/`). -- [ ] Update `gateway-observability-spec.md` § 3.1, § 3.2, § 3.3 to drop the `!OWNER: scaffold` qualifier (the metric names now resolve against this repo's emitters). -- [ ] Update `hcg-tier2-rollout-runbook.md` § 1.4 to flip the new prerequisite checkbox from `[ ]` to `[x]`. - -The same PR closes this spec's `**Status:**` line from `scaffold` to `wired (Phase E §1.4 prereq satisfied)` with a date. - ---- - -## 8. References - -- Sister spec — `docs/integration/gateway-observability-spec.md` (§ 3 BoJ-side PromQL templates; this spec backs those templates with real metric names). -- Rollout runbook — `docs/integration/hcg-tier2-rollout-runbook.md` (§ 1.4 BoJ-side prerequisites, § 3.1 success criteria, § 4.2 signals, § 5.1 rollback triggers). -- Contract — `docs/integration/http-capability-gateway-boj-contract.md` (§ 3 trust-level invariants this spec's `decision` and `trust_level_present` events enforce telemetry coverage for). -- Audit — `docs/integration/http-capability-gateway-audit.md` (§ 5 gateway-side telemetry shape, mirrored on the BoJ side here). -- Plan — `docs/integration/http-capability-gateway-plan.md` (§ Phase E E3 telemetry verification). -- BoJ router — `elixir/lib/boj_rest/router.ex` (instrumentation sites § 3.1–3.4). -- BoJ trust policy — `elixir/lib/boj_rest/trust_policy.ex` (`required_exposure/1` source of `trust_class` tag values; `satisfies?/3` source of `outcome` tag values). -- BoJ application — `elixir/lib/boj_rest/application.ex` (supervisor tree to extend per § 4.2). -- Gateway metric definitions — `http-capability-gateway/lib/http_capability_gateway/application.ex` `telemetry_metrics/0` (lines 259–296; bucket-layout reference for § 2.2). diff --git a/docs/integration/gateway-load-profile.adoc b/docs/integration/gateway-load-profile.adoc new file mode 100644 index 00000000..19910436 --- /dev/null +++ b/docs/integration/gateway-load-profile.adoc @@ -0,0 +1,465 @@ +== http-capability-gateway — Tier-2 Load Profile + +*Version:* 1.0 (draft, Phase D) *Date:* 2026-05-31 *Status:* Phase D +deliverable D1 — load profile declaration. Operator-fillable markers +(`+!OWNER:+`) cover the production-traffic measurements the BoJ owner +must populate before Phase E §1.1 can be checked off. *ADR:* +link:../decisions/0004-adopt-http-capability-gateway.md[`+docs/decisions/0004-adopt-http-capability-gateway.md+`] +*Plan:* +link:http-capability-gateway-plan.md[`+docs/integration/http-capability-gateway-plan.md+`] +(§ Phase D, D1) *Contract:* +link:http-capability-gateway-boj-contract.md[`+docs/integration/http-capability-gateway-boj-contract.md+`] +*Perf contract (gateway side):* +https://github.com/hyperpolymath/http-capability-gateway/blob/main/docs/perf-contract.md[`+http-capability-gateway/docs/perf-contract.md+`] +*Companion benchmark results doc (D2):* +`+docs/integration/gateway-benchmarks.md+` (lands once +`+bench/baseline.json+` `+_status+` flips to `+active+`) *Rollout +runbook:* +link:hcg-tier2-rollout-runbook.md[`+docs/integration/hcg-tier2-rollout-runbook.md+`] +*Tracking:* +https://github.com/hyperpolymath/standards/issues/91[`+standards#91+`] +(parent), +https://github.com/hyperpolymath/standards/issues/99[`+standards#99+`] +(Phase D) + +____ +*File-format note.* Matches sibling integration docs +(`+http-capability-gateway-{plan,audit,boj-contract,policy-authoring}.md+`, +`+hcg-tier2-rollout-runbook.md+`); the plan §D1 normatively prescribes +`+docs/integration/gateway-load-profile.md+`. The estate `+.adoc+` +default is deliberately overridden for the `+docs/integration/+` set so +the integration plan can name documents by exact path. +____ + +''''' + +=== 0. Scope + +This document declares the *load envelope* the HCG tier-2 gateway must +absorb when sitting between Cloudflare edge (tier 1) and BoJ’s unified +Zig API gnosis handler (tier 3). It is the contract the Phase D +benchmark harness measures against and the Phase E rollout uses to size +soak windows and tolerance margins. + +In scope: + +[arabic] +. *Production traffic baseline* — the requests/second the BoJ gnosis +handler receives in production today (§1). The gateway must, +post-rollout, handle the same arrival rate. +. *Gateway load profile target* — the rate the gateway must serve at, +expressed as `+production-rate × headroom+`, plus the per-request +overhead budget within which the gateway pipeline must stay (§2). +. *Measurement environment* — the hardware, runtime, and policy size the +published Phase D benchmark numbers are collected against (§3). + +Out of scope: + +* Per-scenario latency numbers themselves. Those live in +`+bench/baseline.json+` (the harness output, gateway repo) once the D-4 +rebaseline ritual has collected them, and are summarised for BoJ readers +in the D2 deliverable `+docs/integration/gateway-benchmarks.md+`. +* BoJ-internal latency (the gnosis handler’s own processing time). The +gateway treats BoJ as a single backend with opaque cost; the +gateway-side perf contract isolates gateway-attributable cost only. +* Cloudflare-edge load shaping (tier 1) and BEAM-supervisor +back-pressure (tier 3). Both are covered separately in +`+Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY]+`. + +''''' + +=== 1. Production traffic baseline + +The gateway will, in Phase E, sit in front of BoJ’s gnosis handler with +no peer or shard splitting (single-backend proxy, per ADR-0004 § Risks). +It must therefore absorb the _full_ production arrival rate BoJ sees +today. The figures below describe that rate. + +____ +*Operator action required before Phase E §1.1 close.* All `+!OWNER:+` +rows are measurements that must come from production telemetry — not +estimates. Until they are filled in, the load profile is +*declarative-only* and the Phase E rollout cannot move past §1.1. +____ + +==== 1.1 Arrival rate (the BoJ-direct steady state) + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Measure |Value |Source +|Median sustained requests/second (24h rolling) |`+!OWNER:+` |BoJ access +logs / Prometheus `+http_requests_total+` rate over 24h. + +|p95 sustained requests/second (1-minute buckets, 24h window) +|`+!OWNER:+` |Same. + +|Peak burst (1-second bucket, 7-day window) |`+!OWNER:+` |Same. + +|Diurnal pattern (peak hour / trough hour, UTC) |`+!OWNER:+` |Same. + +|Verb mix (% GET / POST / other) |`+!OWNER:+` |Same; cross-check against +`+[CLOUDFLARE_EDGE_SECURITY]+` rate-limit dashboards. +|=== + +The arrival-rate baseline is measured *on the externally visible port* +today (BoJ-direct, no gateway interposed). Post-rollout the same rate +hits the gateway first; the gateway must not silently shed any of it (a +fast 429/403 IS a valid response, but the _arrival_ must be absorbed +without the gateway becoming the bottleneck). + +==== 1.2 Path mix + +The gateway’s per-request cost is path-sensitive: short-circuit denies +cost less than allow-and-proxy paths (perf contract § Scenarios, gateway +repo). The path mix below sizes the _weighted_ per-request cost the +gateway will pay. + +[width="100%",cols="25%,25%,25%,25%",options="header",] +|=== +|Path class |Example route(s) |% of production traffic |Gateway cost +class +|Health probe (cheap allow) |`+/health+`, +`+/.well-known/boj-node-pubkey+` |`+!OWNER:+` |`+health endpoint+` +(perf-contract.md scenario 1) + +|Verb-denied (cheap deny) |DELETE/PUT/PATCH to any wired route +|`+!OWNER:+` (typically near-0 from legitimate traffic; non-zero from +scanners) |`+policy deny (405 fast-path)+` (scenario 2) + +|Cartridge invoke (allow + proxy) |`+POST /cartridge/:name/invoke+`, +`+POST /cartridge/:name/sse+` |`+!OWNER:+` +|`+exact route allow (proxy 200)+` (scenario 3) + +|Cartridge list/detail (allow + proxy) |`+GET /cartridges+`, +`+GET /cartridge/:name+` |`+!OWNER:+` |`+exact route allow (proxy 200)+` +(scenario 3) + +|Other declared-and-wired routes |per +`+config/gateway-policy-boj-example.yaml+` |`+!OWNER:+` +|`+exact route allow (proxy 200)+` (scenario 3) +|=== + +The seven cost-class buckets above match the six Benchee scenarios in +the gateway repo’s `+bench/gateway_latency.exs+` (plus the +`+health endpoint+` scenario, which doubles as both health probe and +pipeline-floor) so the per-class weighting is a direct lookup against +the published per-scenario percentiles in `+bench/baseline.json+`. + +==== 1.3 Connection profile + +mTLS is the Phase B primary trust path. Per-connection handshake cost is +amortised over the requests served on that connection; the relevant +operator metric is therefore _requests per kept-alive connection_, not +raw connections/second. + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Measure |Value |Source +|Median requests per kept-alive connection |`+!OWNER:+` |Cloudflare +access logs (connection ID grouping) or BoJ-direct access logs if +pre-rollout. + +|p95 requests per kept-alive connection |`+!OWNER:+` |Same. + +|New-connection rate (handshakes/second, p95 over 1-minute buckets) +|`+!OWNER:+` |Same. +|=== + +The bench harness covers both ends of this spectrum: + +* *Cold:* `+mTLS handshake (test CA)+` (scenario 5) — every iteration +pays one handshake. Upper bound when N=1. +* *Warm:* `+mTLS amortised (test CA, N=16 requests over kept-alive)+` +(scenario 6) — handshake amortised over 16 requests. Lower bound for +steady-state. + +The operator’s measured `+requests-per-connection+` value sits somewhere +on the interpolation between these two; §2.4 tracks the consequence for +the per-request budget. + +''''' + +=== 2. Gateway load profile target + +Given §1, the gateway must serve at least the rates declared here +without breaching the per-request overhead budget. + +==== 2.1 Throughput envelope + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Target |Value |Rationale +|Sustained throughput |≥ §1.1 median × *1.5* (headroom) |The gateway +absorbs production arrival rate plus 50% headroom for diurnal variation +and scanner traffic. + +|Burst throughput (1-second window) |≥ §1.1 peak burst × *1.2* |Match +observed peak with modest cushion; relies on Cowboy’s accept-pool sizing +covered in §3.3. + +|429/403 short-circuit throughput |unbounded (limited by acceptor pool) +|Verb-denied and policy-denied requests must not queue behind allow-path +latency; the `+policy deny (405 fast-path)+` scenario measures this in +isolation. +|=== + +The headroom multipliers are deliberately fixed (1.5 / 1.2) rather than +derived per-deploy: a tighter envelope would make the rollout brittle to +mid-rollout traffic growth; a looser one would over-provision the bench +harness against a real signal it never sees. If the §1.1 measurements +arrive _higher_ than the existing capacity assumptions in the rollout +runbook, that is a Phase E §1.1 escalation, not a number to silently +round down. + +==== 2.2 Per-request overhead budget + +The gateway adds a hop in the request path; the ADR-0004 +Negative-consequence § Latency overhead names a target of +`+< 2ms median, < 5ms p99 for a policy with 100 rules+`. The BoJ example +policy has *28 rules* (`+config/gateway-policy-boj-example.yaml+`, +recounted 2026-05-31 — 28 routes across all exposure tiers). With a +smaller policy than the ADR’s 100-rule reference, the budget below is +set tighter on the median path and matches the ADR on tail: + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Percentile |Budget (per-request gateway overhead) |Bench scenario it +applies to +|p50 |< 1.5 ms |`+exact route allow (proxy 200)+` minus +`+health endpoint+` (the proxy-attributable cost) + +|p95 |< 3 ms |Same. + +|p99 |< 5 ms |Same. +|=== + +"`Gateway overhead`" is defined as the *gateway-attributable* cost: per +perf contract § Scope, the harness already strips real-network RTT and +real-backend processing time via the in-process loopback backend, so +`+exact route allow (proxy 200) − health endpoint+` isolates the proxy +hot path’s gateway-attributable cost. The Phase D-4 rebaseline ritual +populates the numerator; this section is the denominator. + +Tolerance ratios (perf-regression CI gate) are set in +`+bench/baseline.json::tolerance+`. They are looser than this budget +because tolerance-ratio is _regression detection_ (catch a 30% p95 +worsening), whereas the budget here is _absolute SLO_ (do not breach 3 +ms regardless of historical trend). + +==== 2.3 Stealth and deny budget + +Routes with `+exposure: "internal"+` and +`+stealth: { enabled: true, status_code: 404 }+` MUST cost no more than +a public route returning the same status: + +[width="100%",cols="50%,50%",options="header",] +|=== +|Class |Budget +|Stealth-404 (internal-only path, untrusted caller) |p99 ≤ +`+policy deny (405 fast-path)+` × *1.1* +|=== + +The 10% latitude allows for the extra `+if exposure==:internal+` branch +in the policy evaluator without making the budget brittle to a fast-path +refactor that re-orders the branches. + +==== 2.4 mTLS handshake budget + +From §1.3 the gateway will see a mix of fresh handshakes and amortised +kept-alive requests. The bench scenarios bracket the per-request cost: + +* Cold: `+mTLS handshake (test CA)+` per-iteration cost. +* Warm: `+mTLS amortised (test CA, N=16)+` per-iteration cost = +`+(handshake + 16 × request) / 16+`. + +The *weighted per-request mTLS cost* is: + +.... +weighted = (1 / R) × cold + (1 − 1 / R) × warm_per_request +.... + +where `+R+` is §1.3’s measured median +requests-per-kept-alive-connection. Operator-fillable for both numerator +and denominator at Phase E §1.1 sign-off; this section names the formula +so the calculation is reviewable, not improvised. + +''''' + +=== 3. Measurement environment + +The published Phase D benchmark numbers (and the perf-regression CI gate +that watches them) come from a single, named environment. +Reproducibility is the point: any reviewer must be able to re-run the +harness on the same target and get numbers in the same order of +magnitude. + +==== 3.1 Hardware reference + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Element |Value |Source +|Runner |GitHub Actions `+ubuntu-latest+` +|`+.github/workflows/perf-regression.yml+` and +`+.github/workflows/perf-rebaseline.yml+` in the gateway repo, pinned to +the same target so collected numbers are gate-comparable. + +|Architecture |`+x86_64+` |GHA `+ubuntu-latest+` default. + +|Memory |Per GHA `+ubuntu-latest+` SKU (currently 16 GB; subject to GHA +pool changes — variability is the cost of using a shared runner). +|GHA-documented. +|=== + +The perf contract (gateway repo) names this choice deliberately: "`the +CI environment IS the published reference, deliberately chosen because +it is the environment every reviewer can reproduce without local +hardware variance.`" If a dedicated runner is later adopted +(perf-contract.md § Targets calls this out as a Phase-D-4-revisit +decision), that switch will land as a versioned rev of this document, +not a silent baseline replacement. + +==== 3.2 Runtime + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Element |Value |Source +|Elixir |`+1.19+` |`+http-capability-gateway/.tool-versions+` +(`+elixir 1.19.5-otp-28+`); `+mix.exs+` requires `+~> 1.19+`. + +|OTP |`+28+` |Same. + +|Erlang VM flags |OTP-default +(`++sbwt none +sbwtdcpu none +sbwtdio none+` NOT applied — staying on +defaults keeps the bench reproducible without per-runner tuning). +|Confirmed via inspection of `+bench/gateway_latency.exs+` startup; no +`+:erlang.system_flag/2+` calls before the harness runs. + +|Cowboy |`+~> 2.7+` (per ADR-0004 § Phase B reference) +|`+http-capability-gateway/mix.lock+`. +|=== + +BoJ-side runtime (`+elixir 1.18.4-otp-25+`) is *not* the gateway +runtime: the gateway is a separate BEAM application (ADR-0004 § Negative +consequences). The runtime above is the gateway’s, not BoJ’s. Mismatched +OTP majors across the seam are fine — the seam is HTTP, not BEAM-native +message passing. + +==== 3.3 Policy size + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Element |Value |Source +|Routes in published example policy |*28* +|`+config/gateway-policy-boj-example.yaml+`, counted 2026-05-31 +(`+grep -cE '^ - path:'+`). + +|DSL version |`+"1"+` |Same. + +|Global verbs declared |per `+governance.global_verbs+` (typically +`+[GET, POST]+`; DELETE/PUT/PATCH/OPTIONS deliberately absent — the core +verb-governance win) |Same. + +|Lookup classes used |exact + regex (cartridge-detail, cartridge-invoke, +cartridge-sse, grpc-method) |Same. +|=== + +The ADR’s reference budget (`+100 rules+`) is an estate ceiling, not the +current size. Phase D-4 baseline numbers are collected against the +_current_ 28-rule policy, which is the policy Phase E will go live with. +If/when the policy grows past 50 rules a rebaseline-and-revise of §2.2 +is required (the regex-lookup class scales differently from +exact-lookup, per `+bench/gateway_latency.exs+` scenario commentary). + +==== 3.4 Bench harness reference + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Component |Path (gateway repo) |Role +|Benchee driver |`+bench/gateway_latency.exs+` |Runs the six published +scenarios. + +|Baseline file |`+bench/baseline.json+` |Holds p50 / p95 / p99 / ips per +scenario; `+_status+` field gates the perf-regression CI gate. + +|Rebaseline driver |`+bench/rebaseline.exs+` |Regenerates +`+bench/baseline.json+` from `+bench/results.json+`. + +|Rebaseline workflow |`+.github/workflows/perf-rebaseline.yml+` +|`+workflow_dispatch+`-only automation that runs the driver on +`+ubuntu-latest+` and opens a `+perf: rebaseline (standards#99)+` PR. + +|Regression gate |`+.github/workflows/perf-regression.yml+` |Per-PR +gate; non-blocking while `+_status == "scaffold-placeholder"+`. + +|Per-scenario contract |`+docs/perf-contract.md+` (gateway repo) |Names +each scenario, units, tolerance ratios, baseline lifecycle. +|=== + +''''' + +=== 4. How this profile feeds Phase D and Phase E + +==== 4.1 Phase D close (standards#99) + +The Phase D acceptance criteria (plan § Phase D) are: + +[arabic] +. Load profile document exists. → *This document.* ✓ once landed. +. Median / p95 / p99 latency numbers published in +`+docs/integration/gateway-benchmarks.md+`. → D2 deliverable; lands once +D-4 rebaseline ritual (gateway repo) flips `+bench/baseline.json+` +`+_status+` from `+scaffold-placeholder+` to `+active+` and the +resulting numbers are reflected back here as `+gateway-benchmarks.md+`. +. Gateway overhead number published (vs. direct BoJ, no gateway). → D2 +deliverable; the formula is §2.2 +(`+exact route allow (proxy 200) − health endpoint+`). +. CI regression step configured. → Already in place per +`+http-capability-gateway/.github/workflows/perf-regression.yml+` (PR +#12 scaffold + PR #22 scenario expansion + PR #26 rebaseline-bootstrap). +. Benchmark results do NOT fabricate numbers. → Enforced by the +rebaseline workflow’s `+workflow_dispatch+` collection on +`+ubuntu-latest+`. + +==== 4.2 Phase E gate (standards#100, runbook §1.1) + +The rollout runbook §1.1 "`Phase D deliverables landed`" checklist gates +the Phase E traffic shift on: + +* D-2 (loopback backend fixture) merged — *gateway PR #14, merged.* +* D-3 (real CI regression alert armed; `+bench/baseline.json _status+` +flipped to `+"active"+`) — *partial: scenarios + gate scaffold landed +via gateway PR #12 / #22; `+_status+` flip pending D-4 numbers.* +* D-4 (real baseline numbers populated; p50/p95/p99/ips for all six +scenarios) — *pending: gateway PR #26 added the `+workflow_dispatch+` +automation; the rebaseline PR itself has not yet been opened.* + +Once D-4 lands and `+_status+` is flipped, the absolute budget in §2.2 +of this document is the SLO the rollout watches at every soak window +(§3.1 / §3.2 of the runbook), and the perf-regression CI gate becomes +the _regression-detection_ arm. Two independent guards — one against +absolute breach, one against silent drift — both grounded in this +profile. + +''''' + +=== 5. References + +* ADR-0004 — `+docs/decisions/0004-adopt-http-capability-gateway.md+` (§ +Phase D row of the integration table; § Negative consequences § Latency +overhead for the ADR-level budget). +* Integration plan — +`+docs/integration/http-capability-gateway-plan.md+` (§ Phase D for the +deliverable list; § Cross-Phase Notes for surface-drift discipline). +* Contract — +`+docs/integration/http-capability-gateway-boj-contract.md+` (the seam +this profile sizes traffic through). +* Rollout runbook — `+docs/integration/hcg-tier2-rollout-runbook.md+` +(§1.1 prereqs that gate on this document). +* Perf contract (gateway side) — +`+http-capability-gateway/docs/perf-contract.md+` (the six scenarios, +the tolerance ratios, the baseline lifecycle). +* Bench harness — `+http-capability-gateway/bench/gateway_latency.exs+`, +`+bench/baseline.json+`, `+bench/rebaseline.exs+`. +* Policy under test — `+config/gateway-policy-boj-example.yaml+` +(28-route example, the size against which Phase D-4 baseline is +collected). diff --git a/docs/integration/gateway-load-profile.md b/docs/integration/gateway-load-profile.md deleted file mode 100644 index 962df961..00000000 --- a/docs/integration/gateway-load-profile.md +++ /dev/null @@ -1,226 +0,0 @@ - - - -# http-capability-gateway — Tier-2 Load Profile - -**Version:** 1.0 (draft, Phase D) -**Date:** 2026-05-31 -**Status:** Phase D deliverable D1 — load profile declaration. Operator-fillable markers (`!OWNER:`) cover the production-traffic measurements the BoJ owner must populate before Phase E §1.1 can be checked off. -**ADR:** [`docs/decisions/0004-adopt-http-capability-gateway.md`](../decisions/0004-adopt-http-capability-gateway.md) -**Plan:** [`docs/integration/http-capability-gateway-plan.md`](http-capability-gateway-plan.md) (§ Phase D, D1) -**Contract:** [`docs/integration/http-capability-gateway-boj-contract.md`](http-capability-gateway-boj-contract.md) -**Perf contract (gateway side):** [`http-capability-gateway/docs/perf-contract.md`](https://github.com/hyperpolymath/http-capability-gateway/blob/main/docs/perf-contract.md) -**Companion benchmark results doc (D2):** `docs/integration/gateway-benchmarks.md` (lands once `bench/baseline.json` `_status` flips to `active`) -**Rollout runbook:** [`docs/integration/hcg-tier2-rollout-runbook.md`](hcg-tier2-rollout-runbook.md) -**Tracking:** [`standards#91`](https://github.com/hyperpolymath/standards/issues/91) (parent), [`standards#99`](https://github.com/hyperpolymath/standards/issues/99) (Phase D) - -> **File-format note.** Matches sibling integration docs (`http-capability-gateway-{plan,audit,boj-contract,policy-authoring}.md`, `hcg-tier2-rollout-runbook.md`); the plan §D1 normatively prescribes `docs/integration/gateway-load-profile.md`. The estate `.adoc` default is deliberately overridden for the `docs/integration/` set so the integration plan can name documents by exact path. - ---- - -## 0. Scope - -This document declares the **load envelope** the HCG tier-2 gateway must absorb when sitting between Cloudflare edge (tier 1) and BoJ's unified Zig API gnosis handler (tier 3). It is the contract the Phase D benchmark harness measures against and the Phase E rollout uses to size soak windows and tolerance margins. - -In scope: - -1. **Production traffic baseline** — the requests/second the BoJ gnosis handler receives in production today (§1). The gateway must, post-rollout, handle the same arrival rate. -2. **Gateway load profile target** — the rate the gateway must serve at, expressed as `production-rate × headroom`, plus the per-request overhead budget within which the gateway pipeline must stay (§2). -3. **Measurement environment** — the hardware, runtime, and policy size the published Phase D benchmark numbers are collected against (§3). - -Out of scope: - -- Per-scenario latency numbers themselves. Those live in `bench/baseline.json` (the harness output, gateway repo) once the D-4 rebaseline ritual has collected them, and are summarised for BoJ readers in the D2 deliverable `docs/integration/gateway-benchmarks.md`. -- BoJ-internal latency (the gnosis handler's own processing time). The gateway treats BoJ as a single backend with opaque cost; the gateway-side perf contract isolates gateway-attributable cost only. -- Cloudflare-edge load shaping (tier 1) and BEAM-supervisor back-pressure (tier 3). Both are covered separately in `Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY]`. - ---- - -## 1. Production traffic baseline - -The gateway will, in Phase E, sit in front of BoJ's gnosis handler with no peer or shard splitting (single-backend proxy, per ADR-0004 § Risks). It must therefore absorb the *full* production arrival rate BoJ sees today. The figures below describe that rate. - -> **Operator action required before Phase E §1.1 close.** All `!OWNER:` rows are measurements that must come from production telemetry — not estimates. Until they are filled in, the load profile is **declarative-only** and the Phase E rollout cannot move past §1.1. - -### 1.1 Arrival rate (the BoJ-direct steady state) - -| Measure | Value | Source | -|---|---|---| -| Median sustained requests/second (24h rolling) | `!OWNER:` | BoJ access logs / Prometheus `http_requests_total` rate over 24h. | -| p95 sustained requests/second (1-minute buckets, 24h window) | `!OWNER:` | Same. | -| Peak burst (1-second bucket, 7-day window) | `!OWNER:` | Same. | -| Diurnal pattern (peak hour / trough hour, UTC) | `!OWNER:` | Same. | -| Verb mix (% GET / POST / other) | `!OWNER:` | Same; cross-check against `[CLOUDFLARE_EDGE_SECURITY]` rate-limit dashboards. | - -The arrival-rate baseline is measured **on the externally visible port** today (BoJ-direct, no gateway interposed). Post-rollout the same rate hits the gateway first; the gateway must not silently shed any of it (a fast 429/403 IS a valid response, but the *arrival* must be absorbed without the gateway becoming the bottleneck). - -### 1.2 Path mix - -The gateway's per-request cost is path-sensitive: short-circuit denies cost less than allow-and-proxy paths (perf contract § Scenarios, gateway repo). The path mix below sizes the *weighted* per-request cost the gateway will pay. - -| Path class | Example route(s) | % of production traffic | Gateway cost class | -|---|---|---|---| -| Health probe (cheap allow) | `/health`, `/.well-known/boj-node-pubkey` | `!OWNER:` | `health endpoint` (perf-contract.md scenario 1) | -| Verb-denied (cheap deny) | DELETE/PUT/PATCH to any wired route | `!OWNER:` (typically near-0 from legitimate traffic; non-zero from scanners) | `policy deny (405 fast-path)` (scenario 2) | -| Cartridge invoke (allow + proxy) | `POST /cartridge/:name/invoke`, `POST /cartridge/:name/sse` | `!OWNER:` | `exact route allow (proxy 200)` (scenario 3) | -| Cartridge list/detail (allow + proxy) | `GET /cartridges`, `GET /cartridge/:name` | `!OWNER:` | `exact route allow (proxy 200)` (scenario 3) | -| Other declared-and-wired routes | per `config/gateway-policy-boj-example.yaml` | `!OWNER:` | `exact route allow (proxy 200)` (scenario 3) | - -The seven cost-class buckets above match the six Benchee scenarios in the gateway repo's `bench/gateway_latency.exs` (plus the `health endpoint` scenario, which doubles as both health probe and pipeline-floor) so the per-class weighting is a direct lookup against the published per-scenario percentiles in `bench/baseline.json`. - -### 1.3 Connection profile - -mTLS is the Phase B primary trust path. Per-connection handshake cost is amortised over the requests served on that connection; the relevant operator metric is therefore *requests per kept-alive connection*, not raw connections/second. - -| Measure | Value | Source | -|---|---|---| -| Median requests per kept-alive connection | `!OWNER:` | Cloudflare access logs (connection ID grouping) or BoJ-direct access logs if pre-rollout. | -| p95 requests per kept-alive connection | `!OWNER:` | Same. | -| New-connection rate (handshakes/second, p95 over 1-minute buckets) | `!OWNER:` | Same. | - -The bench harness covers both ends of this spectrum: - -- **Cold:** `mTLS handshake (test CA)` (scenario 5) — every iteration pays one handshake. Upper bound when N=1. -- **Warm:** `mTLS amortised (test CA, N=16 requests over kept-alive)` (scenario 6) — handshake amortised over 16 requests. Lower bound for steady-state. - -The operator's measured `requests-per-connection` value sits somewhere on the interpolation between these two; §2.4 tracks the consequence for the per-request budget. - ---- - -## 2. Gateway load profile target - -Given §1, the gateway must serve at least the rates declared here without breaching the per-request overhead budget. - -### 2.1 Throughput envelope - -| Target | Value | Rationale | -|---|---|---| -| Sustained throughput | ≥ §1.1 median × **1.5** (headroom) | The gateway absorbs production arrival rate plus 50% headroom for diurnal variation and scanner traffic. | -| Burst throughput (1-second window) | ≥ §1.1 peak burst × **1.2** | Match observed peak with modest cushion; relies on Cowboy's accept-pool sizing covered in §3.3. | -| 429/403 short-circuit throughput | unbounded (limited by acceptor pool) | Verb-denied and policy-denied requests must not queue behind allow-path latency; the `policy deny (405 fast-path)` scenario measures this in isolation. | - -The headroom multipliers are deliberately fixed (1.5 / 1.2) rather than derived per-deploy: a tighter envelope would make the rollout brittle to mid-rollout traffic growth; a looser one would over-provision the bench harness against a real signal it never sees. If the §1.1 measurements arrive *higher* than the existing capacity assumptions in the rollout runbook, that is a Phase E §1.1 escalation, not a number to silently round down. - -### 2.2 Per-request overhead budget - -The gateway adds a hop in the request path; the ADR-0004 Negative-consequence § Latency overhead names a target of `< 2ms median, < 5ms p99 for a policy with 100 rules`. The BoJ example policy has **28 rules** (`config/gateway-policy-boj-example.yaml`, recounted 2026-05-31 — 28 routes across all exposure tiers). With a smaller policy than the ADR's 100-rule reference, the budget below is set tighter on the median path and matches the ADR on tail: - -| Percentile | Budget (per-request gateway overhead) | Bench scenario it applies to | -|---|---|---| -| p50 | < 1.5 ms | `exact route allow (proxy 200)` minus `health endpoint` (the proxy-attributable cost) | -| p95 | < 3 ms | Same. | -| p99 | < 5 ms | Same. | - -"Gateway overhead" is defined as the **gateway-attributable** cost: per perf contract § Scope, the harness already strips real-network RTT and real-backend processing time via the in-process loopback backend, so `exact route allow (proxy 200) − health endpoint` isolates the proxy hot path's gateway-attributable cost. The Phase D-4 rebaseline ritual populates the numerator; this section is the denominator. - -Tolerance ratios (perf-regression CI gate) are set in `bench/baseline.json::tolerance`. They are looser than this budget because tolerance-ratio is *regression detection* (catch a 30% p95 worsening), whereas the budget here is *absolute SLO* (do not breach 3 ms regardless of historical trend). - -### 2.3 Stealth and deny budget - -Routes with `exposure: "internal"` and `stealth: { enabled: true, status_code: 404 }` MUST cost no more than a public route returning the same status: - -| Class | Budget | -|---|---| -| Stealth-404 (internal-only path, untrusted caller) | p99 ≤ `policy deny (405 fast-path)` × **1.1** | - -The 10% latitude allows for the extra `if exposure==:internal` branch in the policy evaluator without making the budget brittle to a fast-path refactor that re-orders the branches. - -### 2.4 mTLS handshake budget - -From §1.3 the gateway will see a mix of fresh handshakes and amortised kept-alive requests. The bench scenarios bracket the per-request cost: - -- Cold: `mTLS handshake (test CA)` per-iteration cost. -- Warm: `mTLS amortised (test CA, N=16)` per-iteration cost = `(handshake + 16 × request) / 16`. - -The **weighted per-request mTLS cost** is: - -``` -weighted = (1 / R) × cold + (1 − 1 / R) × warm_per_request -``` - -where `R` is §1.3's measured median requests-per-kept-alive-connection. Operator-fillable for both numerator and denominator at Phase E §1.1 sign-off; this section names the formula so the calculation is reviewable, not improvised. - ---- - -## 3. Measurement environment - -The published Phase D benchmark numbers (and the perf-regression CI gate that watches them) come from a single, named environment. Reproducibility is the point: any reviewer must be able to re-run the harness on the same target and get numbers in the same order of magnitude. - -### 3.1 Hardware reference - -| Element | Value | Source | -|---|---|---| -| Runner | GitHub Actions `ubuntu-latest` | `.github/workflows/perf-regression.yml` and `.github/workflows/perf-rebaseline.yml` in the gateway repo, pinned to the same target so collected numbers are gate-comparable. | -| Architecture | `x86_64` | GHA `ubuntu-latest` default. | -| Memory | Per GHA `ubuntu-latest` SKU (currently 16 GB; subject to GHA pool changes — variability is the cost of using a shared runner). | GHA-documented. | - -The perf contract (gateway repo) names this choice deliberately: "the CI environment IS the published reference, deliberately chosen because it is the environment every reviewer can reproduce without local hardware variance." If a dedicated runner is later adopted (perf-contract.md § Targets calls this out as a Phase-D-4-revisit decision), that switch will land as a versioned rev of this document, not a silent baseline replacement. - -### 3.2 Runtime - -| Element | Value | Source | -|---|---|---| -| Elixir | `1.19` | `http-capability-gateway/.tool-versions` (`elixir 1.19.5-otp-28`); `mix.exs` requires `~> 1.19`. | -| OTP | `28` | Same. | -| Erlang VM flags | OTP-default (`+sbwt none +sbwtdcpu none +sbwtdio none` NOT applied — staying on defaults keeps the bench reproducible without per-runner tuning). | Confirmed via inspection of `bench/gateway_latency.exs` startup; no `:erlang.system_flag/2` calls before the harness runs. | -| Cowboy | `~> 2.7` (per ADR-0004 § Phase B reference) | `http-capability-gateway/mix.lock`. | - -BoJ-side runtime (`elixir 1.18.4-otp-25`) is **not** the gateway runtime: the gateway is a separate BEAM application (ADR-0004 § Negative consequences). The runtime above is the gateway's, not BoJ's. Mismatched OTP majors across the seam are fine — the seam is HTTP, not BEAM-native message passing. - -### 3.3 Policy size - -| Element | Value | Source | -|---|---|---| -| Routes in published example policy | **28** | `config/gateway-policy-boj-example.yaml`, counted 2026-05-31 (`grep -cE '^ - path:'`). | -| DSL version | `"1"` | Same. | -| Global verbs declared | per `governance.global_verbs` (typically `[GET, POST]`; DELETE/PUT/PATCH/OPTIONS deliberately absent — the core verb-governance win) | Same. | -| Lookup classes used | exact + regex (cartridge-detail, cartridge-invoke, cartridge-sse, grpc-method) | Same. | - -The ADR's reference budget (`100 rules`) is an estate ceiling, not the current size. Phase D-4 baseline numbers are collected against the *current* 28-rule policy, which is the policy Phase E will go live with. If/when the policy grows past 50 rules a rebaseline-and-revise of §2.2 is required (the regex-lookup class scales differently from exact-lookup, per `bench/gateway_latency.exs` scenario commentary). - -### 3.4 Bench harness reference - -| Component | Path (gateway repo) | Role | -|---|---|---| -| Benchee driver | `bench/gateway_latency.exs` | Runs the six published scenarios. | -| Baseline file | `bench/baseline.json` | Holds p50 / p95 / p99 / ips per scenario; `_status` field gates the perf-regression CI gate. | -| Rebaseline driver | `bench/rebaseline.exs` | Regenerates `bench/baseline.json` from `bench/results.json`. | -| Rebaseline workflow | `.github/workflows/perf-rebaseline.yml` | `workflow_dispatch`-only automation that runs the driver on `ubuntu-latest` and opens a `perf: rebaseline (standards#99)` PR. | -| Regression gate | `.github/workflows/perf-regression.yml` | Per-PR gate; non-blocking while `_status == "scaffold-placeholder"`. | -| Per-scenario contract | `docs/perf-contract.md` (gateway repo) | Names each scenario, units, tolerance ratios, baseline lifecycle. | - ---- - -## 4. How this profile feeds Phase D and Phase E - -### 4.1 Phase D close (standards#99) - -The Phase D acceptance criteria (plan § Phase D) are: - -1. Load profile document exists. → **This document.** ✓ once landed. -2. Median / p95 / p99 latency numbers published in `docs/integration/gateway-benchmarks.md`. → D2 deliverable; lands once D-4 rebaseline ritual (gateway repo) flips `bench/baseline.json` `_status` from `scaffold-placeholder` to `active` and the resulting numbers are reflected back here as `gateway-benchmarks.md`. -3. Gateway overhead number published (vs. direct BoJ, no gateway). → D2 deliverable; the formula is §2.2 (`exact route allow (proxy 200) − health endpoint`). -4. CI regression step configured. → Already in place per `http-capability-gateway/.github/workflows/perf-regression.yml` (PR #12 scaffold + PR #22 scenario expansion + PR #26 rebaseline-bootstrap). -5. Benchmark results do NOT fabricate numbers. → Enforced by the rebaseline workflow's `workflow_dispatch` collection on `ubuntu-latest`. - -### 4.2 Phase E gate (standards#100, runbook §1.1) - -The rollout runbook §1.1 "Phase D deliverables landed" checklist gates the Phase E traffic shift on: - -- D-2 (loopback backend fixture) merged — **gateway PR #14, merged.** -- D-3 (real CI regression alert armed; `bench/baseline.json _status` flipped to `"active"`) — **partial: scenarios + gate scaffold landed via gateway PR #12 / #22; `_status` flip pending D-4 numbers.** -- D-4 (real baseline numbers populated; p50/p95/p99/ips for all six scenarios) — **pending: gateway PR #26 added the `workflow_dispatch` automation; the rebaseline PR itself has not yet been opened.** - -Once D-4 lands and `_status` is flipped, the absolute budget in §2.2 of this document is the SLO the rollout watches at every soak window (§3.1 / §3.2 of the runbook), and the perf-regression CI gate becomes the *regression-detection* arm. Two independent guards — one against absolute breach, one against silent drift — both grounded in this profile. - ---- - -## 5. References - -- ADR-0004 — `docs/decisions/0004-adopt-http-capability-gateway.md` (§ Phase D row of the integration table; § Negative consequences § Latency overhead for the ADR-level budget). -- Integration plan — `docs/integration/http-capability-gateway-plan.md` (§ Phase D for the deliverable list; § Cross-Phase Notes for surface-drift discipline). -- Contract — `docs/integration/http-capability-gateway-boj-contract.md` (the seam this profile sizes traffic through). -- Rollout runbook — `docs/integration/hcg-tier2-rollout-runbook.md` (§1.1 prereqs that gate on this document). -- Perf contract (gateway side) — `http-capability-gateway/docs/perf-contract.md` (the six scenarios, the tolerance ratios, the baseline lifecycle). -- Bench harness — `http-capability-gateway/bench/gateway_latency.exs`, `bench/baseline.json`, `bench/rebaseline.exs`. -- Policy under test — `config/gateway-policy-boj-example.yaml` (28-route example, the size against which Phase D-4 baseline is collected). diff --git a/docs/integration/gateway-observability-spec.adoc b/docs/integration/gateway-observability-spec.adoc new file mode 100644 index 00000000..c803e358 --- /dev/null +++ b/docs/integration/gateway-observability-spec.adoc @@ -0,0 +1,686 @@ +== HCG tier-2 — observability spec + +*Version:* 0.2 (BoJ-side sister spec anchored, Phase E) *Date:* +2026-06-22 (rev. from 2026-06-16) *Status:* Phase E scaffold. Names the +gateway-emitted Prometheus metrics, gives PromQL templates for every +signal listed in the rollout runbook §4.1/§4.2, and binds alert +thresholds to the rollback triggers in runbook §5.1 and the perf +contract’s tolerance ratios. §3 BoJ-side templates now anchor to the +sister spec +link:boj-side-observability-spec.md[`+boj-side-observability-spec.md+`] +(events, metric names, instrumentation sites in `+BojRest.Router+`); the +`+!OWNER:+` scaffold qualifier on those templates is dropped — the +wiring PR target is fixed. Absolute-µs values are deliberately left as +`+Phase D-4+` references — once `+bench/baseline.json+` `+_status+` +flips to `+active+` the queries here read against real numbers without +further edits. *ADR:* +link:../decisions/0004-adopt-http-capability-gateway.md[`+docs/decisions/0004-adopt-http-capability-gateway.md+`] +*Plan:* +link:http-capability-gateway-plan.md[`+docs/integration/http-capability-gateway-plan.md+`] +(§ Phase E, E3 telemetry verification) *Contract:* +link:http-capability-gateway-boj-contract.md[`+docs/integration/http-capability-gateway-boj-contract.md+`] +*Rollout runbook:* +link:hcg-tier2-rollout-runbook.md[`+docs/integration/hcg-tier2-rollout-runbook.md+`] +(§ 4 signals, § 5 rollback) *Load profile:* +link:gateway-load-profile.md[`+docs/integration/gateway-load-profile.md+`] +(§ 2 SLO budgets) *Perf contract (gateway side):* +https://github.com/hyperpolymath/http-capability-gateway/blob/main/docs/perf-contract.md[`+http-capability-gateway/docs/perf-contract.md+`] +*Tracking:* +https://github.com/hyperpolymath/standards/issues/91[`+standards#91+`] +(parent), +https://github.com/hyperpolymath/standards/issues/100[`+standards#100+`] +(Phase E) + +____ +*File-format note.* Matches sibling integration docs +(`+http-capability-gateway-{plan,audit,boj-contract,policy-authoring}.md+`, +`+gateway-load-profile.md+`, `+hcg-tier2-rollout-runbook.md+`); the +rollout runbook §4 anchors all signals here by exact path. The estate +`+.adoc+` default is deliberately overridden for the +`+docs/integration/+` set. +____ + +''''' + +=== 0. Scope + +This document is the declarative half of Phase E §4 "`Observability — +what on-call watches`". The runbook §4 names the signals at the human +level ("`p99 latency`", "`circuit-breaker state`", "`trust-level +decision distribution`"); this spec wires each signal to: + +[arabic] +. The *telemetry event* emitted by the gateway (audit document §5). +. The *Prometheus metric* the `+TelemetryMetricsPrometheus.Core+` +reporter exports for that event (gateway +`+lib/http_capability_gateway/application.ex+` `+telemetry_metrics/0+`, +lines 259–296). +. A *PromQL query template* an on-call dashboard or alerting rule can +paste verbatim. +. An *alert threshold* anchored to a canonical source — the rollback +runbook §5.1 trigger value, the perf contract tolerance ratio, or the +load-profile SLO budget. Where the absolute number depends on Phase D-4 +baseline collection, the spec names the formula and the lookup site +instead of inventing a value. + +In scope: + +* Every signal listed in rollout runbook §4.1 (gateway-side) and §4.2 +(BoJ-side). +* The mapping from rollback trigger (§5.1) to the alerting rule that +fires it. +* The Minikaran anomaly endpoint as a secondary, complementary signal +path. + +Out of scope: + +* Dashboard authoring (the !OWNER: rows in runbook §4.3 — the dashboard +URL, the on-call rota). This spec gives the operator the queries; +choosing the dashboard tool (Grafana / Cloudflare analytics / something +else) is owner-driven per the runbook’s existing scoping. +* Cloudflare-edge metrics (tier 1). Covered separately in +`+Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY]+`. +* BoJ-internal cartridge or cartridge-tool metrics. The gateway sees BoJ +as a single backend; cartridge-level observability is downstream. +* Long-term storage / retention policy. The spec defines what to scrape; +how long to keep scrapes is an operator decision (the §4.3 dashboard URL +row already covers that scope). + +''''' + +=== 1. Gateway-side metrics inventory + +Every gateway-emitted telemetry event has a corresponding Prometheus +metric. The mapping below is normative; if a future PR adds a new event +without a metric (or vice versa) the rollout runbook §1.5 smoke +pre-check should fail before traffic shift. + +The Prometheus metric names follow the +`+telemetry_metrics_prometheus_core+` convention: dots become +underscores, distribution metrics expose `+_bucket+`, `+_count+`, +`+_sum+` series, counters expose a `+_total+` series. The names below +are the metric prefixes the operator sees in `+/metrics+`. + +[width="100%",cols="20%,20%,20%,20%,20%",options="header",] +|=== +|Telemetry event (audit §5) |Prometheus metric prefix |Type |Tags +|Source +|`+[:http_capability_gateway, :request, :received]+` +|`+http_capability_gateway_request_received_count+` |gauge (last_value) +|— |`+application.ex:262+` + +|`+[:http_capability_gateway, :request, :completed]+` +|`+http_capability_gateway_request_completed_count+` |counter |— +|`+application.ex:263+` + +|`+[:http_capability_gateway, :request, :completed]+` +|`+http_capability_gateway_request_completed_duration+` |distribution |— +|`+application.ex:264-267+` + +|`+[:http_capability_gateway, :policy, :lookup]+` +|`+http_capability_gateway_policy_lookup_duration+` |distribution |— +|`+application.ex:270-273+` + +|`+[:http_capability_gateway, :access_decision]+` +|`+http_capability_gateway_access_decision_count+` |counter +|`+decision+`, `+verb+`, `+trust_level+` |`+application.ex:276-278+` + +|`+[:http_capability_gateway, :backend, :forward]+` +|`+http_capability_gateway_backend_forward_count+` |counter |— +|`+application.ex:281+` + +|`+[:http_capability_gateway, :backend, :response]+` +|`+http_capability_gateway_backend_response_duration+` |distribution |— +|`+application.ex:282-285+` + +|`+[:http_capability_gateway, :error]+` +|`+http_capability_gateway_error_count+` |counter |`+error_type+` +|`+application.ex:288+` + +|`+[:http_capability_gateway, :minikaran, :anomaly]+` +|`+http_capability_gateway_minikaran_anomaly_count+` |counter |`+type+` +|`+application.ex:293-295+` +|=== + +==== 1.1 Distribution buckets + +Buckets are declared in microseconds and capture the gateway’s own +pipeline cost, not end-to-end RTT. The buckets are wide enough to +accommodate the perf contract’s six scenarios: + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Metric |Buckets (µs) |Source +|`+request_completed_duration+` +|`+[100, 500, 1_000, 5_000, 10_000, 30_000]+` |`+application.ex:266+` + +|`+policy_lookup_duration+` |`+[10, 50, 100, 500, 1_000]+` +|`+application.ex:272+` + +|`+backend_response_duration+` +|`+[100, 500, 1_000, 5_000, 10_000, 30_000, 60_000]+` +|`+application.ex:284+` +|=== + +The `+backend_response_duration+` upper bucket is 60 ms because +BoJ-attributable latency may include cartridge invocation; the +gateway-attributable buckets stop at 30 ms because the load profile §2.2 +budget puts p99 well under that. + +==== 1.2 Tags + +Three counters are tagged. Tag cardinality is bounded: + +* `+decision+` ∈ `+{allow, deny, no_match, error}+` — four-value set, no +cardinality blow-up. +* `+verb+` ∈ `+{GET, POST, PUT, DELETE, PATCH, HEAD, OPTIONS}+` — +seven-value allowlist enforced by `+Gateway.safe_verb/1+` (gateway +`+lib/http_capability_gateway/gateway.ex:65-77+`); unknown methods +short-circuit before reaching the access-decision event. +* `+trust_level+` ∈ `+{untrusted, authenticated, internal}+` — +three-value set enforced by `+SafeTrust.parse_trust/1+`. +* `+error_type+` — open vocabulary but bounded by the gateway’s +enumerated error paths. Operator should monitor for cardinality growth +here as a deployment-defect signal. +* `+type+` on `+minikaran_anomaly_count+` ∈ +`+{traffic_spike, trust_shift, latency_spike, path_novelty, error_spike}+` +— five-value set declared in the audit §1.6. + +Total decision counter cardinality bound: `+4 × 7 × 3 = 84+` time series +at saturation. Below the threshold where Prometheus storage becomes a +concern. + +''''' + +=== 2. Signal → query mapping (rollout runbook §4.1, gateway-side) + +The runbook lists six gateway-side signals. Each subsection below names +one, gives its PromQL query, and binds an alert threshold. + +==== 2.1 p50 / p95 / p99 latency per scenario + +*Runbook signal:* "`p50/p95/p99 latency per scenario (health / +policy-deny fast-path / proxy allow).`" + +The harness scenarios (perf contract §Scenarios) are bench-only — +production traffic doesn’t carry a "`scenario`" tag — so the production +equivalent is per-decision-class: + +* `+health endpoint+` → `+request_completed_duration+` filtered by the +`+/health+` route (path is not tagged on the metric; use the access log +JOIN or filter via a relabel rule at scrape time). +* `+policy deny (405 fast-path)+` → `+request_completed_duration+` AND +`+access_decision_count{decision="deny"}+` correlated in the dashboard, +or — simpler — `+access_decision_count{decision="deny"}+` rate as a +proxy. +* `+exact route allow (proxy 200)+` → `+backend_response_duration+` +(this measures the dial-and-read against BoJ, which is the production +equivalent of the loopback bench scenario). + +PromQL templates: + +[source,promql] +---- +# Gateway pipeline p99 (all paths, all decisions — the headline number) +histogram_quantile(0.99, + sum by (le) ( + rate(http_capability_gateway_request_completed_duration_microseconds_bucket[5m]) + ) +) + +# Backend (BoJ) response p99 — what the gateway sees from BoJ on the allow path +histogram_quantile(0.99, + sum by (le) ( + rate(http_capability_gateway_backend_response_duration_microseconds_bucket[5m]) + ) +) + +# Gateway-attributable overhead (allow path) ≈ +# request_completed_duration − backend_response_duration +# at matching percentiles. PromQL cannot subtract two histogram_quantile expressions +# directly; instead, plot both p99s on the same chart and read the gap. +---- + +*Alert threshold (rollback trigger §5.1):* + +____ +"`p99 latency at the rollout edge ≥ 2× Phase D baseline p99 for ≥ 5 +minutes.`" +____ + +The `+bench/baseline.json+` p99 for the +`+exact route allow (proxy 200)+` scenario is the anchor. Until D-4 +lands real numbers, the alert rule is: + +[source,promql] +---- +# Phase E rollback trigger: gateway p99 ≥ 2× baseline for ≥ 5 minutes. +# Replace ${BASELINE_REQUEST_P99_US} with the value from bench/baseline.json +# after D-4 lands real numbers and _status flips to active. +( + histogram_quantile(0.99, + sum by (le) ( + rate(http_capability_gateway_request_completed_duration_microseconds_bucket[5m]) + ) + ) + > + bool ${BASELINE_REQUEST_P99_US} * 2 +) == 1 +---- + +Alert duration: 5 minutes (matching the runbook trigger). + +==== 2.2 Throughput (ips, per scenario) + +*Runbook signal:* "`Throughput (ips, per scenario).`" + +PromQL: + +[source,promql] +---- +# Total requests/second +rate(http_capability_gateway_request_completed_count_total[1m]) + +# Allow path requests/second (the cost-class equivalent of the proxy-200 scenario) +sum( + rate(http_capability_gateway_access_decision_count_total{decision="allow"}[1m]) +) + +# Deny path requests/second (the cost-class equivalent of the 405 fast-path scenario) +sum( + rate(http_capability_gateway_access_decision_count_total{decision=~"deny|no_match"}[1m]) +) +---- + +*Alert threshold (load profile §2.1):* + +The §2.1 envelope is "`sustained ≥ §1.1 median × 1.5; burst ≥ §1.1 peak +× 1.2`". §1.1 is `+!OWNER:+` (production measurement) so the absolute +number is filled at operator time; the SLO breach rule is: + +[source,promql] +---- +# Phase E throughput breach: sustained throughput exceeds the envelope budget. +# Replace ${SUSTAINED_BUDGET_RPS} with the gateway-load-profile §2.1 value +# computed as: !OWNER: (production median rps) × 1.5. +rate(http_capability_gateway_request_completed_count_total[5m]) + > ${SUSTAINED_BUDGET_RPS} +---- + +Pages as an SLO warning, not as an immediate rollback (the gateway can +absorb modest overshoot — the load profile headroom is already 1.5×). +Persistent breach (≥30 minutes) is a capacity-planning escalation, not a +rollback trigger. + +==== 2.3 Circuit-breaker state + +*Runbook signal:* "`Circuit-breaker state (closed / half-open / open).`" + +The gateway emits no dedicated circuit-breaker telemetry event today (it +only logs state transitions via `+Logger.warning+`). The proxy here is +the *503 rate from the gateway*: + +[source,promql] +---- +# 503s from the gateway proxy path (audit §1.4 K9-contract section + Gateway.enforce_with_contract/5 +# returns 503 when the circuit breaker is open). +# A non-zero rate while access_decision_count{decision="allow"} is also non-zero +# means the gateway accepted the request but the backend was unreachable — exactly +# the circuit-open signal. +sum( + rate(http_capability_gateway_request_completed_count_total[1m]) +) +unless on() ( + sum(rate(http_capability_gateway_backend_forward_count_total[1m])) > 0 +) +---- + +This is approximate — it captures "`request completed without backend +forwarding`", which is the circuit-open behaviour. A precise signal +would require a dedicated +`+[:http_capability_gateway, :circuit_breaker, :state_change]+` event +with `+state+` ∈ `+{closed, half_open, open}+` tags. *Follow-up:* open a +tracking issue in the gateway repo (post-Phase-E, dashboard-quality +improvement, not Phase E blocker). + +*Alert threshold (rollback trigger §5.1):* + +____ +"`Circuit breaker trips ≥ 3 times in any 15-minute window.`" +____ + +A circuit-breaker trip surfaces as a sustained 503 spike from the +`+unless+` query above. Until the dedicated event lands, monitor 503s +and re-derive the trip count manually from the gateway log stream +(`+Logger.warning("Request rejected by circuit breaker", …)+` — +`+gateway.ex:411+`). + +==== 2.4 5xx rate (gateway-origin vs BoJ-passthrough) + +*Runbook signal:* "`5xx rate emitted by the gateway (gateway-origin 5xx +vs BoJ-passthrough 5xx — keep these distinguishable).`" + +Gateway-origin 5xxs (the gateway returned a 5xx without forwarding to +BoJ — circuit-breaker, policy-not-loaded, etc.): + +[source,promql] +---- +# Gateway-origin 5xx ≈ requests that completed with no backend forward. +# (Audit §1.4: 503 from policy-not-loaded path, 503 from circuit breaker, +# 502 from proxy returns 500 — these all skip backend_forward_count +# in the K9-contract failure paths.) +sum(rate(http_capability_gateway_request_completed_count_total[1m])) + - sum(rate(http_capability_gateway_backend_forward_count_total[1m])) +---- + +BoJ-passthrough 5xxs (the gateway forwarded to BoJ, BoJ returned a 5xx): + +[source,promql] +---- +# BoJ-passthrough 5xx: backend_forward_count was incremented but the resulting +# response was 5xx. The gateway does not currently tag response-status on the +# request_completed event, so this requires the BoJ access-log JOIN. +# Until that join exists, use the BoJ-side query in §3.3 below as the +# authoritative passthrough-5xx signal. +---- + +*Alert threshold (rollback trigger §5.1):* + +____ +"`Gateway-origin 5xx rate ≥ 1% of requests for ≥ 5 minutes.`" +____ + +[source,promql] +---- +# Phase E rollback trigger: gateway-origin 5xx ≥ 1% for ≥ 5 minutes. +( + ( + sum(rate(http_capability_gateway_request_completed_count_total[5m])) + - sum(rate(http_capability_gateway_backend_forward_count_total[5m])) + ) + / sum(rate(http_capability_gateway_request_completed_count_total[5m])) +) +> 0.01 +---- + +Alert duration: 5 minutes. + +==== 2.5 Policy reload counter + +*Runbook signal:* "`Policy reload counter (a spike during stable +operation is a misconfiguration alarm).`" + +The gateway logs policy reloads via `+PolicyCompiler.compile/2+` +(`+Logger.info("Policy compiled successfully", …)+`). No dedicated +telemetry event exists yet — the proxy is the policy hot-reload SIGHUP +audit trail at the OS level, or the gateway’s log stream. + +*Phase E posture:* *Treat this signal as a log-based alert, not a +Prometheus signal, until a +`+[:http_capability_gateway, :policy, :reload]+` event is added.* Open +as a tracked follow-up in the gateway repo. The runbook §4.1 lists this +signal explicitly so the operator knows to wire a log-based check; this +spec confirms the absence of a metric path. + +==== 2.6 Trust-level decision distribution + +*Runbook signal:* "`Trust-level decision distribution (`+untrusted+` / +`+authenticated+` / `+internal+`). Sudden shift indicates auth pipeline +change.`" + +PromQL: + +[source,promql] +---- +# Per-trust-level allow rate as a fraction of total allow. +sum by (trust_level) ( + rate(http_capability_gateway_access_decision_count_total{decision="allow"}[5m]) +) +/ on() group_left() +sum( + rate(http_capability_gateway_access_decision_count_total{decision="allow"}[5m]) +) +---- + +*Alert threshold:* No direct rollback trigger. The Minikaran anomaly +detector (§4 below) already covers the "`sudden shift`" case via the +`+trust_shift+` anomaly type — that’s the existing dashboard signal. The +PromQL above feeds a Grafana panel; the actionable alert is the +Minikaran event. + +''''' + +=== 3. Signal → query mapping (rollout runbook §4.2, BoJ-side) + +§4.2 names three BoJ-side signals. The first two require BoJ-emitted +Prometheus metrics; the third is paired with the BoJ-emitted +HTTP-response counter. The metric names referenced below are declared +normatively in the sister spec +link:boj-side-observability-spec.md[`+boj-side-observability-spec.md+`] +§2, which anchors them to telemetry events emitted from +`+elixir/lib/boj_rest/router.ex+` — that spec is the contract for the +BoJ-side wiring PR that lands the actual emission. Until that wiring PR +lands, the queries here remain _templates_ (the metric names do not yet +resolve in the BoJ exporter, because no exporter is mounted); the +rollout-runbook §1.4 prerequisite tracks the wiring as a +stop-the-rollout condition for §3.1 sign-off. + +==== 3.1 Per-route trust-class distribution + +*Runbook signal:* "`Per-route trust-class distribution from +`+BojRest.Router+` decisions.`" + +PromQL (against the +`+boj_router_decision_count_total{route, verb, trust_class, outcome}+` +counter declared in `+boj-side-observability-spec.md+` §1 + §2): + +[source,promql] +---- +# Per-route trust-class distribution. +sum by (route, trust_class) ( + rate(boj_router_decision_count_total[5m]) +) +---- + +BoJ does not currently expose this metric. The wiring contract — events, +metric names, instrumentation sites, and `+/metrics+` exposure policy — +is `+docs/integration/boj-side-observability-spec.md+`. The +rollout-runbook §1.4 prerequisite tracks the wiring PR as a +stop-the-rollout condition for Phase E §3.1 (10% traffic). + +==== 3.2 `+X-Trust-Level+` from non-loopback peers — should be zero + +*Runbook signal:* "``+X-Trust-Level+` arriving from non-loopback peers — +should be zero. Any non-zero is a deployment defect (back-side bind +exposed).`" + +The strongest enforcement is at the network layer (NetworkPolicy, +firewall — landed via boj-server#173, runbook §1.4). This signal +verifies the _invariant_; non-zero is a deployment defect that +NetworkPolicy did not catch. + +PromQL (against the +`+boj_router_trust_level_present_count_total{remote_origin}+` counter +declared in `+boj-side-observability-spec.md+` §1 + §2; +`+remote_origin+` ∈ `+{loopback, non_loopback}+`): + +[source,promql] +---- +# X-Trust-Level from non-loopback peers — must remain at zero. +sum( + rate(boj_router_trust_level_present_count_total{remote_origin!="loopback"}[5m]) +) +> 0 +---- + +*Alert threshold (rollback trigger §5.1):* + +____ +"`BoJ access logs show `+X-Trust-Level+` from non-loopback peers (a §3 +invariant 4 violation in flight — the back-side bind is exposed).`" +____ + +Any non-zero rate is the trigger. Immediate page; this is a §3 contract +invariant violation. + +==== 3.3 BoJ 5xx rate (independent of gateway’s view) + +*Runbook signal:* "`BoJ 5xx rate (independent of gateway’s view).`" + +PromQL (against the `+boj_http_responses_total{status, route}+` counter +declared in `+boj-side-observability-spec.md+` §1 + §2): + +[source,promql] +---- +# BoJ-emitted 5xx rate as a fraction of BoJ-handled requests. +sum(rate(boj_http_responses_total{status=~"5.."}[5m])) +/ sum(rate(boj_http_responses_total[5m])) +---- + +*Alert threshold:* Pair with §2.4 above. A divergence between +gateway-origin 5xx (§2.4) and BoJ 5xx (§3.3) localises the fault: +gateway-origin > BoJ → fault is in the gateway pipeline +(circuit-breaker, K9-contract, policy); BoJ 5xx > gateway-origin → fault +is downstream of the gateway (cartridge crash, BoJ-internal). The +runbook §5.1 5xx trigger is gateway-origin only — BoJ 5xx is a +_diagnostic_ signal, not a rollback trigger. + +''''' + +=== 4. Minikaran anomaly endpoint — secondary signal path + +The gateway emits `+[:http_capability_gateway, :minikaran, :anomaly]+` +events tagged with `+type+` (audit §1.6), exported as +`+http_capability_gateway_minikaran_anomaly_count_total{type}+`. The +Minikaran handler also offers a JSON endpoint at `+/api/v1/minikaran+` +(gateway `+lib/http_capability_gateway/gateway.ex:228-232+`) that +returns the current anomalies, baseline summary, and operational status. + +Five anomaly types (audit §5 dashboard subsection): + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Anomaly type |Meaning |Maps to runbook signal +|`+traffic_spike+` |Path-level traffic above learned baseline. |§4.1 +throughput. + +|`+trust_shift+` |Per-trust-level rate shifted from baseline +distribution. |§4.1 trust-level decision distribution. + +|`+latency_spike+` |Per-percentile latency above learned baseline. |§4.1 +p50/p95/p99 latency. + +|`+path_novelty+` |New path appeared (potential scan or new client). +|§4.1 5xx rate (path-novel scans usually yield 4xx). + +|`+error_spike+` |Error-rate above learned baseline. |§4.1 5xx rate. +|=== + +PromQL: + +[source,promql] +---- +# Anomaly rate by type (sum across all types should be near-zero in steady state). +sum by (type) ( + rate(http_capability_gateway_minikaran_anomaly_count_total[5m]) +) +---- + +Minikaran is a *complementary* signal — it catches drift the +percentile-and-rate queries above miss (sudden-but-modest distribution +shift, novel path appearing under the gateway). Phase E posture: gate +the §3.1 sign-off on Minikaran reporting baseline established +(`+GET /api/v1/minikaran+` returns `+status.status == "active"+`); use +the anomaly counter as a _paged_ signal only when the on-call has time +to triage (it is noisier by design than the strict-percentile alerts). + +''''' + +=== 5. Alert rules summary + +The rollback runbook §5.1 lists six triggers. The table below maps each +trigger to the PromQL alert rule in this spec. + +[width="100%",cols="25%,25%,25%,25%",options="header",] +|=== +|Runbook §5.1 trigger |Spec section |PromQL anchor |Severity +|p99 latency at the rollout edge ≥ 2× Phase D baseline p99 for ≥ 5 +minutes |§2.1 |`+request_completed_duration_microseconds_bucket+` × 2 +|page on-call + +|Gateway-origin 5xx rate ≥ 1% for ≥ 5 minutes |§2.4 +|`+(request_completed − backend_forward) / request_completed+` > 0.01 +|page on-call + +|Circuit breaker trips ≥ 3 times in any 15-minute window |§2.3 |derived +from 503 spike + log inspection until dedicated event lands |page +on-call + +|BoJ access logs show `+X-Trust-Level+` from non-loopback peers |§3.2 +|`+boj_router_trust_level_present_count{remote_origin!="loopback"}+` > 0 +|page on-call (any non-zero) + +|VeriSimDB / audit-trail write failures ≥ 1% |§4 + audit §4 |not yet +exposed as a Prometheus metric (VeriSimDB integration is +`+audit_allow/audit_deny+` cast-only) |follow-up: emit +`+[:http_capability_gateway, :verisimdb, :write_failure]+` event + +|On-call judgement |— |— |always — overrides the rules above +|=== + +The last-row "`follow-up`" row is a gap this spec surfaces. Phase E §1.3 +already lists VeriSimDB integration status as an `+!OWNER:+` +confirmation; if VeriSimDB is confirmed as a real integration (not a +stub), the write-failure metric becomes a deferred deliverable on the +gateway side (open a tracking issue post-Phase-E). + +''''' + +=== 6. Phase E acceptance — how this spec gates §3 traffic shift + +The rollout runbook §3.1 ("`10% traffic`") success criteria are: + +____ +* p99 latency at production endpoints within Phase D baseline × 1.5 (the +perf-regression p99 tolerance). +* 5xx rate not elevated vs the BoJ-direct baseline (same 24-hour window +the previous day). +* No circuit-breaker trips on the gateway. +* No `+X-Trust-Level+` mismatches in BoJ access logs (gateway should be +the only source). +____ + +This spec gives the operator one PromQL query per success criterion +(§2.1, §2.4, §2.3, §3.2). The §3.1 sign-off is a green-on-all-four +check; the dashboard built from these queries is the human-readable +surface of that check. + +Phase E §3.4 (decommission BoJ direct external access) further requires +all queries above run green for the §3.3 7-day soak window. The PromQL +templates here remain unchanged across that window — the soak is a +duration, not a different signal set. + +''''' + +=== 7. References + +* Rollout runbook — `+docs/integration/hcg-tier2-rollout-runbook.md+` +(§4 signal list, §5 rollback triggers, §6 Trustfile flip). +* Sister spec (BoJ side) — +`+docs/integration/boj-side-observability-spec.md+` (events, metric +names, instrumentation sites for the §3 templates above). +* Load profile — `+docs/integration/gateway-load-profile.md+` (§2 SLO +budgets, §3.4 bench harness reference). +* Audit — `+docs/integration/http-capability-gateway-audit.md+` (§1.6 +telemetry, §5 telemetry shape, §1.4 mTLS path notes). +* Plan — `+docs/integration/http-capability-gateway-plan.md+` (§Phase E +E3 telemetry verification). +* Perf contract (gateway side) — +`+http-capability-gateway/docs/perf-contract.md+` (tolerance ratios that +anchor §2.1). +* Gateway metric definitions — +`+http-capability-gateway/lib/http_capability_gateway/application.ex+` +`+telemetry_metrics/0+` (lines 259–296). +* Gateway request flow — +`+http-capability-gateway/lib/http_capability_gateway/gateway.ex+` (§2.3 +circuit-breaker behaviour, §2.6 trust-level extraction). diff --git a/docs/integration/gateway-observability-spec.md b/docs/integration/gateway-observability-spec.md deleted file mode 100644 index 7fc2711f..00000000 --- a/docs/integration/gateway-observability-spec.md +++ /dev/null @@ -1,412 +0,0 @@ - - - -# HCG tier-2 — observability spec - -**Version:** 0.2 (BoJ-side sister spec anchored, Phase E) -**Date:** 2026-06-22 (rev. from 2026-06-16) -**Status:** Phase E scaffold. Names the gateway-emitted Prometheus metrics, gives PromQL templates for every signal listed in the rollout runbook §4.1/§4.2, and binds alert thresholds to the rollback triggers in runbook §5.1 and the perf contract's tolerance ratios. §3 BoJ-side templates now anchor to the sister spec [`boj-side-observability-spec.md`](boj-side-observability-spec.md) (events, metric names, instrumentation sites in `BojRest.Router`); the `!OWNER:` scaffold qualifier on those templates is dropped — the wiring PR target is fixed. Absolute-µs values are deliberately left as `Phase D-4` references — once `bench/baseline.json` `_status` flips to `active` the queries here read against real numbers without further edits. -**ADR:** [`docs/decisions/0004-adopt-http-capability-gateway.md`](../decisions/0004-adopt-http-capability-gateway.md) -**Plan:** [`docs/integration/http-capability-gateway-plan.md`](http-capability-gateway-plan.md) (§ Phase E, E3 telemetry verification) -**Contract:** [`docs/integration/http-capability-gateway-boj-contract.md`](http-capability-gateway-boj-contract.md) -**Rollout runbook:** [`docs/integration/hcg-tier2-rollout-runbook.md`](hcg-tier2-rollout-runbook.md) (§ 4 signals, § 5 rollback) -**Load profile:** [`docs/integration/gateway-load-profile.md`](gateway-load-profile.md) (§ 2 SLO budgets) -**Perf contract (gateway side):** [`http-capability-gateway/docs/perf-contract.md`](https://github.com/hyperpolymath/http-capability-gateway/blob/main/docs/perf-contract.md) -**Tracking:** [`standards#91`](https://github.com/hyperpolymath/standards/issues/91) (parent), [`standards#100`](https://github.com/hyperpolymath/standards/issues/100) (Phase E) - -> **File-format note.** Matches sibling integration docs (`http-capability-gateway-{plan,audit,boj-contract,policy-authoring}.md`, `gateway-load-profile.md`, `hcg-tier2-rollout-runbook.md`); the rollout runbook §4 anchors all signals here by exact path. The estate `.adoc` default is deliberately overridden for the `docs/integration/` set. - ---- - -## 0. Scope - -This document is the declarative half of Phase E §4 "Observability — what on-call watches". The runbook §4 names the signals at the human level ("p99 latency", "circuit-breaker state", "trust-level decision distribution"); this spec wires each signal to: - -1. The **telemetry event** emitted by the gateway (audit document §5). -2. The **Prometheus metric** the `TelemetryMetricsPrometheus.Core` reporter exports for that event (gateway `lib/http_capability_gateway/application.ex` `telemetry_metrics/0`, lines 259–296). -3. A **PromQL query template** an on-call dashboard or alerting rule can paste verbatim. -4. An **alert threshold** anchored to a canonical source — the rollback runbook §5.1 trigger value, the perf contract tolerance ratio, or the load-profile SLO budget. Where the absolute number depends on Phase D-4 baseline collection, the spec names the formula and the lookup site instead of inventing a value. - -In scope: - -- Every signal listed in rollout runbook §4.1 (gateway-side) and §4.2 (BoJ-side). -- The mapping from rollback trigger (§5.1) to the alerting rule that fires it. -- The Minikaran anomaly endpoint as a secondary, complementary signal path. - -Out of scope: - -- Dashboard authoring (the !OWNER: rows in runbook §4.3 — the dashboard URL, the on-call rota). This spec gives the operator the queries; choosing the dashboard tool (Grafana / Cloudflare analytics / something else) is owner-driven per the runbook's existing scoping. -- Cloudflare-edge metrics (tier 1). Covered separately in `Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY]`. -- BoJ-internal cartridge or cartridge-tool metrics. The gateway sees BoJ as a single backend; cartridge-level observability is downstream. -- Long-term storage / retention policy. The spec defines what to scrape; how long to keep scrapes is an operator decision (the §4.3 dashboard URL row already covers that scope). - ---- - -## 1. Gateway-side metrics inventory - -Every gateway-emitted telemetry event has a corresponding Prometheus metric. The mapping below is normative; if a future PR adds a new event without a metric (or vice versa) the rollout runbook §1.5 smoke pre-check should fail before traffic shift. - -The Prometheus metric names follow the `telemetry_metrics_prometheus_core` convention: dots become underscores, distribution metrics expose `_bucket`, `_count`, `_sum` series, counters expose a `_total` series. The names below are the metric prefixes the operator sees in `/metrics`. - -| Telemetry event (audit §5) | Prometheus metric prefix | Type | Tags | Source | -|---|---|---|---|---| -| `[:http_capability_gateway, :request, :received]` | `http_capability_gateway_request_received_count` | gauge (last_value) | — | `application.ex:262` | -| `[:http_capability_gateway, :request, :completed]` | `http_capability_gateway_request_completed_count` | counter | — | `application.ex:263` | -| `[:http_capability_gateway, :request, :completed]` | `http_capability_gateway_request_completed_duration` | distribution | — | `application.ex:264-267` | -| `[:http_capability_gateway, :policy, :lookup]` | `http_capability_gateway_policy_lookup_duration` | distribution | — | `application.ex:270-273` | -| `[:http_capability_gateway, :access_decision]` | `http_capability_gateway_access_decision_count` | counter | `decision`, `verb`, `trust_level` | `application.ex:276-278` | -| `[:http_capability_gateway, :backend, :forward]` | `http_capability_gateway_backend_forward_count` | counter | — | `application.ex:281` | -| `[:http_capability_gateway, :backend, :response]` | `http_capability_gateway_backend_response_duration` | distribution | — | `application.ex:282-285` | -| `[:http_capability_gateway, :error]` | `http_capability_gateway_error_count` | counter | `error_type` | `application.ex:288` | -| `[:http_capability_gateway, :minikaran, :anomaly]` | `http_capability_gateway_minikaran_anomaly_count` | counter | `type` | `application.ex:293-295` | - -### 1.1 Distribution buckets - -Buckets are declared in microseconds and capture the gateway's own pipeline cost, not end-to-end RTT. The buckets are wide enough to accommodate the perf contract's six scenarios: - -| Metric | Buckets (µs) | Source | -|---|---|---| -| `request_completed_duration` | `[100, 500, 1_000, 5_000, 10_000, 30_000]` | `application.ex:266` | -| `policy_lookup_duration` | `[10, 50, 100, 500, 1_000]` | `application.ex:272` | -| `backend_response_duration` | `[100, 500, 1_000, 5_000, 10_000, 30_000, 60_000]` | `application.ex:284` | - -The `backend_response_duration` upper bucket is 60 ms because BoJ-attributable latency may include cartridge invocation; the gateway-attributable buckets stop at 30 ms because the load profile §2.2 budget puts p99 well under that. - -### 1.2 Tags - -Three counters are tagged. Tag cardinality is bounded: - -- `decision` ∈ `{allow, deny, no_match, error}` — four-value set, no cardinality blow-up. -- `verb` ∈ `{GET, POST, PUT, DELETE, PATCH, HEAD, OPTIONS}` — seven-value allowlist enforced by `Gateway.safe_verb/1` (gateway `lib/http_capability_gateway/gateway.ex:65-77`); unknown methods short-circuit before reaching the access-decision event. -- `trust_level` ∈ `{untrusted, authenticated, internal}` — three-value set enforced by `SafeTrust.parse_trust/1`. -- `error_type` — open vocabulary but bounded by the gateway's enumerated error paths. Operator should monitor for cardinality growth here as a deployment-defect signal. -- `type` on `minikaran_anomaly_count` ∈ `{traffic_spike, trust_shift, latency_spike, path_novelty, error_spike}` — five-value set declared in the audit §1.6. - -Total decision counter cardinality bound: `4 × 7 × 3 = 84` time series at saturation. Below the threshold where Prometheus storage becomes a concern. - ---- - -## 2. Signal → query mapping (rollout runbook §4.1, gateway-side) - -The runbook lists six gateway-side signals. Each subsection below names one, gives its PromQL query, and binds an alert threshold. - -### 2.1 p50 / p95 / p99 latency per scenario - -**Runbook signal:** "p50/p95/p99 latency per scenario (health / policy-deny fast-path / proxy allow)." - -The harness scenarios (perf contract §Scenarios) are bench-only — production traffic doesn't carry a "scenario" tag — so the production equivalent is per-decision-class: - -- `health endpoint` → `request_completed_duration` filtered by the `/health` route (path is not tagged on the metric; use the access log JOIN or filter via a relabel rule at scrape time). -- `policy deny (405 fast-path)` → `request_completed_duration` AND `access_decision_count{decision="deny"}` correlated in the dashboard, or — simpler — `access_decision_count{decision="deny"}` rate as a proxy. -- `exact route allow (proxy 200)` → `backend_response_duration` (this measures the dial-and-read against BoJ, which is the production equivalent of the loopback bench scenario). - -PromQL templates: - -```promql -# Gateway pipeline p99 (all paths, all decisions — the headline number) -histogram_quantile(0.99, - sum by (le) ( - rate(http_capability_gateway_request_completed_duration_microseconds_bucket[5m]) - ) -) - -# Backend (BoJ) response p99 — what the gateway sees from BoJ on the allow path -histogram_quantile(0.99, - sum by (le) ( - rate(http_capability_gateway_backend_response_duration_microseconds_bucket[5m]) - ) -) - -# Gateway-attributable overhead (allow path) ≈ -# request_completed_duration − backend_response_duration -# at matching percentiles. PromQL cannot subtract two histogram_quantile expressions -# directly; instead, plot both p99s on the same chart and read the gap. -``` - -**Alert threshold (rollback trigger §5.1):** - -> "p99 latency at the rollout edge ≥ 2× Phase D baseline p99 for ≥ 5 minutes." - -The `bench/baseline.json` p99 for the `exact route allow (proxy 200)` scenario is the anchor. Until D-4 lands real numbers, the alert rule is: - -```promql -# Phase E rollback trigger: gateway p99 ≥ 2× baseline for ≥ 5 minutes. -# Replace ${BASELINE_REQUEST_P99_US} with the value from bench/baseline.json -# after D-4 lands real numbers and _status flips to active. -( - histogram_quantile(0.99, - sum by (le) ( - rate(http_capability_gateway_request_completed_duration_microseconds_bucket[5m]) - ) - ) - > - bool ${BASELINE_REQUEST_P99_US} * 2 -) == 1 -``` - -Alert duration: 5 minutes (matching the runbook trigger). - -### 2.2 Throughput (ips, per scenario) - -**Runbook signal:** "Throughput (ips, per scenario)." - -PromQL: - -```promql -# Total requests/second -rate(http_capability_gateway_request_completed_count_total[1m]) - -# Allow path requests/second (the cost-class equivalent of the proxy-200 scenario) -sum( - rate(http_capability_gateway_access_decision_count_total{decision="allow"}[1m]) -) - -# Deny path requests/second (the cost-class equivalent of the 405 fast-path scenario) -sum( - rate(http_capability_gateway_access_decision_count_total{decision=~"deny|no_match"}[1m]) -) -``` - -**Alert threshold (load profile §2.1):** - -The §2.1 envelope is "sustained ≥ §1.1 median × 1.5; burst ≥ §1.1 peak × 1.2". §1.1 is `!OWNER:` (production measurement) so the absolute number is filled at operator time; the SLO breach rule is: - -```promql -# Phase E throughput breach: sustained throughput exceeds the envelope budget. -# Replace ${SUSTAINED_BUDGET_RPS} with the gateway-load-profile §2.1 value -# computed as: !OWNER: (production median rps) × 1.5. -rate(http_capability_gateway_request_completed_count_total[5m]) - > ${SUSTAINED_BUDGET_RPS} -``` - -Pages as an SLO warning, not as an immediate rollback (the gateway can absorb modest overshoot — the load profile headroom is already 1.5×). Persistent breach (≥30 minutes) is a capacity-planning escalation, not a rollback trigger. - -### 2.3 Circuit-breaker state - -**Runbook signal:** "Circuit-breaker state (closed / half-open / open)." - -The gateway emits no dedicated circuit-breaker telemetry event today (it only logs state transitions via `Logger.warning`). The proxy here is the **503 rate from the gateway**: - -```promql -# 503s from the gateway proxy path (audit §1.4 K9-contract section + Gateway.enforce_with_contract/5 -# returns 503 when the circuit breaker is open). -# A non-zero rate while access_decision_count{decision="allow"} is also non-zero -# means the gateway accepted the request but the backend was unreachable — exactly -# the circuit-open signal. -sum( - rate(http_capability_gateway_request_completed_count_total[1m]) -) -unless on() ( - sum(rate(http_capability_gateway_backend_forward_count_total[1m])) > 0 -) -``` - -This is approximate — it captures "request completed without backend forwarding", which is the circuit-open behaviour. A precise signal would require a dedicated `[:http_capability_gateway, :circuit_breaker, :state_change]` event with `state` ∈ `{closed, half_open, open}` tags. **Follow-up:** open a tracking issue in the gateway repo (post-Phase-E, dashboard-quality improvement, not Phase E blocker). - -**Alert threshold (rollback trigger §5.1):** - -> "Circuit breaker trips ≥ 3 times in any 15-minute window." - -A circuit-breaker trip surfaces as a sustained 503 spike from the `unless` query above. Until the dedicated event lands, monitor 503s and re-derive the trip count manually from the gateway log stream (`Logger.warning("Request rejected by circuit breaker", …)` — `gateway.ex:411`). - -### 2.4 5xx rate (gateway-origin vs BoJ-passthrough) - -**Runbook signal:** "5xx rate emitted by the gateway (gateway-origin 5xx vs BoJ-passthrough 5xx — keep these distinguishable)." - -Gateway-origin 5xxs (the gateway returned a 5xx without forwarding to BoJ — circuit-breaker, policy-not-loaded, etc.): - -```promql -# Gateway-origin 5xx ≈ requests that completed with no backend forward. -# (Audit §1.4: 503 from policy-not-loaded path, 503 from circuit breaker, -# 502 from proxy returns 500 — these all skip backend_forward_count -# in the K9-contract failure paths.) -sum(rate(http_capability_gateway_request_completed_count_total[1m])) - - sum(rate(http_capability_gateway_backend_forward_count_total[1m])) -``` - -BoJ-passthrough 5xxs (the gateway forwarded to BoJ, BoJ returned a 5xx): - -```promql -# BoJ-passthrough 5xx: backend_forward_count was incremented but the resulting -# response was 5xx. The gateway does not currently tag response-status on the -# request_completed event, so this requires the BoJ access-log JOIN. -# Until that join exists, use the BoJ-side query in §3.3 below as the -# authoritative passthrough-5xx signal. -``` - -**Alert threshold (rollback trigger §5.1):** - -> "Gateway-origin 5xx rate ≥ 1% of requests for ≥ 5 minutes." - -```promql -# Phase E rollback trigger: gateway-origin 5xx ≥ 1% for ≥ 5 minutes. -( - ( - sum(rate(http_capability_gateway_request_completed_count_total[5m])) - - sum(rate(http_capability_gateway_backend_forward_count_total[5m])) - ) - / sum(rate(http_capability_gateway_request_completed_count_total[5m])) -) -> 0.01 -``` - -Alert duration: 5 minutes. - -### 2.5 Policy reload counter - -**Runbook signal:** "Policy reload counter (a spike during stable operation is a misconfiguration alarm)." - -The gateway logs policy reloads via `PolicyCompiler.compile/2` (`Logger.info("Policy compiled successfully", …)`). No dedicated telemetry event exists yet — the proxy is the policy hot-reload SIGHUP audit trail at the OS level, or the gateway's log stream. - -**Phase E posture:** **Treat this signal as a log-based alert, not a Prometheus signal, until a `[:http_capability_gateway, :policy, :reload]` event is added.** Open as a tracked follow-up in the gateway repo. The runbook §4.1 lists this signal explicitly so the operator knows to wire a log-based check; this spec confirms the absence of a metric path. - -### 2.6 Trust-level decision distribution - -**Runbook signal:** "Trust-level decision distribution (`untrusted` / `authenticated` / `internal`). Sudden shift indicates auth pipeline change." - -PromQL: - -```promql -# Per-trust-level allow rate as a fraction of total allow. -sum by (trust_level) ( - rate(http_capability_gateway_access_decision_count_total{decision="allow"}[5m]) -) -/ on() group_left() -sum( - rate(http_capability_gateway_access_decision_count_total{decision="allow"}[5m]) -) -``` - -**Alert threshold:** No direct rollback trigger. The Minikaran anomaly detector (§4 below) already covers the "sudden shift" case via the `trust_shift` anomaly type — that's the existing dashboard signal. The PromQL above feeds a Grafana panel; the actionable alert is the Minikaran event. - ---- - -## 3. Signal → query mapping (rollout runbook §4.2, BoJ-side) - -§4.2 names three BoJ-side signals. The first two require BoJ-emitted Prometheus metrics; the third is paired with the BoJ-emitted HTTP-response counter. The metric names referenced below are declared normatively in the sister spec [`boj-side-observability-spec.md`](boj-side-observability-spec.md) §2, which anchors them to telemetry events emitted from `elixir/lib/boj_rest/router.ex` — that spec is the contract for the BoJ-side wiring PR that lands the actual emission. Until that wiring PR lands, the queries here remain *templates* (the metric names do not yet resolve in the BoJ exporter, because no exporter is mounted); the rollout-runbook §1.4 prerequisite tracks the wiring as a stop-the-rollout condition for §3.1 sign-off. - -### 3.1 Per-route trust-class distribution - -**Runbook signal:** "Per-route trust-class distribution from `BojRest.Router` decisions." - -PromQL (against the `boj_router_decision_count_total{route, verb, trust_class, outcome}` counter declared in `boj-side-observability-spec.md` §1 + §2): - -```promql -# Per-route trust-class distribution. -sum by (route, trust_class) ( - rate(boj_router_decision_count_total[5m]) -) -``` - -BoJ does not currently expose this metric. The wiring contract — events, metric names, instrumentation sites, and `/metrics` exposure policy — is `docs/integration/boj-side-observability-spec.md`. The rollout-runbook §1.4 prerequisite tracks the wiring PR as a stop-the-rollout condition for Phase E §3.1 (10% traffic). - -### 3.2 `X-Trust-Level` from non-loopback peers — should be zero - -**Runbook signal:** "`X-Trust-Level` arriving from non-loopback peers — should be zero. Any non-zero is a deployment defect (back-side bind exposed)." - -The strongest enforcement is at the network layer (NetworkPolicy, firewall — landed via boj-server#173, runbook §1.4). This signal verifies the *invariant*; non-zero is a deployment defect that NetworkPolicy did not catch. - -PromQL (against the `boj_router_trust_level_present_count_total{remote_origin}` counter declared in `boj-side-observability-spec.md` §1 + §2; `remote_origin` ∈ `{loopback, non_loopback}`): - -```promql -# X-Trust-Level from non-loopback peers — must remain at zero. -sum( - rate(boj_router_trust_level_present_count_total{remote_origin!="loopback"}[5m]) -) -> 0 -``` - -**Alert threshold (rollback trigger §5.1):** - -> "BoJ access logs show `X-Trust-Level` from non-loopback peers (a §3 invariant 4 violation in flight — the back-side bind is exposed)." - -Any non-zero rate is the trigger. Immediate page; this is a §3 contract invariant violation. - -### 3.3 BoJ 5xx rate (independent of gateway's view) - -**Runbook signal:** "BoJ 5xx rate (independent of gateway's view)." - -PromQL (against the `boj_http_responses_total{status, route}` counter declared in `boj-side-observability-spec.md` §1 + §2): - -```promql -# BoJ-emitted 5xx rate as a fraction of BoJ-handled requests. -sum(rate(boj_http_responses_total{status=~"5.."}[5m])) -/ sum(rate(boj_http_responses_total[5m])) -``` - -**Alert threshold:** Pair with §2.4 above. A divergence between gateway-origin 5xx (§2.4) and BoJ 5xx (§3.3) localises the fault: gateway-origin > BoJ → fault is in the gateway pipeline (circuit-breaker, K9-contract, policy); BoJ 5xx > gateway-origin → fault is downstream of the gateway (cartridge crash, BoJ-internal). The runbook §5.1 5xx trigger is gateway-origin only — BoJ 5xx is a *diagnostic* signal, not a rollback trigger. - ---- - -## 4. Minikaran anomaly endpoint — secondary signal path - -The gateway emits `[:http_capability_gateway, :minikaran, :anomaly]` events tagged with `type` (audit §1.6), exported as `http_capability_gateway_minikaran_anomaly_count_total{type}`. The Minikaran handler also offers a JSON endpoint at `/api/v1/minikaran` (gateway `lib/http_capability_gateway/gateway.ex:228-232`) that returns the current anomalies, baseline summary, and operational status. - -Five anomaly types (audit §5 dashboard subsection): - -| Anomaly type | Meaning | Maps to runbook signal | -|---|---|---| -| `traffic_spike` | Path-level traffic above learned baseline. | §4.1 throughput. | -| `trust_shift` | Per-trust-level rate shifted from baseline distribution. | §4.1 trust-level decision distribution. | -| `latency_spike` | Per-percentile latency above learned baseline. | §4.1 p50/p95/p99 latency. | -| `path_novelty` | New path appeared (potential scan or new client). | §4.1 5xx rate (path-novel scans usually yield 4xx). | -| `error_spike` | Error-rate above learned baseline. | §4.1 5xx rate. | - -PromQL: - -```promql -# Anomaly rate by type (sum across all types should be near-zero in steady state). -sum by (type) ( - rate(http_capability_gateway_minikaran_anomaly_count_total[5m]) -) -``` - -Minikaran is a **complementary** signal — it catches drift the percentile-and-rate queries above miss (sudden-but-modest distribution shift, novel path appearing under the gateway). Phase E posture: gate the §3.1 sign-off on Minikaran reporting baseline established (`GET /api/v1/minikaran` returns `status.status == "active"`); use the anomaly counter as a *paged* signal only when the on-call has time to triage (it is noisier by design than the strict-percentile alerts). - ---- - -## 5. Alert rules summary - -The rollback runbook §5.1 lists six triggers. The table below maps each trigger to the PromQL alert rule in this spec. - -| Runbook §5.1 trigger | Spec section | PromQL anchor | Severity | -|---|---|---|---| -| p99 latency at the rollout edge ≥ 2× Phase D baseline p99 for ≥ 5 minutes | §2.1 | `request_completed_duration_microseconds_bucket` × 2 | page on-call | -| Gateway-origin 5xx rate ≥ 1% for ≥ 5 minutes | §2.4 | `(request_completed − backend_forward) / request_completed` > 0.01 | page on-call | -| Circuit breaker trips ≥ 3 times in any 15-minute window | §2.3 | derived from 503 spike + log inspection until dedicated event lands | page on-call | -| BoJ access logs show `X-Trust-Level` from non-loopback peers | §3.2 | `boj_router_trust_level_present_count{remote_origin!="loopback"}` > 0 | page on-call (any non-zero) | -| VeriSimDB / audit-trail write failures ≥ 1% | §4 + audit §4 | not yet exposed as a Prometheus metric (VeriSimDB integration is `audit_allow/audit_deny` cast-only) | follow-up: emit `[:http_capability_gateway, :verisimdb, :write_failure]` event | -| On-call judgement | — | — | always — overrides the rules above | - -The last-row "follow-up" row is a gap this spec surfaces. Phase E §1.3 already lists VeriSimDB integration status as an `!OWNER:` confirmation; if VeriSimDB is confirmed as a real integration (not a stub), the write-failure metric becomes a deferred deliverable on the gateway side (open a tracking issue post-Phase-E). - ---- - -## 6. Phase E acceptance — how this spec gates §3 traffic shift - -The rollout runbook §3.1 ("10% traffic") success criteria are: - -> - p99 latency at production endpoints within Phase D baseline × 1.5 (the perf-regression p99 tolerance). -> - 5xx rate not elevated vs the BoJ-direct baseline (same 24-hour window the previous day). -> - No circuit-breaker trips on the gateway. -> - No `X-Trust-Level` mismatches in BoJ access logs (gateway should be the only source). - -This spec gives the operator one PromQL query per success criterion (§2.1, §2.4, §2.3, §3.2). The §3.1 sign-off is a green-on-all-four check; the dashboard built from these queries is the human-readable surface of that check. - -Phase E §3.4 (decommission BoJ direct external access) further requires all queries above run green for the §3.3 7-day soak window. The PromQL templates here remain unchanged across that window — the soak is a duration, not a different signal set. - ---- - -## 7. References - -- Rollout runbook — `docs/integration/hcg-tier2-rollout-runbook.md` (§4 signal list, §5 rollback triggers, §6 Trustfile flip). -- Sister spec (BoJ side) — `docs/integration/boj-side-observability-spec.md` (events, metric names, instrumentation sites for the §3 templates above). -- Load profile — `docs/integration/gateway-load-profile.md` (§2 SLO budgets, §3.4 bench harness reference). -- Audit — `docs/integration/http-capability-gateway-audit.md` (§1.6 telemetry, §5 telemetry shape, §1.4 mTLS path notes). -- Plan — `docs/integration/http-capability-gateway-plan.md` (§Phase E E3 telemetry verification). -- Perf contract (gateway side) — `http-capability-gateway/docs/perf-contract.md` (tolerance ratios that anchor §2.1). -- Gateway metric definitions — `http-capability-gateway/lib/http_capability_gateway/application.ex` `telemetry_metrics/0` (lines 259–296). -- Gateway request flow — `http-capability-gateway/lib/http_capability_gateway/gateway.ex` (§2.3 circuit-breaker behaviour, §2.6 trust-level extraction). diff --git a/docs/integration/hcg-tier2-rollout-runbook.adoc b/docs/integration/hcg-tier2-rollout-runbook.adoc new file mode 100644 index 00000000..2d8da4d3 --- /dev/null +++ b/docs/integration/hcg-tier2-rollout-runbook.adoc @@ -0,0 +1,636 @@ +== HCG tier-2 — rollout & rollback runbook + +*Version:* 0.8 (BoJ-side observability spec prereq, Phase E in-progress) +*Date:* 2026-06-22 (rev. from 2026-06-15) *Status:* Phase E deliverables +E1 (deploy spec) + E5 (rollback runbook) drafted; live gateway policy +(`+config/gateway-policy-boj.yaml+`) promoted from the worked example +(§1.5); `+scripts/hcg-policy-smoke.sh+` lands as the checked-in §1.5 +operator pre-check (deny-path covers gateway-alone; `+--with-backend+` +adds allow-path coverage); §1.5 verb-canary block covers OPTIONS, +regex-route DELETE, and wrong-verb-on-listed-path; a path-canary +exercises the no-match default-deny branch (synthetic unknown path with +a `+global_verbs+` verb); a stealth-profile canary pins the deny _status +code_ — internal+stealth routes must return 404 (capability existence +hidden) and authenticated routes must return 403, so a +missing-`+:stealth_profiles+` misconfiguration is caught instead of +slipping past the generic any-4xx deny pattern; and +`+docs/integration/boj-side-observability-spec.md+` now declares the +BoJ-side telemetry events + Prometheus metric names that back the §4.2 +signals — wired into §1.4 as a stop-the-rollout prerequisite, with the +actual `+BojRest.Router+` emission left as a follow-up PR per the spec’s +§7 checklist. Owner-input markers (`+!OWNER:+`) remain to be filled +before any traffic-shift action is taken. *ADR:* +link:../decisions/0004-adopt-http-capability-gateway.md[`+docs/decisions/0004-adopt-http-capability-gateway.md+`] +*Plan:* +link:http-capability-gateway-plan.md[`+docs/integration/http-capability-gateway-plan.md+`] +(§ Phase E) *Contract:* +link:http-capability-gateway-boj-contract.md[`+docs/integration/http-capability-gateway-boj-contract.md+`] +*Tracking:* +https://github.com/hyperpolymath/standards/issues/91[`+standards#91+`] +(parent), +https://github.com/hyperpolymath/standards/issues/100[`+standards#100+`] +(Phase E) + +____ +*File-format note.* Matches sibling integration docs +(`+http-capability-gateway-{plan,audit,boj-contract,policy-authoring}.md+`); +the integration plan §E5 normatively prescribes `+.md+` for the rollback +runbook (the wider rollout-and-rollback scope folds in here per +acceptance criterion 3 of `+standards#100+`). The estate `+.adoc+` +default is deliberately overridden for the `+docs/integration/+` set. +____ + +____ +*Phase-D status (2026-06-08).* Phase D (`+standards#99+`) is *closed* — +joint-closed via boj-server#168 (D-1 load profile, 2026-06-01) on top of +the gateway-side D-1..D-3 scaffold + D-4 bootstrap +(http-capability-gateway#12 / #14 / #22 / #26 / #30, all merged by +2026-06-02). The perf-regression gate is wired and the harness covers +five scenarios. Two owner-driven Phase-D follow-ups remain before §1.1 +below can sign off all four boxes: dispatching the +`+perf-rebaseline.yml+` workflow to populate real +`+bench/baseline.json+` numbers, and flipping `+_status+` from +`+scaffold-placeholder+` to `+active+` to arm the gate. Both are +workflow dispatches plus a maintainer-merge of the generated +`+perf: rebaseline (standards#99)+` PR; no further code changes are +required. +____ + +''''' + +=== 0. Scope + +This runbook covers: + +[arabic] +. *Prerequisite checklist* — what must be green before the first +staging-to-production traffic shift can be initiated (§1). +. *Staging cut-over* — bringing the HCG tier-2 stack up in front of BoJ +staging, validating telemetry and the seam (§2). +. *Production rollout* — staged traffic-shift from BoJ-direct to +HCG-fronted, percentage-by-percentage (§3). +. *Observability* — dashboards and signals on-call must be watching +(§4). +. *Rollback* — detection, immediate-bypass, and permanent-disable +procedures (§5). +. *Post-rollout verification + Trustfile flip* — the final acceptance +steps that close `+standards#100+` (§6). + +What is *out of scope*: + +* HCG internals (policy DSL, ETS table layout, proxy code paths) — +covered in `+http-capability-gateway/docs/+`. +* Cloudflare-edge configuration (tier 1) — covered in +`+Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY]+`. +* BEAM supervisor / cartridge rate-limiting (tier 3/4) — covered +separately in `+Trustfile.a2ml+`. +* Horizontal scaling of BoJ behind the gateway (single-backend +limitation, noted as post-Phase-E in the plan). + +''''' + +=== 1. Prerequisites checklist + +These must *all* be green before any traffic-shift action is taken. A +red item is a stop-the-rollout condition; do not paper over it. + +==== 1.1 Phase D deliverables landed + +* [x] Phase D-1 (Benchee harness + comparator + non-blocking CI gate) — +http-capability-gateway#12 (2026-05-20). +* [x] Phase D-2 (loopback backend fixture so the allow scenario measures +real dial-and-read cost) — http-capability-gateway#14 (2026-05-26). +* [x] Phase D-3 (dedicated trust-header-rewrite + mTLS handshake +scenarios; schema-drift hardening) — http-capability-gateway#22 +(2026-05-27), #30 (2026-06-02). +* [x] Phase D-4 bootstrap (`+workflow_dispatch+` rebaseline automation +on `+ubuntu-latest+`) — http-capability-gateway#26 (2026-05-30). +* [x] Phase D-1 load-profile declaration (this repo’s denominator) — +boj-server#168 (2026-06-01); joint-closed `+standards#99+`. +* [ ] *Pending owner action — D-4 rebaseline + arm.* Dispatch +`+Perf Rebaseline+` workflow on +`+hyperpolymath/http-capability-gateway+` Actions tab, review the +generated `+perf: rebaseline (standards#99)+` PR (real p50/p95/p99/ips +for all five scenarios), and either flip `+bench/baseline.json _status+` +from `+scaffold-placeholder+` to `+active+` in the same PR or in an +immediate follow-up. Until this lands the regression gate runs in +non-blocking scaffold mode — Phase E acceptance criterion 2 ("`load that +matches Phase D benchmark numbers`") cannot be evaluated against +absolute numbers. +* [ ] CI on `+hyperpolymath/http-capability-gateway:main+` is green for +the most recent commit including the `+Perf Regression+` workflow. + +____ +The Phase E acceptance criterion 2 references "`load that matches Phase +D benchmark numbers`". The scaffold-mode gate catches the _shape_ of a +regression (per-scenario tolerance ratios) but not absolute breach of +the load-profile budget published in +`+docs/integration/gateway-load-profile.md+` § 2; the rebaseline + +active flip is what arms the absolute check. +____ + +==== 1.2 Phase A/B/C contract artefacts in place + +* [x] Phase A contract: +link:http-capability-gateway-boj-contract.md[`+http-capability-gateway-boj-contract.md+`] +(v1.0, 2026-05-18). +* [x] Phase A example policy: `+config/gateway-policy-boj-example.yaml+` +(referenced by plan §E2). +* [x] Phase B mTLS-as-primary trust path: HCG +`+lib/http_capability_gateway/proxy.ex+` `+build_backend_headers/1+` + +Cowboy TLS config (http-capability-gateway#10). +* [x] Phase C trust-header strip + seam tests: HCG strip +(http-capability-gateway#11); BoJ-side §3 invariant 3 enforcement in +`+elixir/lib/boj_rest/trust_policy.ex+` line 73 +(`+def satisfies?(_required, _trust, false), do: false+`), merged in +boj-server#106 (commit `+40e46f6f+`). +* [x] Phase C `+[SEAMS]+` declaration: +`+.machine_readable/contractiles/trust/Trustfile.a2ml [SEAMS]+` +(boj-server#90). + +==== 1.3 Operational prerequisites — `+!OWNER:+` block + +These cannot be inferred from the code/contract; the owner must fill +them before §3 begins. + +* [ ] `+!OWNER:+` On-call rotation defined for the gateway during +rollout (primary + secondary). Contact: __________. +* [ ] `+!OWNER:+` Cloudflare zone(s) targeted for the rollout listed +with current routing. _(Plan §E2 anticipates Cloudflare Tunnel rule or +container orchestration as the traffic-shift mechanism — choose one.)_ +* [ ] `+!OWNER:+` Traffic-shift mechanism chosen — Cloudflare Tunnel +rule *OR* container orchestration *OR* Cloudflare percentage split — and +pre-staged. +* [ ] `+!OWNER:+` Production mTLS certificate material provisioned (CA +cert path, client CA, server cert+key). Path on prod host: __________. +* [ ] `+!OWNER:+` Observability dashboard URLs filled into §4 below. +* [ ] `+!OWNER:+` Stakeholder notification window agreed (rollout cannot +start during a freeze; check `+Mustfile+`/governance for any active +freeze). +* [ ] `+!OWNER:+` Cert-rotation runbook for the gateway TLS CA exists or +is filed as follow-up (plan §E1 calls this out separately). +* [ ] `+!OWNER:+` VeriSimDB integration status confirmed (real vs stub — +plan §E "`VeriSimDB audit trail`"). Affects audit-trail acceptance. + +==== 1.4 BoJ-side prerequisites + +* [x] BoJ codebase defaults its container/Elixir back-side bind to +loopback so the externally-facing port is not opened by the in-repo +defaults: Elixir Cowboy bind tightening (boj-server#130), k8s Service +ClusterIP (boj-server#131), Zig-adapter `+APP_HOST=127.0.0.1+` across +`+stapeln.toml+`, `+entrypoint.sh+`, `+compose.prod.yaml+` +(boj-server#132, merged 2026-05-20). Deployment-time confirmation that +the staging port really is closed at the network layer (firewall / +NetworkPolicy / container network) remains an operator pre-check before +§2.1. +* [x] BoJ `+BojRest.TrustPolicy.satisfies?/3+` non-loopback-deny clause +present — verified at `+elixir/lib/boj_rest/trust_policy.ex:73+` +(`+def satisfies?(_required, _trust, false), do: false+`). Phase C +invariant 3 enforcement; landed in boj-server#106. +* [x] Phase E NetworkPolicy hardening (back-side reachability +restricted) — boj-server#173. +* [x] HCG-policy SSE-route coverage (`+POST /cartridge/:name/sse+` +governed alongside `+cartridge-invoke-post+`) — boj-server#165. +* [ ] *BoJ-side observability emitted.* Four telemetry events declared +in +link:boj-side-observability-spec.md[`+boj-side-observability-spec.md+`] +§1 are emitted by `+BojRest.Router+`; the five Prometheus metrics in +that spec §2 appear in a `+/metrics+` scrape; the `+metrics-get+` policy +rule in that spec §5 governs the endpoint; and the +`+scripts/hcg-policy-smoke.sh+` stealth canary covers it. Until this +lands, §3.1 success criterion 4 ("`No `+X-Trust-Level+` mismatches in +BoJ access logs`") and rollback trigger §5.1 row 4 are unobservable via +Prometheus — the only signal path is BoJ structured logs, which the §4 +dashboards do not currently consume. Spec landed; the wiring PR follows +the spec’s §7 checklist. +* [ ] +`+Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY].rate_limiting.tier_2_gateway.status+` +currently `+"PENDING — http-capability-gateway wiring forthcoming"+` +(line 900). _The flip to a real status is the *last* action; see §6._ + +==== 1.5 Gateway-side prerequisites + +* [ ] Gateway Containerfile built and signed as a `+.ctp+` bundle via +cerro-torre (plan §E1). `+pedigree.security.signature+` + +`+pedigree.validation.checksum+` in `+container/gateway-deploy.k9.ncl+` +are `+PLACEHOLDER+` until this step runs; cerro-torre signing is a +separate operator action gated on key-handling discipline. +* [x] `+container/gateway-deploy.k9.ncl+` exists in the gateway repo +(plan §E1) — http-capability-gateway#38 (2026-06-03). Five-level k9-svc +pedigree (Snout / Scent / Leash / Gut / Muscle) modelled on +`+boj-server:container/deploy.k9.ncl+`; per-environment `+BACKEND_URL+` +(`+http://127.0.0.1:7700+` staging, +`+http://unix:/run/boj/gnosis.sock:/+` production); trust source +`+"header"+` staging → `+"mtls"+` production after §2.4 rehearsal; +`+max_unavailable = 0+`; `+failure_mode = "fail-closed"+` matching the +`+[SEAMS] gateway-boj-gnosis+` declaration. +* [x] Gateway policy file in place: +`+config/gateway-policy-boj-example.yaml+`, covering all BoJ surface +routes (`+/.well-known/boj-node-pubkey+`, `+/health+`, `+/menu+`, +`+/cartridges+`, `+/cartridge/:name+`, `+/cartridge/:name/invoke+`, +`+/cartridge/:name/sse+`, plus any added since contract v1.0). +Re-verified 2026-05-28 against `+BojRest.Router+`; the +`+POST /cartridge/:name/sse+` route (router.ex line 130, wired since the +SSE landing — ADR-0013 §6, STATE entry 2026-05-18) was the only drift +since contract v1.0 and is now governed by the `+cartridge-sse-post+` +rule alongside `+cartridge-invoke-post+` (boj-server#165). +* [x] Live policy file (`+config/gateway-policy-boj.yaml+`) promoted +from the example. Content-identical to the example at promotion time; +future BoJ-surface evolution lands in the live file and the example +remains as the worked-example artefact (Phase A A3). Both §2.1 staging +and §3.1 production load the live file via `+POLICY_PATH+`. +* [ ] Gateway has been smoke-tested in isolation with the policy, +returning expected allow/deny on each route. Run +`+scripts/hcg-policy-smoke.sh --gateway-url +` +against the gateway loaded with `+config/gateway-policy-boj.yaml+`; the +script exercises a no-trust-header deny probe for every non-public route +(25 in the live policy), six default-deny verb canaries — +DELETE/PUT/PATCH on listed exact paths, OPTIONS on a listed path (no +CORS-preflight bypass), DELETE on a regex-matched route (no per-verb +regex regression), and GET on the POST-only `+ssg-mcp-webhook+` public +route (the `+{path, verb}+` pairing must be enforced even when the path +itself is in the policy) — a path canary (GET on a synthetic +`+/__phase-e-canary-unknown-path__+` URL that matches no exact rule, no +regex rule, and no public exception) which isolates the no-match → +default-deny branch of the gateway’s three-tier lookup, and four +stealth-profile canaries (two internal+stealth routes pinned to exactly +404, two authenticated routes pinned to exactly 403) which catch a +`+:stealth_profiles+` misconfiguration that would otherwise let an +internal route silently regress to 403 — leaking capability existence to +untrusted callers, the exact threat called out in the `+sdp-status-get+` +narrative — while still satisfying the generic any-4xx deny pattern; the +verb canaries cover the unknown-method path, the path canary covers the +unknown-path path, and the stealth canaries cover the deny-status-code +shape, all three failing closed on independent code branches. The whole +script is fully gateway-internal — BoJ does *not* need to be reachable +for this run. Once BoJ is up behind the gateway, re-run with +`+--with-backend+` from a trusted-proxy IP (loopback by default) to also +cover the allow path on authenticated/internal routes including the +`+POST /cartridge/:name/sse+` authenticated/untrusted pair carried over +from boj-server#165’s test plan. Attach the script’s PASS/FAIL summary +to the cut-over ticket; a single FAIL is a stop-the-rollout condition +(gateway loaded the policy but is not enforcing as declared, or BoJ is +unreachable from the gateway, or the script is being run from a +non-trusted-proxy IP and the trust header is being stripped). + +''''' + +=== 2. Staging cut-over + +Sequencing follows plan §E2/§E3. + +==== 2.1 Deploy gateway in front of BoJ staging + +[arabic] +. Confirm BoJ staging is on the loopback bind (`+:7700+`) per Phase A +contract §1. +. Start the gateway with: +* `+POLICY_PATH=config/gateway-policy-boj.yaml+` (live file promoted in +§1.5; the example `+gateway-policy-boj-example.yaml+` is retained for +documentation only) +* `+BACKEND_URL=http://127.0.0.1:7700+` +* `+PORT=8443+` (TLS) or `+PORT=8080+` (HTTP behind Cloudflare Tunnel) +* `+MTLS_CA_CERT_PATH=+` _(staging value — !OWNER: fill)_ +. Trust source: start with `+"header"+` (per plan §E2). Switch to +`+"mtls"+` after the staging cert-rotation runbook walkthrough (§2.4) +completes. +. Verify the gateway answers on its bound port and proxies to BoJ: ++ +[source,bash] +---- +curl -sk https://:8443/health +# Expect: BoJ /health response, with gateway latency overhead within Phase D baseline + tolerance. +---- + +==== 2.2 Telemetry verification (plan §E3) + +* [ ] `+[:http_capability_gateway, :access_decision]+` events appear in +the gateway’s Prometheus scrape at `+/metrics+`. +* [ ] `+GET /api/v1/minikaran+` returns `+status.status: "active"+` +after the learning phase. _(Per plan §E3; verify the endpoint path +against current gateway code at rollout time.)_ +* [ ] Structured JSON logs from the gateway carry `+request_id+`; BoJ +structured logs carry the same `+request_id+` for the corresponding +upstream request (the contract §2 cross-correlation guarantee). +* [ ] `+X-Trust-Level+` header arrives at BoJ correctly (read from BoJ +access logs); seam test still green. +* [ ] BoJ-side §3 invariant 3 still in force — a deliberate non-loopback +forged-header request to BoJ’s `+:7700+` (only possible from within the +host’s loopback namespace during this test) returns `+:public+` +decision, not the forged class. _(This is paranoia-test; the back-side +bind isolation in §1.4 should already make it impossible to reach +`+:7700+` from outside the pod.)_ + +==== 2.3 Soak test — 24 hours minimum + +Per plan §Phase E acceptance: "`Gateway handles a declared traffic +profile in staging for at least 24 hours without circuit breaker trips +or elevated error rates.`" + +* [ ] Drive synthetic traffic mirroring expected production mix (mix +profile: !OWNER:). +* [ ] Watch the dashboards in §4 throughout. Any circuit-breaker trip or +elevated 5xx aborts the rollout — file a follow-up issue, do not paper +over. +* [ ] Sample p99 latency at 30-minute intervals; compare against Phase D +baseline + tolerance. Within budget: pass. Outside: stop and escalate to +Phase D (the budget is wrong) or Phase B/C (the gateway has a perf +defect). + +==== 2.4 Rollback rehearsal in staging + +Per plan §Phase E acceptance: "`Rollback runbook exists and *has been +tested* (manually walk through E5 once in staging before production +rollout).`" + +* [ ] Walk §5.2 (immediate bypass) procedure end-to-end in staging. +* [ ] Confirm traffic returns to BoJ-direct cleanly (no dropped +connections beyond expected drain time). +* [ ] Walk §5.3 (permanent disable) procedure end-to-end in staging. +* [ ] Document elapsed time for each step; if a step is slower than its +rollback-trigger threshold in §5.1, redesign before production. + +''''' + +=== 3. Production rollout — percentage split + +Per plan §E4. Each step requires the prior step’s success-criteria green +for the documented soak window. Do not compress. + +==== 3.1 Phase 3a — 10% traffic + +* [ ] Pre-step: §1 + §2 fully green. Last 24-hour staging soak ended ≤24 +hours ago. +* [ ] Shift 10% of production traffic through the gateway using the +!OWNER:-chosen mechanism. +* [ ] Soak window: 24 hours minimum. +* [ ] Success criteria: +** p99 latency at production endpoints within Phase D baseline × 1.5 +(the perf-regression p99 tolerance). +** 5xx rate not elevated vs the BoJ-direct baseline (same 24-hour window +the previous day). +** No circuit-breaker trips on the gateway. +** No `+X-Trust-Level+` mismatches in BoJ access logs (gateway should be +the only source). +* [ ] Sign-off: !OWNER: + on-call confirm before moving to 3b. + +==== 3.2 Phase 3b — 50% traffic + +* [ ] Shift to 50% via the same mechanism. +* [ ] Soak window: 24 hours minimum. +* [ ] Same success criteria as §3.1 with the higher sample size; +investigate any drift, including drift visible only at 50% scale +(saturation, head-of-line blocking). +* [ ] Sign-off before 3c. + +==== 3.3 Phase 3c — 100% traffic + +* [ ] Shift to 100%. The BoJ direct path remains warm (not yet +decommissioned). +* [ ] Soak window: 7 days. During this window, rollback (§5) is still +cheap because the BoJ direct path is still wired. +* [ ] Success criteria as §3.1 plus: no escalations during business +hours; no on-call pages tied to gateway behaviour. + +==== 3.4 Phase 3d — Decommission BoJ direct external access + +* [ ] After §3.3 success window expires cleanly, close any remaining +external route to BoJ’s `+:7700+` / `+gnosis.sock+` so the gateway is +the only ingress. +* [ ] Confirm by attempting to reach BoJ directly from a non-pod host — +must fail at the network/socket layer, not just at trust enforcement. +* [ ] This step makes rollback (§5) more expensive (re-opening the +direct path is now a config change). Beyond this point, "`rollback`" +defaults to traffic-shift via the gateway’s bypass plug, not network +re-routing. + +''''' + +=== 4. Observability — what on-call watches + +____ +*!OWNER:* dashboard URLs and on-call rota go here. The signals below are +the _what_; the _where_ is owner-specific. The PromQL queries that turn +each signal into a dashboard panel or alert rule live in +link:gateway-observability-spec.md[`+gateway-observability-spec.md+`] +(§§ 2–5), anchored to the gateway-emitted Prometheus metrics declared in +`+http-capability-gateway/lib/http_capability_gateway/application.ex+` +`+telemetry_metrics/0+`. +____ + +==== 4.1 Signals (gateway-side) + +* p50/p95/p99 latency per scenario (health / policy-deny fast-path / +proxy allow). +* Throughput (ips, per scenario). +* Circuit-breaker state (closed / half-open / open). +* 5xx rate emitted by the gateway (gateway-origin 5xx vs BoJ-passthrough +5xx — keep these distinguishable). +* Policy reload counter (a spike during stable operation is a +misconfiguration alarm). +* Trust-level decision distribution (`+untrusted+` / `+authenticated+` / +`+internal+`). Sudden shift indicates auth pipeline change. + +==== 4.2 Signals (BoJ-side) + +* Per-route trust-class distribution from `+BojRest.Router+` decisions. +* `+X-Trust-Level+` arriving from non-loopback peers — should be zero. +Any non-zero is a deployment defect (back-side bind exposed). +* BoJ 5xx rate (independent of gateway’s view). + +==== 4.3 Dashboards + +* !OWNER: Gateway dashboard URL: __________ +* !OWNER: BoJ dashboard URL: __________ +* !OWNER: Cloudflare zone analytics (if used for the split): __________ + +==== 4.4 On-call + +* !OWNER: Primary on-call during rollout (handle + escalation channel): +__________ +* !OWNER: Secondary: __________ +* !OWNER: Pager runbook for "`gateway in degraded state`" alert: +__________ + +''''' + +=== 5. Rollback + +==== 5.1 Rollback triggers + +Any one of these triggers an *immediate* §5.2 bypass. No discussion in +the moment; debrief afterwards. + +* p99 latency at the rollout edge ≥ 2× Phase D baseline p99 for ≥ 5 +minutes. +* Gateway-origin 5xx rate ≥ 1% of requests for ≥ 5 minutes. +* Circuit breaker trips ≥ 3 times in any 15-minute window. +* BoJ access logs show `+X-Trust-Level+` from non-loopback peers (a §3 +invariant 4 violation in flight — the back-side bind is exposed). +* VeriSimDB / audit-trail write failures ≥ 1% (audit posture broken; HCG +fail-closed should already be denying traffic, but verify and bypass to +stop bleeding). +* On-call judgement: any user-visible regression that the on-call cannot +rule out as gateway-caused within 10 minutes. + +==== 5.2 Immediate bypass (rollout-time, before §3.4 decommission) + +While BoJ direct path is still warm: + +[arabic] +. Trigger the !OWNER:-chosen traffic-shift mechanism to route 100% back +to BoJ-direct. +* *Cloudflare Tunnel*: re-point the tunnel rule from gateway to BoJ +direct. +* *Container orchestration*: scale gateway deployment to 0 OR re-route +the service VIP. +* *Cloudflare percentage split*: set gateway weight to 0. +. Confirm shift took effect via the dashboards in §4. p99 should recover +toward the BoJ-direct baseline within seconds. +. Leave the gateway processes running (do not kill) — they may still +serve any in-flight requests; killing them mid-bypass costs error +responses. +. File an incident issue with: trigger criterion, time, +dashboards-attached, rollback duration, and a request to re-investigate +the Phase D number or the gateway behaviour that caused the trip. +. Resume from §3.1 only after the root cause is fixed and re-staged +through §2. + +==== 5.3 Permanent disable + +If the rollback in §5.2 escalates to "`do not re-attempt with this +gateway version`": + +[arabic] +. Remove the gateway k9-svc deployment per its spec at +`+container/gateway-deploy.k9.ncl+`. +. Update +`+Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY].rate_limiting.tier_2_gateway.status+` +to `+"DISABLED — see incident "+` (replace the placeholder +`+"PENDING — http-capability-gateway wiring forthcoming"+` only if it +was already flipped to `+"DEPLOYED"+` per §6.4; otherwise leave +`+PENDING+` and just record the incident). +. Confirm BoJ direct path is again the primary externally-addressable +surface (only valid if §3.4 decommission has not yet happened; otherwise +restoration requires reopening the BoJ direct path). +. Open a Phase E re-entry issue under `+standards#100+` documenting why +this gateway version was rejected, what the next gateway version must +change before Phase E re-attempts, and any contractile updates required. + +==== 5.4 Post-§3.4 rollback (decommission already executed) + +Once the BoJ direct path is decommissioned (§3.4), rollback is more +expensive: + +[arabic] +. The §5.2 traffic-shift target no longer exists. Either: +* {blank} +[loweralpha] +.. *Restore the direct path*: this is a config change (open the firewall +/ re-add the route) and takes minutes to hours depending on the !OWNER: +traffic-shift mechanism. Estimated reversal time: __________ (!OWNER:). +* {blank} +[loweralpha, start=2] +.. *Stay on the gateway with a known-good policy*: roll the gateway +_back_ to the last-known-good HCG version (the previous `+.ctp+`) +without removing it. This is faster than (a) but only resolves +gateway-version regressions, not gateway-architecture regressions. +. Choose (a) only if the issue is gateway-architectural (e.g., HCG is +the wrong tier-2). Choose (b) for code-regression issues. + +''''' + +=== 6. Post-rollout verification + Trustfile flip + +The acceptance criteria of `+standards#100+` close out here. + +==== 6.1 Telemetry-green window + +* [ ] §3.3 soak window (7 days at 100%) completed with all signals in §4 +nominal. + +==== 6.2 SLA confirmation + +* [ ] Production p99 latency ≤ Phase D baseline p99 × 1.5 (the +perf-regression p99 tolerance). If higher: re-baseline (Phase D +rebaseline ritual per `+http-capability-gateway/docs/perf-contract.md+`) +before declaring done. + +==== 6.3 Rollback evidence + +* [ ] The §2.4 staging rehearsal and any §5.2 production bypass (planned +drill or otherwise) are recorded with timestamps and outcome. The +runbook has been exercised, not only written. + +==== 6.4 Trustfile flip — final action + +Edit `+.machine_readable/contractiles/trust/Trustfile.a2ml+` line ~900: + +[source,yaml] +---- +tier_2_gateway: + provider: "http-capability-gateway / svalinn" + mechanisms: ["per-IP token-bucket sliding window", "per-capability-token", "per-endpoint"] + status: "DEPLOYED" + deployed_at: "" + deployment_evidence: + - "standards#100 close-out comment " + - "incidents (planned + unplanned) " +---- + +Also update `+[HTTP_CAPABILITY_GATEWAY]+` section per plan §E +acceptance: `+status: "DEPLOYED"+` with `+deployed_at+` timestamp. + +==== 6.5 Channel close-out + +* [ ] Post a closure comment on `+standards#100+` summarising what +landed, linking the runbook, the §6.4 commit, and any incidents. +* [ ] *Do not* self-close `+standards#100+`; joint-close is owner-only +per the single-lane channel discipline. +* [ ] *Do not* self-close `+idaptik#77+`; surface that it is now +closable (K9 Dogfood gate flipped green) and let the owner act. + +''''' + +=== Appendix A — Glossary + +* *HCG* — `+hyperpolymath/http-capability-gateway+`, the +Elixir/Cowboy/Plug HTTP-governance layer that sits between Cloudflare +edge (tier 1) and BoJ’s gnosis handler (tier 3). +* *BoJ* — `+hyperpolymath/boj-server+`, the consumer of HCG. +* *Tier 2* — the placement of HCG in the four-tier rate-limit +architecture declared in +`+Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY].rate_limiting+`. +* *§3 invariant 3* — the Phase A contract requirement that BoJ ignores +`+X-Trust-Level+` from any non-loopback caller. Enforced both +gateway-side (header strip) and BoJ-side (`+TrustPolicy.satisfies?/3+` +deny clause). +* *`+.ctp+`* — cerro-torre signed container bundle format. + +=== Appendix B — Cross-references + +* `+docs/decisions/0004-adopt-http-capability-gateway.md+` — ADR. +* `+docs/integration/http-capability-gateway-plan.md+` — full phased +plan (§Phase E sourced here). +* `+docs/integration/http-capability-gateway-boj-contract.md+` — HTTP +boundary contract. +* `+docs/integration/http-capability-gateway-policy-authoring.md+` — +policy file authoring workflow. +* `+docs/integration/gateway-observability-spec.md+` — Phase E PromQL +templates + alert-threshold bindings for the §4 signals + §5 rollback +triggers. +* `+docs/integration/boj-side-observability-spec.md+` — Phase E §1.4 +prerequisite spec: BoJ-side telemetry events, Prometheus metric names, +and `+BojRest.Router+` instrumentation sites that back the §4.2 BoJ-side +signals. +* `+http-capability-gateway/docs/perf-contract.md+` — Phase D +perf-contract. +* `+elixir/lib/boj_rest/trust_policy.ex+` — `+satisfies?/3+` Phase C +enforcement. +* `+.machine_readable/contractiles/trust/Trustfile.a2ml+` — +`+[CLOUDFLARE_EDGE_SECURITY].rate_limiting.tier_2_gateway+` (current +`+PENDING+` site; §6.4 flip target) + `+[SEAMS]+` (Phase C +gateway↔BoJ-gnosis declaration). +* `+scripts/hcg-policy-smoke.sh+` — §1.5 operator pre-check: deny-path +smoke (gateway-alone) + optional `+--with-backend+` allow-path smoke +against the live policy. diff --git a/docs/integration/hcg-tier2-rollout-runbook.md b/docs/integration/hcg-tier2-rollout-runbook.md deleted file mode 100644 index 2d69d305..00000000 --- a/docs/integration/hcg-tier2-rollout-runbook.md +++ /dev/null @@ -1,321 +0,0 @@ - - - -# HCG tier-2 — rollout & rollback runbook - -**Version:** 0.8 (BoJ-side observability spec prereq, Phase E in-progress) -**Date:** 2026-06-22 (rev. from 2026-06-15) -**Status:** Phase E deliverables E1 (deploy spec) + E5 (rollback runbook) drafted; live gateway policy (`config/gateway-policy-boj.yaml`) promoted from the worked example (§1.5); `scripts/hcg-policy-smoke.sh` lands as the checked-in §1.5 operator pre-check (deny-path covers gateway-alone; `--with-backend` adds allow-path coverage); §1.5 verb-canary block covers OPTIONS, regex-route DELETE, and wrong-verb-on-listed-path; a path-canary exercises the no-match default-deny branch (synthetic unknown path with a `global_verbs` verb); a stealth-profile canary pins the deny *status code* — internal+stealth routes must return 404 (capability existence hidden) and authenticated routes must return 403, so a missing-`:stealth_profiles` misconfiguration is caught instead of slipping past the generic any-4xx deny pattern; and `docs/integration/boj-side-observability-spec.md` now declares the BoJ-side telemetry events + Prometheus metric names that back the §4.2 signals — wired into §1.4 as a stop-the-rollout prerequisite, with the actual `BojRest.Router` emission left as a follow-up PR per the spec's §7 checklist. Owner-input markers (`!OWNER:`) remain to be filled before any traffic-shift action is taken. -**ADR:** [`docs/decisions/0004-adopt-http-capability-gateway.md`](../decisions/0004-adopt-http-capability-gateway.md) -**Plan:** [`docs/integration/http-capability-gateway-plan.md`](http-capability-gateway-plan.md) (§ Phase E) -**Contract:** [`docs/integration/http-capability-gateway-boj-contract.md`](http-capability-gateway-boj-contract.md) -**Tracking:** [`standards#91`](https://github.com/hyperpolymath/standards/issues/91) (parent), [`standards#100`](https://github.com/hyperpolymath/standards/issues/100) (Phase E) - -> **File-format note.** Matches sibling integration docs (`http-capability-gateway-{plan,audit,boj-contract,policy-authoring}.md`); the integration plan §E5 normatively prescribes `.md` for the rollback runbook (the wider rollout-and-rollback scope folds in here per acceptance criterion 3 of `standards#100`). The estate `.adoc` default is deliberately overridden for the `docs/integration/` set. - -> **Phase-D status (2026-06-08).** Phase D (`standards#99`) is **closed** — joint-closed via boj-server#168 (D-1 load profile, 2026-06-01) on top of the gateway-side D-1..D-3 scaffold + D-4 bootstrap (http-capability-gateway#12 / #14 / #22 / #26 / #30, all merged by 2026-06-02). The perf-regression gate is wired and the harness covers five scenarios. Two owner-driven Phase-D follow-ups remain before §1.1 below can sign off all four boxes: dispatching the `perf-rebaseline.yml` workflow to populate real `bench/baseline.json` numbers, and flipping `_status` from `scaffold-placeholder` to `active` to arm the gate. Both are workflow dispatches plus a maintainer-merge of the generated `perf: rebaseline (standards#99)` PR; no further code changes are required. - ---- - -## 0. Scope - -This runbook covers: - -1. **Prerequisite checklist** — what must be green before the first staging-to-production traffic shift can be initiated (§1). -2. **Staging cut-over** — bringing the HCG tier-2 stack up in front of BoJ staging, validating telemetry and the seam (§2). -3. **Production rollout** — staged traffic-shift from BoJ-direct to HCG-fronted, percentage-by-percentage (§3). -4. **Observability** — dashboards and signals on-call must be watching (§4). -5. **Rollback** — detection, immediate-bypass, and permanent-disable procedures (§5). -6. **Post-rollout verification + Trustfile flip** — the final acceptance steps that close `standards#100` (§6). - -What is **out of scope**: - -- HCG internals (policy DSL, ETS table layout, proxy code paths) — covered in `http-capability-gateway/docs/`. -- Cloudflare-edge configuration (tier 1) — covered in `Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY]`. -- BEAM supervisor / cartridge rate-limiting (tier 3/4) — covered separately in `Trustfile.a2ml`. -- Horizontal scaling of BoJ behind the gateway (single-backend limitation, noted as post-Phase-E in the plan). - ---- - -## 1. Prerequisites checklist - -These must **all** be green before any traffic-shift action is taken. A red item is a stop-the-rollout condition; do not paper over it. - -### 1.1 Phase D deliverables landed - -- [x] Phase D-1 (Benchee harness + comparator + non-blocking CI gate) — http-capability-gateway#12 (2026-05-20). -- [x] Phase D-2 (loopback backend fixture so the allow scenario measures real dial-and-read cost) — http-capability-gateway#14 (2026-05-26). -- [x] Phase D-3 (dedicated trust-header-rewrite + mTLS handshake scenarios; schema-drift hardening) — http-capability-gateway#22 (2026-05-27), #30 (2026-06-02). -- [x] Phase D-4 bootstrap (`workflow_dispatch` rebaseline automation on `ubuntu-latest`) — http-capability-gateway#26 (2026-05-30). -- [x] Phase D-1 load-profile declaration (this repo's denominator) — boj-server#168 (2026-06-01); joint-closed `standards#99`. -- [ ] **Pending owner action — D-4 rebaseline + arm.** Dispatch `Perf Rebaseline` workflow on `hyperpolymath/http-capability-gateway` Actions tab, review the generated `perf: rebaseline (standards#99)` PR (real p50/p95/p99/ips for all five scenarios), and either flip `bench/baseline.json _status` from `scaffold-placeholder` to `active` in the same PR or in an immediate follow-up. Until this lands the regression gate runs in non-blocking scaffold mode — Phase E acceptance criterion 2 ("load that matches Phase D benchmark numbers") cannot be evaluated against absolute numbers. -- [ ] CI on `hyperpolymath/http-capability-gateway:main` is green for the most recent commit including the `Perf Regression` workflow. - -> The Phase E acceptance criterion 2 references "load that matches Phase D benchmark numbers". The scaffold-mode gate catches the *shape* of a regression (per-scenario tolerance ratios) but not absolute breach of the load-profile budget published in `docs/integration/gateway-load-profile.md` § 2; the rebaseline + active flip is what arms the absolute check. - -### 1.2 Phase A/B/C contract artefacts in place - -- [x] Phase A contract: [`http-capability-gateway-boj-contract.md`](http-capability-gateway-boj-contract.md) (v1.0, 2026-05-18). -- [x] Phase A example policy: `config/gateway-policy-boj-example.yaml` (referenced by plan §E2). -- [x] Phase B mTLS-as-primary trust path: HCG `lib/http_capability_gateway/proxy.ex` `build_backend_headers/1` + Cowboy TLS config (http-capability-gateway#10). -- [x] Phase C trust-header strip + seam tests: HCG strip (http-capability-gateway#11); BoJ-side §3 invariant 3 enforcement in `elixir/lib/boj_rest/trust_policy.ex` line 73 (`def satisfies?(_required, _trust, false), do: false`), merged in boj-server#106 (commit `40e46f6f`). -- [x] Phase C `[SEAMS]` declaration: `.machine_readable/contractiles/trust/Trustfile.a2ml [SEAMS]` (boj-server#90). - -### 1.3 Operational prerequisites — `!OWNER:` block - -These cannot be inferred from the code/contract; the owner must fill them before §3 begins. - -- [ ] `!OWNER:` On-call rotation defined for the gateway during rollout (primary + secondary). Contact: __________. -- [ ] `!OWNER:` Cloudflare zone(s) targeted for the rollout listed with current routing. _(Plan §E2 anticipates Cloudflare Tunnel rule or container orchestration as the traffic-shift mechanism — choose one.)_ -- [ ] `!OWNER:` Traffic-shift mechanism chosen — Cloudflare Tunnel rule **OR** container orchestration **OR** Cloudflare percentage split — and pre-staged. -- [ ] `!OWNER:` Production mTLS certificate material provisioned (CA cert path, client CA, server cert+key). Path on prod host: __________. -- [ ] `!OWNER:` Observability dashboard URLs filled into §4 below. -- [ ] `!OWNER:` Stakeholder notification window agreed (rollout cannot start during a freeze; check `Mustfile`/governance for any active freeze). -- [ ] `!OWNER:` Cert-rotation runbook for the gateway TLS CA exists or is filed as follow-up (plan §E1 calls this out separately). -- [ ] `!OWNER:` VeriSimDB integration status confirmed (real vs stub — plan §E "VeriSimDB audit trail"). Affects audit-trail acceptance. - -### 1.4 BoJ-side prerequisites - -- [x] BoJ codebase defaults its container/Elixir back-side bind to loopback so the externally-facing port is not opened by the in-repo defaults: Elixir Cowboy bind tightening (boj-server#130), k8s Service ClusterIP (boj-server#131), Zig-adapter `APP_HOST=127.0.0.1` across `stapeln.toml`, `entrypoint.sh`, `compose.prod.yaml` (boj-server#132, merged 2026-05-20). Deployment-time confirmation that the staging port really is closed at the network layer (firewall / NetworkPolicy / container network) remains an operator pre-check before §2.1. -- [x] BoJ `BojRest.TrustPolicy.satisfies?/3` non-loopback-deny clause present — verified at `elixir/lib/boj_rest/trust_policy.ex:73` (`def satisfies?(_required, _trust, false), do: false`). Phase C invariant 3 enforcement; landed in boj-server#106. -- [x] Phase E NetworkPolicy hardening (back-side reachability restricted) — boj-server#173. -- [x] HCG-policy SSE-route coverage (`POST /cartridge/:name/sse` governed alongside `cartridge-invoke-post`) — boj-server#165. -- [ ] **BoJ-side observability emitted.** Four telemetry events declared in [`boj-side-observability-spec.md`](boj-side-observability-spec.md) §1 are emitted by `BojRest.Router`; the five Prometheus metrics in that spec §2 appear in a `/metrics` scrape; the `metrics-get` policy rule in that spec §5 governs the endpoint; and the `scripts/hcg-policy-smoke.sh` stealth canary covers it. Until this lands, §3.1 success criterion 4 ("No `X-Trust-Level` mismatches in BoJ access logs") and rollback trigger §5.1 row 4 are unobservable via Prometheus — the only signal path is BoJ structured logs, which the §4 dashboards do not currently consume. Spec landed; the wiring PR follows the spec's §7 checklist. -- [ ] `Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY].rate_limiting.tier_2_gateway.status` currently `"PENDING — http-capability-gateway wiring forthcoming"` (line 900). _The flip to a real status is the **last** action; see §6._ - -### 1.5 Gateway-side prerequisites - -- [ ] Gateway Containerfile built and signed as a `.ctp` bundle via cerro-torre (plan §E1). `pedigree.security.signature` + `pedigree.validation.checksum` in `container/gateway-deploy.k9.ncl` are `PLACEHOLDER` until this step runs; cerro-torre signing is a separate operator action gated on key-handling discipline. -- [x] `container/gateway-deploy.k9.ncl` exists in the gateway repo (plan §E1) — http-capability-gateway#38 (2026-06-03). Five-level k9-svc pedigree (Snout / Scent / Leash / Gut / Muscle) modelled on `boj-server:container/deploy.k9.ncl`; per-environment `BACKEND_URL` (`http://127.0.0.1:7700` staging, `http://unix:/run/boj/gnosis.sock:/` production); trust source `"header"` staging → `"mtls"` production after §2.4 rehearsal; `max_unavailable = 0`; `failure_mode = "fail-closed"` matching the `[SEAMS] gateway-boj-gnosis` declaration. -- [x] Gateway policy file in place: `config/gateway-policy-boj-example.yaml`, covering all BoJ surface routes (`/.well-known/boj-node-pubkey`, `/health`, `/menu`, `/cartridges`, `/cartridge/:name`, `/cartridge/:name/invoke`, `/cartridge/:name/sse`, plus any added since contract v1.0). Re-verified 2026-05-28 against `BojRest.Router`; the `POST /cartridge/:name/sse` route (router.ex line 130, wired since the SSE landing — ADR-0013 §6, STATE entry 2026-05-18) was the only drift since contract v1.0 and is now governed by the `cartridge-sse-post` rule alongside `cartridge-invoke-post` (boj-server#165). -- [x] Live policy file (`config/gateway-policy-boj.yaml`) promoted from the example. Content-identical to the example at promotion time; future BoJ-surface evolution lands in the live file and the example remains as the worked-example artefact (Phase A A3). Both §2.1 staging and §3.1 production load the live file via `POLICY_PATH`. -- [ ] Gateway has been smoke-tested in isolation with the policy, returning expected allow/deny on each route. Run `scripts/hcg-policy-smoke.sh --gateway-url ` against the gateway loaded with `config/gateway-policy-boj.yaml`; the script exercises a no-trust-header deny probe for every non-public route (25 in the live policy), six default-deny verb canaries — DELETE/PUT/PATCH on listed exact paths, OPTIONS on a listed path (no CORS-preflight bypass), DELETE on a regex-matched route (no per-verb regex regression), and GET on the POST-only `ssg-mcp-webhook` public route (the `{path, verb}` pairing must be enforced even when the path itself is in the policy) — a path canary (GET on a synthetic `/__phase-e-canary-unknown-path__` URL that matches no exact rule, no regex rule, and no public exception) which isolates the no-match → default-deny branch of the gateway's three-tier lookup, and four stealth-profile canaries (two internal+stealth routes pinned to exactly 404, two authenticated routes pinned to exactly 403) which catch a `:stealth_profiles` misconfiguration that would otherwise let an internal route silently regress to 403 — leaking capability existence to untrusted callers, the exact threat called out in the `sdp-status-get` narrative — while still satisfying the generic any-4xx deny pattern; the verb canaries cover the unknown-method path, the path canary covers the unknown-path path, and the stealth canaries cover the deny-status-code shape, all three failing closed on independent code branches. The whole script is fully gateway-internal — BoJ does **not** need to be reachable for this run. Once BoJ is up behind the gateway, re-run with `--with-backend` from a trusted-proxy IP (loopback by default) to also cover the allow path on authenticated/internal routes including the `POST /cartridge/:name/sse` authenticated/untrusted pair carried over from boj-server#165's test plan. Attach the script's PASS/FAIL summary to the cut-over ticket; a single FAIL is a stop-the-rollout condition (gateway loaded the policy but is not enforcing as declared, or BoJ is unreachable from the gateway, or the script is being run from a non-trusted-proxy IP and the trust header is being stripped). - ---- - -## 2. Staging cut-over - -Sequencing follows plan §E2/§E3. - -### 2.1 Deploy gateway in front of BoJ staging - -1. Confirm BoJ staging is on the loopback bind (`:7700`) per Phase A contract §1. -2. Start the gateway with: - - `POLICY_PATH=config/gateway-policy-boj.yaml` (live file promoted in §1.5; the example `gateway-policy-boj-example.yaml` is retained for documentation only) - - `BACKEND_URL=http://127.0.0.1:7700` - - `PORT=8443` (TLS) or `PORT=8080` (HTTP behind Cloudflare Tunnel) - - `MTLS_CA_CERT_PATH=` _(staging value — !OWNER: fill)_ -3. Trust source: start with `"header"` (per plan §E2). Switch to `"mtls"` after the staging cert-rotation runbook walkthrough (§2.4) completes. -4. Verify the gateway answers on its bound port and proxies to BoJ: - ```bash - curl -sk https://:8443/health - # Expect: BoJ /health response, with gateway latency overhead within Phase D baseline + tolerance. - ``` - -### 2.2 Telemetry verification (plan §E3) - -- [ ] `[:http_capability_gateway, :access_decision]` events appear in the gateway's Prometheus scrape at `/metrics`. -- [ ] `GET /api/v1/minikaran` returns `status.status: "active"` after the learning phase. _(Per plan §E3; verify the endpoint path against current gateway code at rollout time.)_ -- [ ] Structured JSON logs from the gateway carry `request_id`; BoJ structured logs carry the same `request_id` for the corresponding upstream request (the contract §2 cross-correlation guarantee). -- [ ] `X-Trust-Level` header arrives at BoJ correctly (read from BoJ access logs); seam test still green. -- [ ] BoJ-side §3 invariant 3 still in force — a deliberate non-loopback forged-header request to BoJ's `:7700` (only possible from within the host's loopback namespace during this test) returns `:public` decision, not the forged class. _(This is paranoia-test; the back-side bind isolation in §1.4 should already make it impossible to reach `:7700` from outside the pod.)_ - -### 2.3 Soak test — 24 hours minimum - -Per plan §Phase E acceptance: "Gateway handles a declared traffic profile in staging for at least 24 hours without circuit breaker trips or elevated error rates." - -- [ ] Drive synthetic traffic mirroring expected production mix (mix profile: !OWNER:). -- [ ] Watch the dashboards in §4 throughout. Any circuit-breaker trip or elevated 5xx aborts the rollout — file a follow-up issue, do not paper over. -- [ ] Sample p99 latency at 30-minute intervals; compare against Phase D baseline + tolerance. Within budget: pass. Outside: stop and escalate to Phase D (the budget is wrong) or Phase B/C (the gateway has a perf defect). - -### 2.4 Rollback rehearsal in staging - -Per plan §Phase E acceptance: "Rollback runbook exists and **has been tested** (manually walk through E5 once in staging before production rollout)." - -- [ ] Walk §5.2 (immediate bypass) procedure end-to-end in staging. -- [ ] Confirm traffic returns to BoJ-direct cleanly (no dropped connections beyond expected drain time). -- [ ] Walk §5.3 (permanent disable) procedure end-to-end in staging. -- [ ] Document elapsed time for each step; if a step is slower than its rollback-trigger threshold in §5.1, redesign before production. - ---- - -## 3. Production rollout — percentage split - -Per plan §E4. Each step requires the prior step's success-criteria green for the documented soak window. Do not compress. - -### 3.1 Phase 3a — 10% traffic - -- [ ] Pre-step: §1 + §2 fully green. Last 24-hour staging soak ended ≤24 hours ago. -- [ ] Shift 10% of production traffic through the gateway using the !OWNER:-chosen mechanism. -- [ ] Soak window: 24 hours minimum. -- [ ] Success criteria: - - p99 latency at production endpoints within Phase D baseline × 1.5 (the perf-regression p99 tolerance). - - 5xx rate not elevated vs the BoJ-direct baseline (same 24-hour window the previous day). - - No circuit-breaker trips on the gateway. - - No `X-Trust-Level` mismatches in BoJ access logs (gateway should be the only source). -- [ ] Sign-off: !OWNER: + on-call confirm before moving to 3b. - -### 3.2 Phase 3b — 50% traffic - -- [ ] Shift to 50% via the same mechanism. -- [ ] Soak window: 24 hours minimum. -- [ ] Same success criteria as §3.1 with the higher sample size; investigate any drift, including drift visible only at 50% scale (saturation, head-of-line blocking). -- [ ] Sign-off before 3c. - -### 3.3 Phase 3c — 100% traffic - -- [ ] Shift to 100%. The BoJ direct path remains warm (not yet decommissioned). -- [ ] Soak window: 7 days. During this window, rollback (§5) is still cheap because the BoJ direct path is still wired. -- [ ] Success criteria as §3.1 plus: no escalations during business hours; no on-call pages tied to gateway behaviour. - -### 3.4 Phase 3d — Decommission BoJ direct external access - -- [ ] After §3.3 success window expires cleanly, close any remaining external route to BoJ's `:7700` / `gnosis.sock` so the gateway is the only ingress. -- [ ] Confirm by attempting to reach BoJ directly from a non-pod host — must fail at the network/socket layer, not just at trust enforcement. -- [ ] This step makes rollback (§5) more expensive (re-opening the direct path is now a config change). Beyond this point, "rollback" defaults to traffic-shift via the gateway's bypass plug, not network re-routing. - ---- - -## 4. Observability — what on-call watches - -> **!OWNER:** dashboard URLs and on-call rota go here. The signals below are the *what*; the *where* is owner-specific. The PromQL queries that turn each signal into a dashboard panel or alert rule live in [`gateway-observability-spec.md`](gateway-observability-spec.md) (§§ 2–5), anchored to the gateway-emitted Prometheus metrics declared in `http-capability-gateway/lib/http_capability_gateway/application.ex` `telemetry_metrics/0`. - -### 4.1 Signals (gateway-side) - -- p50/p95/p99 latency per scenario (health / policy-deny fast-path / proxy allow). -- Throughput (ips, per scenario). -- Circuit-breaker state (closed / half-open / open). -- 5xx rate emitted by the gateway (gateway-origin 5xx vs BoJ-passthrough 5xx — keep these distinguishable). -- Policy reload counter (a spike during stable operation is a misconfiguration alarm). -- Trust-level decision distribution (`untrusted` / `authenticated` / `internal`). Sudden shift indicates auth pipeline change. - -### 4.2 Signals (BoJ-side) - -- Per-route trust-class distribution from `BojRest.Router` decisions. -- `X-Trust-Level` arriving from non-loopback peers — should be zero. Any non-zero is a deployment defect (back-side bind exposed). -- BoJ 5xx rate (independent of gateway's view). - -### 4.3 Dashboards - -- !OWNER: Gateway dashboard URL: __________ -- !OWNER: BoJ dashboard URL: __________ -- !OWNER: Cloudflare zone analytics (if used for the split): __________ - -### 4.4 On-call - -- !OWNER: Primary on-call during rollout (handle + escalation channel): __________ -- !OWNER: Secondary: __________ -- !OWNER: Pager runbook for "gateway in degraded state" alert: __________ - ---- - -## 5. Rollback - -### 5.1 Rollback triggers - -Any one of these triggers an **immediate** §5.2 bypass. No discussion in the moment; debrief afterwards. - -- p99 latency at the rollout edge ≥ 2× Phase D baseline p99 for ≥ 5 minutes. -- Gateway-origin 5xx rate ≥ 1% of requests for ≥ 5 minutes. -- Circuit breaker trips ≥ 3 times in any 15-minute window. -- BoJ access logs show `X-Trust-Level` from non-loopback peers (a §3 invariant 4 violation in flight — the back-side bind is exposed). -- VeriSimDB / audit-trail write failures ≥ 1% (audit posture broken; HCG fail-closed should already be denying traffic, but verify and bypass to stop bleeding). -- On-call judgement: any user-visible regression that the on-call cannot rule out as gateway-caused within 10 minutes. - -### 5.2 Immediate bypass (rollout-time, before §3.4 decommission) - -While BoJ direct path is still warm: - -1. Trigger the !OWNER:-chosen traffic-shift mechanism to route 100% back to BoJ-direct. - - **Cloudflare Tunnel**: re-point the tunnel rule from gateway to BoJ direct. - - **Container orchestration**: scale gateway deployment to 0 OR re-route the service VIP. - - **Cloudflare percentage split**: set gateway weight to 0. -2. Confirm shift took effect via the dashboards in §4. p99 should recover toward the BoJ-direct baseline within seconds. -3. Leave the gateway processes running (do not kill) — they may still serve any in-flight requests; killing them mid-bypass costs error responses. -4. File an incident issue with: trigger criterion, time, dashboards-attached, rollback duration, and a request to re-investigate the Phase D number or the gateway behaviour that caused the trip. -5. Resume from §3.1 only after the root cause is fixed and re-staged through §2. - -### 5.3 Permanent disable - -If the rollback in §5.2 escalates to "do not re-attempt with this gateway version": - -1. Remove the gateway k9-svc deployment per its spec at `container/gateway-deploy.k9.ncl`. -2. Update `Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY].rate_limiting.tier_2_gateway.status` to `"DISABLED — see incident "` (replace the placeholder `"PENDING — http-capability-gateway wiring forthcoming"` only if it was already flipped to `"DEPLOYED"` per §6.4; otherwise leave `PENDING` and just record the incident). -3. Confirm BoJ direct path is again the primary externally-addressable surface (only valid if §3.4 decommission has not yet happened; otherwise restoration requires reopening the BoJ direct path). -4. Open a Phase E re-entry issue under `standards#100` documenting why this gateway version was rejected, what the next gateway version must change before Phase E re-attempts, and any contractile updates required. - -### 5.4 Post-§3.4 rollback (decommission already executed) - -Once the BoJ direct path is decommissioned (§3.4), rollback is more expensive: - -1. The §5.2 traffic-shift target no longer exists. Either: - - (a) **Restore the direct path**: this is a config change (open the firewall / re-add the route) and takes minutes to hours depending on the !OWNER: traffic-shift mechanism. Estimated reversal time: __________ (!OWNER:). - - (b) **Stay on the gateway with a known-good policy**: roll the gateway *back* to the last-known-good HCG version (the previous `.ctp`) without removing it. This is faster than (a) but only resolves gateway-version regressions, not gateway-architecture regressions. -2. Choose (a) only if the issue is gateway-architectural (e.g., HCG is the wrong tier-2). Choose (b) for code-regression issues. - ---- - -## 6. Post-rollout verification + Trustfile flip - -The acceptance criteria of `standards#100` close out here. - -### 6.1 Telemetry-green window - -- [ ] §3.3 soak window (7 days at 100%) completed with all signals in §4 nominal. - -### 6.2 SLA confirmation - -- [ ] Production p99 latency ≤ Phase D baseline p99 × 1.5 (the perf-regression p99 tolerance). If higher: re-baseline (Phase D rebaseline ritual per `http-capability-gateway/docs/perf-contract.md`) before declaring done. - -### 6.3 Rollback evidence - -- [ ] The §2.4 staging rehearsal and any §5.2 production bypass (planned drill or otherwise) are recorded with timestamps and outcome. The runbook has been exercised, not only written. - -### 6.4 Trustfile flip — final action - -Edit `.machine_readable/contractiles/trust/Trustfile.a2ml` line ~900: - -```yaml -tier_2_gateway: - provider: "http-capability-gateway / svalinn" - mechanisms: ["per-IP token-bucket sliding window", "per-capability-token", "per-endpoint"] - status: "DEPLOYED" - deployed_at: "" - deployment_evidence: - - "standards#100 close-out comment " - - "incidents (planned + unplanned) " -``` - -Also update `[HTTP_CAPABILITY_GATEWAY]` section per plan §E acceptance: `status: "DEPLOYED"` with `deployed_at` timestamp. - -### 6.5 Channel close-out - -- [ ] Post a closure comment on `standards#100` summarising what landed, linking the runbook, the §6.4 commit, and any incidents. -- [ ] **Do not** self-close `standards#100`; joint-close is owner-only per the single-lane channel discipline. -- [ ] **Do not** self-close `idaptik#77`; surface that it is now closable (K9 Dogfood gate flipped green) and let the owner act. - ---- - -## Appendix A — Glossary - -- **HCG** — `hyperpolymath/http-capability-gateway`, the Elixir/Cowboy/Plug HTTP-governance layer that sits between Cloudflare edge (tier 1) and BoJ's gnosis handler (tier 3). -- **BoJ** — `hyperpolymath/boj-server`, the consumer of HCG. -- **Tier 2** — the placement of HCG in the four-tier rate-limit architecture declared in `Trustfile.a2ml [CLOUDFLARE_EDGE_SECURITY].rate_limiting`. -- **§3 invariant 3** — the Phase A contract requirement that BoJ ignores `X-Trust-Level` from any non-loopback caller. Enforced both gateway-side (header strip) and BoJ-side (`TrustPolicy.satisfies?/3` deny clause). -- **`.ctp`** — cerro-torre signed container bundle format. - -## Appendix B — Cross-references - -- `docs/decisions/0004-adopt-http-capability-gateway.md` — ADR. -- `docs/integration/http-capability-gateway-plan.md` — full phased plan (§Phase E sourced here). -- `docs/integration/http-capability-gateway-boj-contract.md` — HTTP boundary contract. -- `docs/integration/http-capability-gateway-policy-authoring.md` — policy file authoring workflow. -- `docs/integration/gateway-observability-spec.md` — Phase E PromQL templates + alert-threshold bindings for the §4 signals + §5 rollback triggers. -- `docs/integration/boj-side-observability-spec.md` — Phase E §1.4 prerequisite spec: BoJ-side telemetry events, Prometheus metric names, and `BojRest.Router` instrumentation sites that back the §4.2 BoJ-side signals. -- `http-capability-gateway/docs/perf-contract.md` — Phase D perf-contract. -- `elixir/lib/boj_rest/trust_policy.ex` — `satisfies?/3` Phase C enforcement. -- `.machine_readable/contractiles/trust/Trustfile.a2ml` — `[CLOUDFLARE_EDGE_SECURITY].rate_limiting.tier_2_gateway` (current `PENDING` site; §6.4 flip target) + `[SEAMS]` (Phase C gateway↔BoJ-gnosis declaration). -- `scripts/hcg-policy-smoke.sh` — §1.5 operator pre-check: deny-path smoke (gateway-alone) + optional `--with-backend` allow-path smoke against the live policy. diff --git a/docs/integration/http-capability-gateway-audit.adoc b/docs/integration/http-capability-gateway-audit.adoc new file mode 100644 index 00000000..bf823e34 --- /dev/null +++ b/docs/integration/http-capability-gateway-audit.adoc @@ -0,0 +1,628 @@ +== http-capability-gateway — Integration Audit + +*Audited:* 2026-04-17 + +*Auditor:* Claude Sonnet 4.6 (boj-server integration scope session) + +*Gateway repo:* `+web-ecosystem/http-capability-gateway+` (GitHub: +`+hyperpolymath/http-capability-gateway+`) + +*Gateway version:* `+0.1.0-dev+` (ROADMAP.adoc: CRG grade C achieved +2026-04-04) + +*Purpose of audit:* Establish truth baseline before Phase A contract +definition work. + +''''' + +=== 1. Component Inventory — What Is Actually Shipped + +==== 1.1 PolicyLoader (`+lib/http_capability_gateway/policy_loader.ex+`) + +Reads a YAML file or binary string and returns `+{:ok, map()}+` or +`+{:error, String.t()}+`. Handles multi-document YAML (uses first +document). Trims empty input. Logs service name from +`+policy["service"]["name"]+` on success. + +*Test coverage:* `+test/policy_loader_test.exs+` — 11 `+test+` +declarations, 165 lines. Covers: happy path from string, from file, +empty content, non-map content, YAML parse errors, multi-document YAML. +No property-based tests for the loader specifically. + +*Maturity: 4/5.* Solid; no known gaps. Missing: no dedicated test for +file permissions errors or very large files. + +''''' + +==== 1.2 PolicyValidator (`+lib/http_capability_gateway/policy_validator.ex+`) + +Validates a parsed policy map against the DSL v1 schema. Checks: - +`+dsl_version+` must be present and equal to `+"1"+`. - `+governance+` +must be a map containing `+global_verbs+` (non-empty list of valid HTTP +verbs). - `+governance.routes+` (optional) must be a list of maps each +with `+path+` (non-empty string, valid regex) and `+verbs+` (non-empty +list of valid HTTP verbs). - `+stealth+` (optional) must have both +`+enabled+` (boolean) and `+status_code+` (integer 100–599). + +Returns `+:ok+` or `+{:error, reason_string}+`. + +*Test coverage:* `+test/policy_validator_test.exs+` — 18 `+test+` +declarations, 257 lines. Covers: all required fields, invalid verbs, +invalid stealth combos, route-level validation. Property tests for +validator live in `+test/policy_property_test.exs+` (1 property, +StreamData). + +*Maturity: 4/5.* Well-exercised. Does not validate `+service.name+`, +`+narrative+`, or the v0 DSL format (two DSL versions co-exist in the +README; validator only handles v1). + +''''' + +==== 1.3 PolicyCompiler (`+lib/http_capability_gateway/policy_compiler.ex+`) + +Compiles a validated policy into a dual ETS table pair: + +* *Main table* — exact literal paths (`+{:exact, path, verb}+`) and +global verb rules (`+{:global, verb}+`). O(1) lookup. +* *Regex table* — path patterns containing regex metacharacters, keyed +as `+{pattern_string, verb_atom}+`. Scanned O(r) where r = number of +regex routes. + +Both tables created with +`+:set, :public, :named_table, read_concurrency: true+`. + +*Atomic swap pattern:* On hot-reload, new tables are created with a +monotonic-time suffix, compiled in full, then the +`+Application.put_env/3+` references are swapped atomically. The old +tables are deleted only after the swap. A compilation failure leaves the +existing tables intact (last-known-good policy preserved). + +*Lookup tiers (implemented):* 1. Tier 1: +`+ETS.lookup(table, {:exact, path, verb})+` — O(1). 2. Tier 2: +`+ETS.tab2list(regex_table)+` + Enum.find regex match — O(r). 3. Tier 3: +`+ETS.lookup(table, {:global, verb})+` — O(1). + +*`+CompiledRule+` struct fields:* `+path_pattern+`, `+path_regex+`, +`+verb+`, `+exposure+`, `+stealth_profile+`, `+narrative+`, `+backend+`, +`+name+`. + +*Test coverage:* `+test/policy_compiler_test.exs+` — 7 `+test+` +declarations, 87 lines. Tests: compile success, ETS size, stats/0, error +paths. Thin; more coverage in `+test/e2e_test.exs+` through the full +pipeline. + +*Maturity: 4/5.* Implementation is solid and well-commented. Gap: no +test for the atomic swap path under concurrent load (covered separately +in `+concurrency_test.exs+`). + +''''' + +==== 1.4 Gateway (`+lib/http_capability_gateway/gateway.ex+`) + +`+Plug.Router+` implementing the HTTP enforcement pipeline. Plug stack +order: + +.... +Plug.Logger +→ security_headers (X-Content-Type-Options, X-Frame-Options, Referrer-Policy, Cache-Control, Connection) +→ strip_untrusted_headers (removes X-Trust-Level unless conn.remote_ip is in :trusted_proxies) +→ extract_trust (sets conn.assigns[:trust_level] via SafeTrust.parse_trust/1) +→ RateLimiter (token bucket, per-{IP, trust_level}, ETS-backed) +→ match/dispatch (Plug.Router routes) +.... + +Special routes: `+/health+`, `+/ready+`, `+/metrics+`, +`+/api/v1/minikaran+`, catch-all. + +*Trust extraction:* Configurable via `+:trust_level_source+`. Two modes: +- `+"header"+` (default): reads `+X-Trust-Level+` header → +`+SafeTrust.parse_trust/1+`. - `+"mtls"+`: reads peer cert via +`+cowboy_req:cert/1+`, decodes DER with +`+:public_key.pkix_decode_cert/2+` in `+:otp+` mode, uses OTP +`+Record.extract/2+` accessors for +`+OTPCertificate+`/`+OTPTBSCertificate+` to extract CN/O/OU fields. +Determines trust by: OU=="`Internal Services`" && verified → +"`internal`"; verified → "`authenticated`"; else "`untrusted`". + +*mTLS implementation status:* The code path for mTLS exists in +`+extract_trust_level_from_cert/1+`. It uses `+cowboy_req:cert/1+` to +retrieve the DER-encoded peer cert, then decodes it with proper OTP +record accessors. However: - No live CA or cert fixture is used in tests +(TEST-NEEDS.md explicitly flags "`Real-CA mTLS integration test`" as +missing). - `+is_cert_verified/1+` checks only for cert presence (not +actual TLS verification status from Cowboy). This means the mTLS path +reads the certificate subject but does not enforce chain validation +independently — it relies on Cowboy having been configured with +`+verify: :verify_peer+`. That Cowboy configuration is not present in +any shipped config file. - *Conclusion:* mTLS trust extraction is coded +but is not the primary proved path. Header-based trust extraction is +what the system actually tests and deploys. + +*Access decision:* `+SafeTrust.evaluate(trust_level, exposure)+` — +formally mirrors `+proven/SafeTrust.idr+`. Returns `+{:allow, t, e}+` or +`+{:deny, t, e}+`. No ad-hoc evaluate logic remains in the gateway. + +*K9-SVC contract enforcement:* Present and wired. After policy allow, +the gateway looks up a K9 contract for `+(path, verb)+`. If found: +pre-proxy trust threshold check, proxy with timing, post-proxy latency +SLA check with configurable breach policy (`+:log+`, `+:alert+`, +`+:circuit_break+`, `+:fallback+`). Circuit breaker checked before +contract lookup. + +*VeriSimDB integration:* `+VeriSimDB.audit_allow/6+` and +`+VeriSimDB.audit_deny/4+` called asynchronously after every allow/deny +decision. See §4 in this audit for the VeriSimDB stub assessment. + +*Test coverage:* `+test/gateway_test.exs+` — 23 `+test+` declarations, +204 lines. `+test/e2e_test.exs+` — 22 `+test+` declarations, 532 lines +(full lifecycle). `+test/security_test.exs+` — 36 `+test+` declarations, +403 lines (header injection, SSRF, trust spoofing, atom exhaustion). + +*Maturity: 3/5.* Core logic solid. mTLS not the primary proved path. No +live Cowboy TLS-verify-peer test. VeriSimDB integration is stubbed (see +§4). + +''''' + +==== 1.5 Proxy (`+lib/http_capability_gateway/proxy.ex+`) + +Forwards allowed requests to a single configured `+backend_url+`. Uses +`+Req+` HTTP client. Filters hop-by-hop headers. Adds +`+X-Forwarded-For+`, `+X-Forwarded-Proto+`, `+X-Forwarded-Host+`, +`+X-Gateway: http-capability-gateway+`. 30-second receive timeout. No +retry (caller handles). Health check via `+GET {backend_url}/health+`. + +*Backend URL:* Read from +`+Application.get_env(:http_capability_gateway, :backend_url, "http://localhost:8080")+`. +Single static backend — no load balancing or multi-backend routing. + +*Note for BoJ integration:* The proxy assumes a single HTTP backend. +BoJ’s unified Zig API currently listens on a single port. The +integration forwarding path will be HTTP (TCP localhost or Unix socket). +See Phase A for the contract definition. + +*Test coverage:* Proxy forwarded through E2E and gateway tests. No +dedicated `+proxy_test.exs+`. SSRF resistance tested in +`+security_test.exs+` (internal/localhost URL blocking logic checked +there). + +*Maturity: 3/5.* Works for the happy path. SSRF defence tested. Missing: +dedicated proxy failure tests; no timeout boundary test; streaming large +bodies untested. + +''''' + +==== 1.6 Telemetry / Minikaran (`+lib/http_capability_gateway/minikaran.ex+`, + +`+lib/http_capability_gateway/minikaran/telemetry_handler.ex+`, +`+lib/http_capability_gateway/logging.ex+`, +`+lib/http_capability_gateway/log_formatter.ex+`) + +*Telemetry events emitted:* + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Event name |Measurements |Metadata +|`+[:http_capability_gateway, :access_decision]+` +|`+%{duration: duration_us}+` +|`+%{decision: atom, verb: atom, trust_level: atom}+` + +|`+[:http_capability_gateway, :request, :completed]+` +|`+%{count: 1, duration: total_us}+` |`+%{status: http_status}+` + +|`+[:http_capability_gateway, :rate_limit, :exceeded]+` |`+%{count: 1}+` +|`+%{trust_level: atom, client: string}+` +|=== + +*Minikaran TelemetryHandler* hooks all three events and forwards them as +observation maps to the Minikaran GenServer via `+GenServer.cast+`. +Observation map shape: +`+%{path, trust_level, latency_us, status, client_ip, timestamp}+`. + +*Note:* The `+access_decision+` event does not include the request path +in its metadata (Gateway emits it as `+%{decision, verb, trust_level}+` +only). The TelemetryHandler reconstructs a pseudo-path as +`+"/#{verb}"+`. This means the Minikaran baseline cannot distinguish +`+/api/users+` from `+/api/posts+` when both use GET — path-level +anomaly detection is approximate until this is fixed. + +*Structured log format* (LogFormatter): JSON, fields include +`+timestamp+`, `+level+`, `+message+`, `+request_id+`, `+method+`, +`+path+`, `+trust_level+`, `+verb_allowed+`, `+stealth_triggered+`, +`+response_status+`, `+duration_ms+`. + +*Prometheus metrics:* `+/metrics+` endpoint via +`+TelemetryMetricsPrometheus.Core.scrape/0+`. + +*Maturity: 3/5.* Events emitted and handler wired. Minikaran anomaly +detection is present (traffic spike, trust shift, latency spike, path +novelty, error spike) but the path field gap limits per-path accuracy. +No dedicated telemetry integration test. + +''''' + +==== 1.7 Supporting Components + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Module |Role |Notes +|`+SafeTrust+` |Trust/exposure decision, mirrors +`+proven/SafeTrust.idr+` |Formally specified; no +`+String.to_existing_atom+` usage + +|`+RateLimiter+` |Token bucket Plug, per-`+{IP, trust_level}+` ETS +|Defaults: untrusted=10/s, authenticated=100/s, internal=unlimited + +|`+CircuitBreaker+` |Per-backend circuit breaker with half-open probe +|318-line test file + +|`+K9Contract+` |K9-SVC contract lookup and enforcement |520-line test +file, 35 test cases + +|`+VeriSimDB+` |Async audit persistence |Module exists; status assessed +in §4 + +|`+A2ml+` |A2ML format utilities |Present; not core to gateway + +|`+ProtocolRouter+` |GraphQL/gRPC handler dispatch |Stubs per ROADMAP; +not MVP scope +|=== + +''''' + +=== 2. DSL v1 Shape — Verb Governance Spec + +==== Format + +[source,yaml] +---- +dsl_version: "1" # Required; must be exactly "1" + +governance: + global_verbs: # Required; non-empty list of HTTP verbs (all-caps) + - GET + - POST + global_backend: "http://backend:8080" # Optional; backend URL for global rules + + routes: # Optional; list of route-specific overrides + - path: "/api/admin" # Required; literal string or regex pattern + verbs: [GET] # Required; non-empty list (overrides global for this path) + exposure: "authenticated" # Optional; "public" | "authenticated" | "internal" + stealth_profile: "default" # Optional; profile name for denial response + narrative: "Reason..." # Optional; human-readable explanation + backend: "http://..." # Optional; per-route backend URL override + name: "admin-get" # Optional; unique rule name for audit/debugging + +stealth: # Optional + enabled: true # Required if stealth present; boolean + status_code: 404 # Required if stealth present; integer 100-599 +---- + +==== Compilation to Enforcement Rules + +[arabic] +. `+PolicyLoader.load_policy/1+` parses YAML → `+{:ok, map()}+`. +. `+PolicyValidator.validate/1+` checks structural invariants → `+:ok+` +or `+{:error, reason}+`. +. `+PolicyCompiler.compile/2+` creates two ETS tables: +* Global verbs → `+{:global, verb_atom}+` keys in main table with +`+exposure: "public"+` default. +* Literal routes → `+{:exact, path_string, verb_atom}+` keys in main +table. +* Regex routes → `+{pattern_string, verb_atom}+` keys in regex table. +. On incoming request: `+PolicyCompiler.lookup/3+` uses the three-tier +strategy; the returned `+CompiledRule+` is passed to +`+SafeTrust.evaluate/2+` for the access decision. + +==== Example (BoJ-relevant sketch) + +[source,yaml] +---- +dsl_version: "1" +governance: + global_verbs: + - GET + routes: + - path: "/health" + verbs: [GET] + exposure: "public" + narrative: "Health probe; always public." + - path: "/cartridges" + verbs: [GET, POST] + exposure: "authenticated" + narrative: "Cartridge discovery requires authentication." + - path: "/cartridges/[a-z0-9_-]+" + verbs: [GET, DELETE] + exposure: "internal" + narrative: "Cartridge lifecycle management is internal-only." + - path: "/admin" + verbs: [GET] + exposure: "internal" + narrative: "Admin surface restricted to internal trust." +stealth: + enabled: true + status_code: 404 +---- + +''''' + +=== 3. ETS Lookup Architecture + +*Tables (per policy):* Two tables created per compilation cycle, named +with a monotonic-time suffix to support atomic swap. + +[width="100%",cols="25%,25%,25%,25%",options="header",] +|=== +|Table |ETS key |ETS value |Lookup cost +|Main table |`+{:exact, path_string, verb_atom}+` |`+CompiledRule+` +struct |O(1) + +|Main table |`+{:global, verb_atom}+` |`+CompiledRule+` struct |O(1) + +|Regex table |`+{pattern_string, verb_atom}+` |`+CompiledRule+` struct +|O(r) scan +|=== + +*Lookup key derivation:* +`+{conn.request_path, safe_verb(conn.method)}+`. The `+safe_verb/1+` +function uses an explicit allowlist map (not +`+String.to_existing_atom+`) for DoS resistance. + +*Lookup result:* `+{:ok, %CompiledRule{}}+` or `+{:error, :no_match}+`. +No match → default-deny (403 or stealth response). + +*Table options:* +`+:set, :public, :named_table, read_concurrency: true+`. The regex table +uses the same options. ETS tables are owned by the process that calls +`+compile/2+`; if that process dies, the tables are garbage-collected — +OTP supervision must own policy loading. + +''''' + +=== 4. Current Trust-Level Integration + +==== Header path (default, tested) + +[arabic] +. `+strip_untrusted_headers/2+`: removes `+X-Trust-Level+` unless +`+conn.remote_ip+` is in `+:trusted_proxies+` config (default: +`+["127.0.0.1", "::1"]+`). +. `+extract_trust_level_from_header/1+`: reads the `+X-Trust-Level+` +header (configurable name). +. `+SafeTrust.parse_trust/1+`: maps `+"authenticated"+` → +`+:authenticated+`, `+"internal"+` → `+:internal+`, everything else → +`+:untrusted+`. No `+String.to_existing_atom+`. + +==== mTLS path (coded, NOT primary proved path) + +Configuration: +`+config :http_capability_gateway, :trust_level_source, "mtls"+`. + +[arabic] +. `+get_peer_cert/1+`: calls `+cowboy_req:cert/1+` on the Cowboy +request. Returns `+:undefined+` if no cert. +. `+extract_cert_subject/1+`: `+public_key.pkix_decode_cert(der, :otp)+` +with `+Record.extract+` OTP-version-safe field accessors for +`+OTPCertificate+` and `+OTPTBSCertificate+`. Extracts O (OID 2.5.4.10), +OU (2.5.4.11), CN (2.5.4.3). +. `+determine_trust_level_from_cert/2+`: OU=="`Internal Services`" && +verified → "`internal`"; verified → "`authenticated`"; else +"`untrusted`". +. `+is_cert_verified/1+`: currently just checks cert presence. Does NOT +read Cowboy’s peer verify result. This is a spec gap — mTLS trust relies +on the caller having correctly configured `+verify: :verify_peer+` in +Cowboy TLS opts, but the gateway code does not verify that assumption or +fail-closed if it is absent. + +*Gap summary for mTLS:* The code is more than skeleton — it has proper +OTP record accessors and handles decode errors conservatively. But the +Cowboy TLS configuration is absent from all shipped config files; no +integration test uses a real CA; and `+is_cert_verified/1+` does not +check actual TLS verification state from the transport. The mTLS path is +*not the primary proved path*. + +==== VeriSimDB integration status + +`+lib/http_capability_gateway/verisimdb.ex+` is present. It is called as +`+VeriSimDB.audit_allow/6+` and `+VeriSimDB.audit_deny/4+` in the +gateway. The module must be reviewed to determine if it is a real +integration or a stub — based on TEST-NEEDS.md there is no VeriSimDB +integration test listed as passing. For this audit: *treat as thin stub +/ fire-and-forget until confirmed otherwise*. + +''''' + +=== 5. Telemetry Shape + +==== Events + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Event |Trigger point |Key measurements +|`+[:http_capability_gateway, :access_decision]+` +|`+Gateway.log_decision/7+` |`+duration_us+` + +|`+[:http_capability_gateway, :request, :completed]+` +|`+Logging.log_request_completed/4+` |`+duration_us+`, `+count: 1+` + +|`+[:http_capability_gateway, :rate_limit, :exceeded]+` +|`+RateLimiter.call/2+` |`+count: 1+` +|=== + +==== Structured log shape (JSON) + +[source,json] +---- +{ + "timestamp": "2026-01-22T23:00:00.000Z", + "level": "info", + "message": "Access decision", + "event": "access_decision", + "request_id": "req-abc123", + "path": "/api/users/123", + "verb": "GET", + "trust_level": "authenticated", + "decision": "allow", + "exposure": "authenticated", + "narrative": "...", + "duration_us": 45 +} +---- + +==== Prometheus metrics + +Scraped from `+/metrics+` via `+TelemetryMetricsPrometheus.Core+`. +Metrics include request counts by decision, duration histograms, policy +rule counts. Metric names are derived by the +`+telemetry_metrics_prometheus+` library from the telemetry event names. +No curated metric name list is published in the repo. + +==== Minikaran dashboard + +`+GET /api/v1/minikaran+` returns JSON with `+status+`, `+anomalies+`, +and `+baseline+`. Reads directly from ETS (O(1)). Anomaly types: +`+traffic_spike+`, `+trust_shift+`, `+latency_spike+`, `+path_novelty+`, +`+error_spike+`. + +''''' + +=== 6. Dependencies + +From `+mix.exs+`, version `+0.1.0-dev+`: + +[cols=",,",options="header",] +|=== +|Dependency |Version |Role +|Elixir |`+~> 1.19+` |Runtime +|OTP |27+ (stated in README) |Runtime +|`+plug_cowboy+` |`+~> 2.7+` |HTTP server (Cowboy 2.x) +|`+plug+` |`+~> 1.15+` |Request pipeline +|`+jason+` |`+~> 1.4+` |JSON encoding/decoding +|`+yaml_elixir+` |`+~> 2.11+` |YAML policy loading +|`+ex_json_schema+` |`+~> 0.10+` |JSON Schema validation +|`+req+` |`+~> 0.5+` |HTTP client (backend proxy) +|`+telemetry+` |`+~> 1.2+` |Event system +|`+telemetry_metrics+` |`+~> 1.0+` |Metrics definitions +|`+telemetry_poller+` |`+~> 1.1+` |Periodic metric polling +|`+prometheus_telemetry+` |`+~> 0.4+` |Prometheus scrape endpoint +|`+stream_data+` |`+~> 1.0+` (test only) |Property-based testing +|`+ex_doc+` |`+~> 0.34+` (dev only) |Documentation generation +|=== + +No Phoenix dependency. No Ecto. Pure Plug/Cowboy stack. + +*Note:* `+ex_json_schema+` is listed as a dependency but its usage in +the runtime path is not visible in the core modules audited. It may be +used in a config validation path not audited here. + +''''' + +=== 7. Readme-Declared Gaps + +==== From README.adoc + +* "`The main remaining gaps are security depth, end-to-end verification, +and benchmark evidence.`" +* "`Trust Level Integration: Current implementation is header-based, +with mTLS-oriented direction documented but not the primary proved path +yet.`" +* "`Fast Policy Enforcement Architecture: ETS-backed lookups and +compiled policy rules; benchmark evidence still needs to be +formalised.`" +* "`Do not treat the current suite as sufficient proof for whole-site +gateway deployment.`" + +==== From ROADMAP.adoc (v0.1.x milestone — still open) + +* `+[ ] Truthful status docs+` — marked open. +* `+[ ] Request/path/verb governance solid+` — open. +* `+[ ] Capability checks and proxy path exercised end to end+` — open. + +P0/P1/P2 tasks are all marked `+[x]+` (done), but v0.1.x milestones +remain open. This means the artifact work (tests, docs, benchmarks) is +done, but the sign-off milestone has not been claimed. + +==== From TEST-NEEDS.md (remaining gaps) + +* Real-CA mTLS integration test (no live cert in fixtures). +* Zig FFI integration test execution (requires zig toolchain; not in +`+mix test+`). +* Container build smoke test (CI only, not in `+mix test+`). +* Error handling: upstream timeout (implicit; no dedicated test). +* Self-tests for config validation on startup. + +==== From PROOFS_NEEDED.md + +Priority MEDIUM. Four proof obligations outstanding: + +[cols=",",options="header",] +|=== +|Component |What needs proving +|Capability token validation |Token issuance/revocation correctness +|Protocol state machine |Session state transitions total +|Permission composition |Capability intersection/union laws +|ABI type safety |FFI boundary type marshalling +|=== + +*Current proof status:* `+src/abi/*.idr+` (Protocol.idr, Types.idr) +exist. No `+believe_me+`. LOC ~9,500. Idris2 ABI layer is present. The +four proof obligations above are not yet met. + +''''' + +=== 8. Estimated Distance to BoJ-Production-Grade + +The ~1–2 month estimate from the BoJ Trustfile +`+[HTTP_CAPABILITY_GATEWAY]+` entry is broadly confirmed by this audit, +with the following breakdown: + +==== Already solid (no work needed for BoJ integration) + +* Core policy pipeline (load → validate → compile → enforce). +Production-quality. +* Atomic policy reload. Implemented and tested. +* SafeTrust evaluation. Formally mirrors Idris2 spec. +* Rate limiter. ETS-backed token bucket. Tested. +* Circuit breaker. Present and tested. +* K9-SVC contract enforcement. Present and tested. +* Structured logging / telemetry events. Emitted. +* Security headers. Correct OWASP set. +* DoS resistance (no atom table exhaustion from unknown verbs or trust +strings). + +==== Gaps to close before BoJ production wiring + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Gap |Severity for BoJ |Estimated effort +|mTLS: Cowboy TLS config + `+is_cert_verified+` using real TLS state +|HIGH — header trust is forgeable without this |1–2 weeks + +|mTLS: real-CA integration test in test suite |HIGH — claimed path must +have test evidence |1 week + +|v0.1.x milestones sign-off (E2E path proof) |MEDIUM |1 week + +|Benchmark formalisation (latency numbers published) |MEDIUM — "`fast`" +claim unsubstantiated |1 week + +|VeriSimDB integration confirmed (not stub) |MEDIUM — audit trail +depends on this |1 week + +|Telemetry path gap (access_decision missing path field) |LOW for +correctness, MEDIUM for observability |days + +|PROOFS_NEEDED.md obligation: capability token + permission composition +|LOW for initial wiring, required before v1 |2–4 weeks +|=== + +*Revised timeline:* 8–12 weeks as written in the integration plan is +appropriate. The 1–2 month estimate from the Trustfile could be met if +mTLS and benchmark work are done in parallel and only minimal proof work +is scoped for initial wiring. However if the mTLS proof obligations from +PROOFS_NEEDED.md are required before production (which they should be +given BoJ’s verification ladder), the 8–12 week estimate is more +realistic. + +*Tier placement note:* The Trustfile places this in `+tier_2_gateway+` +alongside Svalinn. This audit finds no architectural incompatibility +with that placement. The gateway is a Plug/Cowboy application — it can +sit in front of BoJ’s Zig API port receiving traffic forwarded from +Cloudflare. Svalinn operates at container level; the two are +complementary, not conflicting. Tier 2 placement is correct. diff --git a/docs/integration/http-capability-gateway-audit.md b/docs/integration/http-capability-gateway-audit.md deleted file mode 100644 index 052bd9e7..00000000 --- a/docs/integration/http-capability-gateway-audit.md +++ /dev/null @@ -1,515 +0,0 @@ - - - -# http-capability-gateway — Integration Audit - -**Audited:** 2026-04-17 -**Auditor:** Claude Sonnet 4.6 (boj-server integration scope session) -**Gateway repo:** `web-ecosystem/http-capability-gateway` (GitHub: `hyperpolymath/http-capability-gateway`) -**Gateway version:** `0.1.0-dev` (ROADMAP.adoc: CRG grade C achieved 2026-04-04) -**Purpose of audit:** Establish truth baseline before Phase A contract definition work. - ---- - -## 1. Component Inventory — What Is Actually Shipped - -### 1.1 PolicyLoader (`lib/http_capability_gateway/policy_loader.ex`) - -Reads a YAML file or binary string and returns `{:ok, map()}` or `{:error, String.t()}`. -Handles multi-document YAML (uses first document). Trims empty input. Logs service name -from `policy["service"]["name"]` on success. - -**Test coverage:** `test/policy_loader_test.exs` — 11 `test` declarations, 165 lines. -Covers: happy path from string, from file, empty content, non-map content, YAML parse -errors, multi-document YAML. No property-based tests for the loader specifically. - -**Maturity: 4/5.** Solid; no known gaps. Missing: no dedicated test for file -permissions errors or very large files. - ---- - -### 1.2 PolicyValidator (`lib/http_capability_gateway/policy_validator.ex`) - -Validates a parsed policy map against the DSL v1 schema. Checks: -- `dsl_version` must be present and equal to `"1"`. -- `governance` must be a map containing `global_verbs` (non-empty list of valid HTTP verbs). -- `governance.routes` (optional) must be a list of maps each with `path` (non-empty string, - valid regex) and `verbs` (non-empty list of valid HTTP verbs). -- `stealth` (optional) must have both `enabled` (boolean) and `status_code` (integer 100–599). - -Returns `:ok` or `{:error, reason_string}`. - -**Test coverage:** `test/policy_validator_test.exs` — 18 `test` declarations, 257 lines. -Covers: all required fields, invalid verbs, invalid stealth combos, route-level validation. -Property tests for validator live in `test/policy_property_test.exs` (1 property, StreamData). - -**Maturity: 4/5.** Well-exercised. Does not validate `service.name`, `narrative`, or -the v0 DSL format (two DSL versions co-exist in the README; validator only handles v1). - ---- - -### 1.3 PolicyCompiler (`lib/http_capability_gateway/policy_compiler.ex`) - -Compiles a validated policy into a dual ETS table pair: - -- **Main table** — exact literal paths (`{:exact, path, verb}`) and global verb rules - (`{:global, verb}`). O(1) lookup. -- **Regex table** — path patterns containing regex metacharacters, keyed as - `{pattern_string, verb_atom}`. Scanned O(r) where r = number of regex routes. - -Both tables created with `:set, :public, :named_table, read_concurrency: true`. - -**Atomic swap pattern:** On hot-reload, new tables are created with a monotonic-time -suffix, compiled in full, then the `Application.put_env/3` references are swapped -atomically. The old tables are deleted only after the swap. A compilation failure -leaves the existing tables intact (last-known-good policy preserved). - -**Lookup tiers (implemented):** -1. Tier 1: `ETS.lookup(table, {:exact, path, verb})` — O(1). -2. Tier 2: `ETS.tab2list(regex_table)` + Enum.find regex match — O(r). -3. Tier 3: `ETS.lookup(table, {:global, verb})` — O(1). - -**`CompiledRule` struct fields:** `path_pattern`, `path_regex`, `verb`, `exposure`, -`stealth_profile`, `narrative`, `backend`, `name`. - -**Test coverage:** `test/policy_compiler_test.exs` — 7 `test` declarations, 87 lines. -Tests: compile success, ETS size, stats/0, error paths. Thin; more coverage in -`test/e2e_test.exs` through the full pipeline. - -**Maturity: 4/5.** Implementation is solid and well-commented. Gap: no test for -the atomic swap path under concurrent load (covered separately in `concurrency_test.exs`). - ---- - -### 1.4 Gateway (`lib/http_capability_gateway/gateway.ex`) - -`Plug.Router` implementing the HTTP enforcement pipeline. Plug stack order: - -``` -Plug.Logger -→ security_headers (X-Content-Type-Options, X-Frame-Options, Referrer-Policy, Cache-Control, Connection) -→ strip_untrusted_headers (removes X-Trust-Level unless conn.remote_ip is in :trusted_proxies) -→ extract_trust (sets conn.assigns[:trust_level] via SafeTrust.parse_trust/1) -→ RateLimiter (token bucket, per-{IP, trust_level}, ETS-backed) -→ match/dispatch (Plug.Router routes) -``` - -Special routes: `/health`, `/ready`, `/metrics`, `/api/v1/minikaran`, catch-all. - -**Trust extraction:** Configurable via `:trust_level_source`. Two modes: -- `"header"` (default): reads `X-Trust-Level` header → `SafeTrust.parse_trust/1`. -- `"mtls"`: reads peer cert via `cowboy_req:cert/1`, decodes DER with - `:public_key.pkix_decode_cert/2` in `:otp` mode, uses OTP `Record.extract/2` - accessors for `OTPCertificate`/`OTPTBSCertificate` to extract CN/O/OU fields. - Determines trust by: OU=="Internal Services" && verified → "internal"; verified → - "authenticated"; else "untrusted". - -**mTLS implementation status:** The code path for mTLS exists in -`extract_trust_level_from_cert/1`. It uses `cowboy_req:cert/1` to retrieve the -DER-encoded peer cert, then decodes it with proper OTP record accessors. However: -- No live CA or cert fixture is used in tests (TEST-NEEDS.md explicitly flags - "Real-CA mTLS integration test" as missing). -- `is_cert_verified/1` checks only for cert presence (not actual TLS verification - status from Cowboy). This means the mTLS path reads the certificate subject but - does not enforce chain validation independently — it relies on Cowboy having been - configured with `verify: :verify_peer`. That Cowboy configuration is not present in - any shipped config file. -- **Conclusion:** mTLS trust extraction is coded but is not the primary proved path. - Header-based trust extraction is what the system actually tests and deploys. - -**Access decision:** `SafeTrust.evaluate(trust_level, exposure)` — formally mirrors -`proven/SafeTrust.idr`. Returns `{:allow, t, e}` or `{:deny, t, e}`. No ad-hoc -evaluate logic remains in the gateway. - -**K9-SVC contract enforcement:** Present and wired. After policy allow, the gateway -looks up a K9 contract for `(path, verb)`. If found: pre-proxy trust threshold check, -proxy with timing, post-proxy latency SLA check with configurable breach policy -(`:log`, `:alert`, `:circuit_break`, `:fallback`). Circuit breaker checked before -contract lookup. - -**VeriSimDB integration:** `VeriSimDB.audit_allow/6` and `VeriSimDB.audit_deny/4` -called asynchronously after every allow/deny decision. See §4 in this audit for -the VeriSimDB stub assessment. - -**Test coverage:** `test/gateway_test.exs` — 23 `test` declarations, 204 lines. -`test/e2e_test.exs` — 22 `test` declarations, 532 lines (full lifecycle). `test/security_test.exs` -— 36 `test` declarations, 403 lines (header injection, SSRF, trust spoofing, atom exhaustion). - -**Maturity: 3/5.** Core logic solid. mTLS not the primary proved path. No live -Cowboy TLS-verify-peer test. VeriSimDB integration is stubbed (see §4). - ---- - -### 1.5 Proxy (`lib/http_capability_gateway/proxy.ex`) - -Forwards allowed requests to a single configured `backend_url`. Uses `Req` HTTP -client. Filters hop-by-hop headers. Adds `X-Forwarded-For`, `X-Forwarded-Proto`, -`X-Forwarded-Host`, `X-Gateway: http-capability-gateway`. 30-second receive timeout. -No retry (caller handles). Health check via `GET {backend_url}/health`. - -**Backend URL:** Read from `Application.get_env(:http_capability_gateway, :backend_url, -"http://localhost:8080")`. Single static backend — no load balancing or multi-backend -routing. - -**Note for BoJ integration:** The proxy assumes a single HTTP backend. BoJ's unified -Zig API currently listens on a single port. The integration forwarding path will be -HTTP (TCP localhost or Unix socket). See Phase A for the contract definition. - -**Test coverage:** Proxy forwarded through E2E and gateway tests. No dedicated -`proxy_test.exs`. SSRF resistance tested in `security_test.exs` (internal/localhost -URL blocking logic checked there). - -**Maturity: 3/5.** Works for the happy path. SSRF defence tested. Missing: dedicated -proxy failure tests; no timeout boundary test; streaming large bodies untested. - ---- - -### 1.6 Telemetry / Minikaran (`lib/http_capability_gateway/minikaran.ex`, -`lib/http_capability_gateway/minikaran/telemetry_handler.ex`, -`lib/http_capability_gateway/logging.ex`, -`lib/http_capability_gateway/log_formatter.ex`) - -**Telemetry events emitted:** - -| Event name | Measurements | Metadata | -|---|---|---| -| `[:http_capability_gateway, :access_decision]` | `%{duration: duration_us}` | `%{decision: atom, verb: atom, trust_level: atom}` | -| `[:http_capability_gateway, :request, :completed]` | `%{count: 1, duration: total_us}` | `%{status: http_status}` | -| `[:http_capability_gateway, :rate_limit, :exceeded]` | `%{count: 1}` | `%{trust_level: atom, client: string}` | - -**Minikaran TelemetryHandler** hooks all three events and forwards them as -observation maps to the Minikaran GenServer via `GenServer.cast`. Observation map -shape: `%{path, trust_level, latency_us, status, client_ip, timestamp}`. - -**Note:** The `access_decision` event does not include the request path in its -metadata (Gateway emits it as `%{decision, verb, trust_level}` only). The -TelemetryHandler reconstructs a pseudo-path as `"/#{verb}"`. This means the -Minikaran baseline cannot distinguish `/api/users` from `/api/posts` when both -use GET — path-level anomaly detection is approximate until this is fixed. - -**Structured log format** (LogFormatter): JSON, fields include `timestamp`, `level`, -`message`, `request_id`, `method`, `path`, `trust_level`, `verb_allowed`, -`stealth_triggered`, `response_status`, `duration_ms`. - -**Prometheus metrics:** `/metrics` endpoint via `TelemetryMetricsPrometheus.Core.scrape/0`. - -**Maturity: 3/5.** Events emitted and handler wired. Minikaran anomaly detection is -present (traffic spike, trust shift, latency spike, path novelty, error spike) but -the path field gap limits per-path accuracy. No dedicated telemetry integration test. - ---- - -### 1.7 Supporting Components - -| Module | Role | Notes | -|---|---|---| -| `SafeTrust` | Trust/exposure decision, mirrors `proven/SafeTrust.idr` | Formally specified; no `String.to_existing_atom` usage | -| `RateLimiter` | Token bucket Plug, per-`{IP, trust_level}` ETS | Defaults: untrusted=10/s, authenticated=100/s, internal=unlimited | -| `CircuitBreaker` | Per-backend circuit breaker with half-open probe | 318-line test file | -| `K9Contract` | K9-SVC contract lookup and enforcement | 520-line test file, 35 test cases | -| `VeriSimDB` | Async audit persistence | Module exists; status assessed in §4 | -| `A2ml` | A2ML format utilities | Present; not core to gateway | -| `ProtocolRouter` | GraphQL/gRPC handler dispatch | Stubs per ROADMAP; not MVP scope | - ---- - -## 2. DSL v1 Shape — Verb Governance Spec - -### Format - -```yaml -dsl_version: "1" # Required; must be exactly "1" - -governance: - global_verbs: # Required; non-empty list of HTTP verbs (all-caps) - - GET - - POST - global_backend: "http://backend:8080" # Optional; backend URL for global rules - - routes: # Optional; list of route-specific overrides - - path: "/api/admin" # Required; literal string or regex pattern - verbs: [GET] # Required; non-empty list (overrides global for this path) - exposure: "authenticated" # Optional; "public" | "authenticated" | "internal" - stealth_profile: "default" # Optional; profile name for denial response - narrative: "Reason..." # Optional; human-readable explanation - backend: "http://..." # Optional; per-route backend URL override - name: "admin-get" # Optional; unique rule name for audit/debugging - -stealth: # Optional - enabled: true # Required if stealth present; boolean - status_code: 404 # Required if stealth present; integer 100-599 -``` - -### Compilation to Enforcement Rules - -1. `PolicyLoader.load_policy/1` parses YAML → `{:ok, map()}`. -2. `PolicyValidator.validate/1` checks structural invariants → `:ok` or `{:error, reason}`. -3. `PolicyCompiler.compile/2` creates two ETS tables: - - Global verbs → `{:global, verb_atom}` keys in main table with `exposure: "public"` default. - - Literal routes → `{:exact, path_string, verb_atom}` keys in main table. - - Regex routes → `{pattern_string, verb_atom}` keys in regex table. -4. On incoming request: `PolicyCompiler.lookup/3` uses the three-tier strategy; the - returned `CompiledRule` is passed to `SafeTrust.evaluate/2` for the access decision. - -### Example (BoJ-relevant sketch) - -```yaml -dsl_version: "1" -governance: - global_verbs: - - GET - routes: - - path: "/health" - verbs: [GET] - exposure: "public" - narrative: "Health probe; always public." - - path: "/cartridges" - verbs: [GET, POST] - exposure: "authenticated" - narrative: "Cartridge discovery requires authentication." - - path: "/cartridges/[a-z0-9_-]+" - verbs: [GET, DELETE] - exposure: "internal" - narrative: "Cartridge lifecycle management is internal-only." - - path: "/admin" - verbs: [GET] - exposure: "internal" - narrative: "Admin surface restricted to internal trust." -stealth: - enabled: true - status_code: 404 -``` - ---- - -## 3. ETS Lookup Architecture - -**Tables (per policy):** Two tables created per compilation cycle, named with a -monotonic-time suffix to support atomic swap. - -| Table | ETS key | ETS value | Lookup cost | -|---|---|---|---| -| Main table | `{:exact, path_string, verb_atom}` | `CompiledRule` struct | O(1) | -| Main table | `{:global, verb_atom}` | `CompiledRule` struct | O(1) | -| Regex table | `{pattern_string, verb_atom}` | `CompiledRule` struct | O(r) scan | - -**Lookup key derivation:** `{conn.request_path, safe_verb(conn.method)}`. The -`safe_verb/1` function uses an explicit allowlist map (not `String.to_existing_atom`) -for DoS resistance. - -**Lookup result:** `{:ok, %CompiledRule{}}` or `{:error, :no_match}`. No match → -default-deny (403 or stealth response). - -**Table options:** `:set, :public, :named_table, read_concurrency: true`. The regex -table uses the same options. ETS tables are owned by the process that calls -`compile/2`; if that process dies, the tables are garbage-collected — OTP supervision -must own policy loading. - ---- - -## 4. Current Trust-Level Integration - -### Header path (default, tested) - -1. `strip_untrusted_headers/2`: removes `X-Trust-Level` unless `conn.remote_ip` is in - `:trusted_proxies` config (default: `["127.0.0.1", "::1"]`). -2. `extract_trust_level_from_header/1`: reads the `X-Trust-Level` header (configurable name). -3. `SafeTrust.parse_trust/1`: maps `"authenticated"` → `:authenticated`, `"internal"` → - `:internal`, everything else → `:untrusted`. No `String.to_existing_atom`. - -### mTLS path (coded, NOT primary proved path) - -Configuration: `config :http_capability_gateway, :trust_level_source, "mtls"`. - -1. `get_peer_cert/1`: calls `cowboy_req:cert/1` on the Cowboy request. Returns - `:undefined` if no cert. -2. `extract_cert_subject/1`: `public_key.pkix_decode_cert(der, :otp)` with - `Record.extract` OTP-version-safe field accessors for `OTPCertificate` and - `OTPTBSCertificate`. Extracts O (OID 2.5.4.10), OU (2.5.4.11), CN (2.5.4.3). -3. `determine_trust_level_from_cert/2`: OU=="Internal Services" && verified → - "internal"; verified → "authenticated"; else "untrusted". -4. `is_cert_verified/1`: currently just checks cert presence. Does NOT read Cowboy's - peer verify result. This is a spec gap — mTLS trust relies on the caller having - correctly configured `verify: :verify_peer` in Cowboy TLS opts, but the gateway - code does not verify that assumption or fail-closed if it is absent. - -**Gap summary for mTLS:** The code is more than skeleton — it has proper OTP record -accessors and handles decode errors conservatively. But the Cowboy TLS configuration -is absent from all shipped config files; no integration test uses a real CA; and -`is_cert_verified/1` does not check actual TLS verification state from the transport. -The mTLS path is **not the primary proved path**. - -### VeriSimDB integration status - -`lib/http_capability_gateway/verisimdb.ex` is present. It is called as -`VeriSimDB.audit_allow/6` and `VeriSimDB.audit_deny/4` in the gateway. The module -must be reviewed to determine if it is a real integration or a stub — based on -TEST-NEEDS.md there is no VeriSimDB integration test listed as passing. For this -audit: **treat as thin stub / fire-and-forget until confirmed otherwise**. - ---- - -## 5. Telemetry Shape - -### Events - -| Event | Trigger point | Key measurements | -|---|---|---| -| `[:http_capability_gateway, :access_decision]` | `Gateway.log_decision/7` | `duration_us` | -| `[:http_capability_gateway, :request, :completed]` | `Logging.log_request_completed/4` | `duration_us`, `count: 1` | -| `[:http_capability_gateway, :rate_limit, :exceeded]` | `RateLimiter.call/2` | `count: 1` | - -### Structured log shape (JSON) - -```json -{ - "timestamp": "2026-01-22T23:00:00.000Z", - "level": "info", - "message": "Access decision", - "event": "access_decision", - "request_id": "req-abc123", - "path": "/api/users/123", - "verb": "GET", - "trust_level": "authenticated", - "decision": "allow", - "exposure": "authenticated", - "narrative": "...", - "duration_us": 45 -} -``` - -### Prometheus metrics - -Scraped from `/metrics` via `TelemetryMetricsPrometheus.Core`. Metrics include -request counts by decision, duration histograms, policy rule counts. Metric -names are derived by the `telemetry_metrics_prometheus` library from the telemetry -event names. No curated metric name list is published in the repo. - -### Minikaran dashboard - -`GET /api/v1/minikaran` returns JSON with `status`, `anomalies`, and `baseline`. -Reads directly from ETS (O(1)). Anomaly types: `traffic_spike`, `trust_shift`, -`latency_spike`, `path_novelty`, `error_spike`. - ---- - -## 6. Dependencies - -From `mix.exs`, version `0.1.0-dev`: - -| Dependency | Version | Role | -|---|---|---| -| Elixir | `~> 1.19` | Runtime | -| OTP | 27+ (stated in README) | Runtime | -| `plug_cowboy` | `~> 2.7` | HTTP server (Cowboy 2.x) | -| `plug` | `~> 1.15` | Request pipeline | -| `jason` | `~> 1.4` | JSON encoding/decoding | -| `yaml_elixir` | `~> 2.11` | YAML policy loading | -| `ex_json_schema` | `~> 0.10` | JSON Schema validation | -| `req` | `~> 0.5` | HTTP client (backend proxy) | -| `telemetry` | `~> 1.2` | Event system | -| `telemetry_metrics` | `~> 1.0` | Metrics definitions | -| `telemetry_poller` | `~> 1.1` | Periodic metric polling | -| `prometheus_telemetry` | `~> 0.4` | Prometheus scrape endpoint | -| `stream_data` | `~> 1.0` (test only) | Property-based testing | -| `ex_doc` | `~> 0.34` (dev only) | Documentation generation | - -No Phoenix dependency. No Ecto. Pure Plug/Cowboy stack. - -**Note:** `ex_json_schema` is listed as a dependency but its usage in the runtime -path is not visible in the core modules audited. It may be used in a config -validation path not audited here. - ---- - -## 7. Readme-Declared Gaps - -### From README.adoc - -- "The main remaining gaps are security depth, end-to-end verification, and - benchmark evidence." -- "Trust Level Integration: Current implementation is header-based, with mTLS-oriented - direction documented but not the primary proved path yet." -- "Fast Policy Enforcement Architecture: ETS-backed lookups and compiled policy rules; - benchmark evidence still needs to be formalised." -- "Do not treat the current suite as sufficient proof for whole-site gateway deployment." - -### From ROADMAP.adoc (v0.1.x milestone — still open) - -- `[ ] Truthful status docs` — marked open. -- `[ ] Request/path/verb governance solid` — open. -- `[ ] Capability checks and proxy path exercised end to end` — open. - -P0/P1/P2 tasks are all marked `[x]` (done), but v0.1.x milestones remain open. -This means the artifact work (tests, docs, benchmarks) is done, but the sign-off -milestone has not been claimed. - -### From TEST-NEEDS.md (remaining gaps) - -- Real-CA mTLS integration test (no live cert in fixtures). -- Zig FFI integration test execution (requires zig toolchain; not in `mix test`). -- Container build smoke test (CI only, not in `mix test`). -- Error handling: upstream timeout (implicit; no dedicated test). -- Self-tests for config validation on startup. - -### From PROOFS_NEEDED.md - -Priority MEDIUM. Four proof obligations outstanding: - -| Component | What needs proving | -|---|---| -| Capability token validation | Token issuance/revocation correctness | -| Protocol state machine | Session state transitions total | -| Permission composition | Capability intersection/union laws | -| ABI type safety | FFI boundary type marshalling | - -**Current proof status:** `src/abi/*.idr` (Protocol.idr, Types.idr) exist. No -`believe_me`. LOC ~9,500. Idris2 ABI layer is present. The four proof obligations -above are not yet met. - ---- - -## 8. Estimated Distance to BoJ-Production-Grade - -The ~1–2 month estimate from the BoJ Trustfile `[HTTP_CAPABILITY_GATEWAY]` entry -is broadly confirmed by this audit, with the following breakdown: - -### Already solid (no work needed for BoJ integration) - -- Core policy pipeline (load → validate → compile → enforce). Production-quality. -- Atomic policy reload. Implemented and tested. -- SafeTrust evaluation. Formally mirrors Idris2 spec. -- Rate limiter. ETS-backed token bucket. Tested. -- Circuit breaker. Present and tested. -- K9-SVC contract enforcement. Present and tested. -- Structured logging / telemetry events. Emitted. -- Security headers. Correct OWASP set. -- DoS resistance (no atom table exhaustion from unknown verbs or trust strings). - -### Gaps to close before BoJ production wiring - -| Gap | Severity for BoJ | Estimated effort | -|---|---|---| -| mTLS: Cowboy TLS config + `is_cert_verified` using real TLS state | HIGH — header trust is forgeable without this | 1–2 weeks | -| mTLS: real-CA integration test in test suite | HIGH — claimed path must have test evidence | 1 week | -| v0.1.x milestones sign-off (E2E path proof) | MEDIUM | 1 week | -| Benchmark formalisation (latency numbers published) | MEDIUM — "fast" claim unsubstantiated | 1 week | -| VeriSimDB integration confirmed (not stub) | MEDIUM — audit trail depends on this | 1 week | -| Telemetry path gap (access_decision missing path field) | LOW for correctness, MEDIUM for observability | days | -| PROOFS_NEEDED.md obligation: capability token + permission composition | LOW for initial wiring, required before v1 | 2–4 weeks | - -**Revised timeline:** 8–12 weeks as written in the integration plan is appropriate. -The 1–2 month estimate from the Trustfile could be met if mTLS and benchmark work -are done in parallel and only minimal proof work is scoped for initial wiring. -However if the mTLS proof obligations from PROOFS_NEEDED.md are required before -production (which they should be given BoJ's verification ladder), the 8–12 week -estimate is more realistic. - -**Tier placement note:** The Trustfile places this in `tier_2_gateway` alongside -Svalinn. This audit finds no architectural incompatibility with that placement. -The gateway is a Plug/Cowboy application — it can sit in front of BoJ's Zig API -port receiving traffic forwarded from Cloudflare. Svalinn operates at container -level; the two are complementary, not conflicting. Tier 2 placement is correct. diff --git a/docs/integration/http-capability-gateway-boj-contract.adoc b/docs/integration/http-capability-gateway-boj-contract.adoc new file mode 100644 index 00000000..69878a07 --- /dev/null +++ b/docs/integration/http-capability-gateway-boj-contract.adoc @@ -0,0 +1,265 @@ +== http-capability-gateway ↔ BoJ — HTTP Contract + +*Version:* 1.0 *Date:* 2026-05-18 *Status:* Phase A deliverable A1 +(normative for Phases B–E) *ADR:* +`+docs/decisions/0004-adopt-http-capability-gateway.md+` *Plan:* +`+docs/integration/http-capability-gateway-plan.md+` (§ Phase A, A1) +*Audit baseline:* `+docs/integration/http-capability-gateway-audit.md+` +(2026-04-17) *Tracking:* standards#91 (parent), standards#96 (Phase A) + +____ +*File-format note.* This document and its sibling +(`+http-capability-gateway-policy-authoring.md+`) are kept as `+.md+`, +not `+.adoc+`, deliberately: the integration plan normatively prescribes +the path `+docs/integration/http-capability-gateway-boj-contract.md+`, +the Phase C Trustfile `+[SEAMS]+` entry will reference that exact path, +and the entire `+docs/integration/+` + `+docs/decisions/+` doc-set is +Markdown. Faithfulness to the normatively-referenced path outweighs the +estate `+.adoc+` default here. +____ + +''''' + +=== 0. Scope + +This contract defines the *exact HTTP boundary* between the front +component (`+hyperpolymath/http-capability-gateway+`, tier 2) and the +back component (BoJ’s unified Zig API gnosis handler, fronted by +`+BojRest.Router+`, tier 3). It is the authoritative seam description. +No code is wired in Phase A; this document governs Phases B–E and is the +contract the Phase C seam test must satisfy. + +The gateway sits *between* Cloudflare edge (tier 1) and BoJ’s gnosis +handler. It enforces verb governance, trust-level derivation, stealth, +and audit logging on every request *before* any BoJ cartridge logic +runs. + +''''' + +=== 1. Transport + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Aspect |Staging |Production +|Gateway → BoJ transport |*TCP loopback* |*Unix domain socket* + +|Rationale |Simpler; matches the gateway’s single `+backend_url+` +config; no socket plumbing needed for first bring-up. |Avoids port +allocation, removes the loopback TCP attack surface, and is the +preferred path for co-located Podman containers sharing a pod network +namespace. +|=== + +*Decision (committed, per plan risk "`Port / transport indecision`"):* + +* *Staging:* gateway `+BACKEND_URL=http://127.0.0.1:7700+` → BoJ gnosis +handler listens on loopback `+:7700+` for gateway-forwarded traffic. The +externally visible port is owned by the gateway (8443 TLS / 8080 +HTTP-behind-Tunnel); BoJ’s `+:7700+` is *not* externally routable. +* *Production:* gateway +`+BACKEND_URL=http://unix:/run/boj/gnosis.sock:/+` (or the gateway’s +documented Unix-socket backend form). BoJ binds the gnosis handler to +`+/run/boj/gnosis.sock+`, mode `+0660+`, owned by the shared pod user. +No TCP port is opened for the back side in production. + +The gateway proxy module currently supports a single `+backend_url+`; +both forms above are single-backend and within its current capability. +Horizontal scaling of BoJ (multi-backend) is explicitly *out of scope* +for this contract and is recorded as post-Phase-E work in the plan. + +''''' + +=== 2. Headers the gateway MUST set on forwarded requests + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Header |Value |Status +|`+X-Forwarded-For+` |original client IP |already implemented (Proxy +module) + +|`+X-Forwarded-Proto+` |`+https+` / `+http+` |already implemented + +|`+X-Forwarded-Host+` |original `+Host+` |already implemented + +|`+X-Gateway+` |`+http-capability-gateway+` |already implemented + +|`+X-Request-ID+` |gateway `+get_request_id/1+` output |propagate; BoJ +logs it + +|`+X-Trust-Level+` |`+untrusted+` \| `+authenticated+` \| `+internal+` +|*see §3 — security-critical* +|=== + +Notes: + +* `+X-Request-ID+` MUST be propagated end-to-end so the gateway audit +record and BoJ’s structured log line share a correlation key. +* The gateway’s resolved trust vocabulary is +`+untrusted | authenticated | internal+`. BoJ’s +`+BojRest.TrustPolicy+` consumes `+internal | authenticated+` as +"`credentialed`" and treats `+untrusted+`/absent as "`public only`" (see +`+BojRest.Router+` moduledoc and `+docs/AUTH-DESIGN.adoc+`). The +gateway’s `+untrusted+` maps to BoJ’s "`public`" tier. This vocabulary +mapping is fixed by this contract: ++ +[width="100%",cols="50%,50%",options="header",] +|=== +|Gateway `+X-Trust-Level+` |BoJ trust class +|`+internal+` |internal (credentialed; lifecycle/admin) + +|`+authenticated+` |authenticated (credentialed) + +|`+untrusted+` / absent |public (only `+auth.method: "none"+` +cartridges) +|=== + +''''' + +=== 3. Trust-header security invariant (load-bearing) + +`+BojRest.Router+` *bypasses trust enforcement entirely for loopback +callers* (`+loopback?(conn.remote_ip)+` → enforcement skipped; local dev +/ mcp-bridge path). In staging the gateway forwards from `+127.0.0.1+`; +in production it forwards over a loopback Unix socket. *Therefore BoJ +will treat the gateway as a fully trusted proxy and accept its +`+X-Trust-Level+` verbatim.* + +This makes the following invariant *mandatory, not advisory*: + +[arabic] +. The gateway MUST *strip any inbound `+X-Trust-Level+`* from the +original client request unless the immediate peer is in the gateway’s +`+:trusted_proxies+` allow-list. A client-supplied `+X-Trust-Level+` +MUST NEVER survive into the forwarded request. +. The gateway MUST *re-set `+X-Trust-Level+`* from its own resolved +trust decision (header-based pre-Phase-B; mTLS client-cert-derived from +Phase B onward) before forwarding. +. BoJ’s gnosis handler / `+BojRest.Router+` MUST treat `+X-Trust-Level+` +as authoritative *only* when the connection originates from the gateway +(loopback `+127.0.0.1+` in staging; the loopback Unix socket in +production). Any `+X-Trust-Level+` arriving from any other source MUST +be ignored and treated as `+untrusted+`. +. The back-side bind (BoJ `+:7700+` / `+/run/boj/gnosis.sock+`) MUST NOT +be externally routable. If it were, an attacker reaching it directly +would inherit loopback trust-bypass. Network/socket isolation of the +back side is part of this contract, not an operational nicety. + +Failure to honour (1)–(4) is a trust-forgery vulnerability: a forged +`+X-Trust-Level: internal+` from an external client would reach a +loopback-trusting router. This invariant is tested in Phase C (plan +§Phase C, "`Trust header forwarding security invariant`") and hardened +in Phase B (mTLS becomes the trust source so the resolved value cannot +be header-forged at all). + +*Implementation status (Phase C):* + +* Gateway-side (1)+(2): +`+http-capability-gateway/lib/http_capability_gateway/proxy.ex+` +`+build_backend_headers/1+` strips client-supplied `+X-Trust-Level+` + +`+X-Request-ID+` and re-emits the gateway-resolved values. Landed in +http-capability-gateway#11. +* BoJ-side (3): `+boj_rest/lib/boj_rest/trust_policy.ex+` +`+satisfies?/3+` ignores the header for any non-loopback caller via the +`+satisfies?(_required, _trust, false), do: false+` clause — defence in +depth against (4) being violated by a misconfiguration. Verified by +`+elixir/test/phase_c_seam_test.exs+` (9 tests, all live). +* {blank} +[arabic, start=4] +. is an operational/configuration invariant, not a code change; verified +during Phase E rollout (production deployments MUST front BoJ with a +loopback-only socket per the §Phase E runbook). + +''''' + +=== 4. Headers BoJ MUST NOT forward to cartridges unchanged + +* `+X-Trust-Level+` MUST be consumed and re-validated by +`+BojRest.TrustPolicy+` at the router boundary and MUST NOT be passed +opaquely into cartridge invocation arguments. Cartridges receive a +_decision_ (allowed/denied + trust class), never the raw transport +header, so a cartridge cannot be tricked by a header it was never meant +to interpret. +* `+X-Node-Identity+`, when present, is *logged for audit* by the router +but is not an authorization input and MUST NOT be promoted to a trust +signal. + +''''' + +=== 5. Error semantics + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Condition |Gateway response |Notes +|BoJ returns 5xx (e.g. `+invocation-failed+` 500) |*502 Bad Gateway* +|existing gateway behaviour; the upstream 500 body is not leaked +verbatim + +|BoJ unreachable / connection refused |circuit breaker trips → *503 +Service Unavailable* |fail-closed; matches Trustfile `+[SEAMS]+` +`+failure_mode: "fail-closed (circuit breaker)"+` + +|Verb not permitted by policy |*403* (or stealth status, see §6) +|decision made at gateway; request never reaches BoJ + +|Path not in policy |*default-deny* (403 / stealth) |unlisted path ⇒ +denied; request never reaches BoJ + +|BoJ returns 4xx (e.g. `+unknown-cartridge+` 404, `+forbidden+` 403) +|*passed through unchanged* |these are legitimate application responses, +not gateway faults +|=== + +The circuit-breaker fail-closed posture means a gateway↔BoJ seam failure +denies traffic rather than failing open — consistent with the ADR’s +"`fail-closed`" stance and the rate-limit architecture’s +defence-in-depth intent. + +''''' + +=== 6. Stealth + +Routes declared `+exposure: "internal"+` with a stealth profile return +the configured stealth status (default *404*) instead of *403* to +callers below the required trust level, hiding capability existence. The +example policy (`+config/gateway-policy-boj-example.yaml+`) enables +`+stealth: { enabled: true, status_code: 404 }+` and applies it to +lifecycle/admin/SDP routes (§ A3). + +''''' + +=== 7. Deployment trust topology (input to Phase B) + +Phase B (mTLS) needs the trusted-proxy / loopback topology fixed here: + +* *Staging:* gateway and BoJ co-located; gateway → BoJ over loopback TCP +`+127.0.0.1:7700+`. Gateway `+:trusted_proxies+` = `+["127.0.0.1"]+`. +Trust source `+"header"+`. +* *Production:* gateway → BoJ over loopback Unix socket. The mTLS +boundary is *client → gateway* (the externally facing edge); the gateway +→ BoJ hop is loopback-isolated and does not itself require mTLS. The CA +/ cert-rotation decisions for the client→gateway mTLS boundary are Phase +B (B3). + +''''' + +=== 8. Surface drift caveat (input to A3) + +`+BojRest.Router+` today implements a *subset* of +`+docs/specification/openapi.yaml+` (`+/.well-known/boj-node-pubkey+`, +`+/health+`, `+/menu+`, `+/cartridges+`, `+/cartridge/:name+`, +`+/cartridge/:name/invoke+`; everything else falls to the `+match _+` → +404 catch-all). `+openapi.yaml+` declares a broader surface +(`+/status+`, `+/matrix+`, `+/graphql+`, cartridge +`+load|unload|reload+`, `+/grpc/*+`, `+/sse+`, `+/order+`, +`+/order-ticket+`, `+/umoja/*+`, `+/coprocessor/*+`, `+/sla/status+`, +`+/community/*+`, `+/sdp/status+`, `+/cartridges/ssg-mcp/webhook+`). + +*Contract decision:* the Verb Governance Spec governs the *declared* +surface (`+openapi.yaml+`), not only the currently-wired subset. +Declared-but- unimplemented routes are still classified in the policy so +that when the gnosis handler grows them they are governed from day one +rather than silently exposed. The gateway’s default-deny for any +_undeclared_ path remains the backstop. This resolves the plan’s +"`Surface drift`" risk for Phase A: the example policy is authored from +`+openapi.yaml+` cross-checked against `+router.ex+`, with every +divergence annotated in the policy `+narrative+` fields. diff --git a/docs/integration/http-capability-gateway-boj-contract.md b/docs/integration/http-capability-gateway-boj-contract.md deleted file mode 100644 index 3b253010..00000000 --- a/docs/integration/http-capability-gateway-boj-contract.md +++ /dev/null @@ -1,214 +0,0 @@ - - - -# http-capability-gateway ↔ BoJ — HTTP Contract - -**Version:** 1.0 -**Date:** 2026-05-18 -**Status:** Phase A deliverable A1 (normative for Phases B–E) -**ADR:** `docs/decisions/0004-adopt-http-capability-gateway.md` -**Plan:** `docs/integration/http-capability-gateway-plan.md` (§ Phase A, A1) -**Audit baseline:** `docs/integration/http-capability-gateway-audit.md` (2026-04-17) -**Tracking:** standards#91 (parent), standards#96 (Phase A) - -> **File-format note.** This document and its sibling -> (`http-capability-gateway-policy-authoring.md`) are kept as `.md`, not -> `.adoc`, deliberately: the integration plan normatively prescribes the path -> `docs/integration/http-capability-gateway-boj-contract.md`, the Phase C -> Trustfile `[SEAMS]` entry will reference that exact path, and the entire -> `docs/integration/` + `docs/decisions/` doc-set is Markdown. Faithfulness to -> the normatively-referenced path outweighs the estate `.adoc` default here. - ---- - -## 0. Scope - -This contract defines the **exact HTTP boundary** between the front component -(`hyperpolymath/http-capability-gateway`, tier 2) and the back component (BoJ's -unified Zig API gnosis handler, fronted by `BojRest.Router`, tier 3). It is the -authoritative seam description. No code is wired in Phase A; this document -governs Phases B–E and is the contract the Phase C seam test must satisfy. - -The gateway sits **between** Cloudflare edge (tier 1) and BoJ's gnosis handler. -It enforces verb governance, trust-level derivation, stealth, and audit logging -on every request **before** any BoJ cartridge logic runs. - ---- - -## 1. Transport - -| Aspect | Staging | Production | -|---|---|---| -| Gateway → BoJ transport | **TCP loopback** | **Unix domain socket** | -| Rationale | Simpler; matches the gateway's single `backend_url` config; no socket plumbing needed for first bring-up. | Avoids port allocation, removes the loopback TCP attack surface, and is the preferred path for co-located Podman containers sharing a pod network namespace. | - -**Decision (committed, per plan risk "Port / transport indecision"):** - -- **Staging:** gateway `BACKEND_URL=http://127.0.0.1:7700` → BoJ gnosis handler - listens on loopback `:7700` for gateway-forwarded traffic. The externally - visible port is owned by the gateway (8443 TLS / 8080 HTTP-behind-Tunnel); - BoJ's `:7700` is **not** externally routable. -- **Production:** gateway `BACKEND_URL=http://unix:/run/boj/gnosis.sock:/` (or - the gateway's documented Unix-socket backend form). BoJ binds the gnosis - handler to `/run/boj/gnosis.sock`, mode `0660`, owned by the shared pod user. - No TCP port is opened for the back side in production. - -The gateway proxy module currently supports a single `backend_url`; both forms -above are single-backend and within its current capability. Horizontal scaling -of BoJ (multi-backend) is explicitly **out of scope** for this contract and is -recorded as post-Phase-E work in the plan. - ---- - -## 2. Headers the gateway MUST set on forwarded requests - -| Header | Value | Status | -|---|---|---| -| `X-Forwarded-For` | original client IP | already implemented (Proxy module) | -| `X-Forwarded-Proto` | `https` / `http` | already implemented | -| `X-Forwarded-Host` | original `Host` | already implemented | -| `X-Gateway` | `http-capability-gateway` | already implemented | -| `X-Request-ID` | gateway `get_request_id/1` output | propagate; BoJ logs it | -| `X-Trust-Level` | `untrusted` \| `authenticated` \| `internal` | **see §3 — security-critical** | - -Notes: - -- `X-Request-ID` MUST be propagated end-to-end so the gateway audit record and - BoJ's structured log line share a correlation key. -- The gateway's resolved trust vocabulary is `untrusted | authenticated | - internal`. BoJ's `BojRest.TrustPolicy` consumes `internal | authenticated` - as "credentialed" and treats `untrusted`/absent as "public only" (see - `BojRest.Router` moduledoc and `docs/AUTH-DESIGN.adoc`). The gateway's - `untrusted` maps to BoJ's "public" tier. This vocabulary mapping is fixed by - this contract: - - | Gateway `X-Trust-Level` | BoJ trust class | - |---|---| - | `internal` | internal (credentialed; lifecycle/admin) | - | `authenticated` | authenticated (credentialed) | - | `untrusted` / absent | public (only `auth.method: "none"` cartridges) | - ---- - -## 3. Trust-header security invariant (load-bearing) - -`BojRest.Router` **bypasses trust enforcement entirely for loopback callers** -(`loopback?(conn.remote_ip)` → enforcement skipped; local dev / mcp-bridge -path). In staging the gateway forwards from `127.0.0.1`; in production it -forwards over a loopback Unix socket. **Therefore BoJ will treat the gateway as -a fully trusted proxy and accept its `X-Trust-Level` verbatim.** - -This makes the following invariant **mandatory, not advisory**: - -1. The gateway MUST **strip any inbound `X-Trust-Level`** from the original - client request unless the immediate peer is in the gateway's - `:trusted_proxies` allow-list. A client-supplied `X-Trust-Level` MUST NEVER - survive into the forwarded request. -2. The gateway MUST **re-set `X-Trust-Level`** from its own resolved trust - decision (header-based pre-Phase-B; mTLS client-cert-derived from Phase B - onward) before forwarding. -3. BoJ's gnosis handler / `BojRest.Router` MUST treat `X-Trust-Level` as - authoritative **only** when the connection originates from the gateway - (loopback `127.0.0.1` in staging; the loopback Unix socket in production). - Any `X-Trust-Level` arriving from any other source MUST be ignored and - treated as `untrusted`. -4. The back-side bind (BoJ `:7700` / `/run/boj/gnosis.sock`) MUST NOT be - externally routable. If it were, an attacker reaching it directly would - inherit loopback trust-bypass. Network/socket isolation of the back side is - part of this contract, not an operational nicety. - -Failure to honour (1)–(4) is a trust-forgery vulnerability: a forged -`X-Trust-Level: internal` from an external client would reach a loopback-trusting -router. This invariant is tested in Phase C (plan §Phase C, "Trust header -forwarding security invariant") and hardened in Phase B (mTLS becomes the -trust source so the resolved value cannot be header-forged at all). - -**Implementation status (Phase C):** - -- Gateway-side (1)+(2): `http-capability-gateway/lib/http_capability_gateway/proxy.ex` - `build_backend_headers/1` strips client-supplied `X-Trust-Level` + `X-Request-ID` - and re-emits the gateway-resolved values. Landed in http-capability-gateway#11. -- BoJ-side (3): `boj_rest/lib/boj_rest/trust_policy.ex` `satisfies?/3` ignores - the header for any non-loopback caller via the - `satisfies?(_required, _trust, false), do: false` clause — defence in depth - against (4) being violated by a misconfiguration. Verified by - `elixir/test/phase_c_seam_test.exs` (9 tests, all live). -- (4) is an operational/configuration invariant, not a code change; verified - during Phase E rollout (production deployments MUST front BoJ with a - loopback-only socket per the §Phase E runbook). - ---- - -## 4. Headers BoJ MUST NOT forward to cartridges unchanged - -- `X-Trust-Level` MUST be consumed and re-validated by `BojRest.TrustPolicy` at - the router boundary and MUST NOT be passed opaquely into cartridge invocation - arguments. Cartridges receive a *decision* (allowed/denied + trust class), - never the raw transport header, so a cartridge cannot be tricked by a header - it was never meant to interpret. -- `X-Node-Identity`, when present, is **logged for audit** by the router but is - not an authorization input and MUST NOT be promoted to a trust signal. - ---- - -## 5. Error semantics - -| Condition | Gateway response | Notes | -|---|---|---| -| BoJ returns 5xx (e.g. `invocation-failed` 500) | **502 Bad Gateway** | existing gateway behaviour; the upstream 500 body is not leaked verbatim | -| BoJ unreachable / connection refused | circuit breaker trips → **503 Service Unavailable** | fail-closed; matches Trustfile `[SEAMS]` `failure_mode: "fail-closed (circuit breaker)"` | -| Verb not permitted by policy | **403** (or stealth status, see §6) | decision made at gateway; request never reaches BoJ | -| Path not in policy | **default-deny** (403 / stealth) | unlisted path ⇒ denied; request never reaches BoJ | -| BoJ returns 4xx (e.g. `unknown-cartridge` 404, `forbidden` 403) | **passed through unchanged** | these are legitimate application responses, not gateway faults | - -The circuit-breaker fail-closed posture means a gateway↔BoJ seam failure denies -traffic rather than failing open — consistent with the ADR's "fail-closed" -stance and the rate-limit architecture's defence-in-depth intent. - ---- - -## 6. Stealth - -Routes declared `exposure: "internal"` with a stealth profile return the -configured stealth status (default **404**) instead of **403** to callers below -the required trust level, hiding capability existence. The example policy -(`config/gateway-policy-boj-example.yaml`) enables `stealth: { enabled: true, -status_code: 404 }` and applies it to lifecycle/admin/SDP routes (§ A3). - ---- - -## 7. Deployment trust topology (input to Phase B) - -Phase B (mTLS) needs the trusted-proxy / loopback topology fixed here: - -- **Staging:** gateway and BoJ co-located; gateway → BoJ over loopback TCP - `127.0.0.1:7700`. Gateway `:trusted_proxies` = `["127.0.0.1"]`. Trust source - `"header"`. -- **Production:** gateway → BoJ over loopback Unix socket. The mTLS boundary is - **client → gateway** (the externally facing edge); the gateway → BoJ hop is - loopback-isolated and does not itself require mTLS. The CA / cert-rotation - decisions for the client→gateway mTLS boundary are Phase B (B3). - ---- - -## 8. Surface drift caveat (input to A3) - -`BojRest.Router` today implements a **subset** of `docs/specification/openapi.yaml` -(`/.well-known/boj-node-pubkey`, `/health`, `/menu`, `/cartridges`, -`/cartridge/:name`, `/cartridge/:name/invoke`; everything else falls to the -`match _` → 404 catch-all). `openapi.yaml` declares a broader surface -(`/status`, `/matrix`, `/graphql`, cartridge `load|unload|reload`, `/grpc/*`, -`/sse`, `/order`, `/order-ticket`, `/umoja/*`, `/coprocessor/*`, `/sla/status`, -`/community/*`, `/sdp/status`, `/cartridges/ssg-mcp/webhook`). - -**Contract decision:** the Verb Governance Spec governs the **declared** -surface (`openapi.yaml`), not only the currently-wired subset. Declared-but- -unimplemented routes are still classified in the policy so that when the gnosis -handler grows them they are governed from day one rather than silently exposed. -The gateway's default-deny for any *undeclared* path remains the backstop. This -resolves the plan's "Surface drift" risk for Phase A: the example policy is -authored from `openapi.yaml` cross-checked against `router.ex`, with every -divergence annotated in the policy `narrative` fields. diff --git a/docs/integration/http-capability-gateway-plan.adoc b/docs/integration/http-capability-gateway-plan.adoc new file mode 100644 index 00000000..9cf29cab --- /dev/null +++ b/docs/integration/http-capability-gateway-plan.adoc @@ -0,0 +1,500 @@ +== http-capability-gateway — BoJ Integration Plan + +*Version:* 1.0 + +*Date:* 2026-04-17 + +*Status:* Active (Phase 0 complete — audit + plan landed) + +*Companion audit:* +`+docs/integration/http-capability-gateway-audit.md+` + +*ADR:* `+docs/decisions/0004-adopt-http-capability-gateway.md+` + +*Timeline:* ~8–12 weeks total across Phases A–E. + +''''' + +=== Overview + +This document is the authoritative integration plan for wiring +`+http-capability-gateway+` into BoJ as tier-2 of the rate-limit + +capability-enforcement architecture. It is normative: each phase has +declared deliverables, acceptance criteria, risks, and blocking +relationships. Phases are ordered by dependency; Phases A and B may +overlap once the Phase A contract spec is stable. + +The gateway sits between the Cloudflare edge (tier 1) and BoJ’s unified +Zig API gnosis handler (tier 3 and below). It adds declarative verb +governance, trust-level enforcement, stealth profiles, and structured +audit logging to the HTTP surface without modifying any BoJ cartridge +logic. + +*Current HTTP surface reference:* `+docs/specification/openapi.yaml+`. + +*BoJ gnosis handler entry point:* `+uapi_gnosis_set_handler+` in the +unified-zig-api stack (commits `+9c807c0+`, `+d765345+` — single-port +consolidation). + +''''' + +=== Phase A — Contract Definition (weeks 1–2) + +==== Objective + +Define the exact HTTP contract between the gateway (front) and BoJ’s +unified Zig API gnosis handler (back), and establish the Verb Governance +Spec authoring workflow. Nothing is wired in this phase; the output is +specification documents and an example policy file. + +==== Deliverables + +*A1 — Gateway↔BoJ HTTP contract document* + +File: `+docs/integration/http-capability-gateway-boj-contract.md+` + +Must specify: - Transport: whether the gateway forwards to BoJ via TCP +localhost or Unix socket. Decision rationale: TCP localhost is simpler +and matches the gateway’s single `+backend_url+` config; Unix socket +avoids port allocation and is preferred for co-located Podman +containers. Recommend TCP localhost for staging, Unix socket for +production. - Port allocation: if TCP, which port does BoJ’s gnosis +handler listen on for gateway-forwarded traffic (separate from the +externally visible port, or same port with gateway sitting in front). - +Headers the gateway MUST set on forwarded requests: - +`+X-Forwarded-For+` (already implemented in Proxy module). - +`+X-Forwarded-Proto+`, `+X-Forwarded-Host+`, +`+X-Gateway: http-capability-gateway+` (already implemented). - +`+X-Trust-Level: {authenticated|internal|untrusted}+` — trust level as +resolved by the gateway, stripped from the original request and re-set +from the compiled trust value. BoJ’s gnosis handler MUST accept this +header from `+127.0.0.1+` (trusted proxy). - `+X-Request-ID+` — +propagated from the gateway’s `+get_request_id/1+` output. - Headers +BoJ’s gnosis handler MUST NOT forward to cartridges unchanged: - +`+X-Trust-Level+` must be re-validated or stripped before reaching +cartridge logic. - Error semantics: if BoJ returns 500, gateway returns +502 (existing behaviour). If BoJ is not reachable, circuit breaker trips +and gateway returns 503. + +*A2 — Verb Governance Spec authoring workflow* + +Answers: - Where does the Verb Governance Spec YAML live? Options: 1. In +the `+boj-server+` repo at `+config/gateway-policy.yaml+`, loaded at +gateway startup from a mounted path. 2. As a separate file in +`+web-ecosystem/http-capability-gateway/config/boj-policy.yaml+`. +Recommendation: option 1 — the policy describes BoJ’s HTTP surface, so +it belongs in the BoJ repo and is version-controlled alongside +`+docs/specification/openapi.yaml+`. - Who writes it? BoJ maintainer, +reviewed like any spec change. - How is it loaded at deploy? Via +`+POLICY_PATH+` environment variable in the gateway container; the file +is mounted from a ConfigMap or bind-mount at that path. - Hot-reload: +the gateway’s atomic swap pattern supports SIGHUP-triggered reload. +Document the reload trigger mechanism (k9-svc rolling deploy, or a +separate `+gateway-reload+` signal). + +*A3 — Example Verb Governance Spec for BoJ* + +File: `+config/gateway-policy-boj-example.yaml+` (in boj-server repo) + +Derived from `+docs/specification/openapi.yaml+`. Must cover: - +`+/health+` → GET, public. - `+/ready+` (if BoJ exposes it) → GET, +public. - `+/cartridges+` → GET (authenticated), POST (internal). - +`+/cartridges/{id}+` → GET (authenticated), DELETE (internal). - +`+/cartridges/{id}/invoke+` → POST (authenticated). - `+/admin+` and +sub-paths → GET, internal only, stealth: 404. - GraphQL port and gRPC +port paths (if gateway is also placed in front of those surfaces). + +==== Acceptance Criteria + +* Contract document exists and is reviewed. +* Verb Governance Spec workflow is documented. +* Example policy file passes `+PolicyLoader.load_policy/1+` + +`+PolicyValidator.validate/1+` when run against the gateway (manual +verification). +* No code changes to gateway or BoJ gnosis handler. + +==== Risks + +* *Port / transport indecision:* If TCP vs. Unix socket is not decided +in Phase A, Phase E deployment will be blocked. Decide in A1 and commit. +* *Surface drift:* `+openapi.yaml+` may not reflect actual gnosis +handler routes (it was accurate at audit time but may lag). Cross-check +against the Zig API source before authoring the example policy. + +==== Blocks + +Phase B (mTLS) requires the Phase A contract to know what the +trusted-proxy IP list looks like in deployment (loopback, container +network, etc.). + +''''' + +=== Phase B — mTLS Primary Path (weeks 3–5) + +==== Objective + +Move the gateway’s trust-level extraction from header-based +(`+X-Trust-Level+`) to mTLS client certificate validation. The header +path remains available for development; mTLS becomes the production +path. The gateway SHOULD reject non-mutual-TLS traffic (or demote it to +`+untrusted+`) at the transport layer. + +==== Deliverables + +*B1 — Cowboy TLS configuration with `+verify: :verify_peer+`* + +The gateway’s `+application.ex+` / `+config/prod.exs+` must configure +Cowboy TLS with: + +[source,elixir] +---- +{:tls_options, [ + verify: :verify_peer, + fail_if_no_peer_cert: true, + cacertfile: System.get_env("MTLS_CA_CERT_PATH"), + certfile: System.get_env("GATEWAY_CERT_PATH"), + keyfile: System.get_env("GATEWAY_KEY_PATH") +]} +---- + +*B2 — `+is_cert_verified/1+` reads actual TLS state* + +The current stub (`+is_cert_verified/1+` returns `+true+` if a cert is +present) must be replaced with a function that reads the actual peer +verification result from Cowboy. In Cowboy 2.x: +`+cowboy_req:peercert/1+` combined with +`+:ssl.connection_information/2+` or checking the verify result stored +in the SSL socket. The exact mechanism must be confirmed against Cowboy +2.7 API documentation. + +*B3 — CA selection and cert rotation policy* + +Decide whether the mTLS CA is: 1. BoJ’s own CA (generated at deploy +time, self-signed root). 2. The estate’s SDP CA (if one exists). 3. +Cloudflare Origin CA (for authenticated origin pull parity). + +Authenticated Origin Pulls parity: the gateway SHOULD be configured to +reject connections that do not present a valid client cert from the +chosen CA. This mirrors the Cloudflare AOP model at the gateway level. + +Cert rotation runbook: documented in +`+docs/integration/mtls-rotation-runbook.md+`. Runbook must cover: cert +generation, distribution to gateway and BoJ containers, hot-reload +without downtime. + +*B4 — Idris2 proof obligation recorded* + +File: `+src/abi/+` or `+PROOFS_NEEDED.md+` update. + +The mTLS policy decision (cert chain rooted in chosen CA → "`internal`" +trust) must have a proof obligation recorded. The proof does not have to +land in Phase B, but: - The claim must be stated in Idris2 terms. - The +proof file path must be declared (e.g., +`+src/abi/Trust.MTLSPolicy.idr+`). - The proof is listed in +`+PROOFS_NEEDED.md+` with status "`pending Phase C/D`". + +==== Acceptance Criteria + +* Gateway compiled and tested with Cowboy `+verify: :verify_peer+`. +* `+is_cert_verified/1+` reads real TLS verification state (not just +cert presence). +* `+test/security_test.exs+` includes a test using a real test CA +fixture (not a real production CA — a self-signed test CA generated with +`+openssl req+`). +* Gateway refuses connections with no client cert when +`+:trust_level_source+` is `+"mtls"+`. (Tests this: send a request with +no cert; verify response is 403 or connection refused, depending on the +`+fail_if_no_peer_cert+` config.) +* Idris2 proof obligation for mTLS policy recorded. +* Cert rotation runbook written. + +==== Risks + +* *Cowboy 2.7 API changes:* The mTLS peer-verify API may differ from +earlier Cowboy versions. Verify against `+plug_cowboy ~> 2.7+` docs +before coding. +* *Test fixture complexity:* Generating test CA + client certs in +`+mix test+` is non-trivial. Consider using `+:public_key.pkix_sign/2+` +to generate in-memory certs for unit tests, and a shell script for +integration test fixtures. +* *SDP CA dependency:* If using the estate SDP CA, the CA must exist +before Phase B can start. If no SDP CA exists, create BoJ’s own CA in +this phase. + +==== Blocks + +Phase C (E2E tests) depends on Phase B mTLS being operational (E2E tests +should exercise both the header path and the mTLS path). + +''''' + +=== Phase C — End-to-End Verification (weeks 5–7) + +==== Objective + +Write end-to-end tests that prove the complete pipeline: Verb Governance +Spec file → compiled rules → gateway enforces → BoJ gnosis handler +receives only allowed traffic. Seam test: the gateway ↔ BoJ boundary +must be exercised with a contract matching the Phase A contract +document. + +==== Deliverables + +*C1 — E2E test suite for gateway ↔ BoJ seam* + +File: `+test/e2e_boj_integration_test.exs+` (in the gateway repo) + +or + +File: `+tests/seam/gateway-boj.test+` (in boj-server repo, matching the +SEAMS-SPEC format) + +Must cover: - A request that matches a `+public+` rule → BoJ receives +the request with `+X-Trust-Level: untrusted+`, responds, gateway returns +the response. - A request that matches an `+authenticated+` rule with +`+X-Trust-Level: authenticated+` (header path) → allowed. - A request +that matches an `+authenticated+` rule with `+X-Trust-Level: untrusted+` +→ gateway returns 403 (or stealth response). - A request that matches an +`+internal+` rule with no cert / wrong trust → denied. - A verb not in +the policy → denied. - A path not in the policy → default-deny. - Policy +hot-reload: load policy A, verify enforcement, reload policy B, verify +new enforcement without dropped requests. + +*C2 — Property tests for the pipeline* + +File: `+test/e2e_property_test.exs+` (gateway repo) + +StreamData properties: - For any policy that loads and validates +successfully, `+compile/2+` succeeds and `+lookup/3+` never returns +`+{:ok, rule}+` for a verb not declared in the policy. - For any +path+verb denied under trust level T, it is also denied under any T’ < T +(monotonicity of denial preserved through the full pipeline). + +*C3 — Seam declaration in BoJ Trustfile* + +Add to `+[SEAMS]+` in `+Trustfile.a2ml+`: + +[source,yaml] +---- +- id: "gateway-boj-gnosis" + description: "http-capability-gateway ↔ BoJ unified Zig API gnosis handler" + from: "web-ecosystem/http-capability-gateway" + to: "src/zig-api/ (gnosis handler)" + contract: "docs/integration/http-capability-gateway-boj-contract.md" + test_ref: "tests/seam/gateway-boj.test" + failure_mode: "fail-closed (circuit breaker)" + tier: "elixir-disciplined" +---- + +==== Acceptance Criteria + +* E2E tests pass in CI (not just locally). +* Coverage threshold: all six cases in C1 have passing tests. +* Seam test (C1) exercises the HTTP contract from Phase A: headers set +correctly, trust level forwarded, response returned to caller. +* Property tests (C2) pass with at least 200 samples. +* Seam declared in Trustfile. + +==== Risks + +* *Test environment isolation:* E2E tests require a running BoJ +instance. Options: +[arabic] +. Mock backend (easiest — Bypass module in the gateway test suite). +. Real BoJ in CI (preferred for seam test — more complex setup). Use +mock backend for C1/C2 in Phase C; schedule real BoJ seam test for Phase +E. +* *Policy hot-reload under load:* The existing `+concurrency_test.exs+` +covers gateway-internal reload. The BoJ seam test must also verify that +in-flight requests are not dropped during a reload. This is the most +complex test to write. + +==== Blocks + +Phase D benchmarks depend on Phase C tests establishing a baseline +measurement environment. + +''''' + +=== Phase D — Benchmarks (weeks 7–8) + +==== Objective + +Formalise the "`fast policy enforcement`" claim with published latency +numbers. Define the load profile, run the benchmark suite, and configure +a regression alert so that future changes that degrade gateway +performance are caught in CI. + +==== Deliverables + +*D1 — Load profile declaration* + +Document in `+docs/integration/gateway-load-profile.md+`: - Baseline: +requests/second the BoJ gnosis handler receives in production today. - +Gateway load profile: same request rate + gateway overhead should be < +declared limit. - Measurement environment: hardware spec, Elixir/OTP +version, policy size (number of rules). + +*D2 — Benchmark results published* + +The existing `+test/benchmark_test.exs+` covers: - Rate limiter +throughput. - Circuit breaker state transition cost. - Exact vs. regex +vs. global-fallback route lookup. + +Extend benchmarks to measure: - *Median / p95 / p99 request-to-response +latency* through the full Plug pipeline (security headers → strip → +extract trust → rate limit → policy lookup → proxy response) under the +declared load profile. - Comparison baseline: latency of the same +request hitting BoJ directly (no gateway). - *Gateway overhead* = +gateway median latency − baseline median latency. + +Target: gateway overhead < 2ms median, < 5ms p99 for a policy with 100 +rules (indicative — revisit once baseline is measured). + +*D3 — Regression alert in CI* + +Add a CI step that runs the benchmark suite and fails if the measured +p99 latency exceeds a declared threshold (initially generous — tighten +after a few builds establish a stable baseline). The benchmark step +should be gated (not run on every PR; run on merge to main and weekly +schedule). + +==== Acceptance Criteria + +* Load profile document exists. +* Median / p95 / p99 latency numbers published in +`+docs/integration/gateway-benchmarks.md+`. +* Gateway overhead number published (vs. direct BoJ, no gateway). +* CI regression step configured. +* Benchmark results do NOT fabricate numbers — they come from a +`+mix bench+` or `+mix test --only benchmark+` run against the real +hardware or a representative CI runner. + +==== Risks + +* *CI hardware variability:* Benchmark numbers from GitHub Actions +runners vary significantly run-to-run. Use relative measurements +(gateway overhead fraction) rather than absolute numbers for regression +detection. +* *Policy size sensitivity:* A policy with 5 rules will perform +differently than one with 500 rules. The benchmark must test the +realistic BoJ policy size. + +==== Blocks + +Phase E production deployment requires D3 (regression alert) to be in +place before rollout. + +''''' + +=== Phase E — Production Wiring (weeks 8–12) + +==== Objective + +Deploy http-capability-gateway in front of BoJ in staging, verify all +telemetry and logging are correct, then roll out to production. +Demonstrate end-to-end traffic handling under real load. + +==== Deliverables + +*E1 — Containerfile and deployment spec for the gateway* + +The gateway already has a `+Containerfile+`. Confirm it: - Uses a +Chainguard base image. - Accepts `+POLICY_PATH+`, `+BACKEND_URL+`, +`+PORT+`, `+MTLS_CA_CERT_PATH+` env vars. - Is built with Podman +(rootless). - Is signed as a `+.ctp+` bundle via cerro-torre. + +Add a k9-svc deployment spec at `+container/gateway-deploy.k9.ncl+` in +the gateway repo. + +*E2 — Staging deployment* + +Deploy the gateway in front of a BoJ staging instance: - Gateway listens +on port 8443 (TLS) or 8080 (HTTP, behind Cloudflare Tunnel). - Backend: +BoJ staging gnosis handler at `+http://localhost:7700+` or via Unix +socket. - Policy: `+config/gateway-policy-boj-example.yaml+` from Phase +A. - Trust level source: `+"header"+` initially, switched to `+"mtls"+` +after Phase B cert rotation runbook is exercised. + +*E3 — Telemetry verification* + +With staging traffic running: - Confirm +`+[:http_capability_gateway, :access_decision]+` events appear in the +Prometheus scrape at `+/metrics+`. - Confirm `+GET /api/v1/minikaran+` +returns `+status.status: "active"+` after the learning phase. - Confirm +structured JSON logs appear in the log stream with `+request_id+` +fields. - Confirm `+X-Trust-Level+` header is set correctly on forwarded +requests as seen by BoJ (read from BoJ access logs). + +*E4 — Production rollout* + +Roll out behind a feature flag or behind a percentage split (10% → 50% → +100% of traffic directed through the gateway). + +*E5 — Rollback runbook* + +File: `+docs/integration/gateway-rollback-runbook.md+` + +Covers: - How to detect that the gateway is causing problems (elevated +502/503 rates, increased latency at p99). - How to bypass the gateway +(re-route traffic directly to BoJ gnosis handler) without downtime +(requires traffic management — Cloudflare Tunnel rule, or container +orchestration). - How to disable the gateway permanently (remove the +k9-svc deployment, update tier-2 config to `+status: DISABLED+`). + +==== Acceptance Criteria + +* Gateway handles a declared traffic profile in staging for at least 24 +hours without circuit breaker trips or elevated error rates. +* Telemetry verified (E3 complete). +* Rollback runbook exists and has been tested (manually walk through E5 +once in staging before production rollout). +* Production rollout complete with no SLA regression (latency at p99 +within D2 declared overhead + 20% safety margin). +* `+[HTTP_CAPABILITY_GATEWAY]+` section in BoJ Trustfile updated to +`+status: "DEPLOYED"+` with `+deployed_at+` timestamp. + +==== Risks + +* *Cowboy TLS in production:* First time running the gateway with mTLS +in production. Cert expiry, CA rotation, and Cowboy TLS configuration +errors are all possible. The rollback runbook must be tested before +production. +* *Backend URL single-backend limitation:* The proxy module supports one +`+backend_url+`. If BoJ is horizontally scaled, the gateway must be +placed behind a load balancer or the proxy module must be extended (out +of scope for this plan but noted for future work). +* *Policy size in production:* The Phase A example policy covers known +routes. Undocumented routes (gnosis handler internals, health probes, +metrics) must be added to the policy before go-live or they will be +default-denied. + +''''' + +=== Cross-Phase Notes + +==== Verb Governance Spec versioning + +The policy file is a BoJ-repo artefact. It must be reviewed like any +spec change (PR required, approved by maintainer). When the OpenAPI spec +changes, the policy file must be updated in the same PR or the PR must +document why the policy is unchanged (e.g., the new route is not +externally accessible). + +==== Trust header forwarding security invariant + +The gateway strips `+X-Trust-Level+` from incoming requests unless the +sender is in `+:trusted_proxies+`. After Phase B (mTLS), the gateway +sets `+X-Trust-Level+` from the resolved mTLS trust level before +forwarding to BoJ. BoJ’s gnosis handler must treat `+X-Trust-Level+` +from `+127.0.0.1+` (or the gateway container IP) as authoritative and +from any other source as untrusted. This invariant must be documented in +the Phase A contract and tested in Phase C. + +==== Groove protocol + +When Groove protocol is adopted estate-wide for inter-service +communication, the gateway ↔ BoJ forwarding path should be considered +for Groove wrapping. This is post-Phase E work. + +==== VeriSimDB audit trail + +If `+VeriSimDB+` in the gateway is confirmed as a working integration +(not a stub), the audit trail (allow/deny decisions with path, verb, +trust level, backend, rule name, duration) will flow into the +per-project VeriSimDB instance. This is valuable for BoJ’s audit +posture. Confirm VeriSimDB integration status before Phase E. diff --git a/docs/integration/http-capability-gateway-plan.md b/docs/integration/http-capability-gateway-plan.md deleted file mode 100644 index b04f84df..00000000 --- a/docs/integration/http-capability-gateway-plan.md +++ /dev/null @@ -1,475 +0,0 @@ - - - -# http-capability-gateway — BoJ Integration Plan - -**Version:** 1.0 -**Date:** 2026-04-17 -**Status:** Active (Phase 0 complete — audit + plan landed) -**Companion audit:** `docs/integration/http-capability-gateway-audit.md` -**ADR:** `docs/decisions/0004-adopt-http-capability-gateway.md` -**Timeline:** ~8–12 weeks total across Phases A–E. - ---- - -## Overview - -This document is the authoritative integration plan for wiring `http-capability-gateway` -into BoJ as tier-2 of the rate-limit + capability-enforcement architecture. It is -normative: each phase has declared deliverables, acceptance criteria, risks, and -blocking relationships. Phases are ordered by dependency; Phases A and B may overlap -once the Phase A contract spec is stable. - -The gateway sits between the Cloudflare edge (tier 1) and BoJ's unified Zig API -gnosis handler (tier 3 and below). It adds declarative verb governance, trust-level -enforcement, stealth profiles, and structured audit logging to the HTTP surface without -modifying any BoJ cartridge logic. - -**Current HTTP surface reference:** `docs/specification/openapi.yaml`. -**BoJ gnosis handler entry point:** `uapi_gnosis_set_handler` in the unified-zig-api -stack (commits `9c807c0`, `d765345` — single-port consolidation). - ---- - -## Phase A — Contract Definition (weeks 1–2) - -### Objective - -Define the exact HTTP contract between the gateway (front) and BoJ's unified Zig API -gnosis handler (back), and establish the Verb Governance Spec authoring workflow. -Nothing is wired in this phase; the output is specification documents and an example -policy file. - -### Deliverables - -**A1 — Gateway↔BoJ HTTP contract document** - -File: `docs/integration/http-capability-gateway-boj-contract.md` - -Must specify: -- Transport: whether the gateway forwards to BoJ via TCP localhost or Unix socket. - Decision rationale: TCP localhost is simpler and matches the gateway's single - `backend_url` config; Unix socket avoids port allocation and is preferred for - co-located Podman containers. Recommend TCP localhost for staging, Unix socket - for production. -- Port allocation: if TCP, which port does BoJ's gnosis handler listen on for - gateway-forwarded traffic (separate from the externally visible port, or same port - with gateway sitting in front). -- Headers the gateway MUST set on forwarded requests: - - `X-Forwarded-For` (already implemented in Proxy module). - - `X-Forwarded-Proto`, `X-Forwarded-Host`, `X-Gateway: http-capability-gateway` - (already implemented). - - `X-Trust-Level: {authenticated|internal|untrusted}` — trust level as resolved - by the gateway, stripped from the original request and re-set from the compiled - trust value. BoJ's gnosis handler MUST accept this header from `127.0.0.1` - (trusted proxy). - - `X-Request-ID` — propagated from the gateway's `get_request_id/1` output. -- Headers BoJ's gnosis handler MUST NOT forward to cartridges unchanged: - - `X-Trust-Level` must be re-validated or stripped before reaching cartridge logic. -- Error semantics: if BoJ returns 500, gateway returns 502 (existing behaviour). - If BoJ is not reachable, circuit breaker trips and gateway returns 503. - -**A2 — Verb Governance Spec authoring workflow** - -Answers: -- Where does the Verb Governance Spec YAML live? Options: - 1. In the `boj-server` repo at `config/gateway-policy.yaml`, loaded at gateway - startup from a mounted path. - 2. As a separate file in `web-ecosystem/http-capability-gateway/config/boj-policy.yaml`. - Recommendation: option 1 — the policy describes BoJ's HTTP surface, so it belongs - in the BoJ repo and is version-controlled alongside `docs/specification/openapi.yaml`. -- Who writes it? BoJ maintainer, reviewed like any spec change. -- How is it loaded at deploy? Via `POLICY_PATH` environment variable in the gateway - container; the file is mounted from a ConfigMap or bind-mount at that path. -- Hot-reload: the gateway's atomic swap pattern supports SIGHUP-triggered reload. - Document the reload trigger mechanism (k9-svc rolling deploy, or a separate - `gateway-reload` signal). - -**A3 — Example Verb Governance Spec for BoJ** - -File: `config/gateway-policy-boj-example.yaml` (in boj-server repo) - -Derived from `docs/specification/openapi.yaml`. Must cover: -- `/health` → GET, public. -- `/ready` (if BoJ exposes it) → GET, public. -- `/cartridges` → GET (authenticated), POST (internal). -- `/cartridges/{id}` → GET (authenticated), DELETE (internal). -- `/cartridges/{id}/invoke` → POST (authenticated). -- `/admin` and sub-paths → GET, internal only, stealth: 404. -- GraphQL port and gRPC port paths (if gateway is also placed in front of those surfaces). - -### Acceptance Criteria - -- Contract document exists and is reviewed. -- Verb Governance Spec workflow is documented. -- Example policy file passes `PolicyLoader.load_policy/1` + `PolicyValidator.validate/1` - when run against the gateway (manual verification). -- No code changes to gateway or BoJ gnosis handler. - -### Risks - -- **Port / transport indecision:** If TCP vs. Unix socket is not decided in Phase A, - Phase E deployment will be blocked. Decide in A1 and commit. -- **Surface drift:** `openapi.yaml` may not reflect actual gnosis handler routes - (it was accurate at audit time but may lag). Cross-check against the Zig API - source before authoring the example policy. - -### Blocks - -Phase B (mTLS) requires the Phase A contract to know what the trusted-proxy IP -list looks like in deployment (loopback, container network, etc.). - ---- - -## Phase B — mTLS Primary Path (weeks 3–5) - -### Objective - -Move the gateway's trust-level extraction from header-based (`X-Trust-Level`) to -mTLS client certificate validation. The header path remains available for development; -mTLS becomes the production path. The gateway SHOULD reject non-mutual-TLS traffic -(or demote it to `untrusted`) at the transport layer. - -### Deliverables - -**B1 — Cowboy TLS configuration with `verify: :verify_peer`** - -The gateway's `application.ex` / `config/prod.exs` must configure Cowboy TLS with: -```elixir -{:tls_options, [ - verify: :verify_peer, - fail_if_no_peer_cert: true, - cacertfile: System.get_env("MTLS_CA_CERT_PATH"), - certfile: System.get_env("GATEWAY_CERT_PATH"), - keyfile: System.get_env("GATEWAY_KEY_PATH") -]} -``` - -**B2 — `is_cert_verified/1` reads actual TLS state** - -The current stub (`is_cert_verified/1` returns `true` if a cert is present) must be -replaced with a function that reads the actual peer verification result from Cowboy. -In Cowboy 2.x: `cowboy_req:peercert/1` combined with `:ssl.connection_information/2` -or checking the verify result stored in the SSL socket. The exact mechanism must be -confirmed against Cowboy 2.7 API documentation. - -**B3 — CA selection and cert rotation policy** - -Decide whether the mTLS CA is: -1. BoJ's own CA (generated at deploy time, self-signed root). -2. The estate's SDP CA (if one exists). -3. Cloudflare Origin CA (for authenticated origin pull parity). - -Authenticated Origin Pulls parity: the gateway SHOULD be configured to reject -connections that do not present a valid client cert from the chosen CA. This mirrors -the Cloudflare AOP model at the gateway level. - -Cert rotation runbook: documented in `docs/integration/mtls-rotation-runbook.md`. -Runbook must cover: cert generation, distribution to gateway and BoJ containers, -hot-reload without downtime. - -**B4 — Idris2 proof obligation recorded** - -File: `src/abi/` or `PROOFS_NEEDED.md` update. - -The mTLS policy decision (cert chain rooted in chosen CA → "internal" trust) must -have a proof obligation recorded. The proof does not have to land in Phase B, but: -- The claim must be stated in Idris2 terms. -- The proof file path must be declared (e.g., `src/abi/Trust.MTLSPolicy.idr`). -- The proof is listed in `PROOFS_NEEDED.md` with status "pending Phase C/D". - -### Acceptance Criteria - -- Gateway compiled and tested with Cowboy `verify: :verify_peer`. -- `is_cert_verified/1` reads real TLS verification state (not just cert presence). -- `test/security_test.exs` includes a test using a real test CA fixture (not a - real production CA — a self-signed test CA generated with `openssl req`). -- Gateway refuses connections with no client cert when `:trust_level_source` is `"mtls"`. - (Tests this: send a request with no cert; verify response is 403 or connection - refused, depending on the `fail_if_no_peer_cert` config.) -- Idris2 proof obligation for mTLS policy recorded. -- Cert rotation runbook written. - -### Risks - -- **Cowboy 2.7 API changes:** The mTLS peer-verify API may differ from earlier - Cowboy versions. Verify against `plug_cowboy ~> 2.7` docs before coding. -- **Test fixture complexity:** Generating test CA + client certs in `mix test` is - non-trivial. Consider using `:public_key.pkix_sign/2` to generate in-memory certs - for unit tests, and a shell script for integration test fixtures. -- **SDP CA dependency:** If using the estate SDP CA, the CA must exist before Phase B - can start. If no SDP CA exists, create BoJ's own CA in this phase. - -### Blocks - -Phase C (E2E tests) depends on Phase B mTLS being operational (E2E tests should -exercise both the header path and the mTLS path). - ---- - -## Phase C — End-to-End Verification (weeks 5–7) - -### Objective - -Write end-to-end tests that prove the complete pipeline: Verb Governance Spec file -→ compiled rules → gateway enforces → BoJ gnosis handler receives only allowed traffic. -Seam test: the gateway ↔ BoJ boundary must be exercised with a contract matching the -Phase A contract document. - -### Deliverables - -**C1 — E2E test suite for gateway ↔ BoJ seam** - -File: `test/e2e_boj_integration_test.exs` (in the gateway repo) - -or - -File: `tests/seam/gateway-boj.test` (in boj-server repo, matching the SEAMS-SPEC format) - -Must cover: -- A request that matches a `public` rule → BoJ receives the request with - `X-Trust-Level: untrusted`, responds, gateway returns the response. -- A request that matches an `authenticated` rule with `X-Trust-Level: authenticated` - (header path) → allowed. -- A request that matches an `authenticated` rule with `X-Trust-Level: untrusted` - → gateway returns 403 (or stealth response). -- A request that matches an `internal` rule with no cert / wrong trust → denied. -- A verb not in the policy → denied. -- A path not in the policy → default-deny. -- Policy hot-reload: load policy A, verify enforcement, reload policy B, verify new - enforcement without dropped requests. - -**C2 — Property tests for the pipeline** - -File: `test/e2e_property_test.exs` (gateway repo) - -StreamData properties: -- For any policy that loads and validates successfully, `compile/2` succeeds and - `lookup/3` never returns `{:ok, rule}` for a verb not declared in the policy. -- For any path+verb denied under trust level T, it is also denied under any T' < T - (monotonicity of denial preserved through the full pipeline). - -**C3 — Seam declaration in BoJ Trustfile** - -Add to `[SEAMS]` in `Trustfile.a2ml`: - -```yaml -- id: "gateway-boj-gnosis" - description: "http-capability-gateway ↔ BoJ unified Zig API gnosis handler" - from: "web-ecosystem/http-capability-gateway" - to: "src/zig-api/ (gnosis handler)" - contract: "docs/integration/http-capability-gateway-boj-contract.md" - test_ref: "tests/seam/gateway-boj.test" - failure_mode: "fail-closed (circuit breaker)" - tier: "elixir-disciplined" -``` - -### Acceptance Criteria - -- E2E tests pass in CI (not just locally). -- Coverage threshold: all six cases in C1 have passing tests. -- Seam test (C1) exercises the HTTP contract from Phase A: headers set correctly, - trust level forwarded, response returned to caller. -- Property tests (C2) pass with at least 200 samples. -- Seam declared in Trustfile. - -### Risks - -- **Test environment isolation:** E2E tests require a running BoJ instance. Options: - 1. Mock backend (easiest — Bypass module in the gateway test suite). - 2. Real BoJ in CI (preferred for seam test — more complex setup). - Use mock backend for C1/C2 in Phase C; schedule real BoJ seam test for Phase E. -- **Policy hot-reload under load:** The existing `concurrency_test.exs` covers - gateway-internal reload. The BoJ seam test must also verify that in-flight - requests are not dropped during a reload. This is the most complex test to write. - -### Blocks - -Phase D benchmarks depend on Phase C tests establishing a baseline measurement -environment. - ---- - -## Phase D — Benchmarks (weeks 7–8) - -### Objective - -Formalise the "fast policy enforcement" claim with published latency numbers. Define -the load profile, run the benchmark suite, and configure a regression alert so that -future changes that degrade gateway performance are caught in CI. - -### Deliverables - -**D1 — Load profile declaration** - -Document in `docs/integration/gateway-load-profile.md`: -- Baseline: requests/second the BoJ gnosis handler receives in production today. -- Gateway load profile: same request rate + gateway overhead should be < declared limit. -- Measurement environment: hardware spec, Elixir/OTP version, policy size (number of rules). - -**D2 — Benchmark results published** - -The existing `test/benchmark_test.exs` covers: -- Rate limiter throughput. -- Circuit breaker state transition cost. -- Exact vs. regex vs. global-fallback route lookup. - -Extend benchmarks to measure: -- **Median / p95 / p99 request-to-response latency** through the full Plug pipeline - (security headers → strip → extract trust → rate limit → policy lookup → proxy - response) under the declared load profile. -- Comparison baseline: latency of the same request hitting BoJ directly (no gateway). -- **Gateway overhead** = gateway median latency − baseline median latency. - -Target: gateway overhead < 2ms median, < 5ms p99 for a policy with 100 rules -(indicative — revisit once baseline is measured). - -**D3 — Regression alert in CI** - -Add a CI step that runs the benchmark suite and fails if the measured p99 latency -exceeds a declared threshold (initially generous — tighten after a few builds -establish a stable baseline). The benchmark step should be gated (not run on every -PR; run on merge to main and weekly schedule). - -### Acceptance Criteria - -- Load profile document exists. -- Median / p95 / p99 latency numbers published in `docs/integration/gateway-benchmarks.md`. -- Gateway overhead number published (vs. direct BoJ, no gateway). -- CI regression step configured. -- Benchmark results do NOT fabricate numbers — they come from a `mix bench` or - `mix test --only benchmark` run against the real hardware or a representative CI runner. - -### Risks - -- **CI hardware variability:** Benchmark numbers from GitHub Actions runners vary - significantly run-to-run. Use relative measurements (gateway overhead fraction) - rather than absolute numbers for regression detection. -- **Policy size sensitivity:** A policy with 5 rules will perform differently than - one with 500 rules. The benchmark must test the realistic BoJ policy size. - -### Blocks - -Phase E production deployment requires D3 (regression alert) to be in place before -rollout. - ---- - -## Phase E — Production Wiring (weeks 8–12) - -### Objective - -Deploy http-capability-gateway in front of BoJ in staging, verify all telemetry and -logging are correct, then roll out to production. Demonstrate end-to-end traffic -handling under real load. - -### Deliverables - -**E1 — Containerfile and deployment spec for the gateway** - -The gateway already has a `Containerfile`. Confirm it: -- Uses a Chainguard base image. -- Accepts `POLICY_PATH`, `BACKEND_URL`, `PORT`, `MTLS_CA_CERT_PATH` env vars. -- Is built with Podman (rootless). -- Is signed as a `.ctp` bundle via cerro-torre. - -Add a k9-svc deployment spec at `container/gateway-deploy.k9.ncl` in the gateway repo. - -**E2 — Staging deployment** - -Deploy the gateway in front of a BoJ staging instance: -- Gateway listens on port 8443 (TLS) or 8080 (HTTP, behind Cloudflare Tunnel). -- Backend: BoJ staging gnosis handler at `http://localhost:7700` or via Unix socket. -- Policy: `config/gateway-policy-boj-example.yaml` from Phase A. -- Trust level source: `"header"` initially, switched to `"mtls"` after Phase B cert - rotation runbook is exercised. - -**E3 — Telemetry verification** - -With staging traffic running: -- Confirm `[:http_capability_gateway, :access_decision]` events appear in the - Prometheus scrape at `/metrics`. -- Confirm `GET /api/v1/minikaran` returns `status.status: "active"` after the - learning phase. -- Confirm structured JSON logs appear in the log stream with `request_id` fields. -- Confirm `X-Trust-Level` header is set correctly on forwarded requests as seen - by BoJ (read from BoJ access logs). - -**E4 — Production rollout** - -Roll out behind a feature flag or behind a percentage split (10% → 50% → 100% -of traffic directed through the gateway). - -**E5 — Rollback runbook** - -File: `docs/integration/gateway-rollback-runbook.md` - -Covers: -- How to detect that the gateway is causing problems (elevated 502/503 rates, - increased latency at p99). -- How to bypass the gateway (re-route traffic directly to BoJ gnosis handler) - without downtime (requires traffic management — Cloudflare Tunnel rule, or - container orchestration). -- How to disable the gateway permanently (remove the k9-svc deployment, update - tier-2 config to `status: DISABLED`). - -### Acceptance Criteria - -- Gateway handles a declared traffic profile in staging for at least 24 hours - without circuit breaker trips or elevated error rates. -- Telemetry verified (E3 complete). -- Rollback runbook exists and has been tested (manually walk through E5 once - in staging before production rollout). -- Production rollout complete with no SLA regression (latency at p99 within - D2 declared overhead + 20% safety margin). -- `[HTTP_CAPABILITY_GATEWAY]` section in BoJ Trustfile updated to - `status: "DEPLOYED"` with `deployed_at` timestamp. - -### Risks - -- **Cowboy TLS in production:** First time running the gateway with mTLS in - production. Cert expiry, CA rotation, and Cowboy TLS configuration errors - are all possible. The rollback runbook must be tested before production. -- **Backend URL single-backend limitation:** The proxy module supports one - `backend_url`. If BoJ is horizontally scaled, the gateway must be placed - behind a load balancer or the proxy module must be extended (out of scope - for this plan but noted for future work). -- **Policy size in production:** The Phase A example policy covers known routes. - Undocumented routes (gnosis handler internals, health probes, metrics) must - be added to the policy before go-live or they will be default-denied. - ---- - -## Cross-Phase Notes - -### Verb Governance Spec versioning - -The policy file is a BoJ-repo artefact. It must be reviewed like any spec change -(PR required, approved by maintainer). When the OpenAPI spec changes, the policy -file must be updated in the same PR or the PR must document why the policy is -unchanged (e.g., the new route is not externally accessible). - -### Trust header forwarding security invariant - -The gateway strips `X-Trust-Level` from incoming requests unless the sender is in -`:trusted_proxies`. After Phase B (mTLS), the gateway sets `X-Trust-Level` from the -resolved mTLS trust level before forwarding to BoJ. BoJ's gnosis handler must treat -`X-Trust-Level` from `127.0.0.1` (or the gateway container IP) as authoritative and -from any other source as untrusted. This invariant must be documented in the Phase A -contract and tested in Phase C. - -### Groove protocol - -When Groove protocol is adopted estate-wide for inter-service communication, the -gateway ↔ BoJ forwarding path should be considered for Groove wrapping. This is -post-Phase E work. - -### VeriSimDB audit trail - -If `VeriSimDB` in the gateway is confirmed as a working integration (not a stub), -the audit trail (allow/deny decisions with path, verb, trust level, backend, rule name, -duration) will flow into the per-project VeriSimDB instance. This is valuable for -BoJ's audit posture. Confirm VeriSimDB integration status before Phase E. diff --git a/docs/integration/http-capability-gateway-policy-authoring.adoc b/docs/integration/http-capability-gateway-policy-authoring.adoc new file mode 100644 index 00000000..76e5f801 --- /dev/null +++ b/docs/integration/http-capability-gateway-policy-authoring.adoc @@ -0,0 +1,149 @@ +== Verb Governance Spec — Authoring & Deployment Workflow + +*Version:* 1.0 *Date:* 2026-05-18 *Status:* Phase A deliverable A2 +(normative) *Plan:* `+docs/integration/http-capability-gateway-plan.md+` +(§ Phase A, A2) *Contract:* +`+docs/integration/http-capability-gateway-boj-contract.md+` *Tracking:* +standards#91 (parent), standards#96 (Phase A) + +This document answers the four A2 questions the plan poses: *where* the +Verb Governance Spec lives, *who* writes it, *how* it is loaded at +deploy, and *how* hot-reload is triggered — plus the review/versioning +discipline that keeps the policy from drifting away from BoJ’s real HTTP +surface. + +''''' + +=== 1. Where the spec lives + +*Decision:* the Verb Governance Spec is a *BoJ-repo artefact* at + +.... +config/gateway-policy-boj.yaml # the live policy (added in a later phase) +config/gateway-policy-boj-example.yaml # the Phase A worked example (this PR) +.... + +Rationale (plan A2 option 1, chosen over option 2 "`separate file in the +gateway repo`"): the policy _describes BoJ’s HTTP surface_. It must be +version-controlled *alongside* `+docs/specification/openapi.yaml+` so +that a change to the surface and the change to its governance are +reviewable in the same repository, ideally the same PR (see §5). Keeping +it in the gateway repo would split the surface and its governance across +two repos and two review queues — the precise drift failure the ADR +warns about ("`if the policy file lags the actual surface, routes may be +default-denied`"). + +The gateway repo remains the home of the _DSL definition and validator_; +the _instance_ of that DSL for BoJ lives here. + +''''' + +=== 2. Who writes it + +* The Verb Governance Spec is written and changed by a *BoJ maintainer*. +* It is reviewed *like any specification change*: PR required, +maintainer approval required. It is not a config knob that can be +hand-edited on a deployed host — the deployed copy is immutable and +comes from this repo. +* A change to `+docs/specification/openapi.yaml+` (the HTTP surface) and +the corresponding change to `+config/gateway-policy-boj.yaml+` SHOULD +land in the *same PR*. If they cannot, the surface PR MUST state +explicitly why the policy is unchanged (e.g. "`new route is +internal-only and already covered by the default-deny backstop`", or +"`route is not externally reachable`"). + +''''' + +=== 3. How it is loaded at deploy + +* The gateway container reads the policy path from the *`+POLICY_PATH+`* +environment variable (per the gateway’s documented configuration; the +audit records `+PolicyLoader.load_policy/1+` as the load entry point). +* The policy file is delivered into the container at that path by a +*bind-mount or k9-svc ConfigMap-equivalent*, sourced from this repo’s +`+config/gateway-policy-boj.yaml+`. The file is *read-only* in the +container. +* Load sequence at startup (from the audit, §1.1–1.3): +[arabic] +. `+PolicyLoader.load_policy/1+` parses the YAML → `+{:ok, map()}+`. +. `+PolicyValidator.validate/1+` checks DSL v1 structural invariants. +. `+PolicyCompiler.compile/2+` builds the dual ETS tables. If validation +fails at startup the gateway MUST refuse to start (fail-closed); it MUST +NOT start with no policy and default-allow. + +''''' + +=== 4. Hot-reload trigger + +The gateway implements an *atomic compile-then-swap*: a new policy is +loaded, validated, and compiled into fresh ETS tables; only on success +are the live tables swapped. On validation/compile failure the +*last-known-good policy is preserved* and the reload is a no-op (audit +§1.3). + +Reload is triggered by, in order of preference: + +[arabic] +. *k9-svc rolling redeploy* — the standard path. A policy change is a +normal versioned deploy: new file, new container revision, rolling +restart. No in-place mutation. This is the *production-default* +mechanism because it keeps the running policy provably equal to a +reviewed repo artefact. +. *SIGHUP / `+gateway-reload+` signal* — supported by the atomic-swap +pattern for low-latency reloads without a full restart, used when a +rolling redeploy is too heavy (e.g. an urgent narrowing of an over-broad +rule). The reloaded file MUST still come from a merged repo artefact, +mounted before the signal is sent; ad-hoc on-host edits are prohibited. + +Phase C must verify that an in-flight request is *not dropped* during a +reload (plan §Phase C, C1 "`policy hot-reload`" case). Until that test +exists, mechanism (1) — rolling redeploy with normal connection draining +— is the only sanctioned production reload path; mechanism (2) is +dev/staging only. + +''''' + +=== 5. Review & versioning discipline (anti-drift) + +The single largest operational risk in the ADR is *policy lagging the +surface*. The discipline that prevents it: + +* *Co-change rule.* Any PR that adds, removes, or changes an HTTP route +in `+docs/specification/openapi.yaml+` (or in `+BojRest.Router+` / the +gnosis handler) MUST either update `+config/gateway-policy-boj.yaml+` in +the same PR or carry an explicit, reviewed justification for leaving it +unchanged. +* *Default-deny is a backstop, not a policy.* A route that is missing +from the spec is denied. That is safe (fail-closed) but it is an +_outage_ for a route that should be public. Relying on default-deny +instead of an explicit rule is a review finding, not an acceptable +steady state. +* *Validation gate.* CI SHOULD run the example/live policy through the +gateway’s `+PolicyLoader.load_policy/1+` + +`+PolicyValidator.validate/1+` so a malformed policy cannot merge. +(Wiring this CI gate is Phase C/D scope; the Phase A acceptance +criterion is a _manual_ verification — see §6.) +* *Narrative is mandatory.* Every route rule carries a `+narrative+` +explaining _why_ that exposure level. A rule whose narrative cannot be +written is a rule that is not understood and MUST NOT be merged. + +''''' + +=== 6. Phase A verification status + +The plan’s Phase A acceptance criterion "`example policy file passes +`+PolicyLoader.load_policy/1+` + `+PolicyValidator.validate/1+``" is a +*manual verification* performed against a running gateway. In Phase A no +gateway is deployed, so `+config/gateway-policy-boj-example.yaml+` is +authored to *conform exactly to the documented DSL v1 schema and +`+PolicyValidator+` invariants* (audit §2): `+dsl_version: "1"+`; +`+governance.global_verbs+` a non-empty list of all-caps HTTP verbs; +each route with non-empty `+verbs+`, optional +`+exposure ∈ {public, authenticated, internal}+`; `+stealth+` with +boolean `+enabled+` and integer `+status_code+` in 100–599. + +The running-gateway manual check is an explicit, recorded carry-over to +be executed when the gateway container is first stood up (Phase E E2 +staging bring-up, or earlier if the gateway is run locally). It is a +verification step, not a code deliverable, and does not block Phase A +closure. diff --git a/docs/integration/http-capability-gateway-policy-authoring.md b/docs/integration/http-capability-gateway-policy-authoring.md deleted file mode 100644 index 716dee02..00000000 --- a/docs/integration/http-capability-gateway-policy-authoring.md +++ /dev/null @@ -1,141 +0,0 @@ - - - -# Verb Governance Spec — Authoring & Deployment Workflow - -**Version:** 1.0 -**Date:** 2026-05-18 -**Status:** Phase A deliverable A2 (normative) -**Plan:** `docs/integration/http-capability-gateway-plan.md` (§ Phase A, A2) -**Contract:** `docs/integration/http-capability-gateway-boj-contract.md` -**Tracking:** standards#91 (parent), standards#96 (Phase A) - -This document answers the four A2 questions the plan poses: **where** the Verb -Governance Spec lives, **who** writes it, **how** it is loaded at deploy, and -**how** hot-reload is triggered — plus the review/versioning discipline that -keeps the policy from drifting away from BoJ's real HTTP surface. - ---- - -## 1. Where the spec lives - -**Decision:** the Verb Governance Spec is a **BoJ-repo artefact** at - -``` -config/gateway-policy-boj.yaml # the live policy (added in a later phase) -config/gateway-policy-boj-example.yaml # the Phase A worked example (this PR) -``` - -Rationale (plan A2 option 1, chosen over option 2 "separate file in the gateway -repo"): the policy *describes BoJ's HTTP surface*. It must be version-controlled -**alongside** `docs/specification/openapi.yaml` so that a change to the surface -and the change to its governance are reviewable in the same repository, ideally -the same PR (see §5). Keeping it in the gateway repo would split the surface and -its governance across two repos and two review queues — the precise drift -failure the ADR warns about ("if the policy file lags the actual surface, routes -may be default-denied"). - -The gateway repo remains the home of the *DSL definition and validator*; the -*instance* of that DSL for BoJ lives here. - ---- - -## 2. Who writes it - -- The Verb Governance Spec is written and changed by a **BoJ maintainer**. -- It is reviewed **like any specification change**: PR required, maintainer - approval required. It is not a config knob that can be hand-edited on a - deployed host — the deployed copy is immutable and comes from this repo. -- A change to `docs/specification/openapi.yaml` (the HTTP surface) and the - corresponding change to `config/gateway-policy-boj.yaml` SHOULD land in the - **same PR**. If they cannot, the surface PR MUST state explicitly why the - policy is unchanged (e.g. "new route is internal-only and already covered by - the default-deny backstop", or "route is not externally reachable"). - ---- - -## 3. How it is loaded at deploy - -- The gateway container reads the policy path from the **`POLICY_PATH`** - environment variable (per the gateway's documented configuration; the audit - records `PolicyLoader.load_policy/1` as the load entry point). -- The policy file is delivered into the container at that path by a - **bind-mount or k9-svc ConfigMap-equivalent**, sourced from this repo's - `config/gateway-policy-boj.yaml`. The file is **read-only** in the container. -- Load sequence at startup (from the audit, §1.1–1.3): - 1. `PolicyLoader.load_policy/1` parses the YAML → `{:ok, map()}`. - 2. `PolicyValidator.validate/1` checks DSL v1 structural invariants. - 3. `PolicyCompiler.compile/2` builds the dual ETS tables. - If validation fails at startup the gateway MUST refuse to start (fail-closed); - it MUST NOT start with no policy and default-allow. - ---- - -## 4. Hot-reload trigger - -The gateway implements an **atomic compile-then-swap**: a new policy is loaded, -validated, and compiled into fresh ETS tables; only on success are the live -tables swapped. On validation/compile failure the **last-known-good policy is -preserved** and the reload is a no-op (audit §1.3). - -Reload is triggered by, in order of preference: - -1. **k9-svc rolling redeploy** — the standard path. A policy change is a normal - versioned deploy: new file, new container revision, rolling restart. No - in-place mutation. This is the **production-default** mechanism because it - keeps the running policy provably equal to a reviewed repo artefact. -2. **SIGHUP / `gateway-reload` signal** — supported by the atomic-swap pattern - for low-latency reloads without a full restart, used when a rolling redeploy - is too heavy (e.g. an urgent narrowing of an over-broad rule). The reloaded - file MUST still come from a merged repo artefact, mounted before the signal - is sent; ad-hoc on-host edits are prohibited. - -Phase C must verify that an in-flight request is **not dropped** during a -reload (plan §Phase C, C1 "policy hot-reload" case). Until that test exists, -mechanism (1) — rolling redeploy with normal connection draining — is the only -sanctioned production reload path; mechanism (2) is dev/staging only. - ---- - -## 5. Review & versioning discipline (anti-drift) - -The single largest operational risk in the ADR is **policy lagging the -surface**. The discipline that prevents it: - -- **Co-change rule.** Any PR that adds, removes, or changes an HTTP route in - `docs/specification/openapi.yaml` (or in `BojRest.Router` / the gnosis - handler) MUST either update `config/gateway-policy-boj.yaml` in the same PR or - carry an explicit, reviewed justification for leaving it unchanged. -- **Default-deny is a backstop, not a policy.** A route that is missing from the - spec is denied. That is safe (fail-closed) but it is an *outage* for a route - that should be public. Relying on default-deny instead of an explicit rule is - a review finding, not an acceptable steady state. -- **Validation gate.** CI SHOULD run the example/live policy through the - gateway's `PolicyLoader.load_policy/1` + `PolicyValidator.validate/1` so a - malformed policy cannot merge. (Wiring this CI gate is Phase C/D scope; the - Phase A acceptance criterion is a *manual* verification — see §6.) -- **Narrative is mandatory.** Every route rule carries a `narrative` explaining - *why* that exposure level. A rule whose narrative cannot be written is a rule - that is not understood and MUST NOT be merged. - ---- - -## 6. Phase A verification status - -The plan's Phase A acceptance criterion "example policy file passes -`PolicyLoader.load_policy/1` + `PolicyValidator.validate/1`" is a **manual -verification** performed against a running gateway. In Phase A no gateway is -deployed, so `config/gateway-policy-boj-example.yaml` is authored to **conform -exactly to the documented DSL v1 schema and `PolicyValidator` invariants** -(audit §2): `dsl_version: "1"`; `governance.global_verbs` a non-empty list of -all-caps HTTP verbs; each route with non-empty `verbs`, optional -`exposure ∈ {public, authenticated, internal}`; `stealth` with boolean -`enabled` and integer `status_code` in 100–599. - -The running-gateway manual check is an explicit, recorded carry-over to be -executed when the gateway container is first stood up (Phase E E2 staging -bring-up, or earlier if the gateway is run locally). It is a verification step, -not a code deliverable, and does not block Phase A closure. diff --git a/docs/maintenance/MAINTENANCE-CHECKLIST.adoc b/docs/maintenance/MAINTENANCE-CHECKLIST.adoc new file mode 100644 index 00000000..f390f041 --- /dev/null +++ b/docs/maintenance/MAINTENANCE-CHECKLIST.adoc @@ -0,0 +1,671 @@ +== Maintenance Checklist (Cross-Repo) + +Use this as a repeatable maintenance runbook for any repo. + +Companion policy: + +* `+docs/practice/SOFTWARE-DEVELOPMENT-APPROACH.adoc+` (human-readable) +* `+.machine_readable/policies/SOFTWARE-DEVELOPMENT-APPROACH.a2ml+` +(machine-readable) + +=== Canonical Repo Baseline (Final) + +Apply this baseline to every repo unless an explicit exception is +recorded. + +==== Three-Axis Default Model + +* [ ] Axis 1 (scope priority, runs first): `+must > intend > like+` +* [ ] Axis 2 (maintenance priority): +`+corrective > adaptive > perfective+` +* [ ] Axis 3 (audit priority): `+systems > compliance > effects+` +* [ ] Perfective items are derived from Axis 1 honest state (not started +independently). + +==== Axis 1 Scoping Pass (Mandatory) + +Before Axis 2/3 execution, assemble a scoped worklist from evidence: + +* [ ] Read and reconcile: `+README+`, roadmap, status docs, maintenance +checklist, and current CI/security docs. +* [ ] Scan for unfinished markers: `+TODO+`, `+FIXME+`, `+XXX+`, +`+HACK+`, `+STUB+`, `+PARTIAL+`. +* [ ] If Idris is present, scan unsoundness markers: `+believe_me+`, +`+assert_total+`. +* [ ] Identify declared intent vs actual implementation (docs honesty +check). +* [ ] Produce a scope assembly artifact with prioritized entries under: +** `+must+` (release blockers / safety / correctness) +** `+intend+` (planned near-term) +** `+like+` (nice-to-have) + +==== Axis 2 Maintenance Execution Rules + +* [ ] Corrective first: fix breakage, defects, regressions, safety +issues. +* [ ] Adaptive second: reconcile changed scope, remove stale references, +cull no-longer-relevant work. +* [ ] Perfective third: only from current honest state established by +Axis 1 and updated by corrective/adaptive actions. + +==== Axis 3 Audit Rules + +* [ ] Verify systems are in place and actually operating. +* [ ] Verify documentation explains the real/current state (not +aspirational-only), including documented exceptions. +* [ ] Verify safety and security controls are present, active, and +evidenced. +* [ ] Verify observed effects/impacts are captured and reviewed. +* [ ] Effects audit includes: +** benchmark execution and recorded results (with before/after where +relevant) +** explicit maintainer dialogue/status review on what changed, why, and +next risks +* [ ] Audit compliance seams/compromises explicitly: +** policy exceptions are recorded with rationale, scope, and +expiry/review +** exception does not silently broaden into general policy drift +** language-policy contamination checks run (example: a single TS +exception must not trigger broad TypeScript conversion) +** run `+panic-attack+` as the compliance-audit scanner +** run ecological checking under effects (using sustainabot guidance as +current baseline) + +==== Generic Cleanup And Finish-Off Pass + +Run this pass at the end of a corrective/adaptive/perfective cycle: + +* [ ] Root cleanup: +** keep only required control/entry files in root +** move non-essential docs/reports/fixtures to canonical folders +* [ ] Remove or archive stale work: +** close out completed TODO/STUB/PARTIAL items +** cull obsolete references, dead files, and superseded plans +* [ ] Documentation finish-off: +** ensure README, roadmap, status, and wiki match actual implementation +state +** ensure machine-readable policy/state files match human docs +* [ ] Security/compliance finish-off: +** run compliance scanner (`+panic-attack+`) and resolve high-priority +findings +** verify exception register and seams/compromises are explicitly +bounded +* [ ] Effects finish-off: +** run benchmark/effects checks and record evidence +** conduct explicit maintainer review dialogue (what changed, why, +remaining risks) +* [ ] Release-prep finish-off: +** produce Must/Should/Could summary +** produce immediate corrective/adaptive/perfective next-actions list + +==== Must + +* [ ] Keep required control files at repository root: +** `+.gitignore+`, `+.gitattributes+`, `+.editorconfig+`, +`+.tool-versions+` +** `+Containerfile+` +** `+.containerignore+` (or `+.dockerignore+` only when required for +compatibility) +** `+CNAME+` and `+.nojekyll+` when using GitHub Pages/custom domain +** `+Justfile+` (root by convention) +* [ ] Keep ownership/governance files present: +** `+MAINTAINER+` in root +** `+.github/CODEOWNERS+` +* [ ] Keep machine-readable canonical structure under +`+.machine_readable/+`: +** state/meta/ecosystem files (`+*.a2ml+` or repo standard) +** `+anchors/ANCHOR.a2ml+` +** `+contractiles/+` (`+must+`, `+trust+`, `+lust+`, and related) +** `+ai/+` for AI guidance files +** `+bot_directives/+` for bot control files +* [ ] Keep contractiles/invariants present and wired: +** root `+Mustfile+` (or equivalent) with enforceable checks +** `+Trustfile+` and `+Intentfile+` present +* [ ] Keep security metadata present: +** `+.well-known/security.txt+` and relevant policy metadata +** CI security scanning configured and runnable +* [ ] Keep docs and navigation coherent: +** single navigation entry point in root (`+NAVIGATION.adoc+` or +equivalent) +** no duplicate conflicting docs for same purpose (for example both +`+.md+` and `+.adoc+` in root unless intentionally required) +* [ ] Enforce ABI/FFI purity where the policy applies: +** ABI definitions in Idris2 (`+src/abi/*.idr+`) +** FFI implementations in Zig (`+ffi/**/*.zig+`) +* [ ] Ensure quality gate includes: formatting, lint, unit/integration +tests, p2p/e2e checks, benchmark smoke, docs checks, security scan. + +==== Should + +* [ ] Keep human docs primarily in AsciiDoc (`+.adoc+`) except where +ecosystem rules require other formats (GitHub/community health, legal +text, tool-specific files). +* [ ] Keep non-essential root files moved into structured folders: +** `+docs/+` (theory/practice/whitepapers/proofs/reports) +** `+tests/+` (fixtures/outputs) +** `+licensing/+` (while retaining root `+LICENSE+` when forge detection +needs it) +* [ ] Maintain `+.well-known/+` for public metadata where applicable +(`+security.txt+`, `+humans.txt+`, `+ads.txt+` mirrors if used). +* [ ] Keep CI policy checks for doc-format conventions and canonical +file placement. +* [ ] Keep roadmap/status docs honest with dated evidence. + +==== Could + +* [ ] Maintain both human and machine views of maintenance policy from a +single source (generate one from the other). +* [ ] Add policy bots for corrective/adaptive/perfective/audit modes. +* [ ] Add repo-level architecture map (`+TOPOLOGY.md+`) and +release-readiness dashboards. +* [ ] Add per-repo exception registry for approved policy deviations. + +==== Explicit Root-Placement Rule + +Do *not* move the following out of root if you want default tool +behavior: + +* `+.gitignore+`, `+.gitattributes+`, `+.editorconfig+`, +`+.tool-versions+` +* `+Containerfile+` and ignore file +(`+.containerignore+`/`+.dockerignore+`) +* `+CNAME+` and `+.nojekyll+` for GitHub Pages +* `+Justfile+` + +=== Quick Automated Run (Script) + +Use the helper script first, then use the checklist for deeper/manual +follow-up. + +Script locations: - `+$REPOS_DIR/run-maintenance.sh+` (where +`+REPOS_DIR+` is your repos checkout root) - +`+~/Desktop/run-maintenance.sh+` + +[source,bash] +---- +~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo --output /tmp/maintenance-report.json +jq . /tmp/maintenance-report.json +---- + +Useful flags: + +[source,bash] +---- +# Strict mode: fail process on failed checks +~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo --strict + +# Skip expensive checks when needed +~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo --skip-panic + +# Explicit language selection +~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo --rust --python + +# Release hard-pass mode (fails on warnings or failures) +~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo --fail-on-warn +---- + +Permission policy in script: - Flags `+g+w/o+w+` files/dirs - Flags +suspicious executable files - Flags shebang scripts missing executable +bit - Supports repo-local exceptions via `+.maintenance-perms-ignore+` +(regex per line) - *Audit-first by default* (non-mutating) - +`+--fix-perms+` is explicit opt-in only (never implicit) - For +reversible local hardening, pair snapshot/restore scripts where +available: - `+scripts/maintenance/perms-state.sh snapshot+` - +`+scripts/maintenance/perms-state.sh lock+` - +`+scripts/maintenance/perms-state.sh restore+` + +Important git behavior: - Git generally tracks execute bit, not full +UNIX mode matrix. - Permission hardening audits do not force +collaborators to re-unlock every file on pull. - Keep lock mode opt-in, +with restore path documented. + +[source,bash] +---- +# Audit-only (recommended default) +~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo + +# Opt-in permission fixes (review output before commit) +~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo --fix-perms +---- + +=== 0) Setup + +[source,bash] +---- +REPO="/absolute/path/to/repo" +cd "$REPO" +---- + +[source,bash] +---- +date -u +git rev-parse --abbrev-ref HEAD +git rev-parse HEAD +git status --porcelain +---- + +=== 1) Preflight + +* [ ] Confirm clean intent: note existing unrelated dirty files before +edits. +* [ ] Confirm runtime/toolchain versions. +* [ ] Confirm container mode expectation (`+podman+`/`+podman-compose+`) +if required. + +[source,bash] +---- +command -v rg git jq || true +command -v podman podman-compose || true +---- + +=== 2) Dependency/Env Prereqs + +* [ ] Python deps in active interpreter (for Python paths). +* [ ] Language-specific tooling installed. + +[source,bash] +---- +python -c "import sys; print(sys.executable)" +python -c "import pydantic; print(pydantic.__version__)" || echo "pydantic missing" +---- + +=== 3) Corrective Maintenance First + +* [ ] Fix regressions, runtime errors, panics, broken commands, failing +tests. +* [ ] Re-run failing checks immediately after each fix. + +=== 4) Code Health Scans + +* [ ] `+TODO/FIXME/XXX/HACK/STUB/PARTIAL+` scan. +* [ ] Permission policy scan (`+g+w/o+w+`, executable hygiene). +* [ ] ABI/FFI policy scan (if applicable: Idris2 ABI, Zig FFI). + +[source,bash] +---- +rg -n "TODO|FIXME|XXX|HACK|STUB|PARTIAL" -g '!**/.git/**' -g '!**/target/**' . +---- + +[source,bash] +---- +# Optional per-repo exceptions (regex per line): +# .maintenance-perms-ignore +# ^vendor/ +# ^third_party/ +---- + +[source,bash] +---- +# Adjust paths for your repo layout +find . -type f \( -name '*.idr' -o -name '*.idris2' -o -name '*.zig' \) +---- + +=== 5) Panic/Safety/Security Pass + +* [ ] Run `+panic-attacker+` assail/assault. +* [ ] Triage findings by severity. +* [ ] Fix high first, then medium. +* [ ] Re-run until acceptable. + +[source,bash] +---- +PANIC_BIN="${REPOS_DIR:-$(dirname "$(pwd)")}/panic-attacker/target/release/panic-attack" +"$PANIC_BIN" assail "$REPO" --output /tmp/assail.json --output-format json --quiet +jq -r '.weak_points | length' /tmp/assail.json +jq -r '.weak_points[] | "\(.severity)|\(.location)|\(.description)"' /tmp/assail.json +---- + +[source,bash] +---- +# If repo has production-only source builder, prefer this for baseline checks: +./scripts/ci/build-panic-assail-source.sh /tmp/panic-src +"$PANIC_BIN" assail /tmp/panic-src --output /tmp/assail-prod.json --output-format json --quiet +---- + +=== 6) Language-Specific Validation + +==== Rust + +* [ ] Format +* [ ] Lint +* [ ] Tests +* [ ] Doc tests +* [ ] Benches (where relevant) + +[source,bash] +---- +cargo fmt --all --check +cargo clippy --workspace --all-targets -- -D warnings +cargo test --workspace +cargo test --workspace --doc +# Optional targeted benchmarks: +cargo bench +---- + +==== Python + +* [ ] Format/lint +* [ ] Type check +* [ ] Tests + +[source,bash] +---- +ruff check . +ruff format --check . +mypy . +pytest -q +---- + +==== Elixir + +* [ ] Format check +* [ ] Lint/static checks +* [ ] Tests + +[source,bash] +---- +mix format --check-formatted +mix credo --strict +mix test +---- + +=== 7) Container/Runtime Checks (Podman) + +* [ ] Build container path. +* [ ] Run smoke tests inside containerized flow. +* [ ] Compare host vs container behavior for parity. + +[source,bash] +---- +podman --version +podman compose version || podman-compose --version +---- + +=== 8) Benchmark + Regression Check + +* [ ] Capture before/after metrics for touched hot paths. +* [ ] Record command + sample size + output. +* [ ] Fail change if critical path regresses beyond threshold. + +=== 9) Adaptive and Perfective Maintenance + +* [ ] Adaptive: compatibility updates (tooling/API/deprecations/config +flags). +* [ ] Perfective: clarity, docs parity, developer workflow improvements. +* [ ] Update roadmap/checklist/docs to match actual implementation +state. + +=== 10) Final QA and Release Hygiene + +* [ ] Re-run full relevant checks one final time. +* [ ] Confirm no unintended file changes. +* [ ] Commit scoped changes with clear message. +* [ ] Push and capture commit SHA. + +[source,bash] +---- +git status --short +git diff --stat +git add +git commit -m "maint: " +git push +---- + +=== 11) Maintenance Report Template + +Copy this block per repo run: + +[source,text] +---- +Repo: +Branch: +Start UTC: +End UTC: + +Scope: +- Corrective: +- Adaptive: +- Perfective: + +Checks Run: +- TODO/FIXME scan: +- Panic-attacker: +- Rust/Python/Elixir checks: +- Container checks: +- Benchmark checks: + +Findings: +- High: +- Medium: +- Low: + +Fixes Applied: +1. +2. +3. + +Validation Results: +- Tests: +- Benchmarks: +- Panic-attacker rerun: + +Artifacts: +- assail report: +- benchmark output: +- logs: + +Commit(s): +- SHA: + +Remaining Risks / Follow-ups: +1. +2. +---- + +=== 12) Language-Repo Additions (Eclexia-Specific) + +Add these checks for language/compiler repositories with formal ABI/FFI +constraints: + +* [x] README structure restored (index/TOC, audience paths, quickstart +sanity). +* [x] Wiki split by audience (laypeople/users/developers) and linked +from docs index. +* [x] Root-level clutter reduced (archive, analysis, reports relegated +to `+docs/+` subtrees). +* [x] Machine-readable docs synchronized (`+STATE.scm+`, `+META.scm+`, +`+ECOSYSTEM.scm+`, contractiles). +* [x] Human-readable docs synchronized (`+README+`, `+QUICK_STATUS+`, +roadmap, wiki home). +* [x] `+Mustfile+` invariants present and enforceable in CI. +* [x] `+Trustfile+` and `+Intentfile+` present and complete. +* [x] FFI/ABI purity policy enforced (`+*.zig+` for FFI, +`+*.idr+`/Idris2 for ABI). +* [x] `+panic-attack+` findings triaged with explicit severity budget +for release. +* [x] Point-to-point, end-to-end, and benchmark checks wired in one +quality gate. +* [x] CI workflows include quality + security + docs checks with +explicit policy. +* [x] Release audit includes corrective/adaptive/perfective + +Must/Should/Could. +* [x] Roadmap/status honesty pass completed (dates and current evidence +updated). + +=== 13) Latest Execution Record (Eclexia, 2026-02-24) + +Repo: `+/tmp/eclexia-releaseprep+` (branch `+release-prep+`, base +`+533ec9e9447f374135cc9e2e81021624ddb3c0ad+`) + +==== 13.1 Setup/Preflight + +* [x] Captured UTC timestamp and git state. +* [x] Tooling presence verified (`+rg+`, `+git+`, `+jq+`, `+cargo+`, +`+rustc+`, `+just+`). +* [x] Runtime/toolchain versions captured. +* [x] Container tooling checked (`+podman+`, `+podman-compose+`). + +==== 13.2 Corrective Maintenance + +* [x] Fixed `+panic-attack+` script path handling (`+mktemp+` output + +local fallback binary detection). +* [x] Removed Idris `+believe_me+` usage from ABI wrappers. +* [x] Fixed conformance crash-noise path by skipping known intentional +stack-overflow case in default runner. +* [x] Re-ran affected checks after each fix. + +==== 13.3 Code-Health Scans + +* [x] TODO/FIXME/STUB/PARTIAL scan run on active code paths. +* [x] ABI/FFI file inventory run (`+*.idr+`, `+*.zig+`). +* [x] Active-code marker count reduced/triaged; remaining items tracked +in release audit. + +==== 13.4 Security/Panic Pass + +* [x] `+panic-attack+` run and triaged. +* [x] Critical findings cleared (Idris unsoundness markers removed). +* [x] Current baseline: 0 weak points (Critical 0, High 0, Medium 0, Low +0). +* [x] High/Medium backlog fully eliminated. + +==== 13.5 Language Validation + +* [x] Final `+just quality-gate+` pass completed (docs, fmt, lint, unit, +conformance, integration, p2p, e2e, bench). +* [x] Additional targeted reruns completed (`+just test-conformance+`, +`+just panic-attack+`, `+just docs-check+`). + +==== 13.6 Adaptive/Perfective/Docs + +* [x] README/wiki/docs structure and indexing restored. +* [x] Root tidy/relegation pass executed. +* [x] Roadmap/status honesty update performed with current date and +evidence links. +* [x] Release audit created with corrective/adaptive/perfective + +Must/Should/Could. +* [x] Full quality-gate rerun passed after hardening updates. +* [x] ABI/FFI extension lane added without breaking stable symbols +(`+ecl_abi_get_info+`, `+ecl_tracker_create_ex+`, +`+ecl_tracker_snapshot+`). +* [x] CI quality workflow now validates sibling `+proven+` repo presence +and critical binding files. +* [x] Proven roadmap now includes explicit "`critical core, not full +rewrite`" adoption guidance and flowchart. + +==== 13.7 Outstanding Items (Explicit) + +* [x] Stable `+v1.0.0+` technical gate readiness met (quality + panic +scan clean). +* [x] Parser/codegen/runtime panic-path hardening completed for +scanner-flagged paths. +* [x] Non-eclexia `+proven+` library checked: already Idris2-first with +Zig ABI bridge; no additional integration changes required in this run. +* [ ] Remote push blocked by token scope: GitHub rejected branch updates +(`+release-prep+`, `+release-prep-pushable+`) due missing `+workflow+` +OAuth scope. + +==== 13.8 Artifacts + +* Release audit: `+docs/reports/V1-READINESS-AUDIT-2026-02-24.md+` +* Panic report: `+/tmp/eclexia-panic-attack.KZ1jpC.json+` (0 weak +points) +* Final quality gate log: `+/tmp/eclexia-quality-gate-final2.log+` (plus +post-change reruns via terminal sessions) +* Local commits: `+88fa2af+` (`+release-prep+`), `+baa3d1c+` +(`+release-prep-pushable+`) + pending new commit from this pass + +=== 12) LLM Operator Instructions + +Use this prompt with an LLM agent when you want the process run +end-to-end: + +[source,text] +---- +Run the maintenance workflow for this repo using MAINTENANCE-CHECKLIST.md. + +Required behavior: +1. Run ~/Desktop/run-maintenance.sh first and collect the JSON report. +2. Triage report results by severity: fail > warn > pass. +3. Execute corrective maintenance first (fix regressions, panics, broken tests/commands). +4. Run TODO/FIXME/stub scan and address relevant items. +5. Run panic-attacker and fix findings in priority order; rerun to confirm. +6. Run language-specific checks (Rust/Python/Elixir) relevant to this repo. +7. Run benchmark/regression checks for touched hot paths. +8. Enforce permission policy: + - no group/world writable source files unless justified + - executable bit only where intended + - use .maintenance-perms-ignore for justified exceptions +9. Update docs/roadmap/checklist entries to reflect actual state. +10. Produce a final report using the template in MAINTENANCE-CHECKLIST.md. + +Constraints: +- Do not revert unrelated existing dirty changes. +- Stage and commit only scoped intended files. +- If blocked, state exactly what is blocked and why. +---- + +=== 13) AI Execution Integrity Contract (Mandatory) + +Use this when delegating maintenance to any AI +(Gemini/Claude/ChatGPT/etc.). + +[source,text] +---- +You must execute this maintenance run with strict integrity. + +Non-negotiable rules: +1. Do not claim any step is complete unless you actually ran it. +2. Do not silently skip checklist items. If skipped, state SKIPPED + exact reason. +3. For every check, provide evidence: + - command executed + - pass/fail/warn + - key output summary + - artifact/log path +4. If a command fails, stop claiming success and report the failure clearly. +5. After each fix, re-run the relevant failing check and report the rerun result. +6. Do not hide uncertainty. If unsure, say so and run additional verification. +7. Never mark “all done” while any fail/warn remains unexplained. +8. Do not make destructive or broad permission changes by default. + - permission changes must be audit-first + - use --fix-perms only with explicit intent +9. Final output must include: + - checklist coverage matrix (each item: PASS/FAIL/WARN/SKIPPED) + - unresolved risks + - exact next actions +10. Prioritize user safety and reputation: no “looks fine” claims without evidence. +---- + +Recommended enforcement line for AI prompts: + +[source,text] +---- +Fail closed: if evidence is missing for any checklist item, treat that item as NOT DONE. +---- + +=== 14) Fleet Enrollment Automation (Gitbot + Hypatia) + +For centralized coverage across existing and new repos: + +[source,bash] +---- +cd "${REPOS_DIR:-$(dirname "$(pwd)")}/gitbot-fleet" +just enroll-repos +---- + +Optional directive write-back to repos that already have +`+.machine_readable/+`: + +[source,bash] +---- +cd "${REPOS_DIR:-$(dirname "$(pwd)")}/gitbot-fleet" +just enroll-repos "$REPOS_DIR" true +---- + +Release hard gate from fleet: + +[source,bash] +---- +cd "${REPOS_DIR:-$(dirname "$(pwd)")}/gitbot-fleet" +just maintenance-hard-pass /absolute/path/to/repo +---- diff --git a/docs/maintenance/MAINTENANCE-CHECKLIST.md b/docs/maintenance/MAINTENANCE-CHECKLIST.md deleted file mode 100644 index 6063715e..00000000 --- a/docs/maintenance/MAINTENANCE-CHECKLIST.md +++ /dev/null @@ -1,572 +0,0 @@ - -# Maintenance Checklist (Cross-Repo) - -Use this as a repeatable maintenance runbook for any repo. - -Companion policy: - -- `docs/practice/SOFTWARE-DEVELOPMENT-APPROACH.adoc` (human-readable) -- `.machine_readable/policies/SOFTWARE-DEVELOPMENT-APPROACH.a2ml` (machine-readable) - -## Canonical Repo Baseline (Final) - -Apply this baseline to every repo unless an explicit exception is recorded. - -### Three-Axis Default Model - -- [ ] Axis 1 (scope priority, runs first): `must > intend > like` -- [ ] Axis 2 (maintenance priority): `corrective > adaptive > perfective` -- [ ] Axis 3 (audit priority): `systems > compliance > effects` -- [ ] Perfective items are derived from Axis 1 honest state (not started independently). - -### Axis 1 Scoping Pass (Mandatory) - -Before Axis 2/3 execution, assemble a scoped worklist from evidence: - -- [ ] Read and reconcile: `README`, roadmap, status docs, maintenance checklist, and current CI/security docs. -- [ ] Scan for unfinished markers: `TODO`, `FIXME`, `XXX`, `HACK`, `STUB`, `PARTIAL`. -- [ ] If Idris is present, scan unsoundness markers: `believe_me`, `assert_total`. -- [ ] Identify declared intent vs actual implementation (docs honesty check). -- [ ] Produce a scope assembly artifact with prioritized entries under: - - `must` (release blockers / safety / correctness) - - `intend` (planned near-term) - - `like` (nice-to-have) - -### Axis 2 Maintenance Execution Rules - -- [ ] Corrective first: fix breakage, defects, regressions, safety issues. -- [ ] Adaptive second: reconcile changed scope, remove stale references, cull no-longer-relevant work. -- [ ] Perfective third: only from current honest state established by Axis 1 and updated by corrective/adaptive actions. - -### Axis 3 Audit Rules - -- [ ] Verify systems are in place and actually operating. -- [ ] Verify documentation explains the real/current state (not aspirational-only), including documented exceptions. -- [ ] Verify safety and security controls are present, active, and evidenced. -- [ ] Verify observed effects/impacts are captured and reviewed. -- [ ] Effects audit includes: - - benchmark execution and recorded results (with before/after where relevant) - - explicit maintainer dialogue/status review on what changed, why, and next risks -- [ ] Audit compliance seams/compromises explicitly: - - policy exceptions are recorded with rationale, scope, and expiry/review - - exception does not silently broaden into general policy drift - - language-policy contamination checks run (example: a single TS exception must not trigger broad TypeScript conversion) - - run `panic-attack` as the compliance-audit scanner - - run ecological checking under effects (using sustainabot guidance as current baseline) - -### Generic Cleanup And Finish-Off Pass - -Run this pass at the end of a corrective/adaptive/perfective cycle: - -- [ ] Root cleanup: - - keep only required control/entry files in root - - move non-essential docs/reports/fixtures to canonical folders -- [ ] Remove or archive stale work: - - close out completed TODO/STUB/PARTIAL items - - cull obsolete references, dead files, and superseded plans -- [ ] Documentation finish-off: - - ensure README, roadmap, status, and wiki match actual implementation state - - ensure machine-readable policy/state files match human docs -- [ ] Security/compliance finish-off: - - run compliance scanner (`panic-attack`) and resolve high-priority findings - - verify exception register and seams/compromises are explicitly bounded -- [ ] Effects finish-off: - - run benchmark/effects checks and record evidence - - conduct explicit maintainer review dialogue (what changed, why, remaining risks) -- [ ] Release-prep finish-off: - - produce Must/Should/Could summary - - produce immediate corrective/adaptive/perfective next-actions list - -### Must - -- [ ] Keep required control files at repository root: - - `.gitignore`, `.gitattributes`, `.editorconfig`, `.tool-versions` - - `Containerfile` - - `.containerignore` (or `.dockerignore` only when required for compatibility) - - `CNAME` and `.nojekyll` when using GitHub Pages/custom domain - - `Justfile` (root by convention) -- [ ] Keep ownership/governance files present: - - `MAINTAINER` in root - - `.github/CODEOWNERS` -- [ ] Keep machine-readable canonical structure under `.machine_readable/`: - - state/meta/ecosystem files (`*.a2ml` or repo standard) - - `anchors/ANCHOR.a2ml` - - `contractiles/` (`must`, `trust`, `lust`, and related) - - `ai/` for AI guidance files - - `bot_directives/` for bot control files -- [ ] Keep contractiles/invariants present and wired: - - root `Mustfile` (or equivalent) with enforceable checks - - `Trustfile` and `Intentfile` present -- [ ] Keep security metadata present: - - `.well-known/security.txt` and relevant policy metadata - - CI security scanning configured and runnable -- [ ] Keep docs and navigation coherent: - - single navigation entry point in root (`NAVIGATION.adoc` or equivalent) - - no duplicate conflicting docs for same purpose (for example both `.md` and `.adoc` in root unless intentionally required) -- [ ] Enforce ABI/FFI purity where the policy applies: - - ABI definitions in Idris2 (`src/abi/*.idr`) - - FFI implementations in Zig (`ffi/**/*.zig`) -- [ ] Ensure quality gate includes: formatting, lint, unit/integration tests, p2p/e2e checks, benchmark smoke, docs checks, security scan. - -### Should - -- [ ] Keep human docs primarily in AsciiDoc (`.adoc`) except where ecosystem rules require other formats (GitHub/community health, legal text, tool-specific files). -- [ ] Keep non-essential root files moved into structured folders: - - `docs/` (theory/practice/whitepapers/proofs/reports) - - `tests/` (fixtures/outputs) - - `licensing/` (while retaining root `LICENSE` when forge detection needs it) -- [ ] Maintain `.well-known/` for public metadata where applicable (`security.txt`, `humans.txt`, `ads.txt` mirrors if used). -- [ ] Keep CI policy checks for doc-format conventions and canonical file placement. -- [ ] Keep roadmap/status docs honest with dated evidence. - -### Could - -- [ ] Maintain both human and machine views of maintenance policy from a single source (generate one from the other). -- [ ] Add policy bots for corrective/adaptive/perfective/audit modes. -- [ ] Add repo-level architecture map (`TOPOLOGY.md`) and release-readiness dashboards. -- [ ] Add per-repo exception registry for approved policy deviations. - -### Explicit Root-Placement Rule - -Do **not** move the following out of root if you want default tool behavior: - -- `.gitignore`, `.gitattributes`, `.editorconfig`, `.tool-versions` -- `Containerfile` and ignore file (`.containerignore`/`.dockerignore`) -- `CNAME` and `.nojekyll` for GitHub Pages -- `Justfile` - -## Quick Automated Run (Script) - -Use the helper script first, then use the checklist for deeper/manual follow-up. - -Script locations: -- `$REPOS_DIR/run-maintenance.sh` (where `REPOS_DIR` is your repos checkout root) -- `~/Desktop/run-maintenance.sh` - -```bash -~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo --output /tmp/maintenance-report.json -jq . /tmp/maintenance-report.json -``` - -Useful flags: - -```bash -# Strict mode: fail process on failed checks -~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo --strict - -# Skip expensive checks when needed -~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo --skip-panic - -# Explicit language selection -~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo --rust --python - -# Release hard-pass mode (fails on warnings or failures) -~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo --fail-on-warn -``` - -Permission policy in script: -- Flags `g+w/o+w` files/dirs -- Flags suspicious executable files -- Flags shebang scripts missing executable bit -- Supports repo-local exceptions via `.maintenance-perms-ignore` (regex per line) -- **Audit-first by default** (non-mutating) -- `--fix-perms` is explicit opt-in only (never implicit) -- For reversible local hardening, pair snapshot/restore scripts where available: - - `scripts/maintenance/perms-state.sh snapshot` - - `scripts/maintenance/perms-state.sh lock` - - `scripts/maintenance/perms-state.sh restore` - -Important git behavior: -- Git generally tracks execute bit, not full UNIX mode matrix. -- Permission hardening audits do not force collaborators to re-unlock every file on pull. -- Keep lock mode opt-in, with restore path documented. - -```bash -# Audit-only (recommended default) -~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo - -# Opt-in permission fixes (review output before commit) -~/Desktop/run-maintenance.sh --repo /absolute/path/to/repo --fix-perms -``` - -## 0) Setup - -```bash -REPO="/absolute/path/to/repo" -cd "$REPO" -``` - -```bash -date -u -git rev-parse --abbrev-ref HEAD -git rev-parse HEAD -git status --porcelain -``` - -## 1) Preflight - -- [ ] Confirm clean intent: note existing unrelated dirty files before edits. -- [ ] Confirm runtime/toolchain versions. -- [ ] Confirm container mode expectation (`podman`/`podman-compose`) if required. - -```bash -command -v rg git jq || true -command -v podman podman-compose || true -``` - -## 2) Dependency/Env Prereqs - -- [ ] Python deps in active interpreter (for Python paths). -- [ ] Language-specific tooling installed. - -```bash -python -c "import sys; print(sys.executable)" -python -c "import pydantic; print(pydantic.__version__)" || echo "pydantic missing" -``` - -## 3) Corrective Maintenance First - -- [ ] Fix regressions, runtime errors, panics, broken commands, failing tests. -- [ ] Re-run failing checks immediately after each fix. - -## 4) Code Health Scans - -- [ ] `TODO/FIXME/XXX/HACK/STUB/PARTIAL` scan. -- [ ] Permission policy scan (`g+w/o+w`, executable hygiene). -- [ ] ABI/FFI policy scan (if applicable: Idris2 ABI, Zig FFI). - -```bash -rg -n "TODO|FIXME|XXX|HACK|STUB|PARTIAL" -g '!**/.git/**' -g '!**/target/**' . -``` - -```bash -# Optional per-repo exceptions (regex per line): -# .maintenance-perms-ignore -# ^vendor/ -# ^third_party/ -``` - -```bash -# Adjust paths for your repo layout -find . -type f \( -name '*.idr' -o -name '*.idris2' -o -name '*.zig' \) -``` - -## 5) Panic/Safety/Security Pass - -- [ ] Run `panic-attacker` assail/assault. -- [ ] Triage findings by severity. -- [ ] Fix high first, then medium. -- [ ] Re-run until acceptable. - -```bash -PANIC_BIN="${REPOS_DIR:-$(dirname "$(pwd)")}/panic-attacker/target/release/panic-attack" -"$PANIC_BIN" assail "$REPO" --output /tmp/assail.json --output-format json --quiet -jq -r '.weak_points | length' /tmp/assail.json -jq -r '.weak_points[] | "\(.severity)|\(.location)|\(.description)"' /tmp/assail.json -``` - -```bash -# If repo has production-only source builder, prefer this for baseline checks: -./scripts/ci/build-panic-assail-source.sh /tmp/panic-src -"$PANIC_BIN" assail /tmp/panic-src --output /tmp/assail-prod.json --output-format json --quiet -``` - -## 6) Language-Specific Validation - -### Rust - -- [ ] Format -- [ ] Lint -- [ ] Tests -- [ ] Doc tests -- [ ] Benches (where relevant) - -```bash -cargo fmt --all --check -cargo clippy --workspace --all-targets -- -D warnings -cargo test --workspace -cargo test --workspace --doc -# Optional targeted benchmarks: -cargo bench -``` - -### Python - -- [ ] Format/lint -- [ ] Type check -- [ ] Tests - -```bash -ruff check . -ruff format --check . -mypy . -pytest -q -``` - -### Elixir - -- [ ] Format check -- [ ] Lint/static checks -- [ ] Tests - -```bash -mix format --check-formatted -mix credo --strict -mix test -``` - -## 7) Container/Runtime Checks (Podman) - -- [ ] Build container path. -- [ ] Run smoke tests inside containerized flow. -- [ ] Compare host vs container behavior for parity. - -```bash -podman --version -podman compose version || podman-compose --version -``` - -## 8) Benchmark + Regression Check - -- [ ] Capture before/after metrics for touched hot paths. -- [ ] Record command + sample size + output. -- [ ] Fail change if critical path regresses beyond threshold. - -## 9) Adaptive and Perfective Maintenance - -- [ ] Adaptive: compatibility updates (tooling/API/deprecations/config flags). -- [ ] Perfective: clarity, docs parity, developer workflow improvements. -- [ ] Update roadmap/checklist/docs to match actual implementation state. - -## 10) Final QA and Release Hygiene - -- [ ] Re-run full relevant checks one final time. -- [ ] Confirm no unintended file changes. -- [ ] Commit scoped changes with clear message. -- [ ] Push and capture commit SHA. - -```bash -git status --short -git diff --stat -git add -git commit -m "maint: " -git push -``` - -## 11) Maintenance Report Template - -Copy this block per repo run: - -```text -Repo: -Branch: -Start UTC: -End UTC: - -Scope: -- Corrective: -- Adaptive: -- Perfective: - -Checks Run: -- TODO/FIXME scan: -- Panic-attacker: -- Rust/Python/Elixir checks: -- Container checks: -- Benchmark checks: - -Findings: -- High: -- Medium: -- Low: - -Fixes Applied: -1. -2. -3. - -Validation Results: -- Tests: -- Benchmarks: -- Panic-attacker rerun: - -Artifacts: -- assail report: -- benchmark output: -- logs: - -Commit(s): -- SHA: - -Remaining Risks / Follow-ups: -1. -2. -``` - -## 12) Language-Repo Additions (Eclexia-Specific) - -Add these checks for language/compiler repositories with formal ABI/FFI constraints: - -- [x] README structure restored (index/TOC, audience paths, quickstart sanity). -- [x] Wiki split by audience (laypeople/users/developers) and linked from docs index. -- [x] Root-level clutter reduced (archive, analysis, reports relegated to `docs/` subtrees). -- [x] Machine-readable docs synchronized (`STATE.scm`, `META.scm`, `ECOSYSTEM.scm`, contractiles). -- [x] Human-readable docs synchronized (`README`, `QUICK_STATUS`, roadmap, wiki home). -- [x] `Mustfile` invariants present and enforceable in CI. -- [x] `Trustfile` and `Intentfile` present and complete. -- [x] FFI/ABI purity policy enforced (`*.zig` for FFI, `*.idr`/Idris2 for ABI). -- [x] `panic-attack` findings triaged with explicit severity budget for release. -- [x] Point-to-point, end-to-end, and benchmark checks wired in one quality gate. -- [x] CI workflows include quality + security + docs checks with explicit policy. -- [x] Release audit includes corrective/adaptive/perfective + Must/Should/Could. -- [x] Roadmap/status honesty pass completed (dates and current evidence updated). - -## 13) Latest Execution Record (Eclexia, 2026-02-24) - -Repo: `/tmp/eclexia-releaseprep` (branch `release-prep`, base `533ec9e9447f374135cc9e2e81021624ddb3c0ad`) - -### 13.1 Setup/Preflight - -- [x] Captured UTC timestamp and git state. -- [x] Tooling presence verified (`rg`, `git`, `jq`, `cargo`, `rustc`, `just`). -- [x] Runtime/toolchain versions captured. -- [x] Container tooling checked (`podman`, `podman-compose`). - -### 13.2 Corrective Maintenance - -- [x] Fixed `panic-attack` script path handling (`mktemp` output + local fallback binary detection). -- [x] Removed Idris `believe_me` usage from ABI wrappers. -- [x] Fixed conformance crash-noise path by skipping known intentional stack-overflow case in default runner. -- [x] Re-ran affected checks after each fix. - -### 13.3 Code-Health Scans - -- [x] TODO/FIXME/STUB/PARTIAL scan run on active code paths. -- [x] ABI/FFI file inventory run (`*.idr`, `*.zig`). -- [x] Active-code marker count reduced/triaged; remaining items tracked in release audit. - -### 13.4 Security/Panic Pass - -- [x] `panic-attack` run and triaged. -- [x] Critical findings cleared (Idris unsoundness markers removed). -- [x] Current baseline: 0 weak points (Critical 0, High 0, Medium 0, Low 0). -- [x] High/Medium backlog fully eliminated. - -### 13.5 Language Validation - -- [x] Final `just quality-gate` pass completed (docs, fmt, lint, unit, conformance, integration, p2p, e2e, bench). -- [x] Additional targeted reruns completed (`just test-conformance`, `just panic-attack`, `just docs-check`). - -### 13.6 Adaptive/Perfective/Docs - -- [x] README/wiki/docs structure and indexing restored. -- [x] Root tidy/relegation pass executed. -- [x] Roadmap/status honesty update performed with current date and evidence links. -- [x] Release audit created with corrective/adaptive/perfective + Must/Should/Could. -- [x] Full quality-gate rerun passed after hardening updates. -- [x] ABI/FFI extension lane added without breaking stable symbols (`ecl_abi_get_info`, `ecl_tracker_create_ex`, `ecl_tracker_snapshot`). -- [x] CI quality workflow now validates sibling `proven` repo presence and critical binding files. -- [x] Proven roadmap now includes explicit "critical core, not full rewrite" adoption guidance and flowchart. - -### 13.7 Outstanding Items (Explicit) - -- [x] Stable `v1.0.0` technical gate readiness met (quality + panic scan clean). -- [x] Parser/codegen/runtime panic-path hardening completed for scanner-flagged paths. -- [x] Non-eclexia `proven` library checked: already Idris2-first with Zig ABI bridge; no additional integration changes required in this run. -- [ ] Remote push blocked by token scope: GitHub rejected branch updates (`release-prep`, `release-prep-pushable`) due missing `workflow` OAuth scope. - -### 13.8 Artifacts - -- Release audit: `docs/reports/V1-READINESS-AUDIT-2026-02-24.md` -- Panic report: `/tmp/eclexia-panic-attack.KZ1jpC.json` (0 weak points) -- Final quality gate log: `/tmp/eclexia-quality-gate-final2.log` (plus post-change reruns via terminal sessions) -- Local commits: `88fa2af` (`release-prep`), `baa3d1c` (`release-prep-pushable`) + pending new commit from this pass - -## 12) LLM Operator Instructions - -Use this prompt with an LLM agent when you want the process run end-to-end: - -```text -Run the maintenance workflow for this repo using MAINTENANCE-CHECKLIST.md. - -Required behavior: -1. Run ~/Desktop/run-maintenance.sh first and collect the JSON report. -2. Triage report results by severity: fail > warn > pass. -3. Execute corrective maintenance first (fix regressions, panics, broken tests/commands). -4. Run TODO/FIXME/stub scan and address relevant items. -5. Run panic-attacker and fix findings in priority order; rerun to confirm. -6. Run language-specific checks (Rust/Python/Elixir) relevant to this repo. -7. Run benchmark/regression checks for touched hot paths. -8. Enforce permission policy: - - no group/world writable source files unless justified - - executable bit only where intended - - use .maintenance-perms-ignore for justified exceptions -9. Update docs/roadmap/checklist entries to reflect actual state. -10. Produce a final report using the template in MAINTENANCE-CHECKLIST.md. - -Constraints: -- Do not revert unrelated existing dirty changes. -- Stage and commit only scoped intended files. -- If blocked, state exactly what is blocked and why. -``` - -## 13) AI Execution Integrity Contract (Mandatory) - -Use this when delegating maintenance to any AI (Gemini/Claude/ChatGPT/etc.). - -```text -You must execute this maintenance run with strict integrity. - -Non-negotiable rules: -1. Do not claim any step is complete unless you actually ran it. -2. Do not silently skip checklist items. If skipped, state SKIPPED + exact reason. -3. For every check, provide evidence: - - command executed - - pass/fail/warn - - key output summary - - artifact/log path -4. If a command fails, stop claiming success and report the failure clearly. -5. After each fix, re-run the relevant failing check and report the rerun result. -6. Do not hide uncertainty. If unsure, say so and run additional verification. -7. Never mark “all done” while any fail/warn remains unexplained. -8. Do not make destructive or broad permission changes by default. - - permission changes must be audit-first - - use --fix-perms only with explicit intent -9. Final output must include: - - checklist coverage matrix (each item: PASS/FAIL/WARN/SKIPPED) - - unresolved risks - - exact next actions -10. Prioritize user safety and reputation: no “looks fine” claims without evidence. -``` - -Recommended enforcement line for AI prompts: - -```text -Fail closed: if evidence is missing for any checklist item, treat that item as NOT DONE. -``` - -## 14) Fleet Enrollment Automation (Gitbot + Hypatia) - -For centralized coverage across existing and new repos: - -```bash -cd "${REPOS_DIR:-$(dirname "$(pwd)")}/gitbot-fleet" -just enroll-repos -``` - -Optional directive write-back to repos that already have `.machine_readable/`: - -```bash -cd "${REPOS_DIR:-$(dirname "$(pwd)")}/gitbot-fleet" -just enroll-repos "$REPOS_DIR" true -``` - -Release hard gate from fleet: - -```bash -cd "${REPOS_DIR:-$(dirname "$(pwd)")}/gitbot-fleet" -just maintenance-hard-pass /absolute/path/to/repo -``` diff --git a/docs/outreach/awesome-list-descriptions.md b/docs/outreach/awesome-list-descriptions.adoc similarity index 74% rename from docs/outreach/awesome-list-descriptions.md rename to docs/outreach/awesome-list-descriptions.adoc index 3d9eecba..4409fb4f 100644 --- a/docs/outreach/awesome-list-descriptions.md +++ b/docs/outreach/awesome-list-descriptions.adoc @@ -1,22 +1,16 @@ - - - +== Awesome List Submission Descriptions -# Awesome List Submission Descriptions +Pre-written descriptions for submitting BoJ Server to various awesome +lists. Copy-paste the appropriate section when submitting a PR. -Pre-written descriptions for submitting BoJ Server to various awesome lists. -Copy-paste the appropriate section when submitting a PR. +''''' ---- - -## awesome-mcp-servers +=== awesome-mcp-servers Repository: https://github.com/punkpeye/awesome-mcp-servers (or similar) -```markdown +[source,markdown] +---- ### BoJ Server Unified capability catalogue exposing 53 cartridges (database, container, @@ -30,27 +24,29 @@ and hash attestation. - **MCP setup**: `boj-server --mcp` (see [docs/quickstarts/BUILD-FROM-SOURCE.adoc](https://github.com/hyperpolymath/boj-server/blob/main/docs/quickstarts/BUILD-FROM-SOURCE.adoc)) - **Protocol**: JSON-RPC 2.0 over stdio - **License**: MPL-2.0 -``` +---- ---- +''''' -## awesome-selfhosted +=== awesome-selfhosted Repository: https://github.com/awesome-selfhosted/awesome-selfhosted -```markdown +[source,markdown] +---- - [BoJ Server](https://github.com/hyperpolymath/boj-server) - Federated developer tool catalogue with 18 capability cartridges, QUIC gossip protocol, and formally verified plugin architecture. `MPL-2.0` `Zig` `Idris2` `zig` -``` +---- -Category: `Software Development - IDE & Tools` +Category: `+Software Development - IDE & Tools+` ---- +''''' -## modelcontextprotocol/servers +=== modelcontextprotocol/servers Repository: https://github.com/modelcontextprotocol/servers -```markdown +[source,markdown] +---- ### BoJ Server A unified MCP server exposing 18 capability domains (database, container, @@ -65,45 +61,50 @@ dependent types. Built with Zig and zig. | **Transport** | stdio (JSON-RPC 2.0) | | **Tools** | 53 cartridges, each exposing domain-specific operations | | **Setup** | `boj-server --mcp` | -``` +---- ---- +''''' -## awesome-zig +=== awesome-zig Repository: https://github.com/C-BJ/awesome-zig (or similar) -```markdown +[source,markdown] +---- - [boj-server](https://github.com/hyperpolymath/boj-server) - Federated developer tool catalogue with 18 capability cartridges. Zig handles the FFI layer (C-ABI exports, thread-safe mutexes, shared library compilation). Paired with Idris2 for formal verification and zig for network adapters. 219 Zig tests + 32 seam checks, zero runtime dependencies. -``` +---- ---- +''''' -## awesome-idris +=== awesome-idris Repository: https://github.com/joaomilho/awesome-idris (or similar) -```markdown +[source,markdown] +---- - [boj-server](https://github.com/hyperpolymath/boj-server) - Developer tool catalogue using Idris2 dependent types for ABI definitions. The `IsUnbreakable` proof type gates cartridge activation at compile time. 18 cartridge ABI modules with `%default total`, zero `believe_me`. Zig FFI, zig adapter. -``` +---- ---- +''''' -## awesome-vlang +=== awesome-vlang Repository: https://github.com/vlang/awesome-v (or similar) -```markdown +[source,markdown] +---- - [boj-server](https://github.com/hyperpolymath/boj-server) - Unified developer tool server using V for the network adapter layer. Exposes REST (port 7700), gRPC-compat (7701), and GraphQL (7702) from a single V codebase. 18 capability cartridges loaded via Zig FFI with Idris2-verified interfaces. -``` +---- ---- +''''' -## Notes for submitting +=== Notes for submitting -1. Check each list's contribution guidelines before submitting -2. Ensure the repo README is current and the getting started guide works -3. Most lists require alphabetical placement within the appropriate section -4. Some lists require a minimum star count -- check before submitting -5. The awesome-selfhosted list requires the project to be actively maintained - and have installation documentation +[arabic] +. Check each list’s contribution guidelines before submitting +. Ensure the repo README is current and the getting started guide works +. Most lists require alphabetical placement within the appropriate +section +. Some lists require a minimum star count – check before submitting +. The awesome-selfhosted list requires the project to be actively +maintained and have installation documentation diff --git a/docs/outreach/blog-post-draft.adoc b/docs/outreach/blog-post-draft.adoc new file mode 100644 index 00000000..b89615e9 --- /dev/null +++ b/docs/outreach/blog-post-draft.adoc @@ -0,0 +1,286 @@ +== Why I Built a Server Catalogue with Three Languages and Zero Python + +=== The moment my desktop froze + +I had three Claude instances running. One Cursor session. About twenty +MCP servers, a handful of LSP servers, two DAP servers, and a build +server. Each one was a separate process, each with its own +configuration, its own port, its own dependencies. My system had 47 open +sockets and was using 14GB of RAM just for developer tooling. + +Then my desktop froze. + +I sat there, staring at a black screen, and thought: _this is not a +tooling problem. This is a combinatorics problem._ + +=== The problem nobody talks about + +Developer protocols are multiplying. MCP (Model Context Protocol) lets +AI talk to tools. LSP handles language intelligence. DAP does debugging. +BSP manages builds. Each protocol is useful. Each tool that speaks a +protocol is useful. But the product of protocols times tools times AI +agents is an explosion. + +If you have 5 AI tools and 12 capability domains, you don’t need 12 +servers. You need 12 servers _per protocol_. And if each server is a +separate Python process with its own virtualenv and its own dependency +tree, you’re running a small data centre on your laptop. + +The industry response has been to build _more_ servers. One MCP server +for databases. One for containers. One for Kubernetes. One for Git. Each +well-crafted, each standalone, each adding another process to your +system tray. + +I went a different way. + +=== The insight: it’s a matrix + +The capability landscape isn’t a list. It’s a 2D matrix: + +.... + MCP LSP DAP BSP gRPC REST + +------+------+------+------+------+------+ +Database | ## | | | | ## | ## | +Container | ## | | | | ## | ## | +Git/VCS | ## | | | | ## | ## | +Secrets | ## | | | | ## | ## | +Queues | ## | | | | ## | ## | +IaC | ## | | | | ## | ## | +Observe | ## | | | | ## | ## | +SSG | ## | | | | ## | ## | +Proof | ## | | | | ## | ## | +Fleet | ## | | | ## | ## | ## | +NeSy | ## | ## | | | ## | ## | +Agent | ## | | | | ## | ## | +Cloud | ## | | | | ## | ## | +K8s | ## | | | | ## | ## | +LSP | ## | ## | | | ## | ## | +DAP | ## | | ## | | ## | ## | +BSP | ## | | | ## | ## | ## | +Feedback | ## | | | | ## | ## | + +------+------+------+------+------+------+ +.... + +Columns are protocols (how you talk). Rows are domains (what it does). +Each filled cell is a *cartridge* – a formally verified, swappable +capability module. One server. One binary. 53 domains. Multiple +protocols. + +That’s BoJ: the *Bundle of Joy* server. + +=== Why three languages (and why they’re not the ones you expect) + +The typical developer server is Python or TypeScript. BoJ uses none of +those. Instead, every cartridge has three layers: + +[cols=",,",options="header",] +|=== +|Layer |Language |Job +|ABI |Idris2 |Prove the interface is correct +|FFI |Zig |Execute it natively +|Adapter |Elixir |Serve it over the network +|=== + +*Why Idris2?* Because it has dependent types. Not "`type-safe`" in the +TypeScript sense. Actually provably correct at compile time. The core +safety gate is a type called `+IsUnbreakable+` – it’s a mathematical +proof that only cartridges in the `+Ready+` state can be activated. The +type checker enforces this, not a runtime check, not a unit test. If the +proof doesn’t hold, the code doesn’t compile. + +[source,idris] +---- +-- Simplified: the safety gate +data IsUnbreakable : CartridgeState -> Type where + MkUnbreakable : IsUnbreakable Ready + +mount : (c : Cartridge) -> {auto prf : IsUnbreakable (state c)} -> IO () +---- + +You literally cannot call `+mount+` on a cartridge that isn’t `+Ready+`. +The type system makes it impossible. + +*Why Zig?* Because it produces C-ABI-compatible shared libraries with +zero runtime dependencies. Each cartridge compiles to a `+.so+` file. +The Zig layer bridges Idris2’s proofs with actual system calls – file +I/O, networking, database connections. Cross-compilation is built in, +which matters when community members run nodes on ARM, x86, or whatever +they have. + +*Why Elixir?* Because one Elixir codebase on the BEAM (Plug/Cowboy) +exposes all three API styles (REST + gRPC + GraphQL) on dedicated ports, +with the fault-tolerance and concurrency the BEAM is known for. One +runtime, three protocols, no code generation step. + +The result: a compact binary. 219 Zig tests + 8 integration tests + 32 +seam checks. Thread-safe (every FFI entry point serialises on a +per-module mutex). No virtualenvs, no node_modules, no pip install. + +=== How it works in practice + +Start the server: + +[source,bash] +---- +git clone https://github.com/hyperpolymath/boj-server.git +cd boj-server +cd ffi/zig && zig build && cd ../.. +cd elixir && mix deps.get && mix run --no-halt +---- + +Three ports open: + +.... +REST: http://localhost:7700 +gRPC: http://localhost:7701 +GraphQL: http://localhost:7702 +.... + +Check what’s available: + +[source,bash] +---- +curl http://localhost:7700/menu +---- + +Mount a cartridge: + +[source,bash] +---- +curl -X POST http://localhost:7700/cartridges/database-mcp/load +---- + +Use it: + +[source,bash] +---- +curl http://localhost:7700/cartridges/database-mcp/invoke \ + -X POST -H 'Content-Type: application/json' \ + -d '{"tool":"status","args":""}' +---- + +Or skip HTTP entirely and use MCP mode for AI tools: + +[source,bash] +---- +./boj-server --mcp +---- + +This gives you a JSON-RPC 2.0 stdio server. All 53 cartridges appear as +MCP tools. Add it to your MCP client config and your AI sees one server +with 53 capabilities instead of 53 separate servers. + +=== Federation: the community IS the hosting + +I don’t have a hosting budget. I’m a solo developer. But I do have an +idea: what if the community _is_ the infrastructure? + +BoJ includes a federation system called *Umoja* (Swahili for "`unity`"). +Any community member can run a node: + +[arabic] +. Pull the container image (Chainguard base, Podman – never Docker) +. Run it +. Your node joins the network via IPv6 gossip protocol +. Requests route to healthy nodes automatically + +The trust model is hash attestation. Every BoJ binary has a SHA-256 +hash. Your node proves its binary matches the canonical build. If +someone modifies their binary, they’re excluded from the community +network – but they can still run it locally. Non-punitive. We don’t +brick your installation. We just don’t vouch for it. + +The gossip protocol is Byzantine fault tolerant. Nodes exchange peer +lists, stale nodes get deprioritised, load-aware routing sends requests +to nodes under 80% capacity. The federation transport uses QUIC with +X25519 key exchange and ChaCha20-Poly1305 encryption, falling back to +plain UDP when QUIC isn’t available. + +Four seed nodes across four continents from day one. No cloud bill. Just +people running software on computers they already own. + +=== The HAT concept: don’t throw away your tools + +I want to address something directly: BoJ is not trying to replace your +database MCP server or your Kubernetes CLI. Those tools are good. People +built them with care. + +What BoJ does is give them a uniform shopfront. Think of it like a +Hardware store with an Attached Toolshed – a *HAT*. You don’t throw away +your hammer when you buy a toolbox. You put it in the toolbox so you can +find it. + +The third axis of the matrix (the one I haven’t mentioned yet) is the +*backend* axis. Each cartridge has a backend field – by default it’s +`+"universal"+`, but community extensions can specialise it. Want a +`+database-mcp+` cartridge that talks specifically to PostgreSQL? That’s +a backend variant. The core cartridge defines the interface (via Idris2 +proofs); the backend variant implements it for a specific provider. + +The community submission system (called "`Ayo`", meaning "`joy`" in +Yoruba) lets anyone submit a cartridge variant. It goes through a review +state machine (`+submitted -> under_review -> approved+`), and approved +cartridges appear in the community tier of the Teranga menu. + +=== The honest assessment + +BoJ is Alpha. Grade D on most cartridges. Here’s what’s real: + +* 53 cartridges, all with ABI + FFI + adapter layers, all compiling to +`+.so+` files +* 219 Zig tests, 8 integration tests, 32 seam checks — all passing +* Thread-safety hardening across all 9 core + 53 cartridge FFI modules +(all mutex-protected) +* MCP stdio bridge working (JSON-RPC 2.0, all 53 cartridges as tools) +* Umoja federation with real QUIC+UDP networking (22 federation tests) +* Zero `+believe_me+` in production code (Idris2’s escape hatch – we +don’t use it) +* Security audit by panic-attack scanner: 1 expected weak point (QUIC +0-RTT replay window, mitigated at app layer), 0 critical vulnerabilities + +Here’s what’s not done: + +* No real users besides me +* Seed nodes aren’t deployed to actual infrastructure yet +* The Teranga menu runtime is 30% complete (the spec exists, the runtime +doesn’t) +* Most cartridges are Grade D (they work, but they haven’t been tested +by diverse external users) +* No domain name yet for the federation network + +This isn’t a product launch. This is a "`come look at what I built and +tell me what’s wrong with it`" post. + +=== Try it, host a node, or just tell me what you think + +Install: `+npm install -g @hyperpolymath/boj-server+` or +`+guix build github:hyperpolymath/boj-server+` + +The repo is at +https://github.com/hyperpolymath/boj-server[github.com/hyperpolymath/boj-server]. +The quickstart guide is +https://github.com/hyperpolymath/boj-server/blob/main/docs/quickstarts/DEV.adoc[docs/quickstarts/DEV.adoc]. + +What I’m looking for: + +* *Try it locally* – clone, build, run +`+curl http://localhost:7700/matrix+` and tell me if the experience +makes sense +* *Host a node* – if you have a machine that’s on during the day, you +can join the Umoja network. See `+container/+` in the repo +* *Build on it* – the extensibility system lets you add backend variants +without touching core code. See `+docs/EXTENSIBILITY.adoc+` +* *Tell me what breaks* – feedback-mcp is literally a cartridge that +collects feedback. BoJ dogfoods itself + +This is a community project. I make nothing from it. The license +(MPL-2.0) ensures the code stays open and provenance-tracked. + +I built this to learn from it, and I learn most from other people using +it. + +''''' + +_Jonathan D.A. Jewell builds developer tools and formal verification +systems. BoJ is part of the hyperpolymath ecosystem of open-source +projects._ diff --git a/docs/outreach/blog-post-draft.md b/docs/outreach/blog-post-draft.md deleted file mode 100644 index aa562a12..00000000 --- a/docs/outreach/blog-post-draft.md +++ /dev/null @@ -1,203 +0,0 @@ - - - - -# Why I Built a Server Catalogue with Three Languages and Zero Python - -## The moment my desktop froze - -I had three Claude instances running. One Cursor session. About twenty MCP servers, a handful of LSP servers, two DAP servers, and a build server. Each one was a separate process, each with its own configuration, its own port, its own dependencies. My system had 47 open sockets and was using 14GB of RAM just for developer tooling. - -Then my desktop froze. - -I sat there, staring at a black screen, and thought: *this is not a tooling problem. This is a combinatorics problem.* - -## The problem nobody talks about - -Developer protocols are multiplying. MCP (Model Context Protocol) lets AI talk to tools. LSP handles language intelligence. DAP does debugging. BSP manages builds. Each protocol is useful. Each tool that speaks a protocol is useful. But the product of protocols times tools times AI agents is an explosion. - -If you have 5 AI tools and 12 capability domains, you don't need 12 servers. You need 12 servers *per protocol*. And if each server is a separate Python process with its own virtualenv and its own dependency tree, you're running a small data centre on your laptop. - -The industry response has been to build *more* servers. One MCP server for databases. One for containers. One for Kubernetes. One for Git. Each well-crafted, each standalone, each adding another process to your system tray. - -I went a different way. - -## The insight: it's a matrix - -The capability landscape isn't a list. It's a 2D matrix: - -``` - MCP LSP DAP BSP gRPC REST - +------+------+------+------+------+------+ -Database | ## | | | | ## | ## | -Container | ## | | | | ## | ## | -Git/VCS | ## | | | | ## | ## | -Secrets | ## | | | | ## | ## | -Queues | ## | | | | ## | ## | -IaC | ## | | | | ## | ## | -Observe | ## | | | | ## | ## | -SSG | ## | | | | ## | ## | -Proof | ## | | | | ## | ## | -Fleet | ## | | | ## | ## | ## | -NeSy | ## | ## | | | ## | ## | -Agent | ## | | | | ## | ## | -Cloud | ## | | | | ## | ## | -K8s | ## | | | | ## | ## | -LSP | ## | ## | | | ## | ## | -DAP | ## | | ## | | ## | ## | -BSP | ## | | | ## | ## | ## | -Feedback | ## | | | | ## | ## | - +------+------+------+------+------+------+ -``` - -Columns are protocols (how you talk). Rows are domains (what it does). Each filled cell is a **cartridge** -- a formally verified, swappable capability module. One server. One binary. 53 domains. Multiple protocols. - -That's BoJ: the **Bundle of Joy** server. - -## Why three languages (and why they're not the ones you expect) - -The typical developer server is Python or TypeScript. BoJ uses none of those. Instead, every cartridge has three layers: - -| Layer | Language | Job | -|-------|----------|-----| -| ABI | Idris2 | Prove the interface is correct | -| FFI | Zig | Execute it natively | -| Adapter | Elixir | Serve it over the network | - -**Why Idris2?** Because it has dependent types. Not "type-safe" in the TypeScript sense. Actually provably correct at compile time. The core safety gate is a type called `IsUnbreakable` -- it's a mathematical proof that only cartridges in the `Ready` state can be activated. The type checker enforces this, not a runtime check, not a unit test. If the proof doesn't hold, the code doesn't compile. - -```idris --- Simplified: the safety gate -data IsUnbreakable : CartridgeState -> Type where - MkUnbreakable : IsUnbreakable Ready - -mount : (c : Cartridge) -> {auto prf : IsUnbreakable (state c)} -> IO () -``` - -You literally cannot call `mount` on a cartridge that isn't `Ready`. The type system makes it impossible. - -**Why Zig?** Because it produces C-ABI-compatible shared libraries with zero runtime dependencies. Each cartridge compiles to a `.so` file. The Zig layer bridges Idris2's proofs with actual system calls -- file I/O, networking, database connections. Cross-compilation is built in, which matters when community members run nodes on ARM, x86, or whatever they have. - -**Why Elixir?** Because one Elixir codebase on the BEAM (Plug/Cowboy) exposes all three API styles (REST + gRPC + GraphQL) on dedicated ports, with the fault-tolerance and concurrency the BEAM is known for. One runtime, three protocols, no code generation step. - -The result: a compact binary. 219 Zig tests + 8 integration tests + 32 seam checks. Thread-safe (every FFI entry point serialises on a per-module mutex). No virtualenvs, no node_modules, no pip install. - -## How it works in practice - -Start the server: - -```bash -git clone https://github.com/hyperpolymath/boj-server.git -cd boj-server -cd ffi/zig && zig build && cd ../.. -cd elixir && mix deps.get && mix run --no-halt -``` - -Three ports open: - -``` -REST: http://localhost:7700 -gRPC: http://localhost:7701 -GraphQL: http://localhost:7702 -``` - -Check what's available: - -```bash -curl http://localhost:7700/menu -``` - -Mount a cartridge: - -```bash -curl -X POST http://localhost:7700/cartridges/database-mcp/load -``` - -Use it: - -```bash -curl http://localhost:7700/cartridges/database-mcp/invoke \ - -X POST -H 'Content-Type: application/json' \ - -d '{"tool":"status","args":""}' -``` - -Or skip HTTP entirely and use MCP mode for AI tools: - -```bash -./boj-server --mcp -``` - -This gives you a JSON-RPC 2.0 stdio server. All 53 cartridges appear as MCP tools. Add it to your MCP client config and your AI sees one server with 53 capabilities instead of 53 separate servers. - -## Federation: the community IS the hosting - -I don't have a hosting budget. I'm a solo developer. But I do have an idea: what if the community *is* the infrastructure? - -BoJ includes a federation system called **Umoja** (Swahili for "unity"). Any community member can run a node: - -1. Pull the container image (Chainguard base, Podman -- never Docker) -2. Run it -3. Your node joins the network via IPv6 gossip protocol -4. Requests route to healthy nodes automatically - -The trust model is hash attestation. Every BoJ binary has a SHA-256 hash. Your node proves its binary matches the canonical build. If someone modifies their binary, they're excluded from the community network -- but they can still run it locally. Non-punitive. We don't brick your installation. We just don't vouch for it. - -The gossip protocol is Byzantine fault tolerant. Nodes exchange peer lists, stale nodes get deprioritised, load-aware routing sends requests to nodes under 80% capacity. The federation transport uses QUIC with X25519 key exchange and ChaCha20-Poly1305 encryption, falling back to plain UDP when QUIC isn't available. - -Four seed nodes across four continents from day one. No cloud bill. Just people running software on computers they already own. - -## The HAT concept: don't throw away your tools - -I want to address something directly: BoJ is not trying to replace your database MCP server or your Kubernetes CLI. Those tools are good. People built them with care. - -What BoJ does is give them a uniform shopfront. Think of it like a Hardware store with an Attached Toolshed -- a **HAT**. You don't throw away your hammer when you buy a toolbox. You put it in the toolbox so you can find it. - -The third axis of the matrix (the one I haven't mentioned yet) is the **backend** axis. Each cartridge has a backend field -- by default it's `"universal"`, but community extensions can specialise it. Want a `database-mcp` cartridge that talks specifically to PostgreSQL? That's a backend variant. The core cartridge defines the interface (via Idris2 proofs); the backend variant implements it for a specific provider. - -The community submission system (called "Ayo", meaning "joy" in Yoruba) lets anyone submit a cartridge variant. It goes through a review state machine (`submitted -> under_review -> approved`), and approved cartridges appear in the community tier of the Teranga menu. - -## The honest assessment - -BoJ is Alpha. Grade D on most cartridges. Here's what's real: - -- 53 cartridges, all with ABI + FFI + adapter layers, all compiling to `.so` files -- 219 Zig tests, 8 integration tests, 32 seam checks — all passing -- Thread-safety hardening across all 9 core + 53 cartridge FFI modules (all mutex-protected) -- MCP stdio bridge working (JSON-RPC 2.0, all 53 cartridges as tools) -- Umoja federation with real QUIC+UDP networking (22 federation tests) -- Zero `believe_me` in production code (Idris2's escape hatch -- we don't use it) -- Security audit by panic-attack scanner: 1 expected weak point (QUIC 0-RTT replay window, mitigated at app layer), 0 critical vulnerabilities - -Here's what's not done: - -- No real users besides me -- Seed nodes aren't deployed to actual infrastructure yet -- The Teranga menu runtime is 30% complete (the spec exists, the runtime doesn't) -- Most cartridges are Grade D (they work, but they haven't been tested by diverse external users) -- No domain name yet for the federation network - -This isn't a product launch. This is a "come look at what I built and tell me what's wrong with it" post. - -## Try it, host a node, or just tell me what you think - -Install: `npm install -g @hyperpolymath/boj-server` or `guix build github:hyperpolymath/boj-server` - -The repo is at [github.com/hyperpolymath/boj-server](https://github.com/hyperpolymath/boj-server). The quickstart guide is [docs/quickstarts/DEV.adoc](https://github.com/hyperpolymath/boj-server/blob/main/docs/quickstarts/DEV.adoc). - -What I'm looking for: - -- **Try it locally** -- clone, build, run `curl http://localhost:7700/matrix` and tell me if the experience makes sense -- **Host a node** -- if you have a machine that's on during the day, you can join the Umoja network. See `container/` in the repo -- **Build on it** -- the extensibility system lets you add backend variants without touching core code. See `docs/EXTENSIBILITY.adoc` -- **Tell me what breaks** -- feedback-mcp is literally a cartridge that collects feedback. BoJ dogfoods itself - -This is a community project. I make nothing from it. The license (MPL-2.0) ensures the code stays open and provenance-tracked. - -I built this to learn from it, and I learn most from other people using it. - ---- - -*Jonathan D.A. Jewell builds developer tools and formal verification systems. BoJ is part of the hyperpolymath ecosystem of open-source projects.* diff --git a/docs/outreach/show-hn-draft.adoc b/docs/outreach/show-hn-draft.adoc new file mode 100644 index 00000000..350fcf7f --- /dev/null +++ b/docs/outreach/show-hn-draft.adoc @@ -0,0 +1,41 @@ +== Show HN: BoJ — 99-cartridge MCP server with formal proofs (Idris2 + Zig) + +BoJ (Bundle of Joy) is an MCP server that bundles 99 tool cartridges — +each with a formally verified ABI (Idris2), a C-compatible FFI (Zig), +and a unified adapter exposing REST, gRPC, GraphQL, and SSE on four +ports. + +What makes it different: + +* *99 cartridges* covering cloud (Cloudflare, Vercel), comms (Gmail, +calendar), GitHub/GitLab, databases, containers, security (DNS Shield, +container hash monitoring, licence-chain provenance via pmpl-mcp), +browsers, and more +* *Formal safety proofs* — every cartridge has an Idris2 ABI module with +dependent types and zero `+believe_me+` postulates. The type system +prevents entire classes of runtime errors +* *Zero Python, zero TypeScript* — built with Zig (FFI), Idris2 +(proofs), and a AffineScript UI. No npm, no pip, no node_modules +* *Glama AAA grade* — Security A, License A, Quality A +* *Federation-ready* — Umoja gossip protocol with QUIC transport, hash +attestation, 4 seed node configs + +The architecture follows the "`ABI/FFI/API triple`" pattern: Idris2 +proves the interface correct at compile time, Zig implements it with C +ABI compatibility, and the adapter provides the user-facing API. Adding +a new cartridge means writing ~600 lines across 3 files. + +Running locally: + +.... +git clone https://github.com/hyperpolymath/boj-server +cd boj-server && just build && just serve +# REST :7700 | gRPC :7701 | GraphQL :7702 | SSE :7703 +.... + +Quickstart: +https://github.com/hyperpolymath/boj-server/blob/main/docs/quickstarts/USER.adoc + +MPL-2.0 license (MPL-2.0 legal fallback; OSI submission pending). + +GitHub: https://github.com/hyperpolymath/boj-server diff --git a/docs/outreach/show-hn-draft.md b/docs/outreach/show-hn-draft.md deleted file mode 100644 index 6ab6783b..00000000 --- a/docs/outreach/show-hn-draft.md +++ /dev/null @@ -1,31 +0,0 @@ - -# Show HN: BoJ — 99-cartridge MCP server with formal proofs (Idris2 + Zig) - -BoJ (Bundle of Joy) is an MCP server that bundles 99 tool cartridges — each with a formally verified ABI (Idris2), a C-compatible FFI (Zig), and a unified adapter exposing REST, gRPC, GraphQL, and SSE on four ports. - -What makes it different: - -- **99 cartridges** covering cloud (Cloudflare, Vercel), comms (Gmail, calendar), GitHub/GitLab, databases, containers, security (DNS Shield, container hash monitoring, licence-chain provenance via pmpl-mcp), browsers, and more -- **Formal safety proofs** — every cartridge has an Idris2 ABI module with dependent types and zero `believe_me` postulates. The type system prevents entire classes of runtime errors -- **Zero Python, zero TypeScript** — built with Zig (FFI), Idris2 (proofs), and a AffineScript UI. No npm, no pip, no node_modules -- **Glama AAA grade** — Security A, License A, Quality A -- **Federation-ready** — Umoja gossip protocol with QUIC transport, hash attestation, 4 seed node configs - -The architecture follows the "ABI/FFI/API triple" pattern: Idris2 proves the interface correct at compile time, Zig implements it with C ABI compatibility, and the adapter provides the user-facing API. Adding a new cartridge means writing ~600 lines across 3 files. - -Running locally: - -``` -git clone https://github.com/hyperpolymath/boj-server -cd boj-server && just build && just serve -# REST :7700 | gRPC :7701 | GraphQL :7702 | SSE :7703 -``` - -Quickstart: https://github.com/hyperpolymath/boj-server/blob/main/docs/quickstarts/USER.adoc - -MPL-2.0 license (MPL-2.0 legal fallback; OSI submission pending). - -GitHub: https://github.com/hyperpolymath/boj-server diff --git a/docs/outreach/show-hn-post.adoc b/docs/outreach/show-hn-post.adoc new file mode 100644 index 00000000..8ec433c0 --- /dev/null +++ b/docs/outreach/show-hn-post.adoc @@ -0,0 +1,43 @@ +== Show HN Draft + +*Title:* Show HN: BoJ – One MCP server, 99 capability cartridges, zero +Python + +''''' + +*Body:* + +I had three Claude instances, a Cursor session, and about twenty +MCP/LSP/DAP servers running. My desktop froze. That was the moment I +realised the problem wasn’t any individual server — it was the +combinatoric explosion of them. + +BoJ (Bundle of Joy) is a single MCP server that covers 99 capability +domains through swappable cartridges. Database, containers, git, +secrets, queues, IaC, observability, static sites, proofs, fleet +management, neurosymbolic AI, agent orchestration, cloud, Kubernetes, +LSP, DAP, BSP, feedback, and more. Each cartridge has a formally +verified interface (Idris2 dependent types prove the safety gate at +compile time), a Zig FFI layer for native execution, and a unified +adapter that exposes REST + gRPC + GraphQL + SSE on four ports. Five +safety modules (SafeHTTP, SafePromptInjection, SafeCORS, SafeAPIKey, +SafeWebSocket) guard the boundary. + +The architecture is a 2D matrix: protocols on one axis, domains on the +other. Instead of N separate servers, you get one catalogue where AI +agents read a menu and mount what they need. Federation is built in — +community nodes discover each other via gossip with hash attestation, so +you can self-host a node and join the network without any central +coordination. + +Install: `+deno install -g npm:@hyperpolymath/boj-server+` or +`+brew install hyperpolymath/tap/boj-server+` + +The whole thing is Alpha — it needs real users doing real things. + +Repo: https://github.com/hyperpolymath/boj-server Quickstart: +https://github.com/hyperpolymath/boj-server/blob/main/docs/quickstarts/USER.adoc + +This is a community project. I make nothing from it. The code is +MPL-2.0-licensed. I built this to learn from it, and I learn most from +other people using it. diff --git a/docs/outreach/show-hn-post.md b/docs/outreach/show-hn-post.md deleted file mode 100644 index 9e505601..00000000 --- a/docs/outreach/show-hn-post.md +++ /dev/null @@ -1,29 +0,0 @@ - - - - -# Show HN Draft - -**Title:** Show HN: BoJ – One MCP server, 99 capability cartridges, zero Python - ---- - -**Body:** - -I had three Claude instances, a Cursor session, and about twenty MCP/LSP/DAP servers running. My desktop froze. That was the moment I realised the problem wasn't any individual server — it was the combinatoric explosion of them. - -BoJ (Bundle of Joy) is a single MCP server that covers 99 capability domains through swappable cartridges. Database, containers, git, secrets, queues, IaC, observability, static sites, proofs, fleet management, neurosymbolic AI, agent orchestration, cloud, Kubernetes, LSP, DAP, BSP, feedback, and more. Each cartridge has a formally verified interface (Idris2 dependent types prove the safety gate at compile time), a Zig FFI layer for native execution, and a unified adapter that exposes REST + gRPC + GraphQL + SSE on four ports. Five safety modules (SafeHTTP, SafePromptInjection, SafeCORS, SafeAPIKey, SafeWebSocket) guard the boundary. - -The architecture is a 2D matrix: protocols on one axis, domains on the other. Instead of N separate servers, you get one catalogue where AI agents read a menu and mount what they need. Federation is built in — community nodes discover each other via gossip with hash attestation, so you can self-host a node and join the network without any central coordination. - -Install: `deno install -g npm:@hyperpolymath/boj-server` or `brew install hyperpolymath/tap/boj-server` - -The whole thing is Alpha — it needs real users doing real things. - -Repo: https://github.com/hyperpolymath/boj-server -Quickstart: https://github.com/hyperpolymath/boj-server/blob/main/docs/quickstarts/USER.adoc - -This is a community project. I make nothing from it. The code is MPL-2.0-licensed. I built this to learn from it, and I learn most from other people using it. diff --git a/docs/papers/boj-architecture-paper.adoc b/docs/papers/boj-architecture-paper.adoc new file mode 100644 index 00000000..58367952 --- /dev/null +++ b/docs/papers/boj-architecture-paper.adoc @@ -0,0 +1,847 @@ +== Formally Verified Capability Catalogues: Dependent Types for Safe Server Plugin Architectures + +*Jonathan D.A. Jewell* The Open University, Milton Keynes, United +Kingdom `+j.d.a.jewell@open.ac.uk+` + +''''' + +=== Abstract + +The proliferation of Model Context Protocol (MCP) servers has created a +combinatoric explosion in the AI tooling ecosystem: N wire protocols +multiplied by M capability domains yields N x M separate server +processes, each with independent failure modes and no formal safety +guarantees. A recent empirical study of 1,899 MCP servers found that 66% +exhibited code smells and only 2.6% included any form of static analysis +[1]. We present the Bundle of Joy (BoJ) server, a single-binary +architecture that collapses the N x M explosion into a two-dimensional +capability matrix of formally verified _cartridges_. Each cartridge +occupies a cell in the matrix (protocol type x capability domain) and +must satisfy an `+IsUnbreakable+` mounting predicate expressed as a +dependent-type proof in Idris 2 before the Zig FFI layer will execute +it. The system comprises 18 cartridges spanning 9 protocol types and 14 +capability domains, validated by 307 tests including 15 cross-layer seam +checks. Static analysis via panic-attack reports zero critical +vulnerabilities and zero cross-language taint paths. The architecture +compiles to a single 18 MB binary with sub-millisecond cartridge +invocation latency and approximately 100 MB memory footprint with all +cartridges loaded. A SWIM-inspired gossip federation layer (Umoja) +enables community-hosted instances with cryptographic hash attestation +for binary integrity. To our knowledge, this is the first system to use +dependent types as a compile-time gate controlling foreign function +interface execution in a production server plugin architecture. + +*Keywords:* dependent types, formal verification, plugin architectures, +Model Context Protocol, foreign function interface, capability-based +security + +*ACM CCS:* Software and its engineering -> Software architectures; +Software and its engineering -> Formal software verification + +''''' + +=== 1. Introduction + +==== 1.1 The Combinatoric Explosion + +Modern AI development environments require servers that expose tools, +resources, and prompts to language model agents. The Model Context +Protocol (MCP) [2] has emerged as a de facto standard for this +integration, but it addresses only one axis of the problem: _how_ an +agent communicates with a server. The _what_ — the capability domain — +is left to individual server implementations. + +In practice, a developer who needs database operations, container +management, and observability must run three separate MCP servers. If +they also need Language Server Protocol (LSP) integration for editor +support and Debug Adapter Protocol (DAP) for debugger attachment, the +count multiplies further. With 9 common wire protocols and 14 +infrastructure domains, the theoretical maximum is 126 distinct servers, +each with its own process, failure mode, configuration surface, and +security boundary. + +This is not a theoretical concern. A survey of the MCP ecosystem [1] +catalogued 1,899 publicly available servers as of mid-2025. The study +found that 66% exhibited at least one code smell, 43% had no input +validation, and only 2.6% employed static analysis. The root cause is +structural: when each protocol-domain intersection is a separate +project, quality assurance effort scales linearly with the number of +servers rather than amortising across a shared architecture. + +==== 1.2 The Matrix Insight + +We observe that the protocol dimension (MCP, LSP, DAP, BSP, gRPC, REST, +etc.) and the domain dimension (database, container, Kubernetes, Git, +secrets, etc.) are _orthogonal_. A database capability should be +expressible over MCP _and_ gRPC _and_ REST without reimplementing the +domain logic. Similarly, the MCP wire protocol should be reusable across +databases _and_ containers _and_ observability without duplicating +framing code. + +This orthogonality suggests a two-dimensional matrix architecture where +each cell is a _cartridge_ — a verified, hot-swappable capability +module. The matrix has a natural extension point: an optional third axis +for backend specialisation (e.g., "`postgresql`" vs. "`mysql`" within +the database domain), enabling community contributions without modifying +core infrastructure. + +==== 1.3 The Safety Gap + +Existing server frameworks provide no formal guarantee that a plugin is +safe to load. Dynamic loading via `+dlopen+` or equivalent mechanisms +trusts the plugin binary unconditionally. Configuration errors, version +mismatches, and corrupted binaries are detected only at runtime — if at +all. + +We close this gap by introducing a three-layer architecture: + +[arabic] +. *Idris 2 ABI layer*: defines cartridge types, lifecycle states, and +the `+IsUnbreakable+` mounting predicate using dependent types with +quantitative type theory [3]. +. *Zig FFI layer*: implements the C-ABI runtime that enforces the +`+IsUnbreakable+` invariant at the native execution boundary, with +thread safety via `+std.Thread.Mutex+`. +. *zig adapter layer*: exposes the verified cartridges over REST, gRPC, +and GraphQL endpoints, plus MCP stdio transport. + +The key contribution is the _safety gate_: the Zig layer refuses to +mount any cartridge whose status integer does not equal `+1+` (Ready), +which is the sole constructor witness for `+IsUnbreakable+` in the Idris +2 proof. This creates a formally grounded compile-time constraint that +is enforced at the FFI execution boundary. + +==== 1.4 Contributions + +This paper makes the following contributions: + +* A *two-dimensional capability matrix* architecture that reduces N x M +server proliferation to a single binary with formally verified +cartridges (Section 3). +* The *IsUnbreakable proof pattern*: a dependent-type mounting predicate +that gates FFI execution, closing the gap between formal verification +and native runtime (Section 3.3). +* *Umoja federation*: a SWIM-inspired gossip protocol with cryptographic +hash attestation for community-hosted server instances (Section 4). +* *Empirical evaluation* across 307 tests, static analysis, and resource +profiling demonstrating the practical viability of the approach (Section +5). +* The *HAT extensibility model* for bridging verified cartridges to +unverified ecosystem tools without compromising core safety invariants +(Section 6). + +''''' + +=== 2. Background and Related Work + +==== 2.1 The Model Context Protocol + +The Model Context Protocol (MCP), originally developed by Anthropic and +now stewarded by the Linux Foundation AI & Data group [2], defines a +JSON-RPC 2.0 framing protocol for exposing tools, resources, and prompts +to language model agents. MCP servers communicate via stdio or HTTP+SSE +transport. + +Qin et al. [1] conducted the first large-scale empirical study of the +MCP ecosystem, analysing 1,899 servers. Their findings are sobering: 66% +contain code smells, 43% lack input validation, and the median server +implements fewer than 5 tools. The study identifies a fundamental +tension between rapid ecosystem growth and software quality. + +==== 2.2 Server Collaboration and Capability Negotiation + +Puttaswamy et al. [4] propose context-aware collaboration between +multiple MCP servers, where servers share contextual information to +provide more coherent responses. Their work assumes servers are +independently deployed black boxes, which is precisely the architectural +constraint we eliminate. + +Li et al. [5] introduce agent-level capability negotiation protocols, +enabling agents to discover and compose server capabilities at runtime. +Our capability matrix formalises the same discovery problem but resolves +it at compile time through the Idris 2 type system rather than at +runtime through protocol negotiation. + +==== 2.3 Dependent Types in Systems Programming + +Brady [3] presents Idris 2 and its foundation in quantitative type +theory (QTT), which tracks resource usage at the type level. QTT enables +erasure of computationally irrelevant terms while preserving their proof +obligations, making dependent types practical for systems programming. + +Prior work has applied dependent types to verified compilers [6], +network protocol specifications [7], and operating system kernels [8]. +However, we are not aware of any system that uses dependent types +specifically to gate foreign function interface execution in a plugin +architecture. The closest related work is Idris 2’s own FFI mechanism, +which provides type safety for individual foreign calls but does not +enforce lifecycle or capability predicates across a catalogue of +plugins. + +==== 2.4 FFI Safety + +Foreign function interface safety has been studied extensively in the +context of Rust’s `+unsafe+` blocks [9], Haskell’s `+unsafePerformIO+` +[10], and OCaml’s `+Obj.magic+` [11]. These mechanisms all share a +common weakness: they create an unchecked boundary where type-level +guarantees are suspended. + +Our architecture differs by maintaining the proof obligation _across_ +the language boundary. The Idris 2 proof constrains which cartridges +_can_ be mounted; the Zig runtime enforces a runtime check that is +_structurally identical_ to the proof witness (status integer = 1). This +is not full formal verification of the Zig code, but it is a disciplined +correspondence that we validate through seam checks (Section 5.1). + +''''' + +=== 3. Architecture + +==== 3.1 The Capability Matrix + +The BoJ capability matrix is a sparse two-dimensional structure. The +columns are _protocol types_ — the wire protocols through which +capabilities are exposed: + +[cols=",,",options="header",] +|=== +|Column |Protocol |Purpose +|1 |MCP |Model Context Protocol (AI agent integration) +|2 |LSP |Language Server Protocol (editor integration) +|3 |DAP |Debug Adapter Protocol (debugger attachment) +|4 |BSP |Build Server Protocol (build system integration) +|5 |NeSy |Neurosymbolic Protocol (hybrid reasoning) +|6 |Agentic |Agentic Protocol (OODA loop orchestration) +|7 |Fleet |Fleet Protocol (multi-bot coordination) +|8 |gRPC |High-performance remote procedure calls +|9 |REST |HTTP/JSON (universal fallback) +|=== + +The rows are _capability domains_ — the classes of infrastructure +operation: + +[cols=",,",options="header",] +|=== +|Row |Domain |Examples +|1 |Cloud |AWS, GCP, Azure provider operations +|2 |Container |Podman, OCI image management +|3 |Database |SQL, NoSQL, VeriSimDB +|4 |Kubernetes |Cluster orchestration +|5 |Git/VCS |GitHub, GitLab, Bitbucket +|6 |Secrets |Vault, SOPS, sealed-secrets +|7 |Queues |NATS, RabbitMQ, Kafka +|8 |IaC |Terraform, Pulumi, Guix +|9 |Observability |Metrics, logs, traces +|10 |SSG |Jekyll, Hugo, Zola +|11 |Proof |Idris 2, Lean, Coq assistants +|12 |Fleet |Gitbot fleet orchestration +|13 |NeSy |Neurosymbolic reasoning +|14 |Feedback |User feedback collection +|=== + +Each cartridge occupies one or more cells. For example, `+database-mcp+` +occupies cells (MCP, Database), (gRPC, Database), and (REST, Database). +The matrix is sparse: not every cell needs to be filled. The current +deployment has 18 cartridges populating approximately 54 cells. + +An optional third axis — the _backend_ dimension — allows community +extensions to specialise a cartridge for a particular provider (e.g., +`+database-mcp+` with backend "`postgresql`" vs. backend "`mysql`") +without forking the core cartridge. By default, all cartridges use the +backend label `+"universal"+`. + +==== 3.2 The Three-Layer Stack + +Each cartridge is implemented as a three-layer vertical slice: + +.... + +-----------------+ + | Idris 2 ABI | Formal specification + proofs + | (SafeX.idr) | IsUnbreakable, lifecycle, types + +-----------------+ + | + | C-ABI integer encoding + v + +-----------------+ + | Zig FFI | Native execution + | (x_ffi.zig) | Thread-safe, zero runtime deps + +-----------------+ + | + | dlopen / direct link + v + +-----------------+ + | zig Adapter | Network exposure + | (x_adapter.v) | REST + gRPC + GraphQL + MCP stdio + +-----------------+ +.... + +The *Idris 2 ABI layer* defines the cartridge’s type signature, +lifecycle states, and domain-specific safety properties. Every ABI +module sets `+%default total+`, requiring that all functions be provably +terminating. The module exports C-ABI encoding functions +(`+statusToInt+`, `+domainToInt+`, `+protocolToInt+`) that establish the +integer contract between the proof layer and the execution layer. + +The *Zig FFI layer* implements the cartridge’s runtime behaviour as +C-ABI exports. All global state is protected by `+std.Thread.Mutex+`, +acquired at the export boundary and released via `+defer+`. The layer +enforces the `+IsUnbreakable+` invariant at mount time: +`+boj_catalogue_mount+` returns `+-1+` if the cartridge’s status integer +is not `+1+` (Ready). + +The *zig adapter layer* exposes cartridges over network protocols. A +single adapter binary serves REST on port 7700, gRPC on port 7701, and +GraphQL on port 7702. MCP transport is available via stdio for direct +agent integration. + +==== 3.3 The Safety Gate + +The core safety mechanism is the `+IsUnbreakable+` dependent type: + +[source,idris] +---- +data IsUnbreakable : Cartridge -> Type where + VerifiedReady : (c : Cartridge) -> + (status c = Ready) -> + IsUnbreakable c +---- + +`+IsUnbreakable+` has exactly one constructor, `+VerifiedReady+`, which +requires a proof that the cartridge’s status field equals `+Ready+`. +This is a _type-level mounting predicate_: any function that requires an +`+IsUnbreakable+` proof can only be called with a cartridge that has +been proven ready at the type level. + +The corresponding Zig enforcement is structurally aligned: + +[source,zig] +---- +pub export fn boj_catalogue_mount(index: usize) c_int { + mutex.lock(); + defer mutex.unlock(); + if (!initialised or index >= catalogue_count) return -2; + if (catalogue[index].status != .ready) return -1; // IsUnbreakable + catalogue[index].mounted = true; + return 0; +} +---- + +The `+status != .ready+` check is the runtime mirror of +`+status c = Ready+`. The integer encoding is fixed by the ABI: +`+CartridgeStatus.ready = 1+` in both Idris 2 +(`+statusToInt Ready = 1+`) and Zig (`+ready = 1+`). This correspondence +is validated by 15 seam checks (Section 5.1) that verify enum encoding +alignment across the language boundary. + +We emphasise what this is _not_: it is not a machine-checked proof that +the Zig code correctly implements the Idris 2 specification. It is a +disciplined architectural pattern where: + +[arabic] +. The proof layer defines the _only_ way a cartridge can be marked safe. +. The encoding layer fixes a stable integer representation. +. The execution layer checks the _same_ integer predicate. +. Seam checks validate that the encodings have not drifted. + +This pattern is weaker than full verified compilation but stronger than +the status quo of unchecked dynamic loading. We discuss paths toward +stronger guarantees in Section 7. + +==== 3.4 Thread Safety + +The BoJ FFI layer protects all mutable global state with module-level +mutexes. Across the 9 core FFI modules (catalogue, loader, federation, +verisimdb, guardian, coprocessor, sla, community, sdp) and 18 cartridge +FFI modules, every C-ABI export follows the same pattern: + +[source,zig] +---- +pub export fn boj_xxx_operation(...) return_type { + mutex.lock(); + defer mutex.unlock(); + // ... operation on protected state ... +} +---- + +This yields 120 mutex-protected export functions across 55 global state +variables. The `+defer+` keyword ensures unlock on all control flow +paths, including early returns and error cases. + +Deadlock prevention follows a simple discipline: internal implementation +functions (suffixed `+_impl+`) are called only while the mutex is held +and never re-acquire it. Re-entrant paths (e.g., mounting a cartridge +that triggers a federation notification) use the `+_impl+` variant to +avoid double-locking. + +''''' + +=== 4. The Umoja Federation + +==== 4.1 Design Goals + +The Umoja federation layer enables community-hosted BoJ instances to +form a decentralised network. The design goals are: + +[arabic] +. *Zero trust by default*: nodes prove their identity through +cryptographic hash attestation of their binary. +. *Eventually consistent*: catalogue state synchronises through +anti-entropy gossip rounds. +. *Partition tolerant*: nodes operate independently during network +partitions and reconcile on reconnection. +. *Georedundant*: seed nodes span multiple continents (EU-West, +EU-Central, US-East, AP-South). + +==== 4.2 SWIM-Inspired Gossip Protocol + +Node liveness detection follows the SWIM protocol model [12]: + +* *Alive*: node is responsive and participating. +* *Suspected*: node has missed heartbeats (configurable threshold). +* *Dead*: node has been confirmed unreachable by multiple peers. + +The gossip layer uses UDP multicast on a link-local IPv6 address +(`+ff02::b04+`) for peer discovery, with point-to-point UDP for +subsequent communication. The wire protocol defines 7 packet types: + +[cols=",,",options="header",] +|=== +|Tag |Packet |Direction +|`+0x01+` |Discover |Broadcast +|`+0x02+` |DiscoverReply |Unicast +|`+0x03+` |GossipDigest |Unicast +|`+0x04+` |GossipDigestReply |Unicast +|`+0x05+` |HandshakeInit |Unicast +|`+0x06+` |HandshakeReply |Unicast +|`+0x07+` |Heartbeat |Unicast +|=== + +==== 4.3 Hash Attestation + +The trust model is built on the `+Attested+` dependent type: + +[source,idris] +---- +data Attested : (n : Node) -> (canonicalHash : String) -> Type where + ValidAttestation : (n : Node) -> + (canonicalHash : String) -> + (binaryHash n = canonicalHash) -> + Attested n canonicalHash +---- + +A node can participate in the community network only if its binary hash +matches the canonical hash published with each release. Nodes with +non-matching hashes are not excluded from running BoJ locally — they +simply cannot join the federated network. This prevents tampered +binaries from serving community requests while preserving the user’s +right to modify their own copy. + +==== 4.4 Anti-Entropy Synchronisation + +Catalogue synchronisation uses SHA-256 digest comparison. Each node +computes a digest over its registered cartridges (name, version, status +triples). During a gossip round, nodes exchange digests; if they differ, +the node with the newer catalogue sends a full catalogue update. This is +a pull-based anti-entropy mechanism [13] that converges in O(log N) +gossip rounds for N nodes. + +==== 4.5 Auto-SDP Zero-Trust Perimeter + +The federation includes a Session Description Protocol (SDP)-inspired +perimeter that maintains an allow-list of attested nodes, automatically +banning nodes that fail attestation or exhibit anomalous behaviour. The +perimeter operates on three principles: (1) all connections require +attestation before data exchange, (2) failed attestations result in +immediate connection termination, and (3) repeated failures trigger +automatic banning with exponential backoff for re-admission. + +''''' + +=== 5. Evaluation + +==== 5.1 Test Coverage + +The BoJ test suite comprises 307 tests across five categories: + +[width="100%",cols="42%,29%,29%",options="header",] +|=== +|Category |Count |Scope +|Core FFI |178 |catalogue (13), loader (14), federation (40), guardian +(12), readiness (28), VeriSimDB (7), e2e order-ticket (3), coprocessor +(14), SLA (11), community (11), SDP (10), seams (15) + +|Cartridge FFI |118 |18 cartridges, each with dedicated test suites + +|Multi-node federation |11 |Peer management, gossip rounds, attestation + +|Total |307 | +|=== + +The 15 seam checks deserve particular attention. These are integration +contract validation tests inspired by the panic-attack diagnostic +pattern [14]. They verify that integer encodings for all enumerations +(`+CartridgeStatus+`, `+ProtocolType+`, `+CapabilityDomain+`, +`+MenuTier+`, `+CircuitState+`, `+Severity+`) are identical between the +Idris 2 ABI definitions and the Zig FFI implementations. A seam check +failure indicates that the proof-to-execution correspondence has been +broken, which is a genuine architectural defect. + +The seam check philosophy is the "`silent signature`": if all checks +pass, the integration surface is verified and there is nothing to +report. Any failure is actionable. + +==== 5.2 Static Analysis + +We analysed the BoJ codebase using panic-attack [14], a static +vulnerability scanner designed for multi-language codebases. The +`+assail+` mode performs taint analysis across language boundaries. + +[cols=",,",options="header",] +|=== +|Finding |Count |Severity +|QUIC crypto dependency (expected) |1 |Weak +|Critical Zig vulnerabilities |0 |— +|Cross-language taint paths |0 |— +|Tainted production paths |0 |— +|=== + +The single weak finding relates to the QUIC transport’s reliance on +X25519 key exchange and ChaCha20-Poly1305 authenticated encryption. +These are industry-standard algorithms, but the scanner correctly flags +any cryptographic dependency as requiring ongoing review. The finding is +expected and documented. + +Notably, the codebase contains zero instances of `+believe_me+` (Idris +2’s escape hatch for admitting unproven propositions). All proofs in the +ABI layer are constructive. + +==== 5.3 Resource Footprint + +[cols=",",options="header",] +|=== +|Metric |Value +|Binary size (all 18 cartridges) |~18 MB +|Memory at rest (all cartridges loaded) |~100 MB +|Cartridge mount latency |< 1 ms +|Cartridge invoke latency |< 1 ms +|Federation gossip round |~50 ms (LAN) +|Concurrent cartridge limit |128 (compile-time constant) +|=== + +The resource profile is modest by contemporary standards. A single BoJ +instance replaces what would otherwise be 18+ separate server processes, +each consuming its own memory, file descriptors, and process table +entries. The "`3 Claude instances + 20 MCP servers = frozen desktop`" +incident that motivated the Guardian module (Section 3.2) consumed 22 GB +of available RAM through unchecked process spawning; a single BoJ +instance would have served the same capability surface in approximately +100 MB. + +==== 5.4 Language Purity + +[cols=",,",options="header",] +|=== +|Language |Source files |Role +|Zig |50 |FFI execution, thread safety, native runtime +|Idris 2 |26 |ABI specification, proofs, type-level safety +|zig |19 |Network adapter (REST, gRPC, GraphQL) +|*Total* |*95* | +|=== + +The codebase contains zero Python, JavaScript, Go, or Rust. Each +language is used for what it does best: Idris 2 for proofs and +type-level reasoning, Zig for zero-overhead native execution with C ABI +compatibility, and zig for ergonomic network server implementation with +built-in HTTP and JSON support. + +==== 5.5 Comparison with the MCP Ecosystem + +To contextualise the BoJ architecture, we compare against the aggregate +statistics from the MCP ecosystem study [1]: + +[width="100%",cols="22%,66%,12%",options="header",] +|=== +|Metric |MCP ecosystem median [1] |BoJ +|Tools per server |< 5 |18 cartridges (54+ matrix cells) + +|Input validation |57% of servers |100% (type-enforced) + +|Static analysis |2.6% of servers |panic-attack + seam checks + +|Formal verification |0% of servers |IsUnbreakable proof on every +cartridge + +|Code smell rate |66% of servers |0 reported by panic-attack +|=== + +We acknowledge that this comparison is not entirely fair: BoJ is a +single carefully engineered system, while the ecosystem study measures a +population of independently developed servers with varying goals and +resources. The comparison illustrates what _is achievable_ with a matrix +architecture, not what is _typical_. + +''''' + +=== 6. The HAT Extensibility Model + +==== 6.1 Motivation + +The BoJ architecture is intentionally restrictive: only cartridges that +pass the `+IsUnbreakable+` proof can be mounted. This safety property +would be undermined if cartridges could invoke arbitrary external tools +directly. Yet practical utility requires integration with real-world +tools (Git CLI, Docker/Podman, Terraform, etc.) that cannot be formally +verified. + +==== 6.2 Hardware Attached on Top + +(The HAT model is also specified normatively in +`+docs/specification/cartridges/README.md+` §5.) + +We resolve this tension with the HAT (Hardware Attached on Top) model, +named by analogy with the Raspberry Pi and BeagleBone hardware extension +ecosystem. A HAT is a bridge script that translates a BoJ cartridge +invocation into a real-world tool call: + +.... + BoJ Cartridge (verified) ---> HAT Bridge ---> External Tool (unverified) + git-mcp git_hat.sh git CLI + container-mcp podman_hat.sh podman + ssg-mcp zola_hat.sh zola +.... + +The bridge is _outside_ the BoJ safety perimeter. The cartridge’s formal +properties (lifecycle, type safety, thread safety) are preserved +regardless of what the HAT does. If a HAT fails, the cartridge’s circuit +breaker (Section 3.2) trips, isolating the failure. + +==== 6.3 Bridge Types + +We define four bridge categories: + +[arabic] +. *CLI wrapper*: the HAT invokes a command-line tool and parses its +output. Simplest and most common (e.g., `+git+`, `+zola+`, `+podman+`). +. *JSON-RPC stdio*: the HAT speaks JSON-RPC 2.0 over stdin/stdout to an +existing MCP-compatible server. This enables composition with the +existing ecosystem. +. *HTTP API*: the HAT calls a REST or GraphQL endpoint on a remote +service. +. *Library FFI*: the HAT links against a native library via C ABI. This +is the highest-performance option and is used for VeriSimDB integration. + +==== 6.4 Safety Properties + +The HAT model preserves BoJ’s safety invariants: + +* *IsUnbreakable still holds*: the cartridge’s mounting proof is +independent of HAT presence. A cartridge without a HAT is safe (it +simply has no external integration). +* *Circuit breaker isolation*: HAT failures trip the per-cartridge +circuit breaker, preventing cascade failures. +* *Thread safety preserved*: HAT invocations go through the same +mutex-protected FFI exports as internal operations. +* *Attestation unaffected*: the HAT is not part of the attested binary; +federation nodes verify only the BoJ core binary hash. + +''''' + +=== 7. Limitations and Future Work + +==== 7.1 Fixed-Size Data Structures + +The catalogue uses compile-time-constant array sizes +(`+MAX_CARTRIDGES = 128+`, `+MAX_NODES = 16+`, `+MAX_PEERS = 16+`). +These are sufficient for the current use case but impose hard upper +bounds. Switching to dynamically allocated structures would require an +allocator in the Zig FFI layer, which we have deliberately avoided to +maintain zero-runtime-dependency guarantees. + +==== 7.2 Single-Process Architecture + +BoJ is a single-process, multi-threaded server. Horizontal scaling is +achieved through federation (multiple BoJ instances) rather than +internal parallelism. Workloads requiring high throughput on a single +capability domain would benefit from a sharded architecture that we have +not yet implemented. + +==== 7.3 Proof-to-Execution Gap + +The `+IsUnbreakable+` pattern provides a disciplined correspondence +between proofs and runtime checks, but it is not a machine-checked proof +that the Zig code correctly implements the Idris 2 specification. Two +paths toward stronger guarantees exist: + +* *Idris 2 C codegen*: Idris 2 can compile to C via its RefC backend. +Extracting the catalogue state machine into C and linking it directly +into the Zig binary would eliminate the proof-to-execution gap for +lifecycle management, at the cost of introducing a C dependency. +* *Zig formal verification*: emerging tools for Zig verification (e.g., +based on LLVM-IR analysis) could provide machine-checked proofs of the +Zig implementation. + +==== 7.4 Linear Types for Thread Safety + +The current thread safety model relies on runtime mutexes. Linear types +— as supported by Idris 2’s QTT and as explored in the Ephapax language +project — could express thread ownership at the type level, making +mutex-based locking unnecessary. This would strengthen the safety +guarantees from "`correctly locked at runtime`" to "`uniquely owned at +compile time.`" + +==== 7.5 Formal Federation Verification + +The Umoja gossip protocol has been tested empirically but not formally +verified. The SWIM protocol has known convergence properties [12]; +proving that the Umoja variant preserves these properties under the hash +attestation constraint is future work. + +''''' + +=== 8. Conclusion + +We have presented the Bundle of Joy server, a formally verified plugin +architecture that collapses the N x M combinatoric explosion of protocol +servers into a single binary with a two-dimensional capability matrix. +The `+IsUnbreakable+` dependent-type predicate gates FFI execution, +ensuring that only proven-ready cartridges can be mounted. The +architecture achieves 100% input validation, zero code smells, and zero +`+believe_me+` escape hatches across 95 source files in three languages. + +The dependent type plus FFI pattern is not limited to server plugin +architectures. Any system that dynamically loads and executes untrusted +modules — browser extensions, database stored procedures, smart +contracts, operating system drivers — faces the same trust boundary. The +pattern of (1) expressing a safety predicate as a dependent type, (2) +fixing a stable encoding, and (3) enforcing the same predicate at the +execution boundary is applicable wherever formal proofs and native +execution must coexist. + +The Umoja federation layer demonstrates that formally verified servers +can participate in decentralised networks without sacrificing safety +guarantees. Hash attestation ties binary integrity to community trust, +enabling a volunteer hosting model analogous to Tor or IPFS but with +stronger provenance guarantees. + +We make the BoJ server available as open-source software under the +MPL-2.0 (MPL-2.0) at `+https://github.com/hyperpolymath/boj-server+`. + +''''' + +=== References + +[1] H. Qin, Y. Zhang, L. Chen, et al., "`An Empirical Study of MCP +Servers: Code Smells, Quality Issues, and Security Vulnerabilities,`" +arXiv preprint arXiv:2506.13538, 2025. + +[2] Model Context Protocol Specification, Linux Foundation AI & Data, +Anthropic, version 2025-11-25. https://spec.modelcontextprotocol.io/ + +[3] E. Brady, "`Idris 2: Quantitative Type Theory in Practice,`" in +_Proceedings of the 35th European Conference on Object-Oriented +Programming (ECOOP)_, 2021. arXiv:2104.00480. + +[4] S. Puttaswamy, M. Kumar, and R. Sharma, "`Context-Aware MCP Server +Collaboration for Coherent AI Agent Responses,`" arXiv preprint +arXiv:2601.11595, 2026. + +[5] J. Li, X. Wang, and H. Chen, "`Agent Capability Negotiation: Dynamic +Discovery and Composition of MCP Server Capabilities,`" arXiv preprint +arXiv:2506.13590, 2025. + +[6] X. Leroy, "`Formal Verification of a Realistic Compiler,`" +_Communications of the ACM_, vol. 52, no. 7, pp. 107-115, 2009. + +[7] N. Swamy, C. Hritcu, C. Keller, et al., “Dependent Types and +Multi-Monadic Effects in F__,” in __Proceedings of POPL*, 2016. + +[8] G. Klein, K. Elphinstone, G. Heiser, et al., "`seL4: Formal +Verification of an OS Kernel,`" in _Proceedings of the 22nd ACM +Symposium on Operating Systems Principles (SOSP)_, 2009. + +[9] R. Jung, J.-H. Jourdan, R. Krebbers, and D. Dreyer, "`RustBelt: +Securing the Foundations of the Rust Programming Language,`" +_Proceedings of the ACM on Programming Languages_, vol. 2, no. POPL, +2018. + +[10] S. Peyton Jones, "`Tackling the Awkward Squad: Monadic +Input/Output, Concurrency, Exceptions, and Foreign-Language Calls in +Haskell,`" in _Engineering Theories of Software Construction_, 2001. + +[11] J. Garrigue, "`Relaxing the Value Restriction,`" in _Functional and +Logic Programming_, Springer, 2004. + +[12] A. Das Gupta, I. Gupta, and M. Agrawal, "`SWIM: Scalable +Weakly-consistent Infection-style Process Group Membership Protocol,`" +in _Proceedings of the International Conference on Dependable Systems +and Networks (DSN)_, 2002. + +[13] A. Demers, D. Greene, C. Hauser, et al., "`Epidemic Algorithms for +Replicated Database Maintenance,`" in _Proceedings of the 6th ACM +Symposium on Principles of Distributed Computing (PODC)_, 1987. + +[14] J. D. A. Jewell, "`panic-attack: Cross-Language Static +Vulnerability Analysis for Multi-Language Codebases,`" +hyperpolymath/panic-attacker, 2026. +https://github.com/hyperpolymath/panic-attacker + +''''' + +=== Appendix A: Idris 2 ABI Module Listing + +[width="100%",cols="34%,37%,29%",options="header",] +|=== +|Module |Purpose |Lines +|`+Boj.Protocol+` |Protocol type enumeration (9 types) |78 + +|`+Boj.Domain+` |Capability domain enumeration (14 domains) |102 + +|`+Boj.Catalogue+` |Cartridge registry, IsUnbreakable proof, matrix +queries |221 + +|`+Boj.Federation+` |Umoja node identity, hash attestation, gossip |165 + +|`+Boj.Guardian+` |Resource monitoring, circuit breaker, +self-diagnostics |299 + +|`+Boj.Menu+` |Teranga menu discovery protocol |~85 + +|18 cartridge ABIs |Per-cartridge safety proofs (`+SafeX.idr+`) |~720 +(~40 each) +|=== + +=== Appendix B: Zig FFI Module Listing + +[cols=",,,",options="header",] +|=== +|Module |C-ABI Exports |Mutex-Protected Globals |Tests +|`+catalogue.zig+` |26 |6 |13 +|`+loader.zig+` |11 |2 |14 +|`+federation.zig+` |44 |21 |40 +|`+guardian.zig+` |29 |9 |12 +|`+readiness.zig+` |8 |3 |28 +|`+verisimdb.zig+` |10 |8 |7 +|`+coprocessor.zig+` |12 |5 |14 +|`+sla.zig+` |15 |4 |11 +|`+community.zig+` |9 |3 |11 +|`+sdp.zig+` |10 |6 |10 +|`+seams.zig+` |0 |0 |15 +|`+e2e_order.zig+` |0 |0 |3 +|`+bench.zig+` |0 |0 |0 (benchmark only) +|18 cartridge FFIs |4 each |1 each |118 total +|=== + +=== Appendix C: Reproducibility + +Build requirements: - Zig >= 0.15.2 - Idris 2 (any recent version with +QTT support) - zig >= 0.5.0 + +[source,sh] +---- +git clone https://github.com/hyperpolymath/boj-server +cd boj-server +just build # Build all layers +just test # Run 307 tests +just assail # Run panic-attack static analysis +just federation # Start a local 3-node federation cluster +---- diff --git a/docs/papers/boj-architecture-paper.md b/docs/papers/boj-architecture-paper.md deleted file mode 100644 index 4b77ec6b..00000000 --- a/docs/papers/boj-architecture-paper.md +++ /dev/null @@ -1,748 +0,0 @@ - - - - -# Formally Verified Capability Catalogues: Dependent Types for Safe Server Plugin Architectures - -**Jonathan D.A. Jewell** -The Open University, Milton Keynes, United Kingdom -`j.d.a.jewell@open.ac.uk` - ---- - -## Abstract - -The proliferation of Model Context Protocol (MCP) servers has created a combinatoric -explosion in the AI tooling ecosystem: N wire protocols multiplied by M capability -domains yields N x M separate server processes, each with independent failure modes and -no formal safety guarantees. A recent empirical study of 1,899 MCP servers found that -66% exhibited code smells and only 2.6% included any form of static analysis [1]. -We present the Bundle of Joy (BoJ) server, a single-binary architecture that collapses -the N x M explosion into a two-dimensional capability matrix of formally verified -*cartridges*. Each cartridge occupies a cell in the matrix (protocol type x capability -domain) and must satisfy an `IsUnbreakable` mounting predicate expressed as a -dependent-type proof in Idris 2 before the Zig FFI layer will execute it. -The system comprises 18 cartridges spanning 9 protocol types and 14 capability -domains, validated by 307 tests including 15 cross-layer seam checks. Static analysis -via panic-attack reports zero critical vulnerabilities and zero cross-language taint -paths. The architecture compiles to a single 18 MB binary with sub-millisecond -cartridge invocation latency and approximately 100 MB memory footprint with all -cartridges loaded. A SWIM-inspired gossip federation layer (Umoja) enables -community-hosted instances with cryptographic hash attestation for binary integrity. -To our knowledge, this is the first system to use dependent types as a compile-time -gate controlling foreign function interface execution in a production server plugin -architecture. - -**Keywords:** dependent types, formal verification, plugin architectures, Model Context -Protocol, foreign function interface, capability-based security - -**ACM CCS:** Software and its engineering -> Software architectures; Software and its -engineering -> Formal software verification - ---- - -## 1. Introduction - -### 1.1 The Combinatoric Explosion - -Modern AI development environments require servers that expose tools, resources, and -prompts to language model agents. The Model Context Protocol (MCP) [2] has emerged as a -de facto standard for this integration, but it addresses only one axis of the problem: -*how* an agent communicates with a server. The *what* --- the capability domain --- is -left to individual server implementations. - -In practice, a developer who needs database operations, container management, and -observability must run three separate MCP servers. If they also need Language Server -Protocol (LSP) integration for editor support and Debug Adapter Protocol (DAP) for -debugger attachment, the count multiplies further. With 9 common wire protocols and 14 -infrastructure domains, the theoretical maximum is 126 distinct servers, each with its -own process, failure mode, configuration surface, and security boundary. - -This is not a theoretical concern. A survey of the MCP ecosystem [1] catalogued 1,899 -publicly available servers as of mid-2025. The study found that 66% exhibited at least -one code smell, 43% had no input validation, and only 2.6% employed static analysis. -The root cause is structural: when each protocol-domain intersection is a separate -project, quality assurance effort scales linearly with the number of servers rather -than amortising across a shared architecture. - -### 1.2 The Matrix Insight - -We observe that the protocol dimension (MCP, LSP, DAP, BSP, gRPC, REST, etc.) and the -domain dimension (database, container, Kubernetes, Git, secrets, etc.) are -*orthogonal*. A database capability should be expressible over MCP *and* gRPC *and* -REST without reimplementing the domain logic. Similarly, the MCP wire protocol should -be reusable across databases *and* containers *and* observability without duplicating -framing code. - -This orthogonality suggests a two-dimensional matrix architecture where each cell is a -*cartridge* --- a verified, hot-swappable capability module. The matrix has a natural -extension point: an optional third axis for backend specialisation (e.g., "postgresql" -vs. "mysql" within the database domain), enabling community contributions without -modifying core infrastructure. - -### 1.3 The Safety Gap - -Existing server frameworks provide no formal guarantee that a plugin is safe to load. -Dynamic loading via `dlopen` or equivalent mechanisms trusts the plugin binary -unconditionally. Configuration errors, version mismatches, and corrupted binaries are -detected only at runtime --- if at all. - -We close this gap by introducing a three-layer architecture: - -1. **Idris 2 ABI layer**: defines cartridge types, lifecycle states, and the - `IsUnbreakable` mounting predicate using dependent types with quantitative type - theory [3]. -2. **Zig FFI layer**: implements the C-ABI runtime that enforces the `IsUnbreakable` - invariant at the native execution boundary, with thread safety via - `std.Thread.Mutex`. -3. **zig adapter layer**: exposes the verified cartridges over REST, gRPC, and - GraphQL endpoints, plus MCP stdio transport. - -The key contribution is the *safety gate*: the Zig layer refuses to mount any cartridge -whose status integer does not equal `1` (Ready), which is the sole constructor witness -for `IsUnbreakable` in the Idris 2 proof. This creates a formally grounded -compile-time constraint that is enforced at the FFI execution boundary. - -### 1.4 Contributions - -This paper makes the following contributions: - -- A **two-dimensional capability matrix** architecture that reduces N x M server - proliferation to a single binary with formally verified cartridges (Section 3). -- The **IsUnbreakable proof pattern**: a dependent-type mounting predicate that gates - FFI execution, closing the gap between formal verification and native runtime - (Section 3.3). -- **Umoja federation**: a SWIM-inspired gossip protocol with cryptographic hash - attestation for community-hosted server instances (Section 4). -- **Empirical evaluation** across 307 tests, static analysis, and resource profiling - demonstrating the practical viability of the approach (Section 5). -- The **HAT extensibility model** for bridging verified cartridges to unverified - ecosystem tools without compromising core safety invariants (Section 6). - ---- - -## 2. Background and Related Work - -### 2.1 The Model Context Protocol - -The Model Context Protocol (MCP), originally developed by Anthropic and now stewarded -by the Linux Foundation AI & Data group [2], defines a JSON-RPC 2.0 framing protocol -for exposing tools, resources, and prompts to language model agents. MCP servers -communicate via stdio or HTTP+SSE transport. - -Qin et al. [1] conducted the first large-scale empirical study of the MCP ecosystem, -analysing 1,899 servers. Their findings are sobering: 66% contain code smells, 43% lack -input validation, and the median server implements fewer than 5 tools. The study -identifies a fundamental tension between rapid ecosystem growth and software quality. - -### 2.2 Server Collaboration and Capability Negotiation - -Puttaswamy et al. [4] propose context-aware collaboration between multiple MCP servers, -where servers share contextual information to provide more coherent responses. Their -work assumes servers are independently deployed black boxes, which is precisely the -architectural constraint we eliminate. - -Li et al. [5] introduce agent-level capability negotiation protocols, enabling agents -to discover and compose server capabilities at runtime. Our capability matrix -formalises the same discovery problem but resolves it at compile time through the -Idris 2 type system rather than at runtime through protocol negotiation. - -### 2.3 Dependent Types in Systems Programming - -Brady [3] presents Idris 2 and its foundation in quantitative type theory (QTT), which -tracks resource usage at the type level. QTT enables erasure of computationally -irrelevant terms while preserving their proof obligations, making dependent types -practical for systems programming. - -Prior work has applied dependent types to verified compilers [6], network protocol -specifications [7], and operating system kernels [8]. However, we are not aware of any -system that uses dependent types specifically to gate foreign function interface -execution in a plugin architecture. The closest related work is Idris 2's own FFI -mechanism, which provides type safety for individual foreign calls but does not enforce -lifecycle or capability predicates across a catalogue of plugins. - -### 2.4 FFI Safety - -Foreign function interface safety has been studied extensively in the context of -Rust's `unsafe` blocks [9], Haskell's `unsafePerformIO` [10], and OCaml's `Obj.magic` -[11]. These mechanisms all share a common weakness: they create an unchecked boundary -where type-level guarantees are suspended. - -Our architecture differs by maintaining the proof obligation *across* the language -boundary. The Idris 2 proof constrains which cartridges *can* be mounted; the Zig -runtime enforces a runtime check that is *structurally identical* to the proof witness -(status integer = 1). This is not full formal verification of the Zig code, but it is -a disciplined correspondence that we validate through seam checks (Section 5.1). - ---- - -## 3. Architecture - -### 3.1 The Capability Matrix - -The BoJ capability matrix is a sparse two-dimensional structure. The columns are -*protocol types* --- the wire protocols through which capabilities are exposed: - -| Column | Protocol | Purpose | -|--------|----------|---------| -| 1 | MCP | Model Context Protocol (AI agent integration) | -| 2 | LSP | Language Server Protocol (editor integration) | -| 3 | DAP | Debug Adapter Protocol (debugger attachment) | -| 4 | BSP | Build Server Protocol (build system integration) | -| 5 | NeSy | Neurosymbolic Protocol (hybrid reasoning) | -| 6 | Agentic | Agentic Protocol (OODA loop orchestration) | -| 7 | Fleet | Fleet Protocol (multi-bot coordination) | -| 8 | gRPC | High-performance remote procedure calls | -| 9 | REST | HTTP/JSON (universal fallback) | - -The rows are *capability domains* --- the classes of infrastructure operation: - -| Row | Domain | Examples | -|-----|--------|----------| -| 1 | Cloud | AWS, GCP, Azure provider operations | -| 2 | Container | Podman, OCI image management | -| 3 | Database | SQL, NoSQL, VeriSimDB | -| 4 | Kubernetes | Cluster orchestration | -| 5 | Git/VCS | GitHub, GitLab, Bitbucket | -| 6 | Secrets | Vault, SOPS, sealed-secrets | -| 7 | Queues | NATS, RabbitMQ, Kafka | -| 8 | IaC | Terraform, Pulumi, Guix | -| 9 | Observability | Metrics, logs, traces | -| 10 | SSG | Jekyll, Hugo, Zola | -| 11 | Proof | Idris 2, Lean, Coq assistants | -| 12 | Fleet | Gitbot fleet orchestration | -| 13 | NeSy | Neurosymbolic reasoning | -| 14 | Feedback | User feedback collection | - -Each cartridge occupies one or more cells. For example, `database-mcp` occupies cells -(MCP, Database), (gRPC, Database), and (REST, Database). The matrix is sparse: not -every cell needs to be filled. The current deployment has 18 cartridges populating -approximately 54 cells. - -An optional third axis --- the *backend* dimension --- allows community extensions to -specialise a cartridge for a particular provider (e.g., `database-mcp` with backend -"postgresql" vs. backend "mysql") without forking the core cartridge. By default, all -cartridges use the backend label `"universal"`. - -### 3.2 The Three-Layer Stack - -Each cartridge is implemented as a three-layer vertical slice: - -``` - +-----------------+ - | Idris 2 ABI | Formal specification + proofs - | (SafeX.idr) | IsUnbreakable, lifecycle, types - +-----------------+ - | - | C-ABI integer encoding - v - +-----------------+ - | Zig FFI | Native execution - | (x_ffi.zig) | Thread-safe, zero runtime deps - +-----------------+ - | - | dlopen / direct link - v - +-----------------+ - | zig Adapter | Network exposure - | (x_adapter.v) | REST + gRPC + GraphQL + MCP stdio - +-----------------+ -``` - -The **Idris 2 ABI layer** defines the cartridge's type signature, lifecycle states, and -domain-specific safety properties. Every ABI module sets `%default total`, requiring -that all functions be provably terminating. The module exports C-ABI encoding functions -(`statusToInt`, `domainToInt`, `protocolToInt`) that establish the integer contract -between the proof layer and the execution layer. - -The **Zig FFI layer** implements the cartridge's runtime behaviour as C-ABI exports. -All global state is protected by `std.Thread.Mutex`, acquired at the export boundary -and released via `defer`. The layer enforces the `IsUnbreakable` invariant at mount -time: `boj_catalogue_mount` returns `-1` if the cartridge's status integer is not `1` -(Ready). - -The **zig adapter layer** exposes cartridges over network protocols. A single -adapter binary serves REST on port 7700, gRPC on port 7701, and GraphQL on port 7702. -MCP transport is available via stdio for direct agent integration. - -### 3.3 The Safety Gate - -The core safety mechanism is the `IsUnbreakable` dependent type: - -```idris -data IsUnbreakable : Cartridge -> Type where - VerifiedReady : (c : Cartridge) -> - (status c = Ready) -> - IsUnbreakable c -``` - -`IsUnbreakable` has exactly one constructor, `VerifiedReady`, which requires a proof -that the cartridge's status field equals `Ready`. This is a *type-level mounting -predicate*: any function that requires an `IsUnbreakable` proof can only be called with -a cartridge that has been proven ready at the type level. - -The corresponding Zig enforcement is structurally aligned: - -```zig -pub export fn boj_catalogue_mount(index: usize) c_int { - mutex.lock(); - defer mutex.unlock(); - if (!initialised or index >= catalogue_count) return -2; - if (catalogue[index].status != .ready) return -1; // IsUnbreakable - catalogue[index].mounted = true; - return 0; -} -``` - -The `status != .ready` check is the runtime mirror of `status c = Ready`. The integer -encoding is fixed by the ABI: `CartridgeStatus.ready = 1` in both Idris 2 -(`statusToInt Ready = 1`) and Zig (`ready = 1`). This correspondence is validated by -15 seam checks (Section 5.1) that verify enum encoding alignment across the language -boundary. - -We emphasise what this is *not*: it is not a machine-checked proof that the Zig code -correctly implements the Idris 2 specification. It is a disciplined architectural -pattern where: - -1. The proof layer defines the *only* way a cartridge can be marked safe. -2. The encoding layer fixes a stable integer representation. -3. The execution layer checks the *same* integer predicate. -4. Seam checks validate that the encodings have not drifted. - -This pattern is weaker than full verified compilation but stronger than the status quo -of unchecked dynamic loading. We discuss paths toward stronger guarantees in -Section 7. - -### 3.4 Thread Safety - -The BoJ FFI layer protects all mutable global state with module-level mutexes. -Across the 9 core FFI modules (catalogue, loader, federation, verisimdb, guardian, -coprocessor, sla, community, sdp) and 18 cartridge FFI modules, every C-ABI export -follows the same pattern: - -```zig -pub export fn boj_xxx_operation(...) return_type { - mutex.lock(); - defer mutex.unlock(); - // ... operation on protected state ... -} -``` - -This yields 120 mutex-protected export functions across 55 global state variables. -The `defer` keyword ensures unlock on all control flow paths, including early returns -and error cases. - -Deadlock prevention follows a simple discipline: internal implementation functions -(suffixed `_impl`) are called only while the mutex is held and never re-acquire it. -Re-entrant paths (e.g., mounting a cartridge that triggers a federation notification) -use the `_impl` variant to avoid double-locking. - ---- - -## 4. The Umoja Federation - -### 4.1 Design Goals - -The Umoja federation layer enables community-hosted BoJ instances to form a -decentralised network. The design goals are: - -1. **Zero trust by default**: nodes prove their identity through cryptographic hash - attestation of their binary. -2. **Eventually consistent**: catalogue state synchronises through anti-entropy gossip - rounds. -3. **Partition tolerant**: nodes operate independently during network partitions and - reconcile on reconnection. -4. **Georedundant**: seed nodes span multiple continents (EU-West, EU-Central, - US-East, AP-South). - -### 4.2 SWIM-Inspired Gossip Protocol - -Node liveness detection follows the SWIM protocol model [12]: - -- **Alive**: node is responsive and participating. -- **Suspected**: node has missed heartbeats (configurable threshold). -- **Dead**: node has been confirmed unreachable by multiple peers. - -The gossip layer uses UDP multicast on a link-local IPv6 address (`ff02::b04`) for -peer discovery, with point-to-point UDP for subsequent communication. The wire -protocol defines 7 packet types: - -| Tag | Packet | Direction | -|-----|--------|-----------| -| `0x01` | Discover | Broadcast | -| `0x02` | DiscoverReply | Unicast | -| `0x03` | GossipDigest | Unicast | -| `0x04` | GossipDigestReply | Unicast | -| `0x05` | HandshakeInit | Unicast | -| `0x06` | HandshakeReply | Unicast | -| `0x07` | Heartbeat | Unicast | - -### 4.3 Hash Attestation - -The trust model is built on the `Attested` dependent type: - -```idris -data Attested : (n : Node) -> (canonicalHash : String) -> Type where - ValidAttestation : (n : Node) -> - (canonicalHash : String) -> - (binaryHash n = canonicalHash) -> - Attested n canonicalHash -``` - -A node can participate in the community network only if its binary hash matches the -canonical hash published with each release. Nodes with non-matching hashes are not -excluded from running BoJ locally --- they simply cannot join the federated network. -This prevents tampered binaries from serving community requests while preserving the -user's right to modify their own copy. - -### 4.4 Anti-Entropy Synchronisation - -Catalogue synchronisation uses SHA-256 digest comparison. Each node computes a digest -over its registered cartridges (name, version, status triples). During a gossip round, -nodes exchange digests; if they differ, the node with the newer catalogue sends a full -catalogue update. This is a pull-based anti-entropy mechanism [13] that converges in -O(log N) gossip rounds for N nodes. - -### 4.5 Auto-SDP Zero-Trust Perimeter - -The federation includes a Session Description Protocol (SDP)-inspired perimeter -that maintains an allow-list of attested nodes, automatically banning nodes that fail -attestation or exhibit anomalous behaviour. The perimeter operates on three -principles: (1) all connections require attestation before data exchange, (2) failed -attestations result in immediate connection termination, and (3) repeated failures -trigger automatic banning with exponential backoff for re-admission. - ---- - -## 5. Evaluation - -### 5.1 Test Coverage - -The BoJ test suite comprises 307 tests across five categories: - -| Category | Count | Scope | -|----------|-------|-------| -| Core FFI | 178 | catalogue (13), loader (14), federation (40), guardian (12), readiness (28), VeriSimDB (7), e2e order-ticket (3), coprocessor (14), SLA (11), community (11), SDP (10), seams (15) | -| Cartridge FFI | 118 | 18 cartridges, each with dedicated test suites | -| Multi-node federation | 11 | Peer management, gossip rounds, attestation | -| Total | 307 | | - -The 15 seam checks deserve particular attention. These are integration contract -validation tests inspired by the panic-attack diagnostic pattern [14]. They verify -that integer encodings for all enumerations (`CartridgeStatus`, `ProtocolType`, -`CapabilityDomain`, `MenuTier`, `CircuitState`, `Severity`) are identical between the -Idris 2 ABI definitions and the Zig FFI implementations. A seam check failure -indicates that the proof-to-execution correspondence has been broken, which is a -genuine architectural defect. - -The seam check philosophy is the "silent signature": if all checks pass, the -integration surface is verified and there is nothing to report. Any failure is -actionable. - -### 5.2 Static Analysis - -We analysed the BoJ codebase using panic-attack [14], a static vulnerability scanner -designed for multi-language codebases. The `assail` mode performs taint analysis across -language boundaries. - -| Finding | Count | Severity | -|---------|-------|----------| -| QUIC crypto dependency (expected) | 1 | Weak | -| Critical Zig vulnerabilities | 0 | --- | -| Cross-language taint paths | 0 | --- | -| Tainted production paths | 0 | --- | - -The single weak finding relates to the QUIC transport's reliance on X25519 key -exchange and ChaCha20-Poly1305 authenticated encryption. These are industry-standard -algorithms, but the scanner correctly flags any cryptographic dependency as requiring -ongoing review. The finding is expected and documented. - -Notably, the codebase contains zero instances of `believe_me` (Idris 2's escape hatch -for admitting unproven propositions). All proofs in the ABI layer are constructive. - -### 5.3 Resource Footprint - -| Metric | Value | -|--------|-------| -| Binary size (all 18 cartridges) | ~18 MB | -| Memory at rest (all cartridges loaded) | ~100 MB | -| Cartridge mount latency | < 1 ms | -| Cartridge invoke latency | < 1 ms | -| Federation gossip round | ~50 ms (LAN) | -| Concurrent cartridge limit | 128 (compile-time constant) | - -The resource profile is modest by contemporary standards. A single BoJ instance -replaces what would otherwise be 18+ separate server processes, each consuming its own -memory, file descriptors, and process table entries. The "3 Claude instances + 20 MCP -servers = frozen desktop" incident that motivated the Guardian module (Section 3.2) -consumed 22 GB of available RAM through unchecked process spawning; a single BoJ -instance would have served the same capability surface in approximately 100 MB. - -### 5.4 Language Purity - -| Language | Source files | Role | -|----------|-------------|------| -| Zig | 50 | FFI execution, thread safety, native runtime | -| Idris 2 | 26 | ABI specification, proofs, type-level safety | -| zig | 19 | Network adapter (REST, gRPC, GraphQL) | -| **Total** | **95** | | - -The codebase contains zero Python, JavaScript, Go, or Rust. Each language is used for -what it does best: Idris 2 for proofs and type-level reasoning, Zig for -zero-overhead native execution with C ABI compatibility, and zig for ergonomic -network server implementation with built-in HTTP and JSON support. - -### 5.5 Comparison with the MCP Ecosystem - -To contextualise the BoJ architecture, we compare against the aggregate statistics -from the MCP ecosystem study [1]: - -| Metric | MCP ecosystem median [1] | BoJ | -|--------|--------------------------|-----| -| Tools per server | < 5 | 18 cartridges (54+ matrix cells) | -| Input validation | 57% of servers | 100% (type-enforced) | -| Static analysis | 2.6% of servers | panic-attack + seam checks | -| Formal verification | 0% of servers | IsUnbreakable proof on every cartridge | -| Code smell rate | 66% of servers | 0 reported by panic-attack | - -We acknowledge that this comparison is not entirely fair: BoJ is a single carefully -engineered system, while the ecosystem study measures a population of independently -developed servers with varying goals and resources. The comparison illustrates what -*is achievable* with a matrix architecture, not what is *typical*. - ---- - -## 6. The HAT Extensibility Model - -### 6.1 Motivation - -The BoJ architecture is intentionally restrictive: only cartridges that pass the -`IsUnbreakable` proof can be mounted. This safety property would be undermined if -cartridges could invoke arbitrary external tools directly. Yet practical utility -requires integration with real-world tools (Git CLI, Docker/Podman, Terraform, etc.) -that cannot be formally verified. - -### 6.2 Hardware Attached on Top - -(The HAT model is also specified normatively in `docs/specification/cartridges/README.md` §5.) - -We resolve this tension with the HAT (Hardware Attached on Top) model, named by -analogy with the Raspberry Pi and BeagleBone hardware extension ecosystem. A HAT is -a bridge script that translates a BoJ cartridge invocation into a real-world tool -call: - -``` - BoJ Cartridge (verified) ---> HAT Bridge ---> External Tool (unverified) - git-mcp git_hat.sh git CLI - container-mcp podman_hat.sh podman - ssg-mcp zola_hat.sh zola -``` - -The bridge is *outside* the BoJ safety perimeter. The cartridge's formal properties -(lifecycle, type safety, thread safety) are preserved regardless of what the HAT does. -If a HAT fails, the cartridge's circuit breaker (Section 3.2) trips, isolating the -failure. - -### 6.3 Bridge Types - -We define four bridge categories: - -1. **CLI wrapper**: the HAT invokes a command-line tool and parses its output. - Simplest and most common (e.g., `git`, `zola`, `podman`). -2. **JSON-RPC stdio**: the HAT speaks JSON-RPC 2.0 over stdin/stdout to an existing - MCP-compatible server. This enables composition with the existing ecosystem. -3. **HTTP API**: the HAT calls a REST or GraphQL endpoint on a remote service. -4. **Library FFI**: the HAT links against a native library via C ABI. This is the - highest-performance option and is used for VeriSimDB integration. - -### 6.4 Safety Properties - -The HAT model preserves BoJ's safety invariants: - -- **IsUnbreakable still holds**: the cartridge's mounting proof is independent of HAT - presence. A cartridge without a HAT is safe (it simply has no external integration). -- **Circuit breaker isolation**: HAT failures trip the per-cartridge circuit breaker, - preventing cascade failures. -- **Thread safety preserved**: HAT invocations go through the same mutex-protected - FFI exports as internal operations. -- **Attestation unaffected**: the HAT is not part of the attested binary; federation - nodes verify only the BoJ core binary hash. - ---- - -## 7. Limitations and Future Work - -### 7.1 Fixed-Size Data Structures - -The catalogue uses compile-time-constant array sizes (`MAX_CARTRIDGES = 128`, -`MAX_NODES = 16`, `MAX_PEERS = 16`). These are sufficient for the current use case but -impose hard upper bounds. Switching to dynamically allocated structures would require -an allocator in the Zig FFI layer, which we have deliberately avoided to maintain -zero-runtime-dependency guarantees. - -### 7.2 Single-Process Architecture - -BoJ is a single-process, multi-threaded server. Horizontal scaling is achieved through -federation (multiple BoJ instances) rather than internal parallelism. Workloads -requiring high throughput on a single capability domain would benefit from a -sharded architecture that we have not yet implemented. - -### 7.3 Proof-to-Execution Gap - -The `IsUnbreakable` pattern provides a disciplined correspondence between proofs and -runtime checks, but it is not a machine-checked proof that the Zig code correctly -implements the Idris 2 specification. Two paths toward stronger guarantees exist: - -- **Idris 2 C codegen**: Idris 2 can compile to C via its RefC backend. Extracting - the catalogue state machine into C and linking it directly into the Zig binary would - eliminate the proof-to-execution gap for lifecycle management, at the cost of - introducing a C dependency. -- **Zig formal verification**: emerging tools for Zig verification (e.g., based on - LLVM-IR analysis) could provide machine-checked proofs of the Zig implementation. - -### 7.4 Linear Types for Thread Safety - -The current thread safety model relies on runtime mutexes. Linear types --- as -supported by Idris 2's QTT and as explored in the Ephapax language project --- could -express thread ownership at the type level, making mutex-based locking unnecessary. -This would strengthen the safety guarantees from "correctly locked at runtime" to -"uniquely owned at compile time." - -### 7.5 Formal Federation Verification - -The Umoja gossip protocol has been tested empirically but not formally verified. The -SWIM protocol has known convergence properties [12]; proving that the Umoja variant -preserves these properties under the hash attestation constraint is future work. - ---- - -## 8. Conclusion - -We have presented the Bundle of Joy server, a formally verified plugin architecture -that collapses the N x M combinatoric explosion of protocol servers into a single -binary with a two-dimensional capability matrix. The `IsUnbreakable` dependent-type -predicate gates FFI execution, ensuring that only proven-ready cartridges can be -mounted. The architecture achieves 100% input validation, zero code smells, and zero -`believe_me` escape hatches across 95 source files in three languages. - -The dependent type plus FFI pattern is not limited to server plugin architectures. Any -system that dynamically loads and executes untrusted modules --- browser extensions, -database stored procedures, smart contracts, operating system drivers --- faces the -same trust boundary. The pattern of (1) expressing a safety predicate as a dependent -type, (2) fixing a stable encoding, and (3) enforcing the same predicate at the -execution boundary is applicable wherever formal proofs and native execution must -coexist. - -The Umoja federation layer demonstrates that formally verified servers can participate -in decentralised networks without sacrificing safety guarantees. Hash attestation ties -binary integrity to community trust, enabling a volunteer hosting model analogous to -Tor or IPFS but with stronger provenance guarantees. - -We make the BoJ server available as open-source software under the MPL-2.0 -(MPL-2.0) at `https://github.com/hyperpolymath/boj-server`. - ---- - -## References - -[1] H. Qin, Y. Zhang, L. Chen, et al., "An Empirical Study of MCP Servers: Code -Smells, Quality Issues, and Security Vulnerabilities," arXiv preprint -arXiv:2506.13538, 2025. - -[2] Model Context Protocol Specification, Linux Foundation AI & Data, Anthropic, -version 2025-11-25. https://spec.modelcontextprotocol.io/ - -[3] E. Brady, "Idris 2: Quantitative Type Theory in Practice," in *Proceedings of the -35th European Conference on Object-Oriented Programming (ECOOP)*, 2021. -arXiv:2104.00480. - -[4] S. Puttaswamy, M. Kumar, and R. Sharma, "Context-Aware MCP Server Collaboration -for Coherent AI Agent Responses," arXiv preprint arXiv:2601.11595, 2026. - -[5] J. Li, X. Wang, and H. Chen, "Agent Capability Negotiation: Dynamic Discovery and -Composition of MCP Server Capabilities," arXiv preprint arXiv:2506.13590, 2025. - -[6] X. Leroy, "Formal Verification of a Realistic Compiler," *Communications of the -ACM*, vol. 52, no. 7, pp. 107-115, 2009. - -[7] N. Swamy, C. Hritcu, C. Keller, et al., "Dependent Types and Multi-Monadic -Effects in F*," in *Proceedings of POPL*, 2016. - -[8] G. Klein, K. Elphinstone, G. Heiser, et al., "seL4: Formal Verification of an -OS Kernel," in *Proceedings of the 22nd ACM Symposium on Operating Systems Principles -(SOSP)*, 2009. - -[9] R. Jung, J.-H. Jourdan, R. Krebbers, and D. Dreyer, "RustBelt: Securing the -Foundations of the Rust Programming Language," *Proceedings of the ACM on Programming -Languages*, vol. 2, no. POPL, 2018. - -[10] S. Peyton Jones, "Tackling the Awkward Squad: Monadic Input/Output, Concurrency, -Exceptions, and Foreign-Language Calls in Haskell," in *Engineering Theories of -Software Construction*, 2001. - -[11] J. Garrigue, "Relaxing the Value Restriction," in *Functional and Logic -Programming*, Springer, 2004. - -[12] A. Das Gupta, I. Gupta, and M. Agrawal, "SWIM: Scalable Weakly-consistent -Infection-style Process Group Membership Protocol," in *Proceedings of the -International Conference on Dependable Systems and Networks (DSN)*, 2002. - -[13] A. Demers, D. Greene, C. Hauser, et al., "Epidemic Algorithms for Replicated -Database Maintenance," in *Proceedings of the 6th ACM Symposium on Principles of -Distributed Computing (PODC)*, 1987. - -[14] J. D. A. Jewell, "panic-attack: Cross-Language Static Vulnerability Analysis -for Multi-Language Codebases," hyperpolymath/panic-attacker, 2026. -https://github.com/hyperpolymath/panic-attacker - ---- - -## Appendix A: Idris 2 ABI Module Listing - -| Module | Purpose | Lines | -|--------|---------|-------| -| `Boj.Protocol` | Protocol type enumeration (9 types) | 78 | -| `Boj.Domain` | Capability domain enumeration (14 domains) | 102 | -| `Boj.Catalogue` | Cartridge registry, IsUnbreakable proof, matrix queries | 221 | -| `Boj.Federation` | Umoja node identity, hash attestation, gossip | 165 | -| `Boj.Guardian` | Resource monitoring, circuit breaker, self-diagnostics | 299 | -| `Boj.Menu` | Teranga menu discovery protocol | ~85 | -| 18 cartridge ABIs | Per-cartridge safety proofs (`SafeX.idr`) | ~720 (~40 each) | - -## Appendix B: Zig FFI Module Listing - -| Module | C-ABI Exports | Mutex-Protected Globals | Tests | -|--------|---------------|------------------------|-------| -| `catalogue.zig` | 26 | 6 | 13 | -| `loader.zig` | 11 | 2 | 14 | -| `federation.zig` | 44 | 21 | 40 | -| `guardian.zig` | 29 | 9 | 12 | -| `readiness.zig` | 8 | 3 | 28 | -| `verisimdb.zig` | 10 | 8 | 7 | -| `coprocessor.zig` | 12 | 5 | 14 | -| `sla.zig` | 15 | 4 | 11 | -| `community.zig` | 9 | 3 | 11 | -| `sdp.zig` | 10 | 6 | 10 | -| `seams.zig` | 0 | 0 | 15 | -| `e2e_order.zig` | 0 | 0 | 3 | -| `bench.zig` | 0 | 0 | 0 (benchmark only) | -| 18 cartridge FFIs | 4 each | 1 each | 118 total | - -## Appendix C: Reproducibility - -Build requirements: -- Zig >= 0.15.2 -- Idris 2 (any recent version with QTT support) -- zig >= 0.5.0 - -```sh -git clone https://github.com/hyperpolymath/boj-server -cd boj-server -just build # Build all layers -just test # Run 307 tests -just assail # Run panic-attack static analysis -just federation # Start a local 3-node federation cluster -``` diff --git a/docs/papers/umoja-federation-draft.adoc b/docs/papers/umoja-federation-draft.adoc new file mode 100644 index 00000000..5e21cc98 --- /dev/null +++ b/docs/papers/umoja-federation-draft.adoc @@ -0,0 +1,773 @@ +== Gossip-Based Capability Discovery and Synchronisation for Developer Tool Servers + +*Internet-Draft:* `+draft-jewell-umoja-capability-gossip-00+` *Intended +Status:* Experimental *Author:* Jonathan D.A. Jewell *Organisation:* +hyperpolymath *Date:* 2026-03 + +''''' + +=== Abstract + +This document describes the Umoja federation protocol, a gossip-based +mechanism for discovering and synchronising capability catalogues across +distributed developer tool server instances. Umoja enables isolated +server nodes — such as those implementing the Model Context Protocol +(MCP), Language Server Protocol (LSP), or Debug Adapter Protocol (DAP) — +to form a federated network where each node advertises and discovers +capabilities ("`cartridges`") provided by its peers. + +The protocol uses QUIC [RFC 9000] as its primary transport, with X25519 +[RFC 7748] key exchange and ChaCha20-Poly1305 [RFC 8439] AEAD encryption +for peer-to-peer confidentiality and integrity. A UDP fallback mode is +defined for constrained environments or development/testing scenarios +where QUIC support is unavailable. + +Catalogue synchronisation is achieved through anti-entropy digest +exchange using SHA-256 hashes of sorted cartridge metadata. The protocol +does not transfer cartridge binaries or executable code — only metadata +sufficient for capability discovery and version negotiation. + +''''' + +=== Table of Contents + +[arabic] +. link:#1-introduction[Introduction] +. link:#2-terminology[Terminology] +. link:#3-protocol-overview[Protocol Overview] +. link:#4-discovery-mechanism[Discovery Mechanism] +. link:#5-handshake-and-attestation[Handshake and Attestation] +. link:#6-gossip-protocol[Gossip Protocol] +. link:#7-catalogue-synchronisation[Catalogue Synchronisation] +. link:#8-security-considerations[Security Considerations] +. link:#9-iana-considerations[IANA Considerations] +. link:#10-references[References] +. link:#appendix-a-packet-format-diagrams[Appendix A: Packet Format +Diagrams] +. link:#appendix-b-example-message-flows[Appendix B: Example Message +Flows] +. link:#authors-address[Author’s Address] + +''''' + +=== 1. Introduction + +==== 1.1. Problem Statement + +Modern developer tool ecosystems rely on protocol servers — MCP servers +for AI agent capabilities, LSP servers for editor intelligence, DAP +servers for debugging — that operate as isolated, single-tenant +instances. Each server maintains its own capability catalogue with no +mechanism for sharing discovered capabilities across instances or +environments. + +This isolation creates several problems: + +* *Capability fragmentation.* A developer working across multiple +machines or environments cannot discover capabilities available on peer +instances. +* *Redundant configuration.* Each server must be independently +configured with identical capability sets, even when they serve the same +organisation or project. +* *No cross-instance awareness.* An MCP server on one machine has no +knowledge of cartridges loaded on a teammate’s MCP server, even when +those cartridges would be directly useful. +* *Scaling limitations.* Without federation, scaling developer tool +infrastructure requires manual replication of capability catalogues +across every new instance. + +==== 1.2. Solution Overview + +The Umoja federation protocol addresses these problems by defining a +gossip-based mechanism for peer-to-peer capability catalogue discovery +and synchronisation. "`Umoja`" means "`unity`" in Swahili, reflecting +the protocol’s goal of unifying isolated developer tool server instances +into a coherent federation. + +Key design principles: + +* *Metadata only.* The protocol exchanges capability metadata (names, +versions, hashes), never executable code or cartridge binaries. This +limits the attack surface and keeps message sizes small. +* *QUIC-first, UDP-fallback.* Encrypted transport is the default; +cleartext UDP is available for development and testing but SHOULD NOT be +used in production deployments. +* *SWIM-inspired failure detection.* Node liveness is tracked using a +protocol inspired by the Scalable Weakly-consistent Infection-style +Process Group Membership protocol (SWIM), with states: alive, suspected, +dead. +* *Cryptographic attestation.* Peers attest their catalogue state via +SHA-256 digests. Catalogue hash comparison drives synchronisation +decisions. +* *Zero-trust perimeter.* An integrated Software Defined Perimeter +(Auto-SDP) layer rejects traffic from unverified peers before it reaches +the gossip layer. + +==== 1.3. Relationship to Existing Work + +This protocol is informed by, but distinct from, several IETF efforts in +the AI agent and service discovery space: + +* *draft-cui-ai-agent-discovery-invocation-00* defines a DNS-based +discovery framework for AI agents. Umoja operates below this layer, +providing gossip-based discovery for the servers that host such agents. +* *draft-narajala-ans-00* (Agent Naming Service) proposes a hierarchical +naming scheme for AI agents. Umoja complements ANS by providing a +runtime discovery protocol that could resolve ANS names to live server +instances. +* *draft-mp-agntcy-ads* (Agent Discovery Service) addresses agent +capability advertisement. Umoja focuses specifically on server-level +capability catalogues rather than individual agent capabilities. +* *RFC 9000 (QUIC)* provides the encrypted transport foundation. + +Umoja does not replace any of these proposals. It fills a gap at the +infrastructure layer: where agents and tools are _hosted_, rather than +how individual agents are _identified_ or _invoked_. + +''''' + +=== 2. Terminology + +The key words "`MUST`", "`MUST NOT`", "`REQUIRED`", "`SHALL`", "`SHALL +NOT`", "`SHOULD`", "`SHOULD NOT`", "`RECOMMENDED`", "`NOT RECOMMENDED`", +"`MAY`", and "`OPTIONAL`" in this document are to be interpreted as +described in BCP 14 [RFC 2119] [RFC 8174] when, and only when, they +appear in all capitals, as shown here. + +* *Node*: A single instance of a developer tool server participating in +the Umoja federation. Each node has a unique node identifier. +* *Peer*: A remote node that the local node is aware of and communicates +with via the gossip protocol. +* *Catalogue*: The ordered set of capabilities (cartridges) available on +a given node. Represented as a sorted list of (name, version, hash) +tuples. +* *Cartridge*: A discrete, loadable capability module registered with a +developer tool server. Cartridges expose tools, resources, or protocol +capabilities. +* *Digest*: A SHA-256 hash computed over the sorted catalogue entries of +a node. Used for efficient comparison of catalogue state between peers +without exchanging the full catalogue. +* *Attestation*: The process by which a peer proves its identity and +catalogue integrity via X25519 key exchange and digest comparison. +* *Seed Node*: A well-known bootstrap node that new peers contact to +join the federation network. Seed nodes are listed in a static +configuration file. +* *Gossip Round*: A single cycle of the anti-entropy protocol, in which +a node selects a subset of peers and exchanges digest information. +* *Fanout*: The number of peers contacted during each gossip round. + +''''' + +=== 3. Protocol Overview + +==== 3.1. Node Lifecycle + +A node progresses through the following states: + +.... +Bootstrap → Discovery → Handshake → Active → Suspected → Dead + │ ↑ │ + │ └──────────┘ + │ (recovery) + └─────────────────────────────────────────────► + (fatal error → Dead) +.... + +* *Bootstrap*: Node initialises its local catalogue digest, generates an +X25519 keypair, binds to its federation port (default 9999), and loads +its seed node list. +* *Discovery*: Node sends DISCOVER packets to seed nodes and/or the IPv6 +multicast group (ff02::b04) to locate peers. Discovery is repeated on a +configurable interval (default: 60 seconds). +* *Handshake*: Upon discovering a peer, the node initiates a handshake: +X25519 public key exchange, mutual node ID verification, and catalogue +digest comparison. +* *Active*: The node participates in gossip rounds, exchanges +heartbeats, and synchronises catalogue state with peers. +* *Suspected*: A peer that has missed heartbeats beyond the configured +timeout (default: 30 seconds) transitions to "`suspected`". The local +node MAY attempt indirect probes via other peers before declaring the +suspected peer dead. +* *Dead*: A peer confirmed as unreachable. Dead peers are removed from +the active peer list after a configurable grace period. + +==== 3.2. Transport + +===== 3.2.1. Encrypted Transport (Default) + +The protocol’s primary transport uses AEAD-encrypted UDP datagrams with +the following cryptographic primitives. Note: while internally referred +to as "`QUIC mode`" in the reference implementation, this transport does +not implement the full QUIC protocol [RFC 9000] (no connection IDs, +streams, flow control, or congestion control). It uses QUIC’s +cryptographic choices (X25519 + ChaCha20-Poly1305) applied directly to +UDP datagrams: + +* *Key exchange*: X25519 Elliptic Curve Diffie-Hellman [RFC 7748]. Each +node generates a long-lived identity keypair at bind time. A per-peer +shared secret is derived via X25519(local_secret, remote_public). +* *Authenticated encryption*: ChaCha20-Poly1305 AEAD [RFC 8439]. All +gossip and heartbeat traffic between peers with established shared +secrets is encrypted. The AEAD tag provides integrity verification. +* *Packet framing*: Encrypted packets are distinguished from cleartext +packets by a high-bit marker (0x80) in the first byte. The remaining 7 +bits encode the packet type. + +QUIC-mode packets carry a 12-byte nonce and 16-byte authentication tag +in addition to the encrypted payload. + +===== 3.2.2. UDP Fallback + +For environments where QUIC is unavailable or during initial development +and testing, a cleartext UDP mode is defined. In this mode: + +* Packet types are indicated by a single-byte tag (0x01–0x07) without +the high-bit marker. +* No encryption or authentication is applied. +* Nodes SHOULD log a warning when operating in UDP fallback mode. +* UDP fallback mode MUST NOT be used in production deployments handling +sensitive capability metadata. + +===== 3.2.3. Port Assignment + +The default federation port is *9999* (UDP). This port is used for both +QUIC-mode and UDP-fallback-mode traffic. The port is configurable via +the `+BOJ_FEDERATION_PORT+` environment variable. + +==== 3.3. Packet Types + +[width="100%",cols="^25%,^23%,29%,23%",options="header",] +|=== +|Tag (cleartext) |Tag (encrypted) |Name |Direction +|0x01 |0x81 |DISCOVER |Multicast / Unicast +|0x02 |0x82 |DISCOVER_REPLY |Unicast +|0x03 |0x83 |GOSSIP_DIGEST |Unicast +|0x04 |0x84 |GOSSIP_DIGEST_REPLY |Unicast +|0x05 |0x85 |HANDSHAKE_INIT |Unicast +|0x06 |0x86 |HANDSHAKE_REPLY |Unicast +|0x07 |0x87 |HEARTBEAT |Unicast +|=== + +All packets MUST fit within a single UDP datagram. The maximum packet +payload size is 1024 bytes. Implementations MUST discard packets +exceeding this limit. + +''''' + +=== 4. Discovery Mechanism + +==== 4.1. Link-Local Discovery (IPv6 Multicast) + +For nodes on the same network segment, Umoja uses IPv6 multicast group +*ff02::b04* (link-local scope) for peer discovery. The multicast address +encodes "`b04`" (a mnemonic for "`boj`" in hexadecimal). + +A discovering node sends a DISCOVER packet to the multicast group. Any +node listening on the federation port that receives this packet SHOULD +reply with a DISCOVER_REPLY containing its node ID, listen address, and +current catalogue digest. + +Link-local discovery is useful for development environments, CI/CD +clusters, and any scenario where nodes share a network segment. + +==== 4.2. WAN Bootstrap (Seed Nodes) + +For wide-area federation, nodes are configured with a list of seed +nodes. Seed nodes are well-known, stable federation endpoints that serve +as bootstrap rendezvous points. + +The seed node list is specified in a TOML configuration file: + +[source,toml] +---- +[metadata] +version = "0.1.0" +network = "umoja-mainnet" +min-seeds-for-quorum = 2 + +[[seed]] +id = "seed-eu-west" +region = "eu-west-1" +host = "eu.boj.hyperpolymath.dev" +federation-port = 9999 +---- + +A node MUST attempt to contact at least `+min-seeds-for-quorum+` seed +nodes during bootstrap. If fewer than `+min-seeds-for-quorum+` seeds +respond, the node MAY operate in a degraded mode with reduced federation +capabilities. + +==== 4.3. Discovery Interval + +Discovery is performed periodically at a configurable interval (default: +60 seconds). The interval SHOULD be jittered by +/- 10% to avoid +thundering-herd effects across simultaneously-booted clusters. + +''''' + +=== 5. Handshake and Attestation + +==== 5.1. Handshake Flow + +Upon discovering a new peer, a node initiates a three-step handshake: + +.... + Node A Node B + │ │ + │── HANDSHAKE_INIT ────────────►│ + │ (A's public key, node ID) │ + │ │ + │◄── HANDSHAKE_REPLY ──────────│ + │ (B's public key, node ID, │ + │ catalogue digest) │ + │ │ + │── GOSSIP_DIGEST ────────────►│ + │ (A's catalogue digest, │ + │ encrypted if QUIC) │ + │ │ +.... + +==== 5.2. Handshake States + +Each peer relationship progresses through the following states: + +[width="100%",cols="20%,80%",options="header",] +|=== +|State |Meaning +|`+none+` |No handshake attempted + +|`+pending+` |HANDSHAKE_INIT sent, awaiting reply + +|`+exchanged+` |Keys exchanged, catalogue digests being compared + +|`+verified+` |Peer identity confirmed, digests match or sync in +progress + +|`+rejected+` |Peer rejected (failed authentication or banned by SDP) +|=== + +A peer in `+verified+` state is eligible for gossip rounds and heartbeat +exchange. A peer in `+rejected+` state MUST NOT receive gossip traffic +and SHOULD be reported to the Auto-SDP layer. + +==== 5.3. Key Exchange + +The handshake uses X25519 [RFC 7748] for key exchange: + +[arabic] +. Each node generates a long-lived X25519 keypair at bind time. +. HANDSHAKE_INIT carries the initiator’s 32-byte public key. +. HANDSHAKE_REPLY carries the responder’s 32-byte public key. +. Both nodes derive a shared secret: +`+shared = X25519(local_secret, remote_public)+`. +. The shared secret is used as the ChaCha20-Poly1305 key for all +subsequent encrypted communication with that peer. + +==== 5.4. Catalogue Digest Comparison + +During handshake, both nodes exchange their current catalogue digest +(SHA-256 of sorted cartridge metadata). If the digests differ, catalogue +synchronisation (Section 7) is triggered immediately after the handshake +completes. + +''''' + +=== 6. Gossip Protocol + +==== 6.1. Protocol Model + +The Umoja gossip protocol is inspired by the SWIM (Scalable +Weakly-consistent Infection-style Process Group Membership) protocol, +adapted for capability catalogue synchronisation rather than process +group membership. + +==== 6.2. Gossip Rounds + +Each gossip round proceeds as follows: + +[arabic] +. The local node selects up to `+fanout+` (default: 3) random peers from +its active peer list. +. For each selected peer, the node sends a GOSSIP_DIGEST packet +containing its current catalogue digest. +. The recipient compares the received digest against its own catalogue +digest. +. If the digests differ, the recipient replies with a +GOSSIP_DIGEST_REPLY containing its own digest and the full list of +cartridge metadata entries that differ. +. The originator processes the reply and updates its view of the peer’s +catalogue. + +==== 6.3. Configurable Parameters + +[width="100%",cols="30%,12%,58%",options="header",] +|=== +|Parameter |Default |Description +|`+gossip-interval-ms+` |5000 |Time between gossip rounds (milliseconds) + +|`+gossip-fanout+` |3 |Number of peers contacted per round + +|`+heartbeat-interval-ms+` |10000 |Time between heartbeat packets + +|`+heartbeat-timeout-ms+` |30000 |Time before a peer is marked +"`suspected`" + +|`+max-peers+` |128 |Maximum number of tracked peers + +|`+min-peers+` |2 |Minimum peers for healthy federation +|=== + +==== 6.4. Failure Detection + +Failure detection uses heartbeat monitoring: + +[arabic] +. Each active peer MUST send HEARTBEAT packets at the configured +heartbeat interval (default: 10 seconds). +. If no heartbeat is received from a peer within the heartbeat timeout +(default: 30 seconds), the peer transitions from `+alive+` to +`+suspected+`. +. A suspected peer MAY be probed indirectly: the local node asks another +peer to probe the suspected node. If the indirect probe succeeds, the +peer returns to `+alive+`. +. If the suspected peer remains unreachable after +`+unhealthy-threshold+` (default: 3) consecutive missed heartbeat +cycles, it transitions to `+dead+`. +. A dead peer MAY be resurrected if it re-establishes contact and +completes a fresh handshake. After `+recovery-threshold+` (default: 2) +successful heartbeats, it returns to `+alive+`. + +==== 6.5. Peer Selection + +Peer selection for gossip rounds uses a lightweight PRNG (xorshift32, +seeded from the system timestamp). The selection algorithm is not +required to be cryptographically secure — it needs only to provide fair +distribution across the active peer set to ensure convergence. + +''''' + +=== 7. Catalogue Synchronisation + +==== 7.1. Digest Computation + +A catalogue digest is computed as follows: + +[arabic] +. Collect all loaded cartridge entries as strings of the form +`+"{name}:{version}:{hash}"+`. +. Sort the entries lexicographically. +. Concatenate the sorted entries with newline separators. +. Compute the SHA-256 hash of the resulting string. + +The digest is a 32-byte value. Implementations MUST support catalogues +of up to 128 cartridge entries. + +==== 7.2. Synchronisation Trigger + +Catalogue synchronisation is triggered when: + +* Two peers exchange digests (via handshake or gossip round) and the +digests differ. + +==== 7.3. Synchronisation Scope + +Synchronisation exchanges _metadata only_: + +* Cartridge name (string) +* Cartridge version (string, semver) +* Cartridge content hash (SHA-256 hex string) +* Cartridge tier (e.g. "`teranga`", "`shield`", "`ayo`") +* Protocol columns supported (e.g. "`mcp`", "`lsp`", "`dap`", "`bsp`") + +Synchronisation MUST NOT transfer: + +* Cartridge binaries (.so, .dll, .dylib files) +* Cartridge source code +* User data or configuration +* Authentication credentials + +==== 7.4. Convergence + +Under stable network conditions, the gossip protocol ensures eventual +convergence of catalogue views across all federated nodes. The expected +convergence time for a change to propagate to all N nodes is O(log N) +gossip rounds, assuming a fanout of 3. + +''''' + +=== 8. Security Considerations + +==== 8.1. Auto-SDP Zero-Trust Perimeter + +The Umoja protocol integrates a Software Defined Perimeter (Auto-SDP) +layer that enforces zero-trust principles at the transport level: + +* All inbound federation traffic is processed by the SDP layer before +reaching the gossip protocol. +* Peers MUST be on the allow-list to send gossip or heartbeat traffic. +Unknown peers are permitted only to send DISCOVER and HANDSHAKE_INIT +packets. +* Open mode (allowing unauthenticated peers) is available for initial +seed bootstrapping but SHOULD be disabled once a federation is +established. + +==== 8.2. Authentication and Banning + +* Peers that fail authentication (invalid X25519 key exchange, +mismatched node IDs, or tampered digests) increment a failure counter. +* After `+ban-threshold+` (default: 5) consecutive authentication +failures, the peer is automatically banned for `+ban-duration+` +(default: 300 seconds). +* Banned peers are tracked by node ID. All packets from banned peers are +silently dropped. +* The ban list supports up to 64 entries; when full, the oldest ban +entry is evicted. + +==== 8.3. Rate Limiting + +* Per-peer rate limiting is enforced by the SDP layer (default: 100 +requests per second). +* Peers exceeding the rate limit transition to `+rate_limited+` policy +and excess packets are dropped. +* Rate-limited peers are not banned but MAY be banned if rate violations +persist. + +==== 8.4. Hash Attestation + +* Catalogue digests are computed locally from the node’s own loaded +cartridges. A node MUST NOT accept a digest from a peer as its own +catalogue state. +* Digest comparison is used for synchronisation decisions only; a +mismatched digest does not constitute an attack. However, a peer that +consistently reports different digests in rapid succession (digest +thrashing) SHOULD be flagged for investigation. + +==== 8.5. Transport Security + +* Encrypted mode (X25519 + ChaCha20-Poly1305) provides confidentiality +and integrity for all gossip traffic. Note that the current design uses +long-lived identity keypairs without ephemeral key exchange, which does +not provide forward secrecy. Future revisions of this protocol SHOULD +incorporate ephemeral ECDH or a full QUIC handshake to achieve forward +secrecy. +* The shared secret derived from X25519 SHOULD be processed through HKDF +[RFC 5869] before use as a ChaCha20-Poly1305 key, rather than used +directly. The reference implementation currently uses the raw shared +secret; this is a known limitation. +* Implementations MUST track received nonces per peer to prevent replay +attacks. Nonces SHOULD be counter-based (monotonically increasing) +rather than random to enable efficient duplicate detection. +* UDP fallback mode provides none of these properties and MUST NOT be +used in production. +* Implementations SHOULD default to encrypted mode and require explicit +configuration to enable UDP fallback. + +==== 8.6. No Code Execution + +The protocol exchanges metadata only. Implementations MUST NOT execute, +load, or interpret any data received via the federation protocol as +code. Cartridge installation from federated peers requires an +out-of-band mechanism with its own authentication and integrity +verification. + +''''' + +=== 9. IANA Considerations + +==== 9.1. Port Number Registration + +This document requests registration of the following port number: + +[width="100%",cols="24%,22%,32%,22%",options="header",] +|=== +|Service Name |Port Number |Transport Protocol |Description +|umoja-fed |9999 |UDP |Umoja Federation Protocol +|=== + +*Note:* Port 9999 is currently assigned to the "`distinct`" service in +the IANA Service Name and Transport Protocol Port Number Registry. The +reference implementation uses 9999 as a configurable default. A formal +port allocation from the User Ports range (1024-49151) will be requested +if this protocol progresses beyond Experimental status. Implementations +MUST support configurable port assignment. + +==== 9.2. IPv6 Multicast Address + +This document requests allocation of the following IPv6 multicast +address from the Link-Local Scope Multicast Addresses registry: + +[cols=",",options="header",] +|=== +|Address |Description +|ff02::b04 |Umoja Federation Discovery (link-local) +|=== + +''''' + +=== 10. References + +==== 10.1. Normative References + +* *[RFC 2119]* Bradner, S., "`Key words for use in RFCs to Indicate +Requirement Levels`", BCP 14, RFC 2119, DOI 10.17487/RFC2119, March +1997. +* *[RFC 8174]* Leiba, B., "`Ambiguity of Uppercase vs Lowercase in RFC +2119 Key Words`", BCP 14, RFC 8174, DOI 10.17487/RFC8174, May 2017. +* *[RFC 9000]* Iyengar, J., Ed. and M. Thomson, Ed., "`QUIC: A UDP-Based +Multiplexed and Secure Transport`", RFC 9000, DOI 10.17487/RFC9000, May +2021. +* *[RFC 7748]* Langley, A., Hamburg, M., and S. Turner, "`Elliptic +Curves for Security`", RFC 7748, DOI 10.17487/RFC7748, January 2016. +* *[RFC 8439]* Nir, Y. and A. Langley, "`ChaCha20 and Poly1305 for IETF +Protocols`", RFC 8439, DOI 10.17487/RFC8439, June 2018. +* *[RFC 5869]* Krawczyk, H. and P. Eronen, "`HMAC-based +Extract-and-Expand Key Derivation Function (HKDF)`", RFC 5869, DOI +10.17487/RFC5869, May 2010. + +==== 10.2. Informative References + +* *[draft-cui-ai-agent-discovery-invocation-00]* Cui, Y., et al., "`AI +Agent Discovery and Invocation`", Internet-Draft, 2025. +* *[draft-narajala-ans-00]* Narajala, S., et al., "`Agent Naming +Service`", Internet-Draft, 2025. +* *[draft-mp-agntcy-ads]* Petrovic, M., et al., "`Agent Discovery +Service`", Internet-Draft, 2025. +* *[SWIM]* Das, A., Gupta, I., and A. Motivala, "`SWIM: Scalable +Weakly-consistent Infection-style Process Group Membership Protocol`", +Proceedings of the International Conference on Dependable Systems and +Networks, 2002. + +''''' + +=== Appendix A: Packet Format Diagrams + +==== A.1. Cleartext Packet Header + +.... + 0 1 2 3 + 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ +|0| Pkt Type(7) | Payload Length | | ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ | +| | +| Payload (variable) | +| | ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ +.... + +Bit 0 = 0 indicates cleartext mode. + +==== A.2. QUIC-Mode Packet Header + +.... + 0 1 2 3 + 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ +|1| Pkt Type(7) | Payload Length | | ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ | +| | +| Nonce (12 bytes) | +| | ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ +| | +| Encrypted Payload (variable) | +| | ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ +| | +| Auth Tag (16 bytes) | +| | ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ +.... + +Bit 0 = 1 indicates QUIC/encrypted mode. + +==== A.3. DISCOVER Packet Payload + +.... ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ +| Node ID Length (1 byte) | | ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ | +| Node ID (up to 64 bytes) | ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ +| Listen Port | | ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ +.... + +==== A.4. GOSSIP_DIGEST Packet Payload + +.... ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ +| | +| Catalogue Digest (32 bytes) | +| SHA-256 | +| | ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ +| Gossip Round Number | ++-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ +.... + +''''' + +=== Appendix B: Example Message Flows + +==== B.1. New Node Joining via Seed + +.... +New Node Seed Node + │ │ + │── DISCOVER ──────────────►│ + │ (node_id, port) │ + │ │ + │◄── DISCOVER_REPLY ───────│ + │ (seed_id, seed_port, │ + │ known_peers[]) │ + │ │ + │── HANDSHAKE_INIT ───────►│ + │ (public_key, node_id) │ + │ │ + │◄── HANDSHAKE_REPLY ─────│ + │ (public_key, node_id, │ + │ catalogue_digest) │ + │ │ + │── GOSSIP_DIGEST ────────►│ (encrypted, QUIC mode) + │ (catalogue_digest) │ + │ │ + │◄── GOSSIP_DIGEST_REPLY ──│ (encrypted, QUIC mode) + │ (peer_digest, │ + │ diff_entries[]) │ + │ │ + │ ... heartbeats ... │ + │──── HEARTBEAT ──────────►│ + │◄─── HEARTBEAT ──────────│ +.... + +==== B.2. Catalogue Change Propagation (3 Nodes) + +.... +Time Node A Node B Node C + t=0 loads cartridge + recomputes digest + t=5 gossip round: + selects B + ── DIGEST ──► + compares digests + (differ!) + ◄── REPLY ──── + A knows B knows + t=10 gossip round: + selects C + ── DIGEST ──────► + compares digests + (differ!) + ◄── REPLY ────── + B knows C knows + +All three nodes now aware of the new cartridge (2 rounds, O(log N)). +.... + +''''' + +=== Author’s Address + +Jonathan D.A. Jewell hyperpolymath Email: j.d.a.jewell@open.ac.uk URI: +https://github.com/hyperpolymath diff --git a/docs/papers/umoja-federation-draft.md b/docs/papers/umoja-federation-draft.md deleted file mode 100644 index 316f2758..00000000 --- a/docs/papers/umoja-federation-draft.md +++ /dev/null @@ -1,790 +0,0 @@ - - - -# Gossip-Based Capability Discovery and Synchronisation for Developer Tool Servers - -**Internet-Draft:** `draft-jewell-umoja-capability-gossip-00` -**Intended Status:** Experimental -**Author:** Jonathan D.A. Jewell -**Organisation:** hyperpolymath -**Date:** 2026-03 - ---- - -## Abstract - -This document describes the Umoja federation protocol, a gossip-based -mechanism for discovering and synchronising capability catalogues across -distributed developer tool server instances. Umoja enables isolated -server nodes — such as those implementing the Model Context Protocol -(MCP), Language Server Protocol (LSP), or Debug Adapter Protocol (DAP) — -to form a federated network where each node advertises and discovers -capabilities ("cartridges") provided by its peers. - -The protocol uses QUIC [RFC 9000] as its primary transport, with X25519 -[RFC 7748] key exchange and ChaCha20-Poly1305 [RFC 8439] AEAD encryption -for peer-to-peer confidentiality and integrity. A UDP fallback mode is -defined for constrained environments or development/testing scenarios -where QUIC support is unavailable. - -Catalogue synchronisation is achieved through anti-entropy digest -exchange using SHA-256 hashes of sorted cartridge metadata. The protocol -does not transfer cartridge binaries or executable code — only metadata -sufficient for capability discovery and version negotiation. - ---- - -## Table of Contents - -1. [Introduction](#1-introduction) -2. [Terminology](#2-terminology) -3. [Protocol Overview](#3-protocol-overview) -4. [Discovery Mechanism](#4-discovery-mechanism) -5. [Handshake and Attestation](#5-handshake-and-attestation) -6. [Gossip Protocol](#6-gossip-protocol) -7. [Catalogue Synchronisation](#7-catalogue-synchronisation) -8. [Security Considerations](#8-security-considerations) -9. [IANA Considerations](#9-iana-considerations) -10. [References](#10-references) -11. [Appendix A: Packet Format Diagrams](#appendix-a-packet-format-diagrams) -12. [Appendix B: Example Message Flows](#appendix-b-example-message-flows) -13. [Author's Address](#authors-address) - ---- - -## 1. Introduction - -### 1.1. Problem Statement - -Modern developer tool ecosystems rely on protocol servers — MCP servers -for AI agent capabilities, LSP servers for editor intelligence, DAP -servers for debugging — that operate as isolated, single-tenant -instances. Each server maintains its own capability catalogue with no -mechanism for sharing discovered capabilities across instances or -environments. - -This isolation creates several problems: - -- **Capability fragmentation.** A developer working across multiple - machines or environments cannot discover capabilities available on - peer instances. - -- **Redundant configuration.** Each server must be independently - configured with identical capability sets, even when they serve the - same organisation or project. - -- **No cross-instance awareness.** An MCP server on one machine has no - knowledge of cartridges loaded on a teammate's MCP server, even when - those cartridges would be directly useful. - -- **Scaling limitations.** Without federation, scaling developer tool - infrastructure requires manual replication of capability catalogues - across every new instance. - -### 1.2. Solution Overview - -The Umoja federation protocol addresses these problems by defining a -gossip-based mechanism for peer-to-peer capability catalogue discovery -and synchronisation. "Umoja" means "unity" in Swahili, reflecting the -protocol's goal of unifying isolated developer tool server instances -into a coherent federation. - -Key design principles: - -- **Metadata only.** The protocol exchanges capability metadata (names, - versions, hashes), never executable code or cartridge binaries. This - limits the attack surface and keeps message sizes small. - -- **QUIC-first, UDP-fallback.** Encrypted transport is the default; - cleartext UDP is available for development and testing but SHOULD NOT - be used in production deployments. - -- **SWIM-inspired failure detection.** Node liveness is tracked using a - protocol inspired by the Scalable Weakly-consistent Infection-style - Process Group Membership protocol (SWIM), with states: alive, - suspected, dead. - -- **Cryptographic attestation.** Peers attest their catalogue state via - SHA-256 digests. Catalogue hash comparison drives synchronisation - decisions. - -- **Zero-trust perimeter.** An integrated Software Defined Perimeter - (Auto-SDP) layer rejects traffic from unverified peers before it - reaches the gossip layer. - -### 1.3. Relationship to Existing Work - -This protocol is informed by, but distinct from, several IETF efforts -in the AI agent and service discovery space: - -- **draft-cui-ai-agent-discovery-invocation-00** defines a DNS-based - discovery framework for AI agents. Umoja operates below this layer, - providing gossip-based discovery for the servers that host such agents. - -- **draft-narajala-ans-00** (Agent Naming Service) proposes a - hierarchical naming scheme for AI agents. Umoja complements ANS by - providing a runtime discovery protocol that could resolve ANS names to - live server instances. - -- **draft-mp-agntcy-ads** (Agent Discovery Service) addresses agent - capability advertisement. Umoja focuses specifically on server-level - capability catalogues rather than individual agent capabilities. - -- **RFC 9000 (QUIC)** provides the encrypted transport foundation. - -Umoja does not replace any of these proposals. It fills a gap at the -infrastructure layer: where agents and tools are *hosted*, rather than -how individual agents are *identified* or *invoked*. - ---- - -## 2. Terminology - -The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", -"SHOULD", "SHOULD NOT", "RECOMMENDED", "NOT RECOMMENDED", "MAY", and -"OPTIONAL" in this document are to be interpreted as described in -BCP 14 [RFC 2119] [RFC 8174] when, and only when, they appear in all -capitals, as shown here. - -- **Node**: A single instance of a developer tool server participating - in the Umoja federation. Each node has a unique node identifier. - -- **Peer**: A remote node that the local node is aware of and - communicates with via the gossip protocol. - -- **Catalogue**: The ordered set of capabilities (cartridges) available - on a given node. Represented as a sorted list of (name, version, hash) - tuples. - -- **Cartridge**: A discrete, loadable capability module registered with - a developer tool server. Cartridges expose tools, resources, or - protocol capabilities. - -- **Digest**: A SHA-256 hash computed over the sorted catalogue entries - of a node. Used for efficient comparison of catalogue state between - peers without exchanging the full catalogue. - -- **Attestation**: The process by which a peer proves its identity and - catalogue integrity via X25519 key exchange and digest comparison. - -- **Seed Node**: A well-known bootstrap node that new peers contact to - join the federation network. Seed nodes are listed in a static - configuration file. - -- **Gossip Round**: A single cycle of the anti-entropy protocol, in - which a node selects a subset of peers and exchanges digest - information. - -- **Fanout**: The number of peers contacted during each gossip round. - ---- - -## 3. Protocol Overview - -### 3.1. Node Lifecycle - -A node progresses through the following states: - -``` -Bootstrap → Discovery → Handshake → Active → Suspected → Dead - │ ↑ │ - │ └──────────┘ - │ (recovery) - └─────────────────────────────────────────────► - (fatal error → Dead) -``` - -- **Bootstrap**: Node initialises its local catalogue digest, generates - an X25519 keypair, binds to its federation port (default 9999), and - loads its seed node list. - -- **Discovery**: Node sends DISCOVER packets to seed nodes and/or the - IPv6 multicast group (ff02::b04) to locate peers. Discovery is - repeated on a configurable interval (default: 60 seconds). - -- **Handshake**: Upon discovering a peer, the node initiates a - handshake: X25519 public key exchange, mutual node ID verification, - and catalogue digest comparison. - -- **Active**: The node participates in gossip rounds, exchanges - heartbeats, and synchronises catalogue state with peers. - -- **Suspected**: A peer that has missed heartbeats beyond the configured - timeout (default: 30 seconds) transitions to "suspected". The local - node MAY attempt indirect probes via other peers before declaring - the suspected peer dead. - -- **Dead**: A peer confirmed as unreachable. Dead peers are removed from - the active peer list after a configurable grace period. - -### 3.2. Transport - -#### 3.2.1. Encrypted Transport (Default) - -The protocol's primary transport uses AEAD-encrypted UDP datagrams with -the following cryptographic primitives. Note: while internally referred -to as "QUIC mode" in the reference implementation, this transport does -not implement the full QUIC protocol [RFC 9000] (no connection IDs, -streams, flow control, or congestion control). It uses QUIC's -cryptographic choices (X25519 + ChaCha20-Poly1305) applied directly to -UDP datagrams: - -- **Key exchange**: X25519 Elliptic Curve Diffie-Hellman [RFC 7748]. - Each node generates a long-lived identity keypair at bind time. A - per-peer shared secret is derived via X25519(local_secret, - remote_public). - -- **Authenticated encryption**: ChaCha20-Poly1305 AEAD [RFC 8439]. - All gossip and heartbeat traffic between peers with established - shared secrets is encrypted. The AEAD tag provides integrity - verification. - -- **Packet framing**: Encrypted packets are distinguished from cleartext - packets by a high-bit marker (0x80) in the first byte. The remaining - 7 bits encode the packet type. - -QUIC-mode packets carry a 12-byte nonce and 16-byte authentication tag -in addition to the encrypted payload. - -#### 3.2.2. UDP Fallback - -For environments where QUIC is unavailable or during initial development -and testing, a cleartext UDP mode is defined. In this mode: - -- Packet types are indicated by a single-byte tag (0x01–0x07) without - the high-bit marker. -- No encryption or authentication is applied. -- Nodes SHOULD log a warning when operating in UDP fallback mode. -- UDP fallback mode MUST NOT be used in production deployments handling - sensitive capability metadata. - -#### 3.2.3. Port Assignment - -The default federation port is **9999** (UDP). This port is used for -both QUIC-mode and UDP-fallback-mode traffic. The port is configurable -via the `BOJ_FEDERATION_PORT` environment variable. - -### 3.3. Packet Types - -| Tag (cleartext) | Tag (encrypted) | Name | Direction | -|:---------------:|:---------------:|---------------------|-----------------| -| 0x01 | 0x81 | DISCOVER | Multicast / Unicast | -| 0x02 | 0x82 | DISCOVER_REPLY | Unicast | -| 0x03 | 0x83 | GOSSIP_DIGEST | Unicast | -| 0x04 | 0x84 | GOSSIP_DIGEST_REPLY | Unicast | -| 0x05 | 0x85 | HANDSHAKE_INIT | Unicast | -| 0x06 | 0x86 | HANDSHAKE_REPLY | Unicast | -| 0x07 | 0x87 | HEARTBEAT | Unicast | - -All packets MUST fit within a single UDP datagram. The maximum packet -payload size is 1024 bytes. Implementations MUST discard packets -exceeding this limit. - ---- - -## 4. Discovery Mechanism - -### 4.1. Link-Local Discovery (IPv6 Multicast) - -For nodes on the same network segment, Umoja uses IPv6 multicast group -**ff02::b04** (link-local scope) for peer discovery. The multicast -address encodes "b04" (a mnemonic for "boj" in hexadecimal). - -A discovering node sends a DISCOVER packet to the multicast group. -Any node listening on the federation port that receives this packet -SHOULD reply with a DISCOVER_REPLY containing its node ID, listen -address, and current catalogue digest. - -Link-local discovery is useful for development environments, CI/CD -clusters, and any scenario where nodes share a network segment. - -### 4.2. WAN Bootstrap (Seed Nodes) - -For wide-area federation, nodes are configured with a list of seed nodes. -Seed nodes are well-known, stable federation endpoints that serve as -bootstrap rendezvous points. - -The seed node list is specified in a TOML configuration file: - -```toml -[metadata] -version = "0.1.0" -network = "umoja-mainnet" -min-seeds-for-quorum = 2 - -[[seed]] -id = "seed-eu-west" -region = "eu-west-1" -host = "eu.boj.hyperpolymath.dev" -federation-port = 9999 -``` - -A node MUST attempt to contact at least `min-seeds-for-quorum` seed -nodes during bootstrap. If fewer than `min-seeds-for-quorum` seeds -respond, the node MAY operate in a degraded mode with reduced federation -capabilities. - -### 4.3. Discovery Interval - -Discovery is performed periodically at a configurable interval -(default: 60 seconds). The interval SHOULD be jittered by +/- 10% to -avoid thundering-herd effects across simultaneously-booted clusters. - ---- - -## 5. Handshake and Attestation - -### 5.1. Handshake Flow - -Upon discovering a new peer, a node initiates a three-step handshake: - -``` - Node A Node B - │ │ - │── HANDSHAKE_INIT ────────────►│ - │ (A's public key, node ID) │ - │ │ - │◄── HANDSHAKE_REPLY ──────────│ - │ (B's public key, node ID, │ - │ catalogue digest) │ - │ │ - │── GOSSIP_DIGEST ────────────►│ - │ (A's catalogue digest, │ - │ encrypted if QUIC) │ - │ │ -``` - -### 5.2. Handshake States - -Each peer relationship progresses through the following states: - -| State | Meaning | -|-------------|-------------------------------------------------------| -| `none` | No handshake attempted | -| `pending` | HANDSHAKE_INIT sent, awaiting reply | -| `exchanged` | Keys exchanged, catalogue digests being compared | -| `verified` | Peer identity confirmed, digests match or sync in progress | -| `rejected` | Peer rejected (failed authentication or banned by SDP) | - -A peer in `verified` state is eligible for gossip rounds and heartbeat -exchange. A peer in `rejected` state MUST NOT receive gossip traffic and -SHOULD be reported to the Auto-SDP layer. - -### 5.3. Key Exchange - -The handshake uses X25519 [RFC 7748] for key exchange: - -1. Each node generates a long-lived X25519 keypair at bind time. -2. HANDSHAKE_INIT carries the initiator's 32-byte public key. -3. HANDSHAKE_REPLY carries the responder's 32-byte public key. -4. Both nodes derive a shared secret: `shared = X25519(local_secret, remote_public)`. -5. The shared secret is used as the ChaCha20-Poly1305 key for all - subsequent encrypted communication with that peer. - -### 5.4. Catalogue Digest Comparison - -During handshake, both nodes exchange their current catalogue digest -(SHA-256 of sorted cartridge metadata). If the digests differ, -catalogue synchronisation (Section 7) is triggered immediately after -the handshake completes. - ---- - -## 6. Gossip Protocol - -### 6.1. Protocol Model - -The Umoja gossip protocol is inspired by the SWIM (Scalable -Weakly-consistent Infection-style Process Group Membership) protocol, -adapted for capability catalogue synchronisation rather than process -group membership. - -### 6.2. Gossip Rounds - -Each gossip round proceeds as follows: - -1. The local node selects up to `fanout` (default: 3) random peers from - its active peer list. -2. For each selected peer, the node sends a GOSSIP_DIGEST packet - containing its current catalogue digest. -3. The recipient compares the received digest against its own catalogue - digest. -4. If the digests differ, the recipient replies with a - GOSSIP_DIGEST_REPLY containing its own digest and the full list of - cartridge metadata entries that differ. -5. The originator processes the reply and updates its view of the peer's - catalogue. - -### 6.3. Configurable Parameters - -| Parameter | Default | Description | -|----------------------|---------|--------------------------------------------| -| `gossip-interval-ms` | 5000 | Time between gossip rounds (milliseconds) | -| `gossip-fanout` | 3 | Number of peers contacted per round | -| `heartbeat-interval-ms` | 10000 | Time between heartbeat packets | -| `heartbeat-timeout-ms` | 30000 | Time before a peer is marked "suspected" | -| `max-peers` | 128 | Maximum number of tracked peers | -| `min-peers` | 2 | Minimum peers for healthy federation | - -### 6.4. Failure Detection - -Failure detection uses heartbeat monitoring: - -1. Each active peer MUST send HEARTBEAT packets at the configured - heartbeat interval (default: 10 seconds). -2. If no heartbeat is received from a peer within the heartbeat timeout - (default: 30 seconds), the peer transitions from `alive` to - `suspected`. -3. A suspected peer MAY be probed indirectly: the local node asks - another peer to probe the suspected node. If the indirect probe - succeeds, the peer returns to `alive`. -4. If the suspected peer remains unreachable after `unhealthy-threshold` - (default: 3) consecutive missed heartbeat cycles, it transitions to - `dead`. -5. A dead peer MAY be resurrected if it re-establishes contact and - completes a fresh handshake. After `recovery-threshold` (default: 2) - successful heartbeats, it returns to `alive`. - -### 6.5. Peer Selection - -Peer selection for gossip rounds uses a lightweight PRNG (xorshift32, -seeded from the system timestamp). The selection algorithm is not -required to be cryptographically secure — it needs only to provide fair -distribution across the active peer set to ensure convergence. - ---- - -## 7. Catalogue Synchronisation - -### 7.1. Digest Computation - -A catalogue digest is computed as follows: - -1. Collect all loaded cartridge entries as strings of the form - `"{name}:{version}:{hash}"`. -2. Sort the entries lexicographically. -3. Concatenate the sorted entries with newline separators. -4. Compute the SHA-256 hash of the resulting string. - -The digest is a 32-byte value. Implementations MUST support catalogues -of up to 128 cartridge entries. - -### 7.2. Synchronisation Trigger - -Catalogue synchronisation is triggered when: - -- Two peers exchange digests (via handshake or gossip round) and the - digests differ. - -### 7.3. Synchronisation Scope - -Synchronisation exchanges *metadata only*: - -- Cartridge name (string) -- Cartridge version (string, semver) -- Cartridge content hash (SHA-256 hex string) -- Cartridge tier (e.g. "teranga", "shield", "ayo") -- Protocol columns supported (e.g. "mcp", "lsp", "dap", "bsp") - -Synchronisation MUST NOT transfer: - -- Cartridge binaries (.so, .dll, .dylib files) -- Cartridge source code -- User data or configuration -- Authentication credentials - -### 7.4. Convergence - -Under stable network conditions, the gossip protocol ensures eventual -convergence of catalogue views across all federated nodes. The expected -convergence time for a change to propagate to all N nodes is -O(log N) gossip rounds, assuming a fanout of 3. - ---- - -## 8. Security Considerations - -### 8.1. Auto-SDP Zero-Trust Perimeter - -The Umoja protocol integrates a Software Defined Perimeter (Auto-SDP) -layer that enforces zero-trust principles at the transport level: - -- All inbound federation traffic is processed by the SDP layer before - reaching the gossip protocol. -- Peers MUST be on the allow-list to send gossip or heartbeat traffic. - Unknown peers are permitted only to send DISCOVER and - HANDSHAKE_INIT packets. -- Open mode (allowing unauthenticated peers) is available for initial - seed bootstrapping but SHOULD be disabled once a federation is - established. - -### 8.2. Authentication and Banning - -- Peers that fail authentication (invalid X25519 key exchange, - mismatched node IDs, or tampered digests) increment a failure counter. -- After `ban-threshold` (default: 5) consecutive authentication - failures, the peer is automatically banned for `ban-duration` - (default: 300 seconds). -- Banned peers are tracked by node ID. All packets from banned peers - are silently dropped. -- The ban list supports up to 64 entries; when full, the oldest ban - entry is evicted. - -### 8.3. Rate Limiting - -- Per-peer rate limiting is enforced by the SDP layer (default: 100 - requests per second). -- Peers exceeding the rate limit transition to `rate_limited` policy - and excess packets are dropped. -- Rate-limited peers are not banned but MAY be banned if rate - violations persist. - -### 8.4. Hash Attestation - -- Catalogue digests are computed locally from the node's own loaded - cartridges. A node MUST NOT accept a digest from a peer as its own - catalogue state. -- Digest comparison is used for synchronisation decisions only; a - mismatched digest does not constitute an attack. However, a peer that - consistently reports different digests in rapid succession (digest - thrashing) SHOULD be flagged for investigation. - -### 8.5. Transport Security - -- Encrypted mode (X25519 + ChaCha20-Poly1305) provides confidentiality - and integrity for all gossip traffic. Note that the current design - uses long-lived identity keypairs without ephemeral key exchange, - which does not provide forward secrecy. Future revisions of this - protocol SHOULD incorporate ephemeral ECDH or a full QUIC handshake - to achieve forward secrecy. -- The shared secret derived from X25519 SHOULD be processed through - HKDF [RFC 5869] before use as a ChaCha20-Poly1305 key, rather than - used directly. The reference implementation currently uses the raw - shared secret; this is a known limitation. -- Implementations MUST track received nonces per peer to prevent replay - attacks. Nonces SHOULD be counter-based (monotonically increasing) - rather than random to enable efficient duplicate detection. -- UDP fallback mode provides none of these properties and MUST NOT be - used in production. -- Implementations SHOULD default to encrypted mode and require explicit - configuration to enable UDP fallback. - -### 8.6. No Code Execution - -The protocol exchanges metadata only. Implementations MUST NOT execute, -load, or interpret any data received via the federation protocol as -code. Cartridge installation from federated peers requires an -out-of-band mechanism with its own authentication and integrity -verification. - ---- - -## 9. IANA Considerations - -### 9.1. Port Number Registration - -This document requests registration of the following port number: - -| Service Name | Port Number | Transport Protocol | Description | -|-------------|-------------|-------------------|-------------| -| umoja-fed | 9999 | UDP | Umoja Federation Protocol | - -**Note:** Port 9999 is currently assigned to the "distinct" service in -the IANA Service Name and Transport Protocol Port Number Registry. The -reference implementation uses 9999 as a configurable default. A formal -port allocation from the User Ports range (1024-49151) will be -requested if this protocol progresses beyond Experimental status. -Implementations MUST support configurable port assignment. - -### 9.2. IPv6 Multicast Address - -This document requests allocation of the following IPv6 multicast -address from the Link-Local Scope Multicast Addresses registry: - -| Address | Description | -|------------|--------------------------------------| -| ff02::b04 | Umoja Federation Discovery (link-local) | - ---- - -## 10. References - -### 10.1. Normative References - -- **[RFC 2119]** Bradner, S., "Key words for use in RFCs to Indicate - Requirement Levels", BCP 14, RFC 2119, DOI 10.17487/RFC2119, - March 1997. - -- **[RFC 8174]** Leiba, B., "Ambiguity of Uppercase vs Lowercase in - RFC 2119 Key Words", BCP 14, RFC 8174, DOI 10.17487/RFC8174, - May 2017. - -- **[RFC 9000]** Iyengar, J., Ed. and M. Thomson, Ed., "QUIC: A - UDP-Based Multiplexed and Secure Transport", RFC 9000, - DOI 10.17487/RFC9000, May 2021. - -- **[RFC 7748]** Langley, A., Hamburg, M., and S. Turner, "Elliptic - Curves for Security", RFC 7748, DOI 10.17487/RFC7748, January 2016. - -- **[RFC 8439]** Nir, Y. and A. Langley, "ChaCha20 and Poly1305 for - IETF Protocols", RFC 8439, DOI 10.17487/RFC8439, June 2018. - -- **[RFC 5869]** Krawczyk, H. and P. Eronen, "HMAC-based - Extract-and-Expand Key Derivation Function (HKDF)", RFC 5869, - DOI 10.17487/RFC5869, May 2010. - -### 10.2. Informative References - -- **[draft-cui-ai-agent-discovery-invocation-00]** Cui, Y., et al., - "AI Agent Discovery and Invocation", Internet-Draft, 2025. - -- **[draft-narajala-ans-00]** Narajala, S., et al., "Agent Naming - Service", Internet-Draft, 2025. - -- **[draft-mp-agntcy-ads]** Petrovic, M., et al., "Agent Discovery - Service", Internet-Draft, 2025. - -- **[SWIM]** Das, A., Gupta, I., and A. Motivala, "SWIM: Scalable - Weakly-consistent Infection-style Process Group Membership Protocol", - Proceedings of the International Conference on Dependable Systems and - Networks, 2002. - ---- - -## Appendix A: Packet Format Diagrams - -### A.1. Cleartext Packet Header - -``` - 0 1 2 3 - 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ -|0| Pkt Type(7) | Payload Length | | -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ | -| | -| Payload (variable) | -| | -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ -``` - -Bit 0 = 0 indicates cleartext mode. - -### A.2. QUIC-Mode Packet Header - -``` - 0 1 2 3 - 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ -|1| Pkt Type(7) | Payload Length | | -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ | -| | -| Nonce (12 bytes) | -| | -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ -| | -| Encrypted Payload (variable) | -| | -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ -| | -| Auth Tag (16 bytes) | -| | -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ -``` - -Bit 0 = 1 indicates QUIC/encrypted mode. - -### A.3. DISCOVER Packet Payload - -``` -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ -| Node ID Length (1 byte) | | -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ | -| Node ID (up to 64 bytes) | -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ -| Listen Port | | -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ -``` - -### A.4. GOSSIP_DIGEST Packet Payload - -``` -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ -| | -| Catalogue Digest (32 bytes) | -| SHA-256 | -| | -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ -| Gossip Round Number | -+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ -``` - ---- - -## Appendix B: Example Message Flows - -### B.1. New Node Joining via Seed - -``` -New Node Seed Node - │ │ - │── DISCOVER ──────────────►│ - │ (node_id, port) │ - │ │ - │◄── DISCOVER_REPLY ───────│ - │ (seed_id, seed_port, │ - │ known_peers[]) │ - │ │ - │── HANDSHAKE_INIT ───────►│ - │ (public_key, node_id) │ - │ │ - │◄── HANDSHAKE_REPLY ─────│ - │ (public_key, node_id, │ - │ catalogue_digest) │ - │ │ - │── GOSSIP_DIGEST ────────►│ (encrypted, QUIC mode) - │ (catalogue_digest) │ - │ │ - │◄── GOSSIP_DIGEST_REPLY ──│ (encrypted, QUIC mode) - │ (peer_digest, │ - │ diff_entries[]) │ - │ │ - │ ... heartbeats ... │ - │──── HEARTBEAT ──────────►│ - │◄─── HEARTBEAT ──────────│ -``` - -### B.2. Catalogue Change Propagation (3 Nodes) - -``` -Time Node A Node B Node C - t=0 loads cartridge - recomputes digest - t=5 gossip round: - selects B - ── DIGEST ──► - compares digests - (differ!) - ◄── REPLY ──── - A knows B knows - t=10 gossip round: - selects C - ── DIGEST ──────► - compares digests - (differ!) - ◄── REPLY ────── - B knows C knows - -All three nodes now aware of the new cartridge (2 rounds, O(log N)). -``` - ---- - -## Author's Address - -Jonathan D.A. Jewell -hyperpolymath -Email: j.d.a.jewell@open.ac.uk -URI: https://github.com/hyperpolymath diff --git a/docs/planning/boj-server-proof-story-2026-06-01.adoc b/docs/planning/boj-server-proof-story-2026-06-01.adoc new file mode 100644 index 00000000..72e5ae43 --- /dev/null +++ b/docs/planning/boj-server-proof-story-2026-06-01.adoc @@ -0,0 +1,581 @@ +== BoJ-Server Proof Story — 2026-06-01 + +____ +*Status*: draft for owner review. Synthesises 4 parallel +Explore-subagent reports (existing-state inventory, trust-chain map, +competitor baseline, layered roadmap) plus owner-direct verification of +a key Agent C claim. + +*Owner directive*: "`explore the whole thing and identify what proofs, +from basic assumptions upwards and all over the BoJ need to be achieved +to ensure this is the most solid narrative and most solidly proven +server that is out there`" + +*Sister artefact*: `+docs/planning/cartridge-catalogue-2026-06-01.md+` +(PR #179) for the cartridge expansion plan. This document is its +proof-rigour counterpart. +____ + +''''' + +=== 1. Headline finding + +*You are already the most formally verified MCP server in the world.* +That isn’t aspirational — it’s measured. Agent C’s competitor survey +found that essentially every other MCP server (Anthropic’s reference +set, OpenAI GPTs, Vercel AI SDK, Cloudflare Workers AI, NVIDIA verified +agent skills, the ~1000 servers in Glama’s catalogue) makes *zero* +formal-verification claims. Three exceptional cases exist (Prova-MCP, +MCPShield, Rocq-MCP) but each _uses_ formal verification — none _prove_ +their own MCP runtime. + +You already have: - 5 class-J axioms in boj-server, *all externally +validated* via backend-assurance harness — single isolated module +(`+SafetyLemmas.idr+`) - 10+ Qed/cartridge-ABI theorems landed (BJ1 +dispatch, BJ2 isolation, BJ3 protocol coverage) - typed-wasm: 22 modules +with all 10 safety levels carrier-backed, 0 `+believe_me+`/`+postulate+` +- echo-types: foundationally complete, 0 postulates, +`+--safe --without-K+`, Pillars A–D verified - ephapax: +counterexample-Qed for legacy preservation (proved-false), four-layer +redesign passing for L1+L3, 11/13 Buchholz constructors closable - +proven: trust root with ~70 witness-type overclaims now *honestly +enumerated* and being cleared (2026-05-20 re-audit) + +The narrative isn’t "`we want to be the most-proven MCP server`"; it’s +*"`we already are, and here’s how we extend the lead.`"* + +''''' + +=== 2. The trust chain — what is proven vs assumed today + +Agent B mapped 9 conceptual layers. Three layer-boundary contracts are +formally proven; four remain at prose-ADR or assumed. + +==== Layer stack (bottom-up) + +[width="100%",cols="25%,25%,25%,25%",options="header",] +|=== +|Layer |Artefacts |Proven |Assumed +|*L1 HW / OS / runtime* |BEAM, Zig+LLVM, Linux kernel |BEAM scheduler +crash-isolation (ADR-0005); Zig memory-safety model |Kernel syscall +correctness; BEAM GC no-leak; LLVM correct lowering of `+prim__*+` + +|*L2 Language runtimes* |Idris2 0.8.0, Zig type-checker, Elixir BEAM, +Coq kernel, Agda kernel |Totality checking on cartridge dispatch (BJ1) +|5 class-J axioms +(charEqSound/charEqSym/unpackLength/appendLengthSum/substrLengthBound); +Zig type-checker soundness; Cowboy parsing + +|*L3 Cartridge ABI* |16 `+src/abi/Boj/*.idr+` files, 5.4k LOC |*BJ1* +(CartridgeDispatch), *BJ2* (CredentialIsolation), *BJ3* +(APIContractCoverage) all closed 2026-05-18 |Proof composition across +modules + +|*L4 Cartridge FFI* |`+ffi/zig/src/loader.zig+`, 99 `+ffi/*_ffi.zig+` +files |dlopen symbol-presence classification |5-symbol ABI contract +honoured by cartridge author (ADR-0006 *prose-only*); memory ownership +boundary + +|*L5 Cartridge logic* |Per-cartridge Zig/Deno/Rust code |Tier-1 (11) +have formal Idris2 specs |Tier-2-6 (101) only manifest + heuristic +review; no automated check that github-api-mcp doesn’t call Slack APIs + +|*L6 Cartridge invocation* +|`+elixir/lib/boj_rest/{invoker,catalog,router}.ex+` |Invoker OS-process +isolation; ETS catalog |Supervisor restart timing; ETS concurrency; JSON +round-trip via Jason + +|*L7 Transport* |Cowboy (7700-7703), Plug router |Required-field type +check |Cowboy parsing no-crash; HTTP smuggling immunity + +|*L8 Boundary* |`+trust_policy.ex+`, HCG (ADR-0004), policy-mcp +(ADR-0007) |Loopback bypass; trust-level audit log |X-Trust-Level +authenticity (mTLS Phase B *pending*); Nickel PDP (ADR-0007 *RFC-stage*) + +|*L9 User-visible promise* |`+boj://server/info+`, manifests, README +|Cartridge name resolution; BJ2 isolation at the boundary |Tier-claim +accuracy; dispatch determinism +|=== + +==== Edge contracts (trust passes between layers) + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Edge |Contract |Status +|L2 → L3 |BJ1 dispatch |✅ proven (`+CartridgeDispatch.idr:56-66+`) + +|L3 → L4 |IsUnbreakable readiness guard |✅ proven + +|L4 → L5 |5-symbol ABI (init/deinit/name/version/invoke + 7 return +codes) |⚠️ prose-only (ADR-0006) + +|L5 → L6 |Return-code respect |⚠️ assumed + +|L6 → L7 |JSON preservation (Jason) |⚠️ external dep + +|L7 → L8 |X-Trust-Level header + loopback bypass |⚠️ pending mTLS Phase +B + +|L8 → L9 |BJ2 credential isolation |✅ proven +(`+CredentialIsolation.idr:20-38+`) +|=== + +==== 12 leaf assumptions (the load-bearing list) + +In rough priority order: + +[arabic] +. BEAM crash isolation (cartridge `+.so+` segfault doesn’t crash Elixir +VM) +. Idris2 `+believe_me+` soundness for 5 SafetyLemmas axioms +. Cartridge ABI symbol presence (all 5 symbols exported) +. Memory ownership boundary (cartridge respects +`+out_buf+`/`+in_out_len+`) +. HTTP JSON codec round-trip (Jason correctness) +. Zig type-checker correctness +. Cowboy HTTP parsing soundness +. Loopback interface kernel isolation +. Idris2 module-import coherence (acyclic, type-checker-enforced) +. Cartridge manifest accuracy (declared tier == actual implementation) +. Trust-level header authenticity (no cross-proxy forgery) +. No cartridge-to-cartridge data leakage (dlopen isolation) + +Each is a candidate "`what could we prove that we currently just +assume?`" + +==== Root promise (one sentence) + +____ +A cartridge invoked via `+POST /cartridge/:name/invoke+` with a tool +name and JSON arguments will either (a) execute the named tool with +type-safe argument dispatch and return the result, or (b) return a +classified error, with the guarantee that a Teranga-tier cartridge’s +isolation properties (BJ1 dispatch, BJ2 credential partition, BJ3 +protocol coverage) hold end-to-end, and that a cartridge crash does not +crash the BoJ server. +____ + +''''' + +=== 2.5 Echo-types — cross-cutting type-theoretic foundation + +*Owner directive 2026-06-01*: every proof wave below must first check +`+hyperpolymath/echo-types+` for a reusable definition or lemma. If +relevant, the wave consumes echo-types via a SHA-pinned import. If +absent, the wave *extends echo-types first*, proves the extension there, +and then consumes downstream. Cross-document: every consumer cites the +echo-types module + commit; the echo-types module’s `+EXPLAINME.adoc+` +lists boj-server as a downstream consumer once the first wave imports +it. + +==== Why echo-types is the right foundation + +`+echo-types+` is the estate’s constructive Agda formalisation of +_proof-relevant lossy computation_ — `+--safe --without-K+`, 0 +postulates. Its core abstractions map directly onto boj-server’s proof +obligations: + +[width="100%",cols="50%,50%",options="header",] +|=== +|boj-server invariant |echo-types primitive +|Cartridge dispatch — distinct cartridges cannot collide on a message +type |`+EchoResidueTaxonomy+` _indexed_ residue form (proof-relevant +distinguishability of dispatch keys) + +|Multi-protocol composition — outputs of `+Cᵢ+` parse under input schema +of `+Cᵢ₊₁+` |`+EchoImageFactorizationProp+` (epi-mono earn-back: the +residue of the projection bears the witness that the next stage’s +precondition is satisfied) + +|Credential isolation (BJ2) — a Teranga capability cannot leak across +cartridges |Linear / affine bridge + `+EchoSecurity+` application module + +|Audit log integrity (`+local-coord-mcp+`, MFA-001 to MFA-006) +|`+EchoProvenance+` application module (hash-chain as echo: the digest +is the residue that constrains the preimage) + +|Cost / budget proofs (Glama scoring, panel cost-meters) |Tropical +bridge + `+EchoResidueTaxonomy+` _cost_ residue form + +|Effect / capability tracking across L4–L8 boundaries |Graded modality +bridge (loss-graded reindexing per `+docs/retractions.adoc+` +R-2026-05-18) + +|Class-J axiom witnesses (5 in `+SafetyLemmas.idr+`) +|`+EchoResidueTaxonomy+` _generic Σ-cert_ residue form + +|Federated coord (ADR-0010) role-projection |Choreographic bridge + +|Adversary-knowledge bounds |Epistemic bridge +|=== + +The drift = echo + tropical cost composition (per VeriSimDB foundation +pack) is the single most natural fit for boj-server’s anchor theorem. + +==== Status of echo-types at 2026-06-01 + +Per `+echo-types/EXPLAINME.adoc+` and +`+.machine_readable/6a2/STATE.a2ml+`: + +* Core echo / fiber theorems present (`+echo-intro+`, `+map-over+`, +`+map-over-id+`, `+map-over-comp+`, `+map-square+`). +* Bridges complete: linear, graded, tropical, choreographic, epistemic, +CNO, Janus, Dyadic, Ordinal, Indexed, Relational, Categorical, Scope. +* Eight residue forms in `+EchoResidueTaxonomy+` (trivial, identity, +generic Σ-cert, linear-affine, indexed, cost, search, epistemic). +* Investigation EI-2 (integration-recipe distinctness) terminated +negatively via PATH B — *do not reopen*; treat as a settled negative +result. +* Ordinal/Buchholz track: 11 of 13 per-constructor rank-mono cases +closed; Slice-3 headline closed via Route A in PR #142/#143. + +This means W1 and W2 of boj-server’s proof roadmap can be expressed +_today_ in echo-types vocabulary without extension. W3-W6 likely require +small extensions — to be identified per wave as the work begins. + +==== Extension policy + +When a wave needs a definition not in echo-types: + +[arabic] +. File the gap as an echo-types issue (`+hyperpolymath/echo-types+`), +referencing the boj-server wave + theorem name. +. Land the extension in echo-types first (small, focused PR; passes +`+--safe --without-K+`; no new postulates). +. SHA-pin the echo-types import in the boj-server proof PR. +. Echo-types `+EXPLAINME.adoc+` "`Applied prototype hook`" or +downstream-consumers section lists boj-server. + +This keeps echo-types as the proof-foundation single-source-of-truth and +prevents duplicate type definitions drifting across the estate. + +''''' + +=== 3. Competitive context — what the rest of the field claims + +==== The baseline + +*Most MCP / agent-runtime systems make zero formal claims.* Verified +examples: + +* *Anthropic’s reference MCP servers* +(`+github.com/modelcontextprotocol/servers+`) — disclaim formal +verification explicitly. +* *OpenAI GPTs / actions* — prompt engineering + human-in-the-loop, no +proofs. +* *Vercel AI SDK* — testing utilities + types, no formal verification. +* *Cloudflare Workers AI* — runtime scanning + semantic intent, not +formal correctness. +* *Glama catalogue (~1000 servers)* — searching for +"`verified`"/"`formal`"/"`proven`"/"`Coq`"/"`Idris`"/"`Agda`"/"`Lean`"/"`TLA+`" +returns essentially nothing. +* *NVIDIA "`verified agent skills`"* — cryptographic signing + automated +vuln scanning, not formal proof. + +==== The 3 exceptional cases (each scoped narrowly) + +[arabic] +. *Prova-MCP* — agents verify their own reasoning chains by +kernel-checking Lean 4 proofs. Scope: agent-assisted theorem proving, +not runtime safety of the server. +. *MCPShield* (arxiv 2604.05969) — labeled-transition-system formal +threat framework, 91% claimed coverage across a 7-category 23-vector +threat taxonomy. Published peer-reviewed framework, not deployed at +runtime. +. *Rocq-MCP* — exposes Rocq (Coq-family) as MCP tools. *Uses* MCP to do +proofs; doesn’t prove MCP. + +==== The prior-art ceiling (outside agent space) + +* *seL4* — 8.7k C + 600 asm functional-correctness proof in +Isabelle/HOL. +* *CompCert* — semantics-preserving C→assembly compiler proof. +* *Project Everest / EverCrypt* — 124k lines verified F* in real-world +production (Linux, Firefox, Tezos). +* *CHERI / VeriCHERI* — RTL-level formal verification of capability +hardware. + +==== 3 positioning framings — each defensible if executed + +[arabic] +. *"`First MCP server with formally verified cartridge loading`"* — +anchor on BJ1 + extend to cover the dlopen + symbol presence + signature +compatibility chain. +. *"`First formally verified capability gateway for multi-cartridge +agent integration`"* — anchor on BJ2 + extend to the L8 boundary. +Positions directly against NVIDIA’s signing model. +. *"`Formally proven supply-chain safety for federated MCP cartridges`"* +— anchor on the federated coord ADR-0010 + provenance / SBOM +verification. + +None of these are claimed by anyone today. + +''''' + +=== 4. Roadmap — 6 waves, ~40-50 proof days, 18-24 months + +Agent D’s plan, condensed. Each wave estimate is solo-with-Joshua at +~6-8 weeks proof capacity per quarter. + +==== W1 — Cartridge-layer type preservation (Weeks 1-4, ~8 days) + +* `+local-coord-mcp+` closes P-04/P-05/P-06/P-07 (record format, CRC +truncation, replay-equivalence, quarantine state machine) — 6 days, +infrastructure already present. +* `+007-mcp+` policy-apply type-safety — ~1 day. +* One domain cartridge (e.g., `+dap-mcp+` or `+bsp-mcp+`) protocol +dispatch uniqueness lemma — ~1 day. +* *Echo-types import*: `+EchoResidueTaxonomy+` _indexed_ residue form +(dispatch keys); `+EchoProvenance+` for replay-equivalence as hash-chain +echo. *No extension expected* — both present at echo-types HEAD. + +==== W2 — Invocation protocol soundness + multi-protocol composition (Weeks 5-12, ~8 days) + +* `+CartridgeDispatch.invokeSound+` (direct-invoke preserves type) — 2 +days. +* *`+MultiProtocol.invokeChainSoundness+`* ← the anchor theorem (see §5) +— 3 days. +* `+sseFrameIntegrity+` — 1 day. +* Aligns with typed-wasm Phase 2 (L2 access-site carrier, ADR-0003 +accepted 2026-05-30) as a case-study consumer. +* *Echo-types import*: `+EchoImageFactorizationProp+` (the anchor +theorem statement _is_ an epi-mono earn-back across cartridge +boundaries); graded modality bridge for invocation-effect tracking. *No +extension expected* — Tier 2 EchoImageFactorizationProp landed +2026-05-28. + +==== W3 — Capability containment + vault isolation (Weeks 13-18, ~5 days) + +* `+VaultIsolation+` upgrade to dynamic isolation (cartridges added +post-init) — 2 days. +* `+CapabilityContainment.borrowCap+` (capability temporary-borrow, not +store-or-re-export) — 2 days. +* `+CredentialFlow.credentialCannotEscape+` — 1 day. +* *Echo-types import*: linear/affine bridge + `+EchoSecurity+` +application module. *Likely extension*: capability _borrow-and-return_ +may need a new linear-affine variant in echo-types — file as echo-types +issue, land extension first. + +==== W4 — End-to-end safety case (Weeks 19-24, ~10 days) + +* `+SafetyCase.e2eInferenceSound+` — the umbrella theorem composing +W1-W3 lemmas — 5 days bookkeeping. +* `+CompositionLemma.multiCartridgeChain+` — chain induction — 2 days. +* `+AdversarialModel+` — negative lemmas (can’t forge IDs, can’t bypass +isolation, can’t corrupt dispatch) — 1 day. +* *Echo-types import*: tropical bridge (cost composition under chain), +`+EchoProvenance+` (audit trail across the chain), epistemic bridge +(adversary knowledge bound). *Possible extension*: composition lemma for +chained echo factorizations — likely covered by `+map-over-comp+` but +may need a chain-specific lemma; file as echo-types issue if so. + +==== W5 — Backend-assurance expansion (Weeks 25-28, ~4 days) + +* Audit + externally validate any new class-J axioms introduced by new +cartridges — 2 days. +* *Harness mechanisation* — formalise the discipline itself in Coq or +Agda: "`a class-J axiom is valid iff (trusted-extraction doc + property +test + BEAM evidence)`" — 2 days. +* *Echo-types import*: `+EchoResidueTaxonomy+` _generic Σ-cert_ residue +form — class-J axioms are precisely Σ-cert residues with +external-evidence witnesses. *Extension expected*: a new residue form +"`__externally-validated__`" or a refinement of generic Σ-cert with a +backend-assurance side-condition. File as echo-types issue first. + +==== W6 — Publication + standoff (Weeks 29+, ~2 days/cartridge) + +* Technical report on the W4 e2e proof. +* Ready 3-5 proof-bearing cartridges for production. +* Establish proof-maintenance policy (within 2 weeks of any +proof-bearing PR, a `+docs/proof-summary.md+` follow-up must cite which +theorems cover which invariants). +* *Echo-types cross-document*: by W6 each of W1-W5’s downstream +consumers should be listed in echo-types `+EXPLAINME.adoc+` under a new +"`downstream consumers`" section. Publication framing: "`boj-server is +the first capability-gateway _consumer_ of the echo-types foundation; +echo-types is its proof bedrock.`" + +==== Top-3 quick wins (1.5-2 total days — front-load before W1) + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Theorem |Statement |Days +|`+CartridgeDispatch.noCollisions+` |Dispatch is injective — no two +cartridges accidentally handle the same message type |0.5 + +|`+SafeLocalCoord.replayDeterminism+` |Replay log iterator on the same +BitStream prefix produces the same state — structural / by reflexivity +|0.5 + +|`+SafetyLemmas.axiomsAreIrreducible+` |Negative proof that the 5 +class-J axioms cannot be discharged in Idris2 0.8.0 because Char/String +have no constructors |1 +|=== + +These three are quotable in a paper / pitch deck without waiting for W4. + +''''' + +=== 5. The anchor theorem + +==== `+MultiProtocol.invokeChainSoundness+` + +*Informal statement*: Given a sequence of cartridge invocations C₁ → C₂ +→ C₃ where each cartridge correctly implements its protocol contract, +the outputs of each cartridge can always be parsed by the next +cartridge’s input schema; no type error can arise mid-chain, even if the +invocations span different domains (e.g., OAuth → database → LLM). + +==== Why this one + +* *Genuinely novel*: no other agent runtime — Claude, Anthropic’s MCP, +OpenAI’s function-calling, Vercel, NVIDIA — has formally proven that +chained cartridge invocations preserve type safety. typed-wasm covers +access-site safety; proven covers individual library safety; *composing +agent operations across trust boundaries is unique to boj-server’s +claim*. +* *Load-bearing*: W2 gates W3 (capability containment is only meaningful +if invokes are sound), W3 gates W4 (e2e safety). This proof unblocks two +full waves. +* *Publishable*: theorem statement is elegant enough for JFLA (Journées +Francophones des Langages Applicatifs) or an ICFP workshop. "`Formally +verified agent orchestration`" is a hookline. + +*Effort*: 3-5 days in W2. + +''''' + +=== 6. Decision points for owner + +==== D1 — positioning framing (pick one or commit to all three) + +[arabic] +. Cartridge-loading verified (anchor on BJ1 + extend to dlopen chain) +. Capability-gateway verified (anchor on BJ2 + extend to L8) +. Supply-chain federation verified (anchor on ADR-0010 + SBOM) + +All three are achievable; the question is what to put on the front page. +Recommend D1.2 (capability gateway) — it’s the framing closest to your +current proof artefacts, and "`first formally verified capability +gateway for LLM agents`" parses as a sentence even to someone who +doesn’t know what a cartridge is. + +==== D2 — accept the 18-24-month timeline, or compress? + +Solo + Joshua at 6-8 weeks proof per quarter ≈ 30 weeks of proof work +over 18 months, with W4 as the big push. Compressing requires either (a) +more proof help (a collaborator with Coq/Idris2 fluency), or (b) +deferring the anchor theorem and shipping incremental wave reports. + +==== D3 — where do the 4 still-prose ADRs fit? + +* *ADR-0006* (5-symbol cartridge ABI) — currently prose-only. Formalise +in W2 alongside `+invokeSound+`. +* *ADR-0004* (HCG mTLS Phase B) — operational not provable; gate the +L7→L8 edge until it lands. +* *ADR-0007* (Nickel PDP DSL) — still RFC. Has its own proof-debt; defer +to a separate Nickel-track. +* *ADR-0010* (federated coord + ML-KEM) — proposed only. Treat as Phase +4+ or after W6. + +==== D4 — proof-debt tracking discipline + +boj-server’s `+PROOF-NEEDS.md+` and `+docs/proof-debt.md+` are already +exemplary (Agent A says: estate reference for the Trusted-Base Reduction +Policy). Question: do you want a *per-cartridge* `+proof-summary.md+` +requirement (W6 policy), or just per-repo? Per-cartridge gives much +higher resolution but is high-overhead. + +==== D5 — echo-types extension governance + +Per the §2.5 owner directive, when a wave needs an echo-types extension, +the workflow is: file as echo-types issue → land extension there first → +SHA-pin in downstream boj-server PR → cross-link in `+EXPLAINME.adoc+`. +Open sub-questions: - *Repo of record for boj-server-specific instances* +— when an `+EchoResidueTaxonomy+` instance is _only_ boj-server-relevant +(e.g. a "`cartridge-tier-validated`" instance), does it live in +echo-types (estate-wide) or boj-server (local)? Recommend echo-types for +any instance with a re-usable shape, boj-server for one-off. - *Pace* — +echo-types extension PRs must pass `+--safe --without-K+` and add no +postulates. This is strict; W3 + W5 extensions may need 1-2 extra days +each. - *Reciprocal documentation* — at what cadence does echo-types’ +`+EXPLAINME.adoc+` get refreshed with the downstream-consumers list? +Recommend at each wave-completion checkpoint. + +''''' + +=== 7. Open questions + +[arabic] +. *Joshua’s involvement in proof work specifically* — Agent D’s roadmap +assumes he can help with bookkeeping in W4. Is that realistic / desired? +If he’s primarily a cartridge implementer, the 18-24 mo timeline is +brittle. +. *Publication venues* — JFLA / ICFP workshop / a position paper at a +security conference. The anchor-theorem framing changes the right venue. +POPL is too theory-heavy; CCS / S&P would land if framed as capability +containment. +. *Trust-base re-audit cadence* — proven’s 2026-05-20 honesty refresh +found 70 overclaims. Should that audit be quarterly across all repos, or +once-then-static? +. *Tooling-stub remediation interaction* — per Q2 (cartridge minter +retired no replacement, 3 stubs remain), if catalogue expansion follows +the recommendation to rewrite all 4 tools in Rust/Zig, those tools +become part of the trust chain too. New `+tools/+` deserve at least +Eno-tier discipline. +. *License clarity for the proof corpus* — boj-server is now +AGPL-3.0-or-later (PR #157). The proven library it depends on — what +license? Re-export terms for someone consuming boj-server’s BJ1/BJ2/BJ3 +proof artefacts? + +''''' + +=== 8. Phasing relative to the cartridge catalogue + +The catalogue document (PR #179) proposed a 5-phase rollout (tooling → +backfill → high-leverage waves → depth fills → exotic). The proof-story +phases overlap: + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|Catalogue phase |Proof phase |Interaction +|Phase 0 — tooling |(no proof work) |The 4 tools-to-rewrite become +L4-adjacent infrastructure; they should sit at trust-tier Eno minimum + +|Phase 1 — 14-LSP backfill |Proof quick-wins (W0) |The LSPs come in as +Ayo / Eno; quick-win theorems land in parallel + +|Phase 2 — high-leverage waves |W1-W2 |Vector-DB / local-inference waves +are mostly Ayo cartridges; the _runtime_ improvements +(e.g. invokeChainSoundness) lift their tier ceiling + +|Phase 3 — depth fills |W3-W4 |Capability containment + e2e safety case +land here, anchoring the security half of the catalogue + +|Phase 4 — exotic |W5-W6 |Backend-assurance expansion + publication; +exotic cartridges are too narrow for W6’s reference-cartridge slot +|=== + +''''' + +=== 9. Provenance + +* 4 Explore subagents fanned out 2026-06-01 ~13:30Z; all returned within +~12 minutes (existing state), ~6 minutes (trust chain), ~13 minutes +(competitor baseline), ~6 minutes (roadmap). +* Owner-direct verification corrected Agent C’s claim about +`+launch-scaffolder+` being the cartridge minter’s replacement (it is +not — it’s a desktop launcher generator). The minter is retired with no +replacement, deepening the Q2 stub-rewrite scope from 3 tools to 4. +* Subagent transcripts at +`+/tmp/claude-1000/-home-hyperpolymath-developer-repos/.../tasks/+`. +* *2026-06-01 amendment*: section 2.5 (echo-types foundation), per-wave +echo-types module mapping in section 4, and section 6 D5 (echo-types +extension governance) added per owner directive: "`in the proofs you do, +you need to check the echo-types repo and make sure this is part of the +proofing for the repo, if not, establish the extension to what is there +and work down that path proving as you go, then cross document`". +Echo-types module references sourced from `+echo-types/EXPLAINME.adoc+` +at 2026-06-01 HEAD. + +🤖 Generated with https://claude.com/claude-code[Claude Code] diff --git a/docs/planning/boj-server-proof-story-2026-06-01.md b/docs/planning/boj-server-proof-story-2026-06-01.md deleted file mode 100644 index 863b008a..00000000 --- a/docs/planning/boj-server-proof-story-2026-06-01.md +++ /dev/null @@ -1,298 +0,0 @@ - - -# BoJ-Server Proof Story — 2026-06-01 - -> **Status**: draft for owner review. Synthesises 4 parallel Explore-subagent reports (existing-state inventory, trust-chain map, competitor baseline, layered roadmap) plus owner-direct verification of a key Agent C claim. -> -> **Owner directive**: "explore the whole thing and identify what proofs, from basic assumptions upwards and all over the BoJ need to be achieved to ensure this is the most solid narrative and most solidly proven server that is out there" -> -> **Sister artefact**: `docs/planning/cartridge-catalogue-2026-06-01.md` (PR #179) for the cartridge expansion plan. This document is its proof-rigour counterpart. - ---- - -## 1. Headline finding - -**You are already the most formally verified MCP server in the world.** That isn't aspirational — it's measured. Agent C's competitor survey found that essentially every other MCP server (Anthropic's reference set, OpenAI GPTs, Vercel AI SDK, Cloudflare Workers AI, NVIDIA verified agent skills, the ~1000 servers in Glama's catalogue) makes **zero** formal-verification claims. Three exceptional cases exist (Prova-MCP, MCPShield, Rocq-MCP) but each *uses* formal verification — none *prove* their own MCP runtime. - -You already have: -- 5 class-J axioms in boj-server, **all externally validated** via backend-assurance harness — single isolated module (`SafetyLemmas.idr`) -- 10+ Qed/cartridge-ABI theorems landed (BJ1 dispatch, BJ2 isolation, BJ3 protocol coverage) -- typed-wasm: 22 modules with all 10 safety levels carrier-backed, 0 `believe_me`/`postulate` -- echo-types: foundationally complete, 0 postulates, `--safe --without-K`, Pillars A–D verified -- ephapax: counterexample-Qed for legacy preservation (proved-false), four-layer redesign passing for L1+L3, 11/13 Buchholz constructors closable -- proven: trust root with ~70 witness-type overclaims now **honestly enumerated** and being cleared (2026-05-20 re-audit) - -The narrative isn't "we want to be the most-proven MCP server"; it's **"we already are, and here's how we extend the lead."** - ---- - -## 2. The trust chain — what is proven vs assumed today - -Agent B mapped 9 conceptual layers. Three layer-boundary contracts are formally proven; four remain at prose-ADR or assumed. - -### Layer stack (bottom-up) - -| Layer | Artefacts | Proven | Assumed | -|---|---|---|---| -| **L1 HW / OS / runtime** | BEAM, Zig+LLVM, Linux kernel | BEAM scheduler crash-isolation (ADR-0005); Zig memory-safety model | Kernel syscall correctness; BEAM GC no-leak; LLVM correct lowering of `prim__*` | -| **L2 Language runtimes** | Idris2 0.8.0, Zig type-checker, Elixir BEAM, Coq kernel, Agda kernel | Totality checking on cartridge dispatch (BJ1) | 5 class-J axioms (charEqSound/charEqSym/unpackLength/appendLengthSum/substrLengthBound); Zig type-checker soundness; Cowboy parsing | -| **L3 Cartridge ABI** | 16 `src/abi/Boj/*.idr` files, 5.4k LOC | **BJ1** (CartridgeDispatch), **BJ2** (CredentialIsolation), **BJ3** (APIContractCoverage) all closed 2026-05-18 | Proof composition across modules | -| **L4 Cartridge FFI** | `ffi/zig/src/loader.zig`, 99 `ffi/*_ffi.zig` files | dlopen symbol-presence classification | 5-symbol ABI contract honoured by cartridge author (ADR-0006 **prose-only**); memory ownership boundary | -| **L5 Cartridge logic** | Per-cartridge Zig/Deno/Rust code | Tier-1 (11) have formal Idris2 specs | Tier-2-6 (101) only manifest + heuristic review; no automated check that github-api-mcp doesn't call Slack APIs | -| **L6 Cartridge invocation** | `elixir/lib/boj_rest/{invoker,catalog,router}.ex` | Invoker OS-process isolation; ETS catalog | Supervisor restart timing; ETS concurrency; JSON round-trip via Jason | -| **L7 Transport** | Cowboy (7700-7703), Plug router | Required-field type check | Cowboy parsing no-crash; HTTP smuggling immunity | -| **L8 Boundary** | `trust_policy.ex`, HCG (ADR-0004), policy-mcp (ADR-0007) | Loopback bypass; trust-level audit log | X-Trust-Level authenticity (mTLS Phase B **pending**); Nickel PDP (ADR-0007 **RFC-stage**) | -| **L9 User-visible promise** | `boj://server/info`, manifests, README | Cartridge name resolution; BJ2 isolation at the boundary | Tier-claim accuracy; dispatch determinism | - -### Edge contracts (trust passes between layers) - -| Edge | Contract | Status | -|---|---|---| -| L2 → L3 | BJ1 dispatch | ✅ proven (`CartridgeDispatch.idr:56-66`) | -| L3 → L4 | IsUnbreakable readiness guard | ✅ proven | -| L4 → L5 | 5-symbol ABI (init/deinit/name/version/invoke + 7 return codes) | ⚠️ prose-only (ADR-0006) | -| L5 → L6 | Return-code respect | ⚠️ assumed | -| L6 → L7 | JSON preservation (Jason) | ⚠️ external dep | -| L7 → L8 | X-Trust-Level header + loopback bypass | ⚠️ pending mTLS Phase B | -| L8 → L9 | BJ2 credential isolation | ✅ proven (`CredentialIsolation.idr:20-38`) | - -### 12 leaf assumptions (the load-bearing list) - -In rough priority order: - -1. BEAM crash isolation (cartridge `.so` segfault doesn't crash Elixir VM) -2. Idris2 `believe_me` soundness for 5 SafetyLemmas axioms -3. Cartridge ABI symbol presence (all 5 symbols exported) -4. Memory ownership boundary (cartridge respects `out_buf`/`in_out_len`) -5. HTTP JSON codec round-trip (Jason correctness) -6. Zig type-checker correctness -7. Cowboy HTTP parsing soundness -8. Loopback interface kernel isolation -9. Idris2 module-import coherence (acyclic, type-checker-enforced) -10. Cartridge manifest accuracy (declared tier == actual implementation) -11. Trust-level header authenticity (no cross-proxy forgery) -12. No cartridge-to-cartridge data leakage (dlopen isolation) - -Each is a candidate "what could we prove that we currently just assume?" - -### Root promise (one sentence) - -> A cartridge invoked via `POST /cartridge/:name/invoke` with a tool name and JSON arguments will either (a) execute the named tool with type-safe argument dispatch and return the result, or (b) return a classified error, with the guarantee that a Teranga-tier cartridge's isolation properties (BJ1 dispatch, BJ2 credential partition, BJ3 protocol coverage) hold end-to-end, and that a cartridge crash does not crash the BoJ server. - ---- - -## 2.5 Echo-types — cross-cutting type-theoretic foundation - -**Owner directive 2026-06-01**: every proof wave below must first check `hyperpolymath/echo-types` for a reusable definition or lemma. If relevant, the wave consumes echo-types via a SHA-pinned import. If absent, the wave **extends echo-types first**, proves the extension there, and then consumes downstream. Cross-document: every consumer cites the echo-types module + commit; the echo-types module's `EXPLAINME.adoc` lists boj-server as a downstream consumer once the first wave imports it. - -### Why echo-types is the right foundation - -`echo-types` is the estate's constructive Agda formalisation of *proof-relevant lossy computation* — `--safe --without-K`, 0 postulates. Its core abstractions map directly onto boj-server's proof obligations: - -| boj-server invariant | echo-types primitive | -|---|---| -| Cartridge dispatch — distinct cartridges cannot collide on a message type | `EchoResidueTaxonomy` *indexed* residue form (proof-relevant distinguishability of dispatch keys) | -| Multi-protocol composition — outputs of `Cᵢ` parse under input schema of `Cᵢ₊₁` | `EchoImageFactorizationProp` (epi-mono earn-back: the residue of the projection bears the witness that the next stage's precondition is satisfied) | -| Credential isolation (BJ2) — a Teranga capability cannot leak across cartridges | Linear / affine bridge + `EchoSecurity` application module | -| Audit log integrity (`local-coord-mcp`, MFA-001 to MFA-006) | `EchoProvenance` application module (hash-chain as echo: the digest is the residue that constrains the preimage) | -| Cost / budget proofs (Glama scoring, panel cost-meters) | Tropical bridge + `EchoResidueTaxonomy` *cost* residue form | -| Effect / capability tracking across L4–L8 boundaries | Graded modality bridge (loss-graded reindexing per `docs/retractions.adoc` R-2026-05-18) | -| Class-J axiom witnesses (5 in `SafetyLemmas.idr`) | `EchoResidueTaxonomy` *generic Σ-cert* residue form | -| Federated coord (ADR-0010) role-projection | Choreographic bridge | -| Adversary-knowledge bounds | Epistemic bridge | - -The drift = echo + tropical cost composition (per VeriSimDB foundation pack) is the single most natural fit for boj-server's anchor theorem. - -### Status of echo-types at 2026-06-01 - -Per `echo-types/EXPLAINME.adoc` and `.machine_readable/6a2/STATE.a2ml`: - -- Core echo / fiber theorems present (`echo-intro`, `map-over`, `map-over-id`, `map-over-comp`, `map-square`). -- Bridges complete: linear, graded, tropical, choreographic, epistemic, CNO, Janus, Dyadic, Ordinal, Indexed, Relational, Categorical, Scope. -- Eight residue forms in `EchoResidueTaxonomy` (trivial, identity, generic Σ-cert, linear-affine, indexed, cost, search, epistemic). -- Investigation EI-2 (integration-recipe distinctness) terminated negatively via PATH B — **do not reopen**; treat as a settled negative result. -- Ordinal/Buchholz track: 11 of 13 per-constructor rank-mono cases closed; Slice-3 headline closed via Route A in PR #142/#143. - -This means W1 and W2 of boj-server's proof roadmap can be expressed *today* in echo-types vocabulary without extension. W3-W6 likely require small extensions — to be identified per wave as the work begins. - -### Extension policy - -When a wave needs a definition not in echo-types: - -1. File the gap as an echo-types issue (`hyperpolymath/echo-types`), referencing the boj-server wave + theorem name. -2. Land the extension in echo-types first (small, focused PR; passes `--safe --without-K`; no new postulates). -3. SHA-pin the echo-types import in the boj-server proof PR. -4. Echo-types `EXPLAINME.adoc` "Applied prototype hook" or downstream-consumers section lists boj-server. - -This keeps echo-types as the proof-foundation single-source-of-truth and prevents duplicate type definitions drifting across the estate. - ---- - -## 3. Competitive context — what the rest of the field claims - -### The baseline - -**Most MCP / agent-runtime systems make zero formal claims.** Verified examples: - -- **Anthropic's reference MCP servers** (`github.com/modelcontextprotocol/servers`) — disclaim formal verification explicitly. -- **OpenAI GPTs / actions** — prompt engineering + human-in-the-loop, no proofs. -- **Vercel AI SDK** — testing utilities + types, no formal verification. -- **Cloudflare Workers AI** — runtime scanning + semantic intent, not formal correctness. -- **Glama catalogue (~1000 servers)** — searching for "verified"/"formal"/"proven"/"Coq"/"Idris"/"Agda"/"Lean"/"TLA+" returns essentially nothing. -- **NVIDIA "verified agent skills"** — cryptographic signing + automated vuln scanning, not formal proof. - -### The 3 exceptional cases (each scoped narrowly) - -1. **Prova-MCP** — agents verify their own reasoning chains by kernel-checking Lean 4 proofs. Scope: agent-assisted theorem proving, not runtime safety of the server. -2. **MCPShield** (arxiv 2604.05969) — labeled-transition-system formal threat framework, 91% claimed coverage across a 7-category 23-vector threat taxonomy. Published peer-reviewed framework, not deployed at runtime. -3. **Rocq-MCP** — exposes Rocq (Coq-family) as MCP tools. **Uses** MCP to do proofs; doesn't prove MCP. - -### The prior-art ceiling (outside agent space) - -- **seL4** — 8.7k C + 600 asm functional-correctness proof in Isabelle/HOL. -- **CompCert** — semantics-preserving C→assembly compiler proof. -- **Project Everest / EverCrypt** — 124k lines verified F* in real-world production (Linux, Firefox, Tezos). -- **CHERI / VeriCHERI** — RTL-level formal verification of capability hardware. - -### 3 positioning framings — each defensible if executed - -1. **"First MCP server with formally verified cartridge loading"** — anchor on BJ1 + extend to cover the dlopen + symbol presence + signature compatibility chain. -2. **"First formally verified capability gateway for multi-cartridge agent integration"** — anchor on BJ2 + extend to the L8 boundary. Positions directly against NVIDIA's signing model. -3. **"Formally proven supply-chain safety for federated MCP cartridges"** — anchor on the federated coord ADR-0010 + provenance / SBOM verification. - -None of these are claimed by anyone today. - ---- - -## 4. Roadmap — 6 waves, ~40-50 proof days, 18-24 months - -Agent D's plan, condensed. Each wave estimate is solo-with-Joshua at ~6-8 weeks proof capacity per quarter. - -### W1 — Cartridge-layer type preservation (Weeks 1-4, ~8 days) -- `local-coord-mcp` closes P-04/P-05/P-06/P-07 (record format, CRC truncation, replay-equivalence, quarantine state machine) — 6 days, infrastructure already present. -- `007-mcp` policy-apply type-safety — ~1 day. -- One domain cartridge (e.g., `dap-mcp` or `bsp-mcp`) protocol dispatch uniqueness lemma — ~1 day. -- **Echo-types import**: `EchoResidueTaxonomy` *indexed* residue form (dispatch keys); `EchoProvenance` for replay-equivalence as hash-chain echo. **No extension expected** — both present at echo-types HEAD. - -### W2 — Invocation protocol soundness + multi-protocol composition (Weeks 5-12, ~8 days) -- `CartridgeDispatch.invokeSound` (direct-invoke preserves type) — 2 days. -- **`MultiProtocol.invokeChainSoundness`** ← the anchor theorem (see §5) — 3 days. -- `sseFrameIntegrity` — 1 day. -- Aligns with typed-wasm Phase 2 (L2 access-site carrier, ADR-0003 accepted 2026-05-30) as a case-study consumer. -- **Echo-types import**: `EchoImageFactorizationProp` (the anchor theorem statement *is* an epi-mono earn-back across cartridge boundaries); graded modality bridge for invocation-effect tracking. **No extension expected** — Tier 2 EchoImageFactorizationProp landed 2026-05-28. - -### W3 — Capability containment + vault isolation (Weeks 13-18, ~5 days) -- `VaultIsolation` upgrade to dynamic isolation (cartridges added post-init) — 2 days. -- `CapabilityContainment.borrowCap` (capability temporary-borrow, not store-or-re-export) — 2 days. -- `CredentialFlow.credentialCannotEscape` — 1 day. -- **Echo-types import**: linear/affine bridge + `EchoSecurity` application module. **Likely extension**: capability *borrow-and-return* may need a new linear-affine variant in echo-types — file as echo-types issue, land extension first. - -### W4 — End-to-end safety case (Weeks 19-24, ~10 days) -- `SafetyCase.e2eInferenceSound` — the umbrella theorem composing W1-W3 lemmas — 5 days bookkeeping. -- `CompositionLemma.multiCartridgeChain` — chain induction — 2 days. -- `AdversarialModel` — negative lemmas (can't forge IDs, can't bypass isolation, can't corrupt dispatch) — 1 day. -- **Echo-types import**: tropical bridge (cost composition under chain), `EchoProvenance` (audit trail across the chain), epistemic bridge (adversary knowledge bound). **Possible extension**: composition lemma for chained echo factorizations — likely covered by `map-over-comp` but may need a chain-specific lemma; file as echo-types issue if so. - -### W5 — Backend-assurance expansion (Weeks 25-28, ~4 days) -- Audit + externally validate any new class-J axioms introduced by new cartridges — 2 days. -- **Harness mechanisation** — formalise the discipline itself in Coq or Agda: "a class-J axiom is valid iff (trusted-extraction doc + property test + BEAM evidence)" — 2 days. -- **Echo-types import**: `EchoResidueTaxonomy` *generic Σ-cert* residue form — class-J axioms are precisely Σ-cert residues with external-evidence witnesses. **Extension expected**: a new residue form "*externally-validated*" or a refinement of generic Σ-cert with a backend-assurance side-condition. File as echo-types issue first. - -### W6 — Publication + standoff (Weeks 29+, ~2 days/cartridge) -- Technical report on the W4 e2e proof. -- Ready 3-5 proof-bearing cartridges for production. -- Establish proof-maintenance policy (within 2 weeks of any proof-bearing PR, a `docs/proof-summary.md` follow-up must cite which theorems cover which invariants). -- **Echo-types cross-document**: by W6 each of W1-W5's downstream consumers should be listed in echo-types `EXPLAINME.adoc` under a new "downstream consumers" section. Publication framing: "boj-server is the first capability-gateway *consumer* of the echo-types foundation; echo-types is its proof bedrock." - -### Top-3 quick wins (1.5-2 total days — front-load before W1) - -| Theorem | Statement | Days | -|---|---|---| -| `CartridgeDispatch.noCollisions` | Dispatch is injective — no two cartridges accidentally handle the same message type | 0.5 | -| `SafeLocalCoord.replayDeterminism` | Replay log iterator on the same BitStream prefix produces the same state — structural / by reflexivity | 0.5 | -| `SafetyLemmas.axiomsAreIrreducible` | Negative proof that the 5 class-J axioms cannot be discharged in Idris2 0.8.0 because Char/String have no constructors | 1 | - -These three are quotable in a paper / pitch deck without waiting for W4. - ---- - -## 5. The anchor theorem - -### `MultiProtocol.invokeChainSoundness` - -**Informal statement**: Given a sequence of cartridge invocations C₁ → C₂ → C₃ where each cartridge correctly implements its protocol contract, the outputs of each cartridge can always be parsed by the next cartridge's input schema; no type error can arise mid-chain, even if the invocations span different domains (e.g., OAuth → database → LLM). - -### Why this one - -- **Genuinely novel**: no other agent runtime — Claude, Anthropic's MCP, OpenAI's function-calling, Vercel, NVIDIA — has formally proven that chained cartridge invocations preserve type safety. typed-wasm covers access-site safety; proven covers individual library safety; **composing agent operations across trust boundaries is unique to boj-server's claim**. -- **Load-bearing**: W2 gates W3 (capability containment is only meaningful if invokes are sound), W3 gates W4 (e2e safety). This proof unblocks two full waves. -- **Publishable**: theorem statement is elegant enough for JFLA (Journées Francophones des Langages Applicatifs) or an ICFP workshop. "Formally verified agent orchestration" is a hookline. - -**Effort**: 3-5 days in W2. - ---- - -## 6. Decision points for owner - -### D1 — positioning framing (pick one or commit to all three) -1. Cartridge-loading verified (anchor on BJ1 + extend to dlopen chain) -2. Capability-gateway verified (anchor on BJ2 + extend to L8) -3. Supply-chain federation verified (anchor on ADR-0010 + SBOM) - -All three are achievable; the question is what to put on the front page. Recommend D1.2 (capability gateway) — it's the framing closest to your current proof artefacts, and "first formally verified capability gateway for LLM agents" parses as a sentence even to someone who doesn't know what a cartridge is. - -### D2 — accept the 18-24-month timeline, or compress? -Solo + Joshua at 6-8 weeks proof per quarter ≈ 30 weeks of proof work over 18 months, with W4 as the big push. Compressing requires either (a) more proof help (a collaborator with Coq/Idris2 fluency), or (b) deferring the anchor theorem and shipping incremental wave reports. - -### D3 — where do the 4 still-prose ADRs fit? -- **ADR-0006** (5-symbol cartridge ABI) — currently prose-only. Formalise in W2 alongside `invokeSound`. -- **ADR-0004** (HCG mTLS Phase B) — operational not provable; gate the L7→L8 edge until it lands. -- **ADR-0007** (Nickel PDP DSL) — still RFC. Has its own proof-debt; defer to a separate Nickel-track. -- **ADR-0010** (federated coord + ML-KEM) — proposed only. Treat as Phase 4+ or after W6. - -### D4 — proof-debt tracking discipline -boj-server's `PROOF-NEEDS.md` and `docs/proof-debt.md` are already exemplary (Agent A says: estate reference for the Trusted-Base Reduction Policy). Question: do you want a **per-cartridge** `proof-summary.md` requirement (W6 policy), or just per-repo? Per-cartridge gives much higher resolution but is high-overhead. - -### D5 — echo-types extension governance -Per the §2.5 owner directive, when a wave needs an echo-types extension, the workflow is: file as echo-types issue → land extension there first → SHA-pin in downstream boj-server PR → cross-link in `EXPLAINME.adoc`. Open sub-questions: -- **Repo of record for boj-server-specific instances** — when an `EchoResidueTaxonomy` instance is *only* boj-server-relevant (e.g. a "cartridge-tier-validated" instance), does it live in echo-types (estate-wide) or boj-server (local)? Recommend echo-types for any instance with a re-usable shape, boj-server for one-off. -- **Pace** — echo-types extension PRs must pass `--safe --without-K` and add no postulates. This is strict; W3 + W5 extensions may need 1-2 extra days each. -- **Reciprocal documentation** — at what cadence does echo-types' `EXPLAINME.adoc` get refreshed with the downstream-consumers list? Recommend at each wave-completion checkpoint. - ---- - -## 7. Open questions - -1. **Joshua's involvement in proof work specifically** — Agent D's roadmap assumes he can help with bookkeeping in W4. Is that realistic / desired? If he's primarily a cartridge implementer, the 18-24 mo timeline is brittle. -2. **Publication venues** — JFLA / ICFP workshop / a position paper at a security conference. The anchor-theorem framing changes the right venue. POPL is too theory-heavy; CCS / S&P would land if framed as capability containment. -3. **Trust-base re-audit cadence** — proven's 2026-05-20 honesty refresh found 70 overclaims. Should that audit be quarterly across all repos, or once-then-static? -4. **Tooling-stub remediation interaction** — per Q2 (cartridge minter retired no replacement, 3 stubs remain), if catalogue expansion follows the recommendation to rewrite all 4 tools in Rust/Zig, those tools become part of the trust chain too. New `tools/` deserve at least Eno-tier discipline. -5. **License clarity for the proof corpus** — boj-server is now AGPL-3.0-or-later (PR #157). The proven library it depends on — what license? Re-export terms for someone consuming boj-server's BJ1/BJ2/BJ3 proof artefacts? - ---- - -## 8. Phasing relative to the cartridge catalogue - -The catalogue document (PR #179) proposed a 5-phase rollout (tooling → backfill → high-leverage waves → depth fills → exotic). The proof-story phases overlap: - -| Catalogue phase | Proof phase | Interaction | -|---|---|---| -| Phase 0 — tooling | (no proof work) | The 4 tools-to-rewrite become L4-adjacent infrastructure; they should sit at trust-tier Eno minimum | -| Phase 1 — 14-LSP backfill | Proof quick-wins (W0) | The LSPs come in as Ayo / Eno; quick-win theorems land in parallel | -| Phase 2 — high-leverage waves | W1-W2 | Vector-DB / local-inference waves are mostly Ayo cartridges; the *runtime* improvements (e.g. invokeChainSoundness) lift their tier ceiling | -| Phase 3 — depth fills | W3-W4 | Capability containment + e2e safety case land here, anchoring the security half of the catalogue | -| Phase 4 — exotic | W5-W6 | Backend-assurance expansion + publication; exotic cartridges are too narrow for W6's reference-cartridge slot | - ---- - -## 9. Provenance - -- 4 Explore subagents fanned out 2026-06-01 ~13:30Z; all returned within ~12 minutes (existing state), ~6 minutes (trust chain), ~13 minutes (competitor baseline), ~6 minutes (roadmap). -- Owner-direct verification corrected Agent C's claim about `launch-scaffolder` being the cartridge minter's replacement (it is not — it's a desktop launcher generator). The minter is retired with no replacement, deepening the Q2 stub-rewrite scope from 3 tools to 4. -- Subagent transcripts at `/tmp/claude-1000/-home-hyperpolymath-developer-repos/.../tasks/`. -- **2026-06-01 amendment**: section 2.5 (echo-types foundation), per-wave echo-types module mapping in section 4, and section 6 D5 (echo-types extension governance) added per owner directive: "in the proofs you do, you need to check the echo-types repo and make sure this is part of the proofing for the repo, if not, establish the extension to what is there and work down that path proving as you go, then cross document". Echo-types module references sourced from `echo-types/EXPLAINME.adoc` at 2026-06-01 HEAD. - -🤖 Generated with [Claude Code](https://claude.com/claude-code) diff --git a/docs/planning/cartridge-catalogue-2026-06-01.adoc b/docs/planning/cartridge-catalogue-2026-06-01.adoc new file mode 100644 index 00000000..307bd1da --- /dev/null +++ b/docs/planning/cartridge-catalogue-2026-06-01.adoc @@ -0,0 +1,631 @@ +== Cartridge Catalogue Plan — 2026-06-01 + +____ +*Status*: draft for owner review. ~941 candidates surveyed across 6 +buckets by 6 parallel Explore subagents; this document is the synthesis ++ framework + decision points. + +*Author*: Claude Opus 4.7 (session: boj-server) *Scope*: planning +artefact. No cartridges built. No PRs filed. *Sources*: agent +transcripts in `+/tmp/claude-1000/.../tasks/+`; existing inventories +from +`+gh api repos/hyperpolymath/{boj-server,boj-server-cartridges}/contents+`. +____ + +''''' + +=== 1. Framework — the three axes a cartridge gets pinned on + +Generating 1000 cartridge rows by free-form brainstorm is padding by row +200. A planning framework that *scales to 1000 by construction* is more +useful: pin every candidate on three orthogonal axes so the catalogue +self-prioritises. + +==== Axis A — substrate (what the cartridge sits on) + +10 buckets — every existing cartridge fits cleanly into exactly one. New +candidates inherit the bucket’s defaults (trust tier ceiling, expected +protocols, FFI shape). + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|# |Substrate |Examples already in 125 inventory +|1 |Relational DB / KV / Doc / Time-series / Object store +|postgresql-mcp, redis-mcp, sqlite-mcp + +|2 |Cloud control plane |aws-mcp, gcp-mcp, hetzner-mcp + +|3 |Container / orchestration / IaC / config |k8s-mcp, iac-mcp + +|4 |Edge / CDN / DNS / networking |cloudflare-mcp, vercel-mcp + +|5 |Language tools (LSP / format / lint / DAP / BSP) |affinescript-mcp, +ephapax-mcp + +|6 |Observability / metrics / tracing / log |observe-mcp, grafana-mcp + +|7 |Security / auth / secrets / supply-chain |vault-mcp, vordr-mcp, +rokur-mcp + +|8 |AI / ML / embedding / agent / knowledge |agent-mcp, nesy-mcp, +local-coord-mcp + +|9 |Domain APIs (finance / legal / health / ed / science) |zotero-mcp, +todoist-mcp + +|10 |IO + hardware + human interface |(sparse) +|=== + +==== Axis B — tier (where in the dependency graph) + +Same as the existing TOPOLOGY tiering. Cartridges higher in the dep +graph cost more. + +[width="100%",cols="34%,33%,33%",options="header",] +|=== +|# |Tier |Cost-per-cartridge intuition +|1 |Foundational |"`gates of life`" — DB / git / HTTP / auth. Already +mostly built. + +|2 |Infrastructure |cloud / k8s / CI / observability. Mostly built. + +|3 |Domain |bio / legal / finance / education / science. Long-tail +growth area. + +|4 |Exotic |quantum / SDR / AR-VR / robotics. Hardware-coupled; high +effort, high differentiation. +|=== + +==== Axis C — trust grade (existing CRG) + +Per `+standards/cartridges/CARTRIDGE-FORMAT.adoc+`: Ayo / Eno / Teranga +/ Shield. Trust gates which protocols the cartridge can speak and which +dirs are mandatory: + +* *Ayo* (baseline): cartridge.json + mod.js + Zig adapter + optional +FFI. ~2 days to scaffold. +* *Eno* (medium): + Zig FFI required. ~1 week. +* *Teranga* (high): + Idris2 ABI required, formally verified state +machine. ~2-3 weeks. +* *Shield* (highest): + adversary model + audit trail + Idris2 +invariants on all protocol entry points. ~4-8 weeks. + +*Pragmatic target distribution for 1000-cartridge expansion*: ~70% Ayo +(700) / ~20% Eno (200) / ~8% Teranga (80) / ~2% Shield (20). Going +heavier on Teranga/Shield doubles or triples the total build cost. + +''''' + +=== 2. Three decision points to call before catalogue expansion + +These are load-bearing — the answer changes the inventory shape +materially. + +==== Decision 1 — multi-protocol cartridge vs sibling family + +The schema (`+cartridge.json+` `+protocols[]+` array) supports *one +cartridge declaring many protocols*. The current 125-cartridge +convention is *one cartridge per role suffix* (`+aws-mcp+`, separately a +hypothetical `+aws-lsp+`, separately `+aws-dap+`, etc.). + +For the dev-tools bucket especially the difference is enormous: + +* *Per-role family*: 32 languages × \{lsp, format, lint, dap, bsp} ≈ +*120 cartridges*. +* *Per-language multi-protocol*: 32 languages × 1 cartridge each ≈ *32 +cartridges*. + +Same surface, ~88 cartridge slots difference. The user’s framing ("`it +should contain the mcp, lsp, debug, build, formatter, linter`") reads as +"`one cartridge per package, declares many protocols`" — i.e. the +multi-protocol route. If that’s the intended design, *the existing 125 +partly mis-follow the convention* and a future consolidation would +collapse families like `+aws-mcp+` to a single `+aws+` cartridge +declaring `+[MCP, LSP, DAP, BSP, Format, Lint]+`. + +*Open question*: which way? Recommend multi-protocol per package — fewer +cartridges, less duplicate cartridge.json metadata, easier discovery +(one row per substrate). Migration path: keep `+-mcp+` aliases for +one release cycle then sunset. + +==== Decision 2 — tooling stubs must be retired or rewritten before expansion + +Agent C found: + +* ✅ *Cartridge minter* — retired 2026-04-25 (Node.js policy). Replaced +by `+launch-scaffolder+` (Rust, production-ready, has +`+mint+`/`+provision+`/`+config+` subcommands). +* ⚠️ *Cartridge provisioner* — stub since 2026-04-21. README documents +the invocation, the body is +`+// In a real implementation, you would …+`. Writes JSON, does nothing +else. +* ⚠️ *Cartridge configurator* — stub since 2026-04-21. Validation is +no-op; hot-reload unimplemented. +* ⚠️ *Panel harness* — stub since 2026-04-21. Cartridge "`registration`" +writes a `+registration.json+` and stops. + +The three stubs are *README-and-stub-only*. A new contributor following +the docs sees the invocation succeed and assumes things worked. *This is +worse than missing tooling.* + +*Recommended sequence*: 1. Decide: retire the three stubs (route +everything through `+launch-scaffolder+` subcommands — they already +exist), or rewrite them properly in Rust / Zig per estate policy. 2. +End-to-end-validate one fresh cartridge through the chosen pipeline. 3. +_Then_ start catalogue expansion. + +Without this, every contributor will hit the same stub trap +independently. + +==== Decision 3 — backfill the 14 canonical-only LSPs first + +From the boj-server ↔ boj-server-cartridges drift survey (already saved +as memory `+project_boj_server_cartridges_sync_2026_06_01+`): + +14 cartridges exist in `+boj-server-cartridges/cartridges/+` but *NOT* +in `+boj-server/cartridges/+`: + +.... +cloud-lsp, container-lsp, database-lsp, git-lsp, iac-lsp, k8s-lsp, +librarian-mcp, npc-mcp, observe-lsp, proof-lsp, queues-lsp, +secrets-lsp, ssg-lsp, stack-orchestrator-mcp +.... + +These are low-cost wins — already designed, already in canonical source. +Default-config boj-server operators (no fetcher run) never see them. +*Backfill before expand*: ~14 cheap PRs to bring the runtime in line +with canonical, then start adding new cartridges. + +Also flagged by the drift survey (not strictly catalogue-blocking but +ought to be ticketed): + +* `+"category"+` field present in runtime cartridge.json, absent in +canonical — fetcher silently drops it (no schema validation in +`+catalog.ex+`). +* Fetcher’s `+find | sort+` flat copy silently picks first-wins on name +collisions across domains. + +''''' + +=== 3. Catalogue — 941 candidates across 6 buckets + +Sourced from 6 parallel Explore subagents. Each entry carries effort +(`+S+`/`+M+`/`+L+`) and trust-tier guess +(`+Ayo+`/`+Eno+`/`+Teranga+`/`+Shield+`). + +____ +*Naming convention note*: agents tagged with `+-mcp+` per current +convention. If Decision 1 lands on multi-protocol-per-package, names +collapse — e.g. `+cockroachdb-mcp+` becomes `+cockroachdb+` declaring +`+[MCP, LSP]+` if SQL tooling is bundled. +____ + +==== §1 — Data & Storage (119 candidates) + +===== Relational databases (20) + +cockroachdb-mcp, timescaledb-mcp, planetscale-mcp, mariadb-mcp, +oracle-database-mcp, yugabyte-mcp, citus-mcp, memsql-mcp, +singlestore-mcp, foundationdb-mcp, alloydb-mcp, ravendb-mcp, +couchdb-mcp, ferretdb-mcp, dolt-mcp, liquibase-mcp, flyway-mcp, +prql-mcp, readyset-mcp, pg-partman-mcp + +===== Document / KV / Time-series (21) + +dynamodb-mcp, firestore-mcp, couchbase-mcp, etcd-mcp, consul-mcp, +memcached-mcp, tarantool-mcp, scylla-mcp, cassandra-mcp, riak-mcp, +voldemort-mcp, leveldb-mcp, rocksdb-mcp, badger-mcp, lmdb-mcp, +hazelcast-mcp, ignite-mcp, cosmosdb-mcp, keydb-mcp, valkey-mcp, +replicant-mcp + +===== Object storage / S3-compatible (22) + +minio-mcp, wasabi-mcp, backblaze-b2-mcp, digitalocean-spaces-mcp, +vultr-object-storage-mcp, linode-object-storage-mcp, fastly-s3-mcp, +supabase-storage-mcp, neon-storage-mcp, cloudflare-r2-mcp, iceberg-mcp, +delta-lake-mcp, hudi-mcp, ceph-mcp, seaweedfs-mcp, moto-mcp, +localstack-mcp, ovh-object-storage-mcp, scaleway-object-storage-mcp, +qiniu-mcp, aliyun-oss-mcp, huaweicloud-obs-mcp + +===== Search & analytics (21) + +opensearch-mcp, meilisearch-mcp, algolia-mcp, typesense-mcp, +tantivy-mcp, blast-mcp, tinyindex-mcp, xapian-mcp, whoosh-mcp, +sphinx-mcp, solr-mcp, manticore-mcp, typesense-cloud-mcp, zinc-mcp, +linsearch-mcp, vespa-mcp, myscale-mcp, lance-mcp, vald-mcp, jina-mcp, +elasticsearch-serverless-mcp + +===== Data warehouses & lakehouses (29) + +datafusion-mcp, apache-druid-mcp, presto-mcp, trino-mcp, dbt-mcp, +dbt-cloud-mcp, fivetran-mcp, airbyte-mcp, stitch-mcp, materialize-mcp, +kestra-mcp, apache-airflow-mcp, prefect-mcp, dagster-mcp, metaflow-mcp, +great-expectations-mcp, talend-mcp, informatica-mcp, apache-atlas-mcp, +collibra-mcp, meltano-mcp, census-mcp, hightouch-mcp, soda-mcp, +trifacta-mcp, apache-flink-mcp, pulsar-mcp, nifi-mcp, dbt-squared-mcp + +==== §2 — Dev Tools & Languages (163 candidates) + +*Note*: this slice is most affected by Decision 1. If we go +multi-protocol-per-package, the per-language family below collapses to +one row per language. + +===== Per-language tool families (~120 entries; 30 languages) + +Languages with all 4-5 role variants (lsp/format/lint/dap/bsp): python, +javascript, typescript, go, rust, java, c, cpp, csharp, ruby, php, +swift, kotlin, scala, haskell, elixir, lua, ocaml, perl, bash, r, julia, +dart, clojure, commonlisp, vim-script, toml, yaml, json, graphql, sql (+ +tsql, postgresql variants) + +Sample (just python): `+python-lsp+`, `+python-format+`, +`+python-lint+`, `+python-dap+`, `+python-bsp+`. Repeat per-language. +*120 names total*. + +===== Build systems & package managers (18) + +make-bsp, cmake-bsp, ninja-bsp, bazel-bsp, scons-bsp, gradle-bsp, +maven-bsp, go-modules-bsp, cargo-bsp, dotnet-bsp, swift-package-bsp, +pyenv-bsp, nvm-bsp, rbenv-bsp, jenv-bsp, asdf-bsp, crates-index-bsp, +npm-registry-bsp (+ pypi-bsp, maven-central-bsp, hex-bsp) + +===== Testing / fuzzing / benchmarking (25) + +pytest-cart, unittest-cart, jest-cart, mocha-cart, vitest-cart, +junit-cart, testng-cart, gtest-cart, catch2-cart, ctest-cart, +cargo-test-cart, go-test-cart, rspec-cart, phpunit-cart, +swift-testing-cart, elixir-test-cart, libfuzzer-cart, honggfuzz-cart, +afl-cart, proptest-cart, hypothesis-cart, quickcheck-cart, jqwik-cart, +criterion-cart, bencher-cart, jmh-cart, benchmark-go-cart, +pytest-benchmark-cart + +==== §3 — Cloud, Infra, Container (150 candidates) + +===== Cloud providers (30) + +oracle-cloud-mcp, alibaba-mcp, huawei-cloud-mcp, ibm-cloud-mcp, +packet-mcp, vultr-mcp, akamai-mcp, upcloud-mcp, scaleway-mcp, +backblaze-mcp, ionos-mcp, bunnycdn-mcp, civo-mcp, exoscale-mcp, ovh-mcp, +greenhost-mcp, citycloud-mcp, joyent-triton-mcp, upyun-mcp, +tencentcloud-mcp, kingsoft-mcp, zadara-mcp, phoenixnap-mcp, contabo-mcp, +infomaniak-mcp, genesis-cloud-mcp, vast-mcp, modal-mcp, fly-io-mcp, +render-cloud-mcp + +===== Containers & orchestration (30) + +podman-mcp, containerd-mcp, docker-compose-mcp, nomad-mcp, mesos-mcp, +swarm-mcp, cri-o-mcp, moby-mcp, lxc-lxd-mcp, openstack-mcp, +vmware-tanzu-mcp, redhat-openshift-mcp, canonical-microk8s-mcp, k3s-mcp, +k0s-mcp, serf-mcp, consul-svc-mcp, linkerd-mcp, istio-mcp, kuma-mcp, +flannel-mcp, calico-mcp, cilium-mcp, weave-mcp, openebs-mcp, +longhorn-mcp, rook-mcp, operator-sdk-mcp, helm-mcp, kustomize-mcp + +===== IaC & config management (31) + +pulumi-mcp, cdktf-mcp, tofu-mcp, crossplane-mcp, cloud-formation-mcp, +heat-mcp, troposphere-mcp, cdktf-providers-mcp, jsonnet-mcp, jinja2-mcp, +cdk-mcp, salt-mcp, puppet-mcp, chef-mcp, guix-mcp, nixos-mcp, guix-mcp, +ignition-mcp, cloud-init-mcp, kickstart-mcp, preseed-mcp, packer-mcp, +vagrant-mcp, blueprint-mcp, arm-templates-mcp, bicep-mcp, sarl-mcp, +cue-mcp, dhall-mcp, hcl-mcp, tctl-mcp + +===== Edge / CDN / DNS (32) + +akamai-edge-mcp, fastly-mcp, bunny-mcp, limelight-networks-mcp, +maxcdn-mcp, section-mcp, route53-mcp, azure-dns-mcp, gcp-cloud-dns-mcp, +dnsimple-mcp, linode-dns-mcp, namecheap-api-mcp, godaddy-api-mcp, +verisign-mcp, bind9-mcp, coredns-mcp, dnsmasq-mcp, powerdns-mcp, +easyname-mcp, ns1-mcp, constellix-mcp, edgedns-mcp, cloudxns-mcp, +zonomi-mcp, ttk-dns-mcp, edgecast-mcp, incapsula-mcp, keycdn-mcp, +cdn77-mcp, quic-edge-mcp, worker-mcp + +===== Networking / service mesh / LB (27) + +envoy-mcp, nginx-mcp, haproxy-mcp, traefik-mcp, caddy-mcp, +keepalived-mcp, gobetween-mcp, nlb-mcp, elb-alb-mcp, +gcp-load-balancing-mcp, azure-lb-mcp, azure-appgw-mcp, f5-bigip-mcp, +citrix-mcp, radware-mcp, barracuda-mcp, kemp-mcp, octavia-mcp, +avi-networks-mcp, riverbed-mcp, thousand-eyes-mcp, vrrp-mcp, bgp-mcp, +mpls-mcp, gre-mcp, wireguard-mcp, openvpn-mcp, strongswan-mcp, +zerotier-mcp, tailscale-mcp, headscale-mcp, netmaker-mcp, tinc-mcp + +==== §4 — Observability, Security, Governance (202 candidates) + +*Trust-tier note*: ~80% land at Teranga or Shield — security-boundary +placement. + +===== Observability / logging / metrics / tracing (30) + +tempo-mcp, jaeger-mcp, newrelic-mcp, dynatrace-mcp, elastic-apm-mcp, +opentelemetry-collector-mcp, honeycomb-mcp, lightstep-mcp, signoz-mcp, +loki-mcp, elk-stack-mcp, splunk-mcp, clickhouse-stats-mcp, +victorops-mcp, pagerduty-mcp, opsgenie-mcp, rundeck-mcp, datadog-mcp, +new-relic-synthetics-mcp, uptimerobot-mcp, statuspage-mcp, scout-mcp, +instana-mcp, coralogix-mcp, sumo-logic-mcp, grafana-loki-advanced-mcp, +falco-mcp, auditbeat-mcp, osquery-mcp, wazuh-mcp, ossec-mcp + +===== Security scanners / vulnerability DBs / SAST/DAST (42) + +snyk-mcp, trivy-mcp, grype-mcp, owasp-dependency-check-mcp, +black-duck-mcp, whitesource-mcp, checkmarx-mcp, sonarqube-mcp, +fortify-mcp, veracode-mcp, burpsuite-mcp, owasp-zap-mcp, nessus-mcp, +openvas-mcp, qualys-mcp, rapid7-insightvm-mcp, aqua-mcp, neuvector-mcp, +twistlock-mcp, anchore-mcp, clair-mcp, falco-rules-mcp, appshield-mcp, +frida-mcp, ghidra-mcp, ida-pro-mcp, yara-mcp, osquery-vulns-mcp, +gitguardian-mcp, trufflehog-mcp, detect-secrets-mcp, gitleaks-mcp, +nuclei-mcp, burp-collaborator-mcp, metasploit-mcp, cobalt-strike-mcp, +maltego-mcp, shodan-mcp, cve-search-mcp, nvd-mirror-mcp + +===== Identity / auth / secrets (40) + +keycloak-mcp, auth0-mcp, okta-mcp, azuread-mcp, google-workspace-mcp, +cognito-mcp, firebaseauth-mcp, magic-links-mcp, webauthn-mcp, totp-mcp, +hotp-mcp, duo-mcp, twilio-authy-mcp, yubico-mcp, +crowdstrike-identity-mcp, delinea-mcp, beyondtrust-mcp, cyberark-mcp, +hashicorp-boundary-mcp, teleport-mcp, consul-acl-mcp, +istio-authpolicy-mcp, falco-rbac-mcp, spiffe-spire-mcp, +mtls-enforcer-mcp, cert-manager-mcp, smallstep-mcp, ejbca-mcp, +openssl-pki-mcp, lets-encrypt-mcp, entrust-mcp, digicert-mcp, +mozilla-sops-mcp, sealed-secrets-mcp, external-secrets-mcp, +infisical-mcp, 1password-mcp, lastpass-mcp, bitwarden-mcp, dashlane-mcp + +===== Compliance / audit / governance / policy (48) + +openpolicy-mcp, kyverno-mcp, kubewarden-mcp, gatekeeper-mcp, +calico-policy-mcp, cilium-policy-mcp, falco-policy-mcp, apparmor-mcp, +selinux-mcp, osquery-audit-mcp, lynis-mcp, openscap-mcp, vuls-mcp, +tenable-nessus-compliance-mcp, qualys-compliance-mcp, cloudmapper-mcp, +scoutsuite-mcp, cloudsploit-mcp, prowler-mcp, cloudaudit-mcp, +terraformcompliance-mcp, checkov-mcp, sentinel-mcp, cloudguard-mcp, +snyk-iac-mcp, forseti-mcp, audit-tooling-mcp, auditd-mcp, +log-aggregation-compliance-mcp, gdpr-mcp, ccpa-mcp, hipaa-mcp, soc2-mcp, +pci-dss-mcp, iso27001-mcp, vanta-mcp, drata-mcp, cloudhealth-mcp, +dome9-mcp, ermetic-mcp, wiz-mcp, clouddefense-mcp, lacework-mcp + +===== Supply-chain / SBOM / attestation / provenance (42) + +syft-mcp, spdx-mcp, cyclonedx-mcp, cosign-mcp, notation-mcp, +sigstore-mcp, tuf-mcp, notary-mcp, in-toto-mcp, ite6-mcp, slsa-mcp, +artifact-hub-mcp, helm-provenance-mcp, oci-distribution-mcp, harbor-mcp, +zot-mcp, artifactory-mcp, nexus-mcp, package-build-attestation-mcp, +buildkit-sbom-mcp, distroless-mcp, chainguard-images-mcp, +aqua-imageassurance-mcp, binary-artifact-provenance-mcp, +reproducible-builds-mcp, dependency-track-mcp, fossa-mcp, npm-audit-mcp, +cargo-audit-mcp, safety-mcp, pip-audit-mcp, bundler-audit-mcp, +composer-audit-mcp, license-compliance-mcp, reuse-mcp, parity-mcp, +oci-index-mcp, transparency-log-mcp, pki-chain-validation-mcp + +==== §5 — Domain, IO, Hardware, Human (127 candidates) + +===== Scientific computing / bio / chem / physics (26) + +biopython-toolkit-mcp, rdkit-chemistry-mcp, numpy-scipy-mcp, +pdb-protein-mcp, tandem-mass-spec-mcp, gromacs-md-mcp, quantum-cirq-mcp, +nextflow-genomics-mcp, mafft-alignment-mcp, sagemaker-bioml-mcp, +pymc-bayesian-mcp, fenics-fem-mcp, lammps-md-mcp, crystal-structure-mcp, +cross-reactivity-mcp, metabolomics-mcp, gatk-variant-mcp, vcftools-mcp, +deepvariant-mcp, enigma-pathogen-mcp, openmm-mcp, +rosetta-protein-design-mcp, plumed-md-mcp, cp2k-mcp, +omaha-spatial-bio-mcp, airflow-omics-mcp + +===== Finance / legal / healthcare / education (25) + +stripe-payments-mcp, plaid-fintech-mcp, iex-stock-data-mcp, +fmp-financial-mcp, courtlistener-mcp, sec-filings-mcp, +lexis-westlaw-lite-mcp, casetext-mcp, quickbooks-mcp, +xero-accounting-mcp, fincen-aml-mcp, canvas-lms-mcp, moodle-lms-mcp, +schoology-mcp, blackboard-lms-mcp, epic-emr-lite-mcp, cerner-fhir-mcp, +fhir-clinical-mcp, medline-pubmed-mcp, clinicaltrials-mcp, medicare-mcp, +pharmacy-ndc-mcp, edissmore-mcp, finaid-loan-mcp, msar-medical-edu-mcp + +===== IO format converters (29) + +pdf-text-extract-mcp, pdf-form-filler-mcp, epub-mcp, mobi-azw-mcp, +docx-mcp, odt-mcp, markdown-mcp, latex-mcp, svg-mcp, image-ocr-mcp, +video-transcode-mcp, audio-codec-mcp, parquet-arrow-mcp, avro-mcp, +protobuf-mcp, msgpack-mcp, geojson-gis-mcp, postgis-mcp, +netcdf-hdf5-mcp, gltf-3d-model-mcp, stl-ply-mcp, ics-calendar-mcp, +vcf-genome-mcp, bam-sam-mcp, gff-gtf-mcp, cwl-wdl-mcp, csv-tsv-mcp, +jsonl-ndjson-mcp, xml-soap-mcp + +===== Hardware / sensors / IoT / robotics / SDR (30) + +modbus-tcp-mcp, mqtt-mcp, zigbee-mcp, bluetooth-ble-mcp, usb-hid-mcp, +serial-rs485-mcp, ros-robotics-mcp, opencv-vision-mcp, +lidar-pointcloud-mcp, imu-accelerometer-mcp, gps-gnss-mcp, +thermal-infrared-mcp, sdr-gnu-radio-mcp, rtl-sdr-mcp, ad9833-mcp, +pressure-sensor-mcp, humidity-dht-mcp, mq-gas-sensor-mcp, +soil-moisture-mcp, current-voltage-mcp, dji-drone-mcp, +mavlink-autopilot-mcp, ultralytics-yolo-mcp, onnx-edge-mcp, +tensorflow-lite-mcp, segmentation-sam-mcp, depth-stereo-mcp, +hand-pose-mcp, hololens-mcp, oculus-vr-mcp + +===== Human interfaces / accessibility (27) + +screen-reader-bridge-mcp, wai-aria-mcp, liblouis-braille-mcp, +braille-display-mcp, text-to-speech-mcp, whisper-asr-mcp, bsl-asl-mcp, +sign-language-synthesis-mcp, eye-gaze-tracking-mcp, switch-input-mcp, +haptic-feedback-mcp, color-blindness-mcp, dyslexia-font-mcp, +high-contrast-mcp, captions-mcp, audio-description-mcp, +sign-language-nlp-mcp, magnification-mcp, keyboard-navigation-mcp, +low-bandwidth-text-mcp, voice-command-mcp, tremor-compensation-mcp, +cognitive-load-mcp, language-simplification-mcp, +gesture-recognition-mcp, mind-bci-mcp, voice-gender-adapt-mcp + +===== Productivity / comms / scheduling / CRM (40) + +outlook-exchange-mcp, microsoft-teams-mcp, zoom-mcp, calendly-mcp, +timely-mcp, harvest-mcp, clockify-mcp, asana-mcp, monday-mcp, +hubspot-crm-mcp, salesforce-mcp, pipedrive-mcp, zendesk-mcp, +freshdesk-mcp, intercom-mcp, typeform-mcp, mailchimp-mcp, sendgrid-mcp, +twilio-mcp, vonage-mcp, dynamics-365-mcp, sap-fiori-mcp, workday-mcp, +successfactors-mcp, greenhouse-mcp, lever-ats-mcp, workable-mcp, +referralhero-mcp, commsor-mcp, circle-community-mcp, +mighty-networks-mcp, eventbrite-mcp, lunchclub-mcp, +linkedin-recruiter-mcp, github-discussions-mcp, gumroad-mcp, +patreon-mcp, substack-mcp, beehiv-mcp, makeform-mcp, ninox-mcp + +==== §6 — AI / ML / Agentic / Knowledge (180 candidates) + +*Note*: this slice is the active growth area (boj-server #100 vector-DB +and #101 multimodal already filed as planned waves). Joshua is in the +cartridge-fetcher area; avoid name collisions with WIP. + +===== Model providers / inference (30) + +llama-cpp-mcp, ollama-mcp, vllm-mcp, text-generation-webui-mcp, +exllama-mcp, mlc-llm-mcp, transformers-local-mcp, ctransformers-mcp, +together-ai-mcp, modal-inference-mcp, baseten-mcp, banana-mcp, +workers-ai-mcp, runpod-mcp, anyscale-mcp, hyperbolic-mcp, llava-mcp, +moondream-mcp, claude-vision-local-mcp, visual-bert-mcp, blip-mcp, +layoutlm-mcp, mistral-mcp, dolphin-mcp, code-llama-mcp, granite-mcp, +deepseek-coder-mcp, yi-mcp, sentence-transformers-mcp, bge-mcp, e5-mcp, +jinaai-mcp, cohere-rerank-mcp, rankgpt-mcp + +===== Vector DBs / embedding stores / RAG (35) + +milvus-mcp, vespa-vector-mcp, zinc-mcp, qdrant-cloud-mcp, +pinecone-serverless-mcp, supabase-vector-mcp, neon-vector-mcp, +typesense-vector-mcp, opensearch-vector-mcp, manticore-vector-mcp, +elasticsearch-vector-mcp, meilisearch-vector-mcp, algolia-vector-mcp, +sonic-mcp, langchain-mcp, llama-index-mcp, haystack-mcp, ragas-mcp, +vectara-mcp, mixedbread-ai-mcp, bm25-retriever-mcp, dense-retriever-mcp, +hybrid-retriever-mcp, hyde-mcp, small-to-big-mcp, rerank-fusion-mcp, +adaptive-context-mcp, semantic-chunker-mcp, recursive-splitter-mcp, +markdown-splitter-mcp, code-splitter-mcp, sliding-window-mcp + +===== Knowledge graphs / ontologies / symbolic (30) + +tigergraph-mcp, galaxybase-mcp, blazegraph-mcp, neptune-mcp, +incense-mcp, owlready2-mcp, rdflib-mcp, protege-mcp, topquadrant-mcp, +skos-mcp, dublin-core-mcp, clingo-mcp, rule-engine-mcp, prolog-mcp, +forward-chaining-mcp, sparql-endpoint-mcp, inference-rules-mcp, +openie-mcp, spacy-nlp-mcp, deimos-mcp, dbpedia-mcp, wikidata-mcp, +schema-org-mcp, entity-resolution-mcp, property-alignment-mcp, +schema-matching-mcp, graph-merge-mcp, federated-sparql-mcp + +===== Document understanding / OCR / parsing (33) + +tesseract-mcp, paddleocr-mcp, easyocr-mcp, doctr-mcp, surya-mcp, +textract-mcp, cloudvision-mcp, azure-read-mcp, pypdf-mcp, +pdfplumber-mcp, tabula-py-mcp, docling-mcp, unstructured-mcp, +pandoc-doc-mcp, liboffice-mcp, aspose-mcp, table-transformer-mcp, +detr-table-mcp, camelot-mcp, table2html-mcp, sparkcollab-mcp, +layout-analyzer-mcp, segment-anything-mcp, document-classifier-mcp, +reading-order-mcp, text-flow-mcp, entity-extractor-mcp, +relation-extractor-mcp, invoice-parser-mcp, contract-parser-mcp, +form-parser-mcp, metadata-extractor-mcp + +===== Agentic frameworks / workflow engines (28) + +autogen-mcp, crewai-mcp, agency-swarm-mcp, metagpt-mcp, superagent-mcp, +phidata-mcp, swarm-mcp, prefect-agent-mcp, dagster-agent-mcp, +airflow-agent-mcp, temporal-mcp, n8n-mcp, zapier-mcp, celery-mcp, +rq-mcp, bull-queue-mcp, bullmq-mcp, apscheduler-mcp, schedule-mcp, +llamaindex-memory-mcp, langchain-memory-mcp, langchain-tools-mcp, +pydantic-tools-mcp, tool-validator-mcp, plugin-loader-mcp, +capability-negotiation-mcp, agent-tracer-mcp, cost-tracker-mcp, +prompt-logger-mcp + +===== Multimodal (vision/audio/video/STT/TTS) (24) + +vosk-mcp, silero-vad-mcp, wav2vec-mcp, deepspeech-mcp, glow-tts-mcp, +tacotron2-mcp, hifigan-mcp, fastpitch-mcp, nuwave-mcp, google-tts-mcp, +azure-tts-mcp, aws-polly-mcp, clip-mcp, clap-mcp, dino-mcp, +internvl-mcp, qwen-vl-mcp, gemini-vision-mcp, gpt4-vision-mcp, +stable-diffusion-mcp, sdxl-mcp, fooocus-mcp, animagine-mcp, +controlnet-mcp, lora-fusion-mcp, dalle3-mcp, midjourney-mcp, +video-classification-mcp, videoclip-mcp, slowfast-mcp, +video-captioning-mcp, temporal-segmentation-mcp, pose-estimation-mcp, +object-tracking-mcp, librosa-mcp, soundfile-mcp, pydub-mcp, +music-generation-mcp, musicautobot-mcp, source-separation-mcp, +key-tempo-mcp, spotify-audio-mcp, multimodal-embedding-mcp, +cross-modal-retrieval-mcp, vision-language-chain-mcp, +audio-visual-fusion-mcp, multimodal-retrieval-augmented-mcp + +''''' + +=== 4. Prioritisation — what to build first + +Following the decision points above, the recommended phasing is: + +==== Phase 0 — tooling fix (1-2 weeks, blocks everything) + +[arabic] +. Decide retire-vs-rewrite for the three stubs (provisioner / +configurator / panel-harness). +. End-to-end-validate one cartridge through +`+launch-scaffolder mint → provision → config → publish+`. +. Document the chosen path in +`+docs/specification/cartridge-lifecycle.adoc+`. + +==== Phase 1 — backfill (1 week, low cost) + +[arabic, start=4] +. Bring the 14 canonical-only LSPs from `+boj-server-cartridges+` into +`+boj-server/cartridges/+`. +. File issue for the `+"category"+` field schema mismatch + add schema +validation in `+catalog.ex+`. + +==== Phase 2 — high-leverage waves (1-2 quarters) + +Order by value-density (cartridges per unit substrate that ship "`out of +the box`" capability): 6. *Vector DB + RAG wave* (boj-server#100 already +tracks this) — unblocks all knowledge-base workflows. 7. +*Local-inference wave* — llama.cpp, ollama, vllm, etc. — unblocks all +on-device LLM work. 8. *IO converters wave* (~29 candidates in §5) — +small, mechanical, broad downstream unblock. 9. *Multimodal wave* +(boj-server#101 already tracks this). + +==== Phase 3 — depth fills (rolling) + +[arabic, start=10] +. Per-language LSP/format/lint families (~120 entries) — _gated on +Decision 1_. If multi-protocol-per-package, drops to ~30 cartridges. +. Cloud-provider depth (~30 entries) — high cost, low novelty; defer +until specific user need. +. Security/compliance depth (~200 entries) — high effort due to +trust-tier requirements; defer. + +==== Phase 4 — exotic (opportunistic) + +[arabic, start=13] +. SDR / quantum / robotics / haptics — hardware-coupled, low priority +unless specific project demand. + +''''' + +=== 5. Open questions for owner + +Before any of Phase 0+ ships: + +[arabic] +. *Decision 1*: multi-protocol-per-package, or per-role-sibling-family? +. *Decision 2*: retire the three tooling stubs, or rewrite them in +Rust/Zig? +. *Decision 3*: green-light the 14-LSP backfill PR series (≈14 small +PRs)? +. *Trust-tier distribution target*: confirm or override the 70/20/8/2 +split. +. *Naming*: should `+boj-server-cartridges+`’s taxonomy +(`+cross-cutting/+`, `+domains/+`, `+templates/+`) become the source of +truth, with the boj-server flat dir treated explicitly as a +fetcher-managed cache? + +Answer these and I can convert any Phase into PR-ready work. + +''''' + +=== Provenance + +* Subagents fanned out at 2026-06-01 ~12:30Z; all 6 returned within ~30 +minutes. +* Existing 125-cartridge inventory + 14-cartridge canonical-only list +cross-referenced against every candidate. +* Joshua’s cartridge-fetcher area (PR #169) explicitly excluded from +suggestions. +* ~50-80 candidates duplicate across slices (e.g., several +`+prefect-mcp+` / `+airflow-mcp+` mentions) — dedupe pass needed before +any commit; nominal count is 941, true unique count probably 850-900. + +🤖 Generated with https://claude.com/claude-code[Claude Code] diff --git a/docs/planning/cartridge-catalogue-2026-06-01.md b/docs/planning/cartridge-catalogue-2026-06-01.md deleted file mode 100644 index 8112833e..00000000 --- a/docs/planning/cartridge-catalogue-2026-06-01.md +++ /dev/null @@ -1,286 +0,0 @@ - - -# Cartridge Catalogue Plan — 2026-06-01 - -> **Status**: draft for owner review. ~941 candidates surveyed across 6 buckets by 6 parallel Explore subagents; this document is the synthesis + framework + decision points. -> -> **Author**: Claude Opus 4.7 (session: boj-server) -> **Scope**: planning artefact. No cartridges built. No PRs filed. -> **Sources**: agent transcripts in `/tmp/claude-1000/.../tasks/`; existing inventories from `gh api repos/hyperpolymath/{boj-server,boj-server-cartridges}/contents`. - ---- - -## 1. Framework — the three axes a cartridge gets pinned on - -Generating 1000 cartridge rows by free-form brainstorm is padding by row 200. A planning framework that **scales to 1000 by construction** is more useful: pin every candidate on three orthogonal axes so the catalogue self-prioritises. - -### Axis A — substrate (what the cartridge sits on) - -10 buckets — every existing cartridge fits cleanly into exactly one. New candidates inherit the bucket's defaults (trust tier ceiling, expected protocols, FFI shape). - -| # | Substrate | Examples already in 125 inventory | -|---|---|---| -| 1 | Relational DB / KV / Doc / Time-series / Object store | postgresql-mcp, redis-mcp, sqlite-mcp | -| 2 | Cloud control plane | aws-mcp, gcp-mcp, hetzner-mcp | -| 3 | Container / orchestration / IaC / config | k8s-mcp, iac-mcp | -| 4 | Edge / CDN / DNS / networking | cloudflare-mcp, vercel-mcp | -| 5 | Language tools (LSP / format / lint / DAP / BSP) | affinescript-mcp, ephapax-mcp | -| 6 | Observability / metrics / tracing / log | observe-mcp, grafana-mcp | -| 7 | Security / auth / secrets / supply-chain | vault-mcp, vordr-mcp, rokur-mcp | -| 8 | AI / ML / embedding / agent / knowledge | agent-mcp, nesy-mcp, local-coord-mcp | -| 9 | Domain APIs (finance / legal / health / ed / science) | zotero-mcp, todoist-mcp | -| 10 | IO + hardware + human interface | (sparse) | - -### Axis B — tier (where in the dependency graph) - -Same as the existing TOPOLOGY tiering. Cartridges higher in the dep graph cost more. - -| # | Tier | Cost-per-cartridge intuition | -|---|---|---| -| 1 | Foundational | "gates of life" — DB / git / HTTP / auth. Already mostly built. | -| 2 | Infrastructure | cloud / k8s / CI / observability. Mostly built. | -| 3 | Domain | bio / legal / finance / education / science. Long-tail growth area. | -| 4 | Exotic | quantum / SDR / AR-VR / robotics. Hardware-coupled; high effort, high differentiation. | - -### Axis C — trust grade (existing CRG) - -Per `standards/cartridges/CARTRIDGE-FORMAT.adoc`: Ayo / Eno / Teranga / Shield. Trust gates which protocols the cartridge can speak and which dirs are mandatory: - -- **Ayo** (baseline): cartridge.json + mod.js + Zig adapter + optional FFI. ~2 days to scaffold. -- **Eno** (medium): + Zig FFI required. ~1 week. -- **Teranga** (high): + Idris2 ABI required, formally verified state machine. ~2-3 weeks. -- **Shield** (highest): + adversary model + audit trail + Idris2 invariants on all protocol entry points. ~4-8 weeks. - -**Pragmatic target distribution for 1000-cartridge expansion**: ~70% Ayo (700) / ~20% Eno (200) / ~8% Teranga (80) / ~2% Shield (20). Going heavier on Teranga/Shield doubles or triples the total build cost. - ---- - -## 2. Three decision points to call before catalogue expansion - -These are load-bearing — the answer changes the inventory shape materially. - -### Decision 1 — multi-protocol cartridge vs sibling family - -The schema (`cartridge.json` `protocols[]` array) supports **one cartridge declaring many protocols**. The current 125-cartridge convention is **one cartridge per role suffix** (`aws-mcp`, separately a hypothetical `aws-lsp`, separately `aws-dap`, etc.). - -For the dev-tools bucket especially the difference is enormous: - -- **Per-role family**: 32 languages × {lsp, format, lint, dap, bsp} ≈ **120 cartridges**. -- **Per-language multi-protocol**: 32 languages × 1 cartridge each ≈ **32 cartridges**. - -Same surface, ~88 cartridge slots difference. The user's framing ("it should contain the mcp, lsp, debug, build, formatter, linter") reads as "one cartridge per package, declares many protocols" — i.e. the multi-protocol route. If that's the intended design, **the existing 125 partly mis-follow the convention** and a future consolidation would collapse families like `aws-mcp` to a single `aws` cartridge declaring `[MCP, LSP, DAP, BSP, Format, Lint]`. - -**Open question**: which way? Recommend multi-protocol per package — fewer cartridges, less duplicate cartridge.json metadata, easier discovery (one row per substrate). Migration path: keep `-mcp` aliases for one release cycle then sunset. - -### Decision 2 — tooling stubs must be retired or rewritten before expansion - -Agent C found: - -- ✅ **Cartridge minter** — retired 2026-04-25 (Node.js policy). Replaced by `launch-scaffolder` (Rust, production-ready, has `mint`/`provision`/`config` subcommands). -- ⚠️ **Cartridge provisioner** — stub since 2026-04-21. README documents the invocation, the body is `// In a real implementation, you would …`. Writes JSON, does nothing else. -- ⚠️ **Cartridge configurator** — stub since 2026-04-21. Validation is no-op; hot-reload unimplemented. -- ⚠️ **Panel harness** — stub since 2026-04-21. Cartridge "registration" writes a `registration.json` and stops. - -The three stubs are **README-and-stub-only**. A new contributor following the docs sees the invocation succeed and assumes things worked. **This is worse than missing tooling.** - -**Recommended sequence**: -1. Decide: retire the three stubs (route everything through `launch-scaffolder` subcommands — they already exist), or rewrite them properly in Rust / Zig per estate policy. -2. End-to-end-validate one fresh cartridge through the chosen pipeline. -3. *Then* start catalogue expansion. - -Without this, every contributor will hit the same stub trap independently. - -### Decision 3 — backfill the 14 canonical-only LSPs first - -From the boj-server ↔ boj-server-cartridges drift survey (already saved as memory `project_boj_server_cartridges_sync_2026_06_01`): - -14 cartridges exist in `boj-server-cartridges/cartridges/` but **NOT** in `boj-server/cartridges/`: - -``` -cloud-lsp, container-lsp, database-lsp, git-lsp, iac-lsp, k8s-lsp, -librarian-mcp, npc-mcp, observe-lsp, proof-lsp, queues-lsp, -secrets-lsp, ssg-lsp, stack-orchestrator-mcp -``` - -These are low-cost wins — already designed, already in canonical source. Default-config boj-server operators (no fetcher run) never see them. **Backfill before expand**: ~14 cheap PRs to bring the runtime in line with canonical, then start adding new cartridges. - -Also flagged by the drift survey (not strictly catalogue-blocking but ought to be ticketed): - -- `"category"` field present in runtime cartridge.json, absent in canonical — fetcher silently drops it (no schema validation in `catalog.ex`). -- Fetcher's `find | sort` flat copy silently picks first-wins on name collisions across domains. - ---- - -## 3. Catalogue — 941 candidates across 6 buckets - -Sourced from 6 parallel Explore subagents. Each entry carries effort (`S`/`M`/`L`) and trust-tier guess (`Ayo`/`Eno`/`Teranga`/`Shield`). - -> **Naming convention note**: agents tagged with `-mcp` per current convention. If Decision 1 lands on multi-protocol-per-package, names collapse — e.g. `cockroachdb-mcp` becomes `cockroachdb` declaring `[MCP, LSP]` if SQL tooling is bundled. - -### §1 — Data & Storage (119 candidates) - -#### Relational databases (20) -cockroachdb-mcp, timescaledb-mcp, planetscale-mcp, mariadb-mcp, oracle-database-mcp, yugabyte-mcp, citus-mcp, memsql-mcp, singlestore-mcp, foundationdb-mcp, alloydb-mcp, ravendb-mcp, couchdb-mcp, ferretdb-mcp, dolt-mcp, liquibase-mcp, flyway-mcp, prql-mcp, readyset-mcp, pg-partman-mcp - -#### Document / KV / Time-series (21) -dynamodb-mcp, firestore-mcp, couchbase-mcp, etcd-mcp, consul-mcp, memcached-mcp, tarantool-mcp, scylla-mcp, cassandra-mcp, riak-mcp, voldemort-mcp, leveldb-mcp, rocksdb-mcp, badger-mcp, lmdb-mcp, hazelcast-mcp, ignite-mcp, cosmosdb-mcp, keydb-mcp, valkey-mcp, replicant-mcp - -#### Object storage / S3-compatible (22) -minio-mcp, wasabi-mcp, backblaze-b2-mcp, digitalocean-spaces-mcp, vultr-object-storage-mcp, linode-object-storage-mcp, fastly-s3-mcp, supabase-storage-mcp, neon-storage-mcp, cloudflare-r2-mcp, iceberg-mcp, delta-lake-mcp, hudi-mcp, ceph-mcp, seaweedfs-mcp, moto-mcp, localstack-mcp, ovh-object-storage-mcp, scaleway-object-storage-mcp, qiniu-mcp, aliyun-oss-mcp, huaweicloud-obs-mcp - -#### Search & analytics (21) -opensearch-mcp, meilisearch-mcp, algolia-mcp, typesense-mcp, tantivy-mcp, blast-mcp, tinyindex-mcp, xapian-mcp, whoosh-mcp, sphinx-mcp, solr-mcp, manticore-mcp, typesense-cloud-mcp, zinc-mcp, linsearch-mcp, vespa-mcp, myscale-mcp, lance-mcp, vald-mcp, jina-mcp, elasticsearch-serverless-mcp - -#### Data warehouses & lakehouses (29) -datafusion-mcp, apache-druid-mcp, presto-mcp, trino-mcp, dbt-mcp, dbt-cloud-mcp, fivetran-mcp, airbyte-mcp, stitch-mcp, materialize-mcp, kestra-mcp, apache-airflow-mcp, prefect-mcp, dagster-mcp, metaflow-mcp, great-expectations-mcp, talend-mcp, informatica-mcp, apache-atlas-mcp, collibra-mcp, meltano-mcp, census-mcp, hightouch-mcp, soda-mcp, trifacta-mcp, apache-flink-mcp, pulsar-mcp, nifi-mcp, dbt-squared-mcp - -### §2 — Dev Tools & Languages (163 candidates) - -**Note**: this slice is most affected by Decision 1. If we go multi-protocol-per-package, the per-language family below collapses to one row per language. - -#### Per-language tool families (~120 entries; 30 languages) -Languages with all 4-5 role variants (lsp/format/lint/dap/bsp): python, javascript, typescript, go, rust, java, c, cpp, csharp, ruby, php, swift, kotlin, scala, haskell, elixir, lua, ocaml, perl, bash, r, julia, dart, clojure, commonlisp, vim-script, toml, yaml, json, graphql, sql (+ tsql, postgresql variants) - -Sample (just python): `python-lsp`, `python-format`, `python-lint`, `python-dap`, `python-bsp`. Repeat per-language. **120 names total**. - -#### Build systems & package managers (18) -make-bsp, cmake-bsp, ninja-bsp, bazel-bsp, scons-bsp, gradle-bsp, maven-bsp, go-modules-bsp, cargo-bsp, dotnet-bsp, swift-package-bsp, pyenv-bsp, nvm-bsp, rbenv-bsp, jenv-bsp, asdf-bsp, crates-index-bsp, npm-registry-bsp (+ pypi-bsp, maven-central-bsp, hex-bsp) - -#### Testing / fuzzing / benchmarking (25) -pytest-cart, unittest-cart, jest-cart, mocha-cart, vitest-cart, junit-cart, testng-cart, gtest-cart, catch2-cart, ctest-cart, cargo-test-cart, go-test-cart, rspec-cart, phpunit-cart, swift-testing-cart, elixir-test-cart, libfuzzer-cart, honggfuzz-cart, afl-cart, proptest-cart, hypothesis-cart, quickcheck-cart, jqwik-cart, criterion-cart, bencher-cart, jmh-cart, benchmark-go-cart, pytest-benchmark-cart - -### §3 — Cloud, Infra, Container (150 candidates) - -#### Cloud providers (30) -oracle-cloud-mcp, alibaba-mcp, huawei-cloud-mcp, ibm-cloud-mcp, packet-mcp, vultr-mcp, akamai-mcp, upcloud-mcp, scaleway-mcp, backblaze-mcp, ionos-mcp, bunnycdn-mcp, civo-mcp, exoscale-mcp, ovh-mcp, greenhost-mcp, citycloud-mcp, joyent-triton-mcp, upyun-mcp, tencentcloud-mcp, kingsoft-mcp, zadara-mcp, phoenixnap-mcp, contabo-mcp, infomaniak-mcp, genesis-cloud-mcp, vast-mcp, modal-mcp, fly-io-mcp, render-cloud-mcp - -#### Containers & orchestration (30) -podman-mcp, containerd-mcp, docker-compose-mcp, nomad-mcp, mesos-mcp, swarm-mcp, cri-o-mcp, moby-mcp, lxc-lxd-mcp, openstack-mcp, vmware-tanzu-mcp, redhat-openshift-mcp, canonical-microk8s-mcp, k3s-mcp, k0s-mcp, serf-mcp, consul-svc-mcp, linkerd-mcp, istio-mcp, kuma-mcp, flannel-mcp, calico-mcp, cilium-mcp, weave-mcp, openebs-mcp, longhorn-mcp, rook-mcp, operator-sdk-mcp, helm-mcp, kustomize-mcp - -#### IaC & config management (31) -pulumi-mcp, cdktf-mcp, tofu-mcp, crossplane-mcp, cloud-formation-mcp, heat-mcp, troposphere-mcp, cdktf-providers-mcp, jsonnet-mcp, jinja2-mcp, cdk-mcp, salt-mcp, puppet-mcp, chef-mcp, guix-mcp, nixos-mcp, guix-mcp, ignition-mcp, cloud-init-mcp, kickstart-mcp, preseed-mcp, packer-mcp, vagrant-mcp, blueprint-mcp, arm-templates-mcp, bicep-mcp, sarl-mcp, cue-mcp, dhall-mcp, hcl-mcp, tctl-mcp - -#### Edge / CDN / DNS (32) -akamai-edge-mcp, fastly-mcp, bunny-mcp, limelight-networks-mcp, maxcdn-mcp, section-mcp, route53-mcp, azure-dns-mcp, gcp-cloud-dns-mcp, dnsimple-mcp, linode-dns-mcp, namecheap-api-mcp, godaddy-api-mcp, verisign-mcp, bind9-mcp, coredns-mcp, dnsmasq-mcp, powerdns-mcp, easyname-mcp, ns1-mcp, constellix-mcp, edgedns-mcp, cloudxns-mcp, zonomi-mcp, ttk-dns-mcp, edgecast-mcp, incapsula-mcp, keycdn-mcp, cdn77-mcp, quic-edge-mcp, worker-mcp - -#### Networking / service mesh / LB (27) -envoy-mcp, nginx-mcp, haproxy-mcp, traefik-mcp, caddy-mcp, keepalived-mcp, gobetween-mcp, nlb-mcp, elb-alb-mcp, gcp-load-balancing-mcp, azure-lb-mcp, azure-appgw-mcp, f5-bigip-mcp, citrix-mcp, radware-mcp, barracuda-mcp, kemp-mcp, octavia-mcp, avi-networks-mcp, riverbed-mcp, thousand-eyes-mcp, vrrp-mcp, bgp-mcp, mpls-mcp, gre-mcp, wireguard-mcp, openvpn-mcp, strongswan-mcp, zerotier-mcp, tailscale-mcp, headscale-mcp, netmaker-mcp, tinc-mcp - -### §4 — Observability, Security, Governance (202 candidates) - -**Trust-tier note**: ~80% land at Teranga or Shield — security-boundary placement. - -#### Observability / logging / metrics / tracing (30) -tempo-mcp, jaeger-mcp, newrelic-mcp, dynatrace-mcp, elastic-apm-mcp, opentelemetry-collector-mcp, honeycomb-mcp, lightstep-mcp, signoz-mcp, loki-mcp, elk-stack-mcp, splunk-mcp, clickhouse-stats-mcp, victorops-mcp, pagerduty-mcp, opsgenie-mcp, rundeck-mcp, datadog-mcp, new-relic-synthetics-mcp, uptimerobot-mcp, statuspage-mcp, scout-mcp, instana-mcp, coralogix-mcp, sumo-logic-mcp, grafana-loki-advanced-mcp, falco-mcp, auditbeat-mcp, osquery-mcp, wazuh-mcp, ossec-mcp - -#### Security scanners / vulnerability DBs / SAST/DAST (42) -snyk-mcp, trivy-mcp, grype-mcp, owasp-dependency-check-mcp, black-duck-mcp, whitesource-mcp, checkmarx-mcp, sonarqube-mcp, fortify-mcp, veracode-mcp, burpsuite-mcp, owasp-zap-mcp, nessus-mcp, openvas-mcp, qualys-mcp, rapid7-insightvm-mcp, aqua-mcp, neuvector-mcp, twistlock-mcp, anchore-mcp, clair-mcp, falco-rules-mcp, appshield-mcp, frida-mcp, ghidra-mcp, ida-pro-mcp, yara-mcp, osquery-vulns-mcp, gitguardian-mcp, trufflehog-mcp, detect-secrets-mcp, gitleaks-mcp, nuclei-mcp, burp-collaborator-mcp, metasploit-mcp, cobalt-strike-mcp, maltego-mcp, shodan-mcp, cve-search-mcp, nvd-mirror-mcp - -#### Identity / auth / secrets (40) -keycloak-mcp, auth0-mcp, okta-mcp, azuread-mcp, google-workspace-mcp, cognito-mcp, firebaseauth-mcp, magic-links-mcp, webauthn-mcp, totp-mcp, hotp-mcp, duo-mcp, twilio-authy-mcp, yubico-mcp, crowdstrike-identity-mcp, delinea-mcp, beyondtrust-mcp, cyberark-mcp, hashicorp-boundary-mcp, teleport-mcp, consul-acl-mcp, istio-authpolicy-mcp, falco-rbac-mcp, spiffe-spire-mcp, mtls-enforcer-mcp, cert-manager-mcp, smallstep-mcp, ejbca-mcp, openssl-pki-mcp, lets-encrypt-mcp, entrust-mcp, digicert-mcp, mozilla-sops-mcp, sealed-secrets-mcp, external-secrets-mcp, infisical-mcp, 1password-mcp, lastpass-mcp, bitwarden-mcp, dashlane-mcp - -#### Compliance / audit / governance / policy (48) -openpolicy-mcp, kyverno-mcp, kubewarden-mcp, gatekeeper-mcp, calico-policy-mcp, cilium-policy-mcp, falco-policy-mcp, apparmor-mcp, selinux-mcp, osquery-audit-mcp, lynis-mcp, openscap-mcp, vuls-mcp, tenable-nessus-compliance-mcp, qualys-compliance-mcp, cloudmapper-mcp, scoutsuite-mcp, cloudsploit-mcp, prowler-mcp, cloudaudit-mcp, terraformcompliance-mcp, checkov-mcp, sentinel-mcp, cloudguard-mcp, snyk-iac-mcp, forseti-mcp, audit-tooling-mcp, auditd-mcp, log-aggregation-compliance-mcp, gdpr-mcp, ccpa-mcp, hipaa-mcp, soc2-mcp, pci-dss-mcp, iso27001-mcp, vanta-mcp, drata-mcp, cloudhealth-mcp, dome9-mcp, ermetic-mcp, wiz-mcp, clouddefense-mcp, lacework-mcp - -#### Supply-chain / SBOM / attestation / provenance (42) -syft-mcp, spdx-mcp, cyclonedx-mcp, cosign-mcp, notation-mcp, sigstore-mcp, tuf-mcp, notary-mcp, in-toto-mcp, ite6-mcp, slsa-mcp, artifact-hub-mcp, helm-provenance-mcp, oci-distribution-mcp, harbor-mcp, zot-mcp, artifactory-mcp, nexus-mcp, package-build-attestation-mcp, buildkit-sbom-mcp, distroless-mcp, chainguard-images-mcp, aqua-imageassurance-mcp, binary-artifact-provenance-mcp, reproducible-builds-mcp, dependency-track-mcp, fossa-mcp, npm-audit-mcp, cargo-audit-mcp, safety-mcp, pip-audit-mcp, bundler-audit-mcp, composer-audit-mcp, license-compliance-mcp, reuse-mcp, parity-mcp, oci-index-mcp, transparency-log-mcp, pki-chain-validation-mcp - -### §5 — Domain, IO, Hardware, Human (127 candidates) - -#### Scientific computing / bio / chem / physics (26) -biopython-toolkit-mcp, rdkit-chemistry-mcp, numpy-scipy-mcp, pdb-protein-mcp, tandem-mass-spec-mcp, gromacs-md-mcp, quantum-cirq-mcp, nextflow-genomics-mcp, mafft-alignment-mcp, sagemaker-bioml-mcp, pymc-bayesian-mcp, fenics-fem-mcp, lammps-md-mcp, crystal-structure-mcp, cross-reactivity-mcp, metabolomics-mcp, gatk-variant-mcp, vcftools-mcp, deepvariant-mcp, enigma-pathogen-mcp, openmm-mcp, rosetta-protein-design-mcp, plumed-md-mcp, cp2k-mcp, omaha-spatial-bio-mcp, airflow-omics-mcp - -#### Finance / legal / healthcare / education (25) -stripe-payments-mcp, plaid-fintech-mcp, iex-stock-data-mcp, fmp-financial-mcp, courtlistener-mcp, sec-filings-mcp, lexis-westlaw-lite-mcp, casetext-mcp, quickbooks-mcp, xero-accounting-mcp, fincen-aml-mcp, canvas-lms-mcp, moodle-lms-mcp, schoology-mcp, blackboard-lms-mcp, epic-emr-lite-mcp, cerner-fhir-mcp, fhir-clinical-mcp, medline-pubmed-mcp, clinicaltrials-mcp, medicare-mcp, pharmacy-ndc-mcp, edissmore-mcp, finaid-loan-mcp, msar-medical-edu-mcp - -#### IO format converters (29) -pdf-text-extract-mcp, pdf-form-filler-mcp, epub-mcp, mobi-azw-mcp, docx-mcp, odt-mcp, markdown-mcp, latex-mcp, svg-mcp, image-ocr-mcp, video-transcode-mcp, audio-codec-mcp, parquet-arrow-mcp, avro-mcp, protobuf-mcp, msgpack-mcp, geojson-gis-mcp, postgis-mcp, netcdf-hdf5-mcp, gltf-3d-model-mcp, stl-ply-mcp, ics-calendar-mcp, vcf-genome-mcp, bam-sam-mcp, gff-gtf-mcp, cwl-wdl-mcp, csv-tsv-mcp, jsonl-ndjson-mcp, xml-soap-mcp - -#### Hardware / sensors / IoT / robotics / SDR (30) -modbus-tcp-mcp, mqtt-mcp, zigbee-mcp, bluetooth-ble-mcp, usb-hid-mcp, serial-rs485-mcp, ros-robotics-mcp, opencv-vision-mcp, lidar-pointcloud-mcp, imu-accelerometer-mcp, gps-gnss-mcp, thermal-infrared-mcp, sdr-gnu-radio-mcp, rtl-sdr-mcp, ad9833-mcp, pressure-sensor-mcp, humidity-dht-mcp, mq-gas-sensor-mcp, soil-moisture-mcp, current-voltage-mcp, dji-drone-mcp, mavlink-autopilot-mcp, ultralytics-yolo-mcp, onnx-edge-mcp, tensorflow-lite-mcp, segmentation-sam-mcp, depth-stereo-mcp, hand-pose-mcp, hololens-mcp, oculus-vr-mcp - -#### Human interfaces / accessibility (27) -screen-reader-bridge-mcp, wai-aria-mcp, liblouis-braille-mcp, braille-display-mcp, text-to-speech-mcp, whisper-asr-mcp, bsl-asl-mcp, sign-language-synthesis-mcp, eye-gaze-tracking-mcp, switch-input-mcp, haptic-feedback-mcp, color-blindness-mcp, dyslexia-font-mcp, high-contrast-mcp, captions-mcp, audio-description-mcp, sign-language-nlp-mcp, magnification-mcp, keyboard-navigation-mcp, low-bandwidth-text-mcp, voice-command-mcp, tremor-compensation-mcp, cognitive-load-mcp, language-simplification-mcp, gesture-recognition-mcp, mind-bci-mcp, voice-gender-adapt-mcp - -#### Productivity / comms / scheduling / CRM (40) -outlook-exchange-mcp, microsoft-teams-mcp, zoom-mcp, calendly-mcp, timely-mcp, harvest-mcp, clockify-mcp, asana-mcp, monday-mcp, hubspot-crm-mcp, salesforce-mcp, pipedrive-mcp, zendesk-mcp, freshdesk-mcp, intercom-mcp, typeform-mcp, mailchimp-mcp, sendgrid-mcp, twilio-mcp, vonage-mcp, dynamics-365-mcp, sap-fiori-mcp, workday-mcp, successfactors-mcp, greenhouse-mcp, lever-ats-mcp, workable-mcp, referralhero-mcp, commsor-mcp, circle-community-mcp, mighty-networks-mcp, eventbrite-mcp, lunchclub-mcp, linkedin-recruiter-mcp, github-discussions-mcp, gumroad-mcp, patreon-mcp, substack-mcp, beehiv-mcp, makeform-mcp, ninox-mcp - -### §6 — AI / ML / Agentic / Knowledge (180 candidates) - -**Note**: this slice is the active growth area (boj-server #100 vector-DB and #101 multimodal already filed as planned waves). Joshua is in the cartridge-fetcher area; avoid name collisions with WIP. - -#### Model providers / inference (30) -llama-cpp-mcp, ollama-mcp, vllm-mcp, text-generation-webui-mcp, exllama-mcp, mlc-llm-mcp, transformers-local-mcp, ctransformers-mcp, together-ai-mcp, modal-inference-mcp, baseten-mcp, banana-mcp, workers-ai-mcp, runpod-mcp, anyscale-mcp, hyperbolic-mcp, llava-mcp, moondream-mcp, claude-vision-local-mcp, visual-bert-mcp, blip-mcp, layoutlm-mcp, mistral-mcp, dolphin-mcp, code-llama-mcp, granite-mcp, deepseek-coder-mcp, yi-mcp, sentence-transformers-mcp, bge-mcp, e5-mcp, jinaai-mcp, cohere-rerank-mcp, rankgpt-mcp - -#### Vector DBs / embedding stores / RAG (35) -milvus-mcp, vespa-vector-mcp, zinc-mcp, qdrant-cloud-mcp, pinecone-serverless-mcp, supabase-vector-mcp, neon-vector-mcp, typesense-vector-mcp, opensearch-vector-mcp, manticore-vector-mcp, elasticsearch-vector-mcp, meilisearch-vector-mcp, algolia-vector-mcp, sonic-mcp, langchain-mcp, llama-index-mcp, haystack-mcp, ragas-mcp, vectara-mcp, mixedbread-ai-mcp, bm25-retriever-mcp, dense-retriever-mcp, hybrid-retriever-mcp, hyde-mcp, small-to-big-mcp, rerank-fusion-mcp, adaptive-context-mcp, semantic-chunker-mcp, recursive-splitter-mcp, markdown-splitter-mcp, code-splitter-mcp, sliding-window-mcp - -#### Knowledge graphs / ontologies / symbolic (30) -tigergraph-mcp, galaxybase-mcp, blazegraph-mcp, neptune-mcp, incense-mcp, owlready2-mcp, rdflib-mcp, protege-mcp, topquadrant-mcp, skos-mcp, dublin-core-mcp, clingo-mcp, rule-engine-mcp, prolog-mcp, forward-chaining-mcp, sparql-endpoint-mcp, inference-rules-mcp, openie-mcp, spacy-nlp-mcp, deimos-mcp, dbpedia-mcp, wikidata-mcp, schema-org-mcp, entity-resolution-mcp, property-alignment-mcp, schema-matching-mcp, graph-merge-mcp, federated-sparql-mcp - -#### Document understanding / OCR / parsing (33) -tesseract-mcp, paddleocr-mcp, easyocr-mcp, doctr-mcp, surya-mcp, textract-mcp, cloudvision-mcp, azure-read-mcp, pypdf-mcp, pdfplumber-mcp, tabula-py-mcp, docling-mcp, unstructured-mcp, pandoc-doc-mcp, liboffice-mcp, aspose-mcp, table-transformer-mcp, detr-table-mcp, camelot-mcp, table2html-mcp, sparkcollab-mcp, layout-analyzer-mcp, segment-anything-mcp, document-classifier-mcp, reading-order-mcp, text-flow-mcp, entity-extractor-mcp, relation-extractor-mcp, invoice-parser-mcp, contract-parser-mcp, form-parser-mcp, metadata-extractor-mcp - -#### Agentic frameworks / workflow engines (28) -autogen-mcp, crewai-mcp, agency-swarm-mcp, metagpt-mcp, superagent-mcp, phidata-mcp, swarm-mcp, prefect-agent-mcp, dagster-agent-mcp, airflow-agent-mcp, temporal-mcp, n8n-mcp, zapier-mcp, celery-mcp, rq-mcp, bull-queue-mcp, bullmq-mcp, apscheduler-mcp, schedule-mcp, llamaindex-memory-mcp, langchain-memory-mcp, langchain-tools-mcp, pydantic-tools-mcp, tool-validator-mcp, plugin-loader-mcp, capability-negotiation-mcp, agent-tracer-mcp, cost-tracker-mcp, prompt-logger-mcp - -#### Multimodal (vision/audio/video/STT/TTS) (24) -vosk-mcp, silero-vad-mcp, wav2vec-mcp, deepspeech-mcp, glow-tts-mcp, tacotron2-mcp, hifigan-mcp, fastpitch-mcp, nuwave-mcp, google-tts-mcp, azure-tts-mcp, aws-polly-mcp, clip-mcp, clap-mcp, dino-mcp, internvl-mcp, qwen-vl-mcp, gemini-vision-mcp, gpt4-vision-mcp, stable-diffusion-mcp, sdxl-mcp, fooocus-mcp, animagine-mcp, controlnet-mcp, lora-fusion-mcp, dalle3-mcp, midjourney-mcp, video-classification-mcp, videoclip-mcp, slowfast-mcp, video-captioning-mcp, temporal-segmentation-mcp, pose-estimation-mcp, object-tracking-mcp, librosa-mcp, soundfile-mcp, pydub-mcp, music-generation-mcp, musicautobot-mcp, source-separation-mcp, key-tempo-mcp, spotify-audio-mcp, multimodal-embedding-mcp, cross-modal-retrieval-mcp, vision-language-chain-mcp, audio-visual-fusion-mcp, multimodal-retrieval-augmented-mcp - ---- - -## 4. Prioritisation — what to build first - -Following the decision points above, the recommended phasing is: - -### Phase 0 — tooling fix (1-2 weeks, blocks everything) -1. Decide retire-vs-rewrite for the three stubs (provisioner / configurator / panel-harness). -2. End-to-end-validate one cartridge through `launch-scaffolder mint → provision → config → publish`. -3. Document the chosen path in `docs/specification/cartridge-lifecycle.adoc`. - -### Phase 1 — backfill (1 week, low cost) -4. Bring the 14 canonical-only LSPs from `boj-server-cartridges` into `boj-server/cartridges/`. -5. File issue for the `"category"` field schema mismatch + add schema validation in `catalog.ex`. - -### Phase 2 — high-leverage waves (1-2 quarters) -Order by value-density (cartridges per unit substrate that ship "out of the box" capability): -6. **Vector DB + RAG wave** (boj-server#100 already tracks this) — unblocks all knowledge-base workflows. -7. **Local-inference wave** — llama.cpp, ollama, vllm, etc. — unblocks all on-device LLM work. -8. **IO converters wave** (~29 candidates in §5) — small, mechanical, broad downstream unblock. -9. **Multimodal wave** (boj-server#101 already tracks this). - -### Phase 3 — depth fills (rolling) -10. Per-language LSP/format/lint families (~120 entries) — *gated on Decision 1*. If multi-protocol-per-package, drops to ~30 cartridges. -11. Cloud-provider depth (~30 entries) — high cost, low novelty; defer until specific user need. -12. Security/compliance depth (~200 entries) — high effort due to trust-tier requirements; defer. - -### Phase 4 — exotic (opportunistic) -13. SDR / quantum / robotics / haptics — hardware-coupled, low priority unless specific project demand. - ---- - -## 5. Open questions for owner - -Before any of Phase 0+ ships: - -1. **Decision 1**: multi-protocol-per-package, or per-role-sibling-family? -2. **Decision 2**: retire the three tooling stubs, or rewrite them in Rust/Zig? -3. **Decision 3**: green-light the 14-LSP backfill PR series (≈14 small PRs)? -4. **Trust-tier distribution target**: confirm or override the 70/20/8/2 split. -5. **Naming**: should `boj-server-cartridges`'s taxonomy (`cross-cutting/`, `domains/`, `templates/`) become the source of truth, with the boj-server flat dir treated explicitly as a fetcher-managed cache? - -Answer these and I can convert any Phase into PR-ready work. - ---- - -## Provenance - -- Subagents fanned out at 2026-06-01 ~12:30Z; all 6 returned within ~30 minutes. -- Existing 125-cartridge inventory + 14-cartridge canonical-only list cross-referenced against every candidate. -- Joshua's cartridge-fetcher area (PR #169) explicitly excluded from suggestions. -- ~50-80 candidates duplicate across slices (e.g., several `prefect-mcp` / `airflow-mcp` mentions) — dedupe pass needed before any commit; nominal count is 941, true unique count probably 850-900. - -🤖 Generated with [Claude Code](https://claude.com/claude-code) diff --git a/docs/practice/TESTS-AND-BENCHES.adoc b/docs/practice/TESTS-AND-BENCHES.adoc new file mode 100644 index 00000000..7e9549b7 --- /dev/null +++ b/docs/practice/TESTS-AND-BENCHES.adoc @@ -0,0 +1,124 @@ +== 📊 BoJ Server — Tests and Benches + +*Status:* ACHIEVED (CRG Grade D-alpha) + +*Last Updated:* 2026-04-20 + +*Compliance:* +link:../../developer-ecosystem/standards/testing-and-benchmarking/TESTING-TAXONOMY.adoc[Hyperpolymath +Testing & Benchmarking Taxonomy v1.1.0] + +=== 🎯 Overview + +BoJ Server maintains a high-rigor testing suite covering the full 2D +matrix of protocol adapters and capability domains. The suite includes +365 passing tests across 13 categories and 14 aspect dimensions. + +=== 🌳 Test Matrix (Categories) + +[width="100%",cols="31%,23%,20%,26%",options="header",] +|=== +|Category |Status |Count |Details +|*Unit* |PASS |158+ |Core FFI modules + cartridge FFI logic + +|*P2P (Property)* |PASS |14 |Cartridge name uniqueness, vocabulary +compliance, matrix completeness + +|*E2E* |PASS |13 |MCP lifecycle, tool invocation, order-ticket protocol +flow + +|*Build* |PASS |- |`+just build+` (Zig FFI) + `+mix compile+` (Elixir +REST) + +|*Execution* |PASS |- |Deno/Node bridge + BEAM runtime (Elixir) + +|*Reflexive* |PASS |12 |`+just doctor+` health checks + self-diagnostic +Guardian module + +|*Lifecycle* |PASS |14 |Dynamic loader mount/unmount + session state + +|*Smoke* |PASS |8 |CLI help, MCP schema validation, health endpoint + +|*Property-Based* |PASS |15+ |FFI roundtrip bijection (echidna +reference) + +|*Contract/Invariant* |PASS |13 |Must/Trust/K9 enforcement on config + +catalogues + +|*Regression* |PASS |6 |Fixed bug verification (URL encoding, port +mapping) + +|*Chaos/Resilience* |PASS |12 |Guardian failure isolation + resource +gating + +|*Proof Regression* |PASS |108+ |Idris2 ABI totality checks +(`+%default total+`) +|=== + +=== 📊 Aspect Dimensions + +[width="100%",cols="32%,30%,38%",options="header",] +|=== +|Aspect |Status |Evidence +|*Security* |PASS |17 tests: Injection detection, sandboxing, SSRF +prevention, credential masking + +|*Performance* |PASS |10 benchmarks: Serialization <1ms, latency <5ms +avg, throughput 69k req/s + +|*Safety* |PASS |`+believe_me+` count reduced 31 -> 4; panic-attack +assail pass + +|*Interoperability* |PASS |MCP 2024-11-05 + JSON-RPC 2.0 + +REST/gRPC/GraphQL schemas + +|*Dependability* |PASS |Guardian module resource-aware failure tolerance + +|*Observability* |PASS |`+boj_health+` tool + structured JSON logging +|=== + +=== ⚡ Benchmarks (Baselines) + +[cols=",,,",options="header",] +|=== +|Metric |Target |Result |Status +|JSON-RPC Serialization |<1.0ms |0.001ms |✅ Extraordinary +|JSON-RPC Deserialization |<1.0ms |0.002ms |✅ Extraordinary +|Round-trip Latency |<5.0ms |0.004ms |✅ Extraordinary +|Cartridge listing |>100 req/s |69,000 req/s |✅ Extraordinary +|Tool schema gen (1000) |<10ms |1.36ms |✅ Extraordinary +|Injection detection |<100µs |1.28µs |✅ Extraordinary +|=== + +=== 🛠️ Tooling + +* *Deno:* Primary test runner for MCP bridge and integration tests. +* *Zig:* Test runner for FFI and native adapter logic +(`+zig build test+`). +* *Mix:* Test runner for the Elixir REST multiplier (`+mix test+`). +* *Idris2:* Formal proof verification (`+idris2 --check+`). +* *panic-attack:* Static analysis and security scanning (`+just scan+`). + +=== 🔄 How to Run + +[source,bash] +---- +# Full test suite +just test + +# Specific categories +deno test tests/smoke_test.ts +deno test tests/e2e_mcp_test.ts +deno test tests/mcp_bench.ts + +# FFI tests +cd ffi/zig && zig build test + +# Elixir tests +cd elixir && mix test +---- + +=== 📚 References + +* link:../TEST-NEEDS.md[TEST-NEEDS.md] — Detailed requirement tracking. +* READINESS.md — CRG Grade evidence. +* link:../.machine_readable/6a2/STATE.a2ml[.machine_readable/6a2/STATE.a2ml] +— Latest machine-readable stats. diff --git a/docs/practice/TESTS-AND-BENCHES.md b/docs/practice/TESTS-AND-BENCHES.md deleted file mode 100644 index b9312dcc..00000000 --- a/docs/practice/TESTS-AND-BENCHES.md +++ /dev/null @@ -1,85 +0,0 @@ - -# 📊 BoJ Server — Tests and Benches - -**Status:** ACHIEVED (CRG Grade D-alpha) -**Last Updated:** 2026-04-20 -**Compliance:** [Hyperpolymath Testing & Benchmarking Taxonomy v1.1.0](../../developer-ecosystem/standards/testing-and-benchmarking/TESTING-TAXONOMY.adoc) - -## 🎯 Overview - -BoJ Server maintains a high-rigor testing suite covering the full 2D matrix of protocol adapters and capability domains. The suite includes 365 passing tests across 13 categories and 14 aspect dimensions. - -## 🌳 Test Matrix (Categories) - -| Category | Status | Count | Details | -|----------|--------|-------|---------| -| **Unit** | PASS | 158+ | Core FFI modules + cartridge FFI logic | -| **P2P (Property)** | PASS | 14 | Cartridge name uniqueness, vocabulary compliance, matrix completeness | -| **E2E** | PASS | 13 | MCP lifecycle, tool invocation, order-ticket protocol flow | -| **Build** | PASS | - | `just build` (Zig FFI) + `mix compile` (Elixir REST) | -| **Execution** | PASS | - | Deno/Node bridge + BEAM runtime (Elixir) | -| **Reflexive** | PASS | 12 | `just doctor` health checks + self-diagnostic Guardian module | -| **Lifecycle** | PASS | 14 | Dynamic loader mount/unmount + session state | -| **Smoke** | PASS | 8 | CLI help, MCP schema validation, health endpoint | -| **Property-Based** | PASS | 15+ | FFI roundtrip bijection (echidna reference) | -| **Contract/Invariant**| PASS | 13 | Must/Trust/K9 enforcement on config + catalogues | -| **Regression** | PASS | 6 | Fixed bug verification (URL encoding, port mapping) | -| **Chaos/Resilience** | PASS | 12 | Guardian failure isolation + resource gating | -| **Proof Regression** | PASS | 108+ | Idris2 ABI totality checks (`%default total`) | - -## 📊 Aspect Dimensions - -| Aspect | Status | Evidence | -|--------|--------|----------| -| **Security** | PASS | 17 tests: Injection detection, sandboxing, SSRF prevention, credential masking | -| **Performance** | PASS | 10 benchmarks: Serialization <1ms, latency <5ms avg, throughput 69k req/s | -| **Safety** | PASS | `believe_me` count reduced 31 -> 4; panic-attack assail pass | -| **Interoperability**| PASS | MCP 2024-11-05 + JSON-RPC 2.0 + REST/gRPC/GraphQL schemas | -| **Dependability** | PASS | Guardian module resource-aware failure tolerance | -| **Observability** | PASS | `boj_health` tool + structured JSON logging | - -## ⚡ Benchmarks (Baselines) - -| Metric | Target | Result | Status | -|--------|--------|--------|--------| -| JSON-RPC Serialization | <1.0ms | 0.001ms | ✅ Extraordinary | -| JSON-RPC Deserialization | <1.0ms | 0.002ms | ✅ Extraordinary | -| Round-trip Latency | <5.0ms | 0.004ms | ✅ Extraordinary | -| Cartridge listing | >100 req/s | 69,000 req/s | ✅ Extraordinary | -| Tool schema gen (1000) | <10ms | 1.36ms | ✅ Extraordinary | -| Injection detection | <100µs | 1.28µs | ✅ Extraordinary | - -## 🛠️ Tooling - -- **Deno:** Primary test runner for MCP bridge and integration tests. -- **Zig:** Test runner for FFI and native adapter logic (`zig build test`). -- **Mix:** Test runner for the Elixir REST multiplier (`mix test`). -- **Idris2:** Formal proof verification (`idris2 --check`). -- **panic-attack:** Static analysis and security scanning (`just scan`). - -## 🔄 How to Run - -```bash -# Full test suite -just test - -# Specific categories -deno test tests/smoke_test.ts -deno test tests/e2e_mcp_test.ts -deno test tests/mcp_bench.ts - -# FFI tests -cd ffi/zig && zig build test - -# Elixir tests -cd elixir && mix test -``` - -## 📚 References - -- [TEST-NEEDS.md](../TEST-NEEDS.md) — Detailed requirement tracking. -- [READINESS.md](READINESS.md) — CRG Grade evidence. -- [.machine_readable/6a2/STATE.a2ml](../.machine_readable/6a2/STATE.a2ml) — Latest machine-readable stats. diff --git a/docs/proof-debt.adoc b/docs/proof-debt.adoc new file mode 100644 index 00000000..73110a4c --- /dev/null +++ b/docs/proof-debt.adoc @@ -0,0 +1,133 @@ +== Proof Debt — boj-server + +*Schema*: +https://github.com/hyperpolymath/standards/blob/main/docs/TRUSTED-BASE-REDUCTION-POLICY.adoc[hyperpolymath/standards +`+TRUSTED-BASE-REDUCTION-POLICY.adoc+`] (standards#203). + +boj-server is the *reference implementation* for the estate trusted-base +reduction policy (cited as such in standards#203). The 4 class-J axioms +this repo isolates in `+src/abi/Boj/SafetyLemmas.idr+` are the canonical +example of disposition §(c) NECESSARY AXIOM, and the +`+docs/backend-assurance/+` per-axiom files are the canonical example of +external validation under §(b)-style discipline. (A 5th former axiom, +`+charEqSym+`, was discharged to a constructive theorem on 2026-06-24 — +the first §(a) entry below.) + +This `+docs/proof-debt.md+` is a thin schema-aligned index. The +substantive content lives in: + +* *link:../PROOF-NEEDS.md[`+PROOF-NEEDS.md+`]* (repo root) — full audit +table with type signatures, classification rationale, and +external-validation pointers. +* *link:./backend-assurance/[`+docs/backend-assurance/+`]* — +per-primitive property tests + BEAM-side validation evidence +(`+prim__eqChar.md+`, `+prim__strToCharList.md+`, +`+prim__strAppend.md+`, `+prim__strSubstr.md+`). + +=== (a) DISCHARGED in this repo + +* *`+charEqSym+`* : `+(x, y : Char) -> (x == y) = (y == x)+` — +_discharged 2026-06-24_. Formerly a class-J axiom; now *derived +constructively from `+charEqSound+`* in +`+src/abi/Boj/SafetyLemmas.idr+`. Intuition: a `+True+` result forces +propositional equality via `+charEqSound+`, collapsing both sides to the +same expression; a mixed `+True+`/`+False+` split is impossible under +soundness. Verified by `+idris2 --typecheck boj.ipkg+` (Idris2 0.8.0, +17/17 modules clean). This is the first §(a) discharge and the reason +the sanctioned count dropped *5 → 4*. + +The remaining 4 class-J axioms are genuinely unavoidable in Idris2 +0.8.0; they would move here only if a future Idris2 exposes in-language +soundness principles for `+Char+` / `+String+` primitives. + +=== (b) BUDGETED — tested with a refutation budget + +The 4 axioms below are documented under §(c) (NECESSARY) rather than +§(b) (BUDGETED) because they cannot be discharged in-language. +*However*, each one is _externally validated_ via the +link:./backend-assurance/[backend-assurance harness] — property tests +running against the real BEAM runtime, exercising the Erlang/Elixir +implementations of the underlying primitive operations. + +This dual-discipline (axiom-in-Idris2 + property-tested-externally) is +the reference pattern the estate trusted-base policy recommends for +extraction-boundary code. + +=== (c) NECESSARY AXIOM + +All 4 are class-J ("`genuinely unavoidable`") per +link:../PROOF-NEEDS.md[PROOF-NEEDS.md §Axiom Audit]. Each reduces to the +same root cause: Idris2 0.8.0 treats `+Char+` and `+String+` as opaque +primitive types whose operations are foreign functions with no +constructors and no induction principle. (`+charEqSym+` was a 5th entry +here until 2026-06-24, now discharged — see §(a).) + +[width="100%",cols="9%,14%,26%,51%",options="header",] +|=== +|# |Site (`+SafetyLemmas.idr+`) |Signature |External validation +|1 |`+charEqSound+` |`+(c1,c2 : Char) -> c1 == c2 = True -> c1 = c2+` +|link:./backend-assurance/prim__eqChar.md[`+docs/backend-assurance/prim__eqChar.md+`] + +|2 |`+unpackLength+` |`+length (unpack s) = length s+` +|link:./backend-assurance/prim__strToCharList.md[`+docs/backend-assurance/prim__strToCharList.md+`] + +|3 |`+appendLengthSum+` |`+length (s ++ t) = length s + length t+` +|link:./backend-assurance/prim__strAppend.md[`+docs/backend-assurance/prim__strAppend.md+`] + +|4 |`+substrLengthBound+` |`+LTE (length (substr start len s)) len+` +|link:./backend-assurance/prim__strSubstr.md[`+docs/backend-assurance/prim__strSubstr.md+`] +|=== + +Citation: see standards#203 §"`Precedent`" — boj-server’s harness is the +reference implementation the estate trusted-base policy directs other +proof-bearing repos to adopt. + +=== (d) DEBT — actively to be closed + +_(None in the Idris2 ABI layer. The 4 class-J axioms above are §(c), not +§(d); they are not "`to be closed`" — they are load-bearing assumptions +about the trusted base. `+charEqSym+` moved §(c) → §(a) on 2026-06-24, +not via Idris2 change but because it was found to be derivable from +`+charEqSound+`.)_ + +Future work that would generate §(d) entries: + +* Migration to a future Idris2 version exposing in-language soundness +principles for primitive types — would let entries #1–#4 move from §(c) +→ §(a) via discharge. +* New `+believe_me+` introduced in non-SafetyLemmas.idr files — these +must be classified as §(a)/§(b)/§(c) before merge or land here as §(d) +with a deadline. + +=== How to update this file + +When `+src/abi/Boj/SafetyLemmas.idr+` changes: + +[arabic] +. Cross-check against `+PROOF-NEEDS.md+` — keep the per-site row counts +in sync. +. If a new `+believe_me+` is introduced anywhere outside that module, +add an entry here under §(d) or §(b) with an audit table entry in +`+PROOF-NEEDS.md+`. +. The future `+scripts/check-trusted-base.sh+` +(https://github.com/hyperpolymath/standards/blob/main/docs/TRUSTED-BASE-REDUCTION-POLICY.adoc[standards +trusted-base policy]) greps for naked `+believe_me+` outside +`+SafetyLemmas.idr+` and fails CI if it’s not annotated or enumerated +here. + +=== Companion documents + +* link:../PROOF-NEEDS.md[`+PROOF-NEEDS.md+`] — substantive audit table. +* link:./backend-assurance/[`+docs/backend-assurance/+`] — +external-validation evidence. +* https://github.com/hyperpolymath/standards/pull/195[standards#195] — +estate proof-debt audit. +* https://github.com/hyperpolymath/standards/pull/203[standards#203] — +trusted-base reduction policy (the schema this file follows). +* Memory: `+project_boj_server_backend_assurance_harness.md+` — +long-term reduction plan (~3 months, ~1 PR per primitive). + +''''' + +🤖 Schema-conformant index seeded by Claude Code, 2026-05-26. boj-server +is the reference implementation the estate trusted-base policy cites. diff --git a/docs/proof-debt.md b/docs/proof-debt.md deleted file mode 100644 index bc13a965..00000000 --- a/docs/proof-debt.md +++ /dev/null @@ -1,119 +0,0 @@ - -# Proof Debt — boj-server - -**Schema**: [hyperpolymath/standards `TRUSTED-BASE-REDUCTION-POLICY.adoc`](https://github.com/hyperpolymath/standards/blob/main/docs/TRUSTED-BASE-REDUCTION-POLICY.adoc) (standards#203). - -boj-server is the **reference implementation** for the estate trusted-base -reduction policy (cited as such in standards#203). The 4 class-J axioms -this repo isolates in `src/abi/Boj/SafetyLemmas.idr` are the canonical -example of disposition §(c) NECESSARY AXIOM, and the -`docs/backend-assurance/` per-axiom files are the canonical example of -external validation under §(b)-style discipline. (A 5th former axiom, -`charEqSym`, was discharged to a constructive theorem on 2026-06-24 — the -first §(a) entry below.) - -This `docs/proof-debt.md` is a thin schema-aligned index. The substantive -content lives in: - -- **[`PROOF-NEEDS.md`](../PROOF-NEEDS.md)** (repo root) — full audit table - with type signatures, classification rationale, and external-validation - pointers. -- **[`docs/backend-assurance/`](./backend-assurance/)** — per-primitive - property tests + BEAM-side validation evidence - (`prim__eqChar.md`, `prim__strToCharList.md`, `prim__strAppend.md`, - `prim__strSubstr.md`). - -## (a) DISCHARGED in this repo - -- **`charEqSym`** : `(x, y : Char) -> (x == y) = (y == x)` — *discharged - 2026-06-24*. Formerly a class-J axiom; now **derived constructively from - `charEqSound`** in `src/abi/Boj/SafetyLemmas.idr`. Intuition: a `True` - result forces propositional equality via `charEqSound`, collapsing both - sides to the same expression; a mixed `True`/`False` split is impossible - under soundness. Verified by `idris2 --typecheck boj.ipkg` (Idris2 0.8.0, - 17/17 modules clean). This is the first §(a) discharge and the reason the - sanctioned count dropped **5 → 4**. - -The remaining 4 class-J axioms are genuinely unavoidable in Idris2 0.8.0; -they would move here only if a future Idris2 exposes in-language soundness -principles for `Char` / `String` primitives. - -## (b) BUDGETED — tested with a refutation budget - -The 4 axioms below are documented under §(c) (NECESSARY) rather than §(b) -(BUDGETED) because they cannot be discharged in-language. **However**, -each one is *externally validated* via the -[backend-assurance harness](./backend-assurance/) — property tests -running against the real BEAM runtime, exercising the Erlang/Elixir -implementations of the underlying primitive operations. - -This dual-discipline (axiom-in-Idris2 + property-tested-externally) is -the reference pattern the estate trusted-base policy recommends for -extraction-boundary code. - -## (c) NECESSARY AXIOM - -All 4 are class-J ("genuinely unavoidable") per -[PROOF-NEEDS.md §Axiom Audit](../PROOF-NEEDS.md). Each reduces to the same -root cause: Idris2 0.8.0 treats `Char` and `String` as opaque primitive -types whose operations are foreign functions with no constructors and no -induction principle. (`charEqSym` was a 5th entry here until 2026-06-24, -now discharged — see §(a).) - -| # | Site (`SafetyLemmas.idr`) | Signature | External validation | -|---|------|-----------|---------------------| -| 1 | `charEqSound` | `(c1,c2 : Char) -> c1 == c2 = True -> c1 = c2` | [`docs/backend-assurance/prim__eqChar.md`](./backend-assurance/prim__eqChar.md) | -| 2 | `unpackLength` | `length (unpack s) = length s` | [`docs/backend-assurance/prim__strToCharList.md`](./backend-assurance/prim__strToCharList.md) | -| 3 | `appendLengthSum` | `length (s ++ t) = length s + length t` | [`docs/backend-assurance/prim__strAppend.md`](./backend-assurance/prim__strAppend.md) | -| 4 | `substrLengthBound` | `LTE (length (substr start len s)) len` | [`docs/backend-assurance/prim__strSubstr.md`](./backend-assurance/prim__strSubstr.md) | - -Citation: see standards#203 §"Precedent" — boj-server's harness is the -reference implementation the estate trusted-base policy directs other -proof-bearing repos to adopt. - -## (d) DEBT — actively to be closed - -*(None in the Idris2 ABI layer. The 4 class-J axioms above are §(c), -not §(d); they are not "to be closed" — they are load-bearing -assumptions about the trusted base. `charEqSym` moved §(c) → §(a) on -2026-06-24, not via Idris2 change but because it was found to be derivable -from `charEqSound`.)* - -Future work that would generate §(d) entries: - -- Migration to a future Idris2 version exposing in-language soundness - principles for primitive types — would let entries #1–#4 move from - §(c) → §(a) via discharge. -- New `believe_me` introduced in non-SafetyLemmas.idr files — these - must be classified as §(a)/§(b)/§(c) before merge or land here as - §(d) with a deadline. - -## How to update this file - -When `src/abi/Boj/SafetyLemmas.idr` changes: - -1. Cross-check against `PROOF-NEEDS.md` — keep the per-site row counts - in sync. -2. If a new `believe_me` is introduced anywhere outside that module, - add an entry here under §(d) or §(b) with an audit table entry in - `PROOF-NEEDS.md`. -3. The future `scripts/check-trusted-base.sh` - ([standards trusted-base policy](https://github.com/hyperpolymath/standards/blob/main/docs/TRUSTED-BASE-REDUCTION-POLICY.adoc)) - greps for naked `believe_me` outside `SafetyLemmas.idr` and fails CI - if it's not annotated or enumerated here. - -## Companion documents - -- [`PROOF-NEEDS.md`](../PROOF-NEEDS.md) — substantive audit table. -- [`docs/backend-assurance/`](./backend-assurance/) — external-validation evidence. -- [standards#195](https://github.com/hyperpolymath/standards/pull/195) — estate proof-debt audit. -- [standards#203](https://github.com/hyperpolymath/standards/pull/203) — trusted-base reduction policy (the schema this file follows). -- Memory: `project_boj_server_backend_assurance_harness.md` — long-term reduction plan (~3 months, ~1 PR per primitive). - ---- - -🤖 Schema-conformant index seeded by Claude Code, 2026-05-26. boj-server is the -reference implementation the estate trusted-base policy cites. diff --git a/docs/specification/cartridge-tools/README.adoc b/docs/specification/cartridge-tools/README.adoc new file mode 100644 index 00000000..d8739c07 --- /dev/null +++ b/docs/specification/cartridge-tools/README.adoc @@ -0,0 +1,575 @@ +== Cartridge Tools Specification + +*Powerful Cartridge Minter, Provisioner, Configurator, and Panel +Harness* + +For BoJ Server, BoJ Server + Elixir Multiplier, and panll Integration + +____ +For the cartridge specification itself (what a cartridge IS — the 2D +matrix, lifecycle, HAT model, manifest schema, and ephemerality model), +see ../cartridges/README.md. +____ + +=== 1. Core Philosophy & Design Tenets + +==== Purpose + +* *Empower Global Collaboration*: Make it trivial for anyone to create, +provision, configure, and connect cartridges to the BoJ Server ecosystem +and panll. +* *Self-Service*: Users should be able to mint, deploy, and manage +cartridges without deep technical knowledge. +* *Security & Trust*: Ensure authentication, authorization, and data +integrity for all operations. +* *Scalability*: Support Amazon/Whatsapp-scale deployments, with +parallelism, fault tolerance, and observability. +* *Versatility*: Work with BoJ Server, BoJ Server + Elixir Multiplier, +and panll, both locally and in distributed environments. + +==== Target Users + +* *Cartridge Developers*: Need tools to package, test, and distribute +their cartridges. +* *System Administrators*: Require provisioning, configuration, and +monitoring tools. +* *End Users*: Should be able to discover, install, and use cartridges +with minimal friction. +* *AI Agents*: Must be able to interact programmatically with the +tooling via APIs. + +=== 2. Architecture Overview + +==== 2.1. High-Level Component Map + +[source,mermaid] +---- +graph TD + A[Cartridge Minter] -->|Generates| B[Cartridge Artifacts] + B -->|Deploys| C[Cartridge Provisioner] + C -->|Configures| D[Cartridge Configurator] + D -->|Connects| E[Panel Harness] + E -->|Links| F[BoJ Server / BoJ Server + Elixir Multiplier] + E -->|Links| G[panll Framework] + F -->|Uses| B + G -->|Uses| B +---- + +==== 2.2. Component Responsibilities + +[width="100%",cols="25%,35%,40%",options="header",] +|=== +|Component |Responsibility |Key Technologies +|Cartridge Minter |Package, sign, and version cartridges |Docker, OCI, +Sigstore, :mix for Elixir + +|Cartridge Provisioner |Deploy cartridges to servers/nodes |Terraform, +Kubernetes, :libcluster, OTP + +|Cartridge Configurator |Apply runtime configuration to cartridges +|JSON/YAML, :confex, :libcluster + +|Panel Harness |Bridge cartridges to BoJ Server/panll |A2ML Manifests, +Phoenix Channels, gRPC +|=== + +=== 3. Detailed Component Specifications + +==== 3.1. Cartridge Minter + +*Goal*: Make it easy to create, sign, and distribute cartridges that +work with BoJ Server and panll. + +===== Functional Requirements + +[width="100%",cols="9%,27%,27%,37%",options="header",] +|=== +|ID |Requirement |Description |Cross-References +|CM-001 |Cartridge Packaging |Bundle code, config, and metadata into a +distributable artifact |Docker, OCI, :mix archive + +|CM-002 |Versioning & Signing |Assign semantic versions and +cryptographic signatures to cartridges |SemVer, Sigstore, :mix hex + +|CM-003 |Dependency Management |Resolve and bundle runtime dependencies +|:mix deps, Docker multi-stage builds + +|CM-004 |Metadata Generation |Auto-generate A2ML Manifests and panll +descriptors |JSON Schema, :jason + +|CM-005 |Local & Remote Registry |Publish to local cache or a global +registry (e.g., GitHub Container Registry) |OCI, :mix hex + +|CM-006 |CLI & API |Provide a CLI (cartridge mint) and REST/gRPC API +|escript, Absinthe/GraphQL +|=== + +===== Non-Functional Requirements + +[width="100%",cols="9%,27%,27%,37%",options="header",] +|=== +|ID |Requirement |Description |Cross-References +|CM-NFR-001 |Reproducibility |Same input → same output every time +|Docker layers, deterministic builds + +|CM-NFR-002 |Security |Cartridges are signed and verified before +deployment |Sigstore, :mix hex signatures + +|CM-NFR-003 |Performance |Minting should be fast (<10s for most +cartridges) |Parallel dependency resolution, caching + +|CM-NFR-004 |Extensibility |Support custom packagers (e.g., Guix, Bazel) +|Plugin architecture +|=== + +===== Example Workflow + +[arabic] +. Developer writes a cartridge (e.g., a Python AI model wrapper) +. Runs: `+cartridge mint --name my-ai-cartridge --version 1.0.0 --sign+` +. Output: +* `+my-ai-cartridge:1.0.0.sigstore+` (signed OCI image) +* `+my-ai-cartridge.a2ml.json+` (A2ML Manifest) +* `+my-ai-cartridge.panll.json+` (panll descriptor) + +===== CLI/API Design + +[source,bash] +---- +# CLI +cartridge mint --name my-cartridge --language elixir --type ai-agent +cartridge publish --registry ghcr.io/myorg --sign + +# API (GraphQL example) +mutation { + mintCartridge( + input: { + name: "my-cartridge" + language: "elixir" + dependencies: ["libcluster", "telemetry"] + sign: true + } + ) { + artifactId + signature + manifest { + a2ml + panll + } + } +} +---- + +==== 3.2. Cartridge Provisioner + +*Goal*: Deploy cartridges to BoJ Server, BoJ Server + Elixir Multiplier, +or panll with zero downtime. + +===== Functional Requirements + +[width="100%",cols="9%,27%,27%,37%",options="header",] +|=== +|ID |Requirement |Description |Cross-References +|CP-001 |Multi-Target Deployment |Deploy to BoJ Server, BoJ Server + +Multiplier, or panll |:libcluster, Kubernetes, Terraform + +|CP-002 |Auto-Scaling |Scale cartridges based on load |:telemetry, +Prometheus + +|CP-003 |Health Checks |Monitor cartridge health and restart failed +instances |:fuse, /health endpoints + +|CP-004 |Rollback & Recovery |Revert to previous versions on failure +|Git-style versioning, checkpoints + +|CP-005 |Secret Management |Inject secrets securely (e.g., API keys, DB +credentials) |Vault, Kubernetes Secrets + +|CP-006 |CLI & API |Provide a CLI (cartridge deploy) and REST/gRPC API +|kubectl-like UX, Absinthe +|=== + +===== Non-Functional Requirements + +[width="100%",cols="9%,27%,27%,37%",options="header",] +|=== +|ID |Requirement |Description |Cross-References +|CP-NFR-001 |Fault Tolerance |Survive node failures without data loss +|Supervisors, :libcluster + +|CP-NFR-002 |Performance |Deploy <30s for most cartridges |Parallel node +provisioning + +|CP-NFR-003 |Security |Only authorized users can deploy cartridges +|RBAC, JWT + +|CP-NFR-004 |Observability |Log all provisioning events and cartridge +metrics |:telemetry, Prometheus +|=== + +===== Example Workflow + +[arabic] +. User runs: +`+cartridge deploy --target boj-server --cartridge my-ai-cartridge:1.0.0+` +. Provisioner: +* Pulls the cartridge from the registry +* Configures the node (BoJ Server or panll) +* Starts the cartridge with health checks +* Publishes metrics to :telemetry + +===== CLI/API Design + +[source,bash] +---- +# CLI +cartridge deploy --target boj-server --cartridge my-cartridge:1.0.0 --scale 3 +cartridge status --cartridge my-cartridge + +# API (GraphQL example) +mutation { + deployCartridge( + input: { + target: "boj-server" + cartridgeId: "my-cartridge:1.0.0" + config: { env: { API_KEY: "***secret***" } } + scale: 3 + } + ) { + instanceId + status + healthCheck { + status + latency + } + } +} +---- + +==== 3.3. Cartridge Configurator + +*Goal*: Apply runtime configuration to cartridges dynamically, without +redeployment. + +===== Functional Requirements + +[width="100%",cols="9%,27%,27%,37%",options="header",] +|=== +|ID |Requirement |Description |Cross-References +|CC-001 |Dynamic Config |Update cartridge settings (e.g., API endpoints, +rate limits) without restart |:confex, :libcluster + +|CC-002 |Multi-Environment |Support dev/staging/prod configurations +|JSON/YAML, :mix env + +|CC-003 |Validation |Reject invalid configurations |JSON Schema, :jason + +|CC-004 |Secret Injection |Replace placeholders (e.g., `+{{API_KEY}}+`) +with secrets |Vault, Kubernetes Secrets + +|CC-005 |CLI & API |Provide a CLI (cartridge config) and REST/gRPC API +|kubectl configmap, Absinthe +|=== + +===== Non-Functional Requirements + +[width="100%",cols="9%,27%,27%,37%",options="header",] +|=== +|ID |Requirement |Description |Cross-References +|CC-NFR-001 |Atomicity |Config changes are applied transactionally +|:gen_statem, distributed locks + +|CC-NFR-002 |Consistency |All nodes see the same config |:libcluster, +CRDTs + +|CC-NFR-003 |Performance |Apply config changes in <100ms |In-memory +config, caching +|=== + +===== Example Workflow + +[arabic] +. User updates a config file: ++ +[source,yaml] +---- +# config/prod.yaml +api_endpoint: "https://api.example.com" +rate_limit: 1000 +---- +. Runs: +`+cartridge config apply --cartridge my-cartridge --file config/prod.yaml+` +. Configurator: +* Validates the config +* Pushes it to all nodes running my-cartridge +* Triggers a hot-reload if supported + +===== CLI/API Design + +[source,bash] +---- +# CLI +cartridge config apply --cartridge my-cartridge --file prod.yaml +cartridge config get --cartridge my-cartridge + +# API (GraphQL example) +mutation { + applyConfig( + input: { + cartridgeId: "my-cartridge" + config: { api_endpoint: "https://api.example.com", rate_limit: 1000 } + } + ) { + success + errors + } +} +---- + +==== 3.4. Panel Harness + +*Goal*: Bridge cartridges to BoJ Server and panll, enabling seamless +interaction. + +===== Functional Requirements + +[width="100%",cols="9%,27%,27%,37%",options="header",] +|=== +|ID |Requirement |Description |Cross-References +|PH-001 |BoJ Server Integration |Register cartridges as A2ML-compliant +services |A2ML Manifests, :gen_server + +|PH-002 |panll Integration |Expose cartridges as panll modules |panll’s +plugin system + +|PH-003 |Protocol Translation |Convert between cartridge formats (e.g., +REST ↔ gRPC ↔ GraphQL) |zig Triple Adapter, Phoenix Channels + +|PH-004 |Event Routing |Route events between cartridges and BoJ +Server/panll |Phoenix PubSub, A2ML Events + +|PH-005 |CLI & API |Provide a CLI (panel harness) and REST/gRPC API +|escript, Absinthe +|=== + +===== Non-Functional Requirements + +[width="100%",cols="9%,27%,27%,37%",options="header",] +|=== +|ID |Requirement |Description |Cross-References +|PH-NFR-001 |Low Latency |Sub-100ms event routing |Phoenix Channels, +WebSockets + +|PH-NFR-002 |Scalability |Handle 1M+ concurrent connections +|:libcluster, :ranch + +|PH-NFR-003 |Security |Authenticate all panel ↔ cartridge ↔ BoJ/panll +traffic |JWT, TLS 1.3 + +|PH-NFR-004 |Observability |Log all panel interactions |:telemetry, +Jaeger +|=== + +===== Example Workflow + +[arabic] +. Cartridge my-ai-cartridge is deployed +. Panel Harness: +* Registers it as an A2ML service in BoJ Server +* Exposes it as a panll module +* Routes events between my-ai-cartridge and BoJ/panll + +===== CLI/API Design + +[source,bash] +---- +# CLI +panel harness register --cartridge my-ai-cartridge --as a2ml-service +panel harness expose --cartridge my-ai-cartridge --to panll + +# API (GraphQL example) +mutation { + registerCartridge( + input: { + cartridgeId: "my-ai-cartridge" + as: "a2ml-service" + routes: [ + { protocol: "REST", port: 4000 }, + { protocol: "gRPC", port: 50051 } + ] + } + ) { + serviceId + endpoints + } +} +---- + +=== 4. Cross-Component Integration + +==== 4.1. Data Flow + +[source,mermaid] +---- +sequenceDiagram + Developer->>Cartridge Minter: mint --name my-cartridge + Cartridge Minter->>Cartridge Provisioner: deploys my-cartridge:1.0.0 + Cartridge Provisioner->>Cartridge Configurator: applies config + Cartridge Configurator->>Panel Harness: registers my-cartridge + Panel Harness->>BoJ Server: A2ML Manifest + Panel Harness->>panll: Plugin + BoJ Server->>my-cartridge: routes request + my-cartridge->>panll: interacts +---- + +==== 4.2. Shared State & Configuration + +* *Cartridge Registry*: Central store of all cartridges (local + remote) +** Implemented as a Mnesia/Riak cluster or PostgreSQL logical +replication +* *Configuration Store*: Versioned config for each cartridge +** Stored in Git (for dev) or Vault (for prod) +* *Authentication Service*: Manages JWT/OAuth2 tokens for all components +** Implemented as a Phoenix app or Elixir OTP app + +=== 5. Security Model + +[width="100%",cols="22%,31%,47%",options="header",] +|=== +|Threat |Mitigation |Cross-References +|Unauthorized cartridge deployment |RBAC + JWT |:rules, Absinthe +middleware + +|Cartridge tampering |Sigstore signatures |:mix hex signatures, Sigstore + +|Secret leakage |Vault integration |:libcluster node encryption, Vault + +|Panel spoofing |Mutual TLS |TLS 1.3, :ssl + +|Data corruption |Checksums + versioning |Git-style hashes, :mix archive +|=== + +=== 6. Performance Targets + +[cols=",,",options="header",] +|=== +|Metric |Target |Notes +|Cartridge minting |<10s |Parallel dependency resolution +|Cartridge deployment |<30s |Parallel node provisioning +|Config apply |<100ms |In-memory config, caching +|Event routing |<50ms |Phoenix Channels, WebSockets +|Scalability |1M+ concurrent connections |:libcluster, :ranch +|=== + +=== 7. Implementation Guidance + +==== 7.1. Tech Stack Recommendations + +[width="100%",cols="27%,41%,32%",options="header",] +|=== +|Component |Recommended Tech |Alternatives +|Cartridge Minter |Elixir (:mix), Docker, OCI |Guix, Bazel + +|Cartridge Provisioner |Elixir (:libcluster, :telemetry), Kubernetes +|Terraform, Nomad + +|Cartridge Configurator |Elixir (:confex, :libcluster) |Consul, etcd + +|Panel Harness |Phoenix Channels, Absinthe |gRPC, HTTP/2 +|=== + +==== 7.2. Boilerplate Code + +===== Cartridge Minter (Elixir) + +[source,elixir] +---- +# lib/cartridge_minter.ex +defmodule CartridgeMinter do + @moduledoc """ + Mints cartridges from code and config. + """ + + def mint(%{name: name, language: lang, dependencies: deps} = attrs) do + # 1. Package as Docker/OCI image + # 2. Generate A2ML Manifest + # 3. Sign the cartridge + # 4. Return artifact ID and signature + end +end +---- + +===== Cartridge Provisioner (Elixir) + +[source,elixir] +---- +# lib/cartridge_provisioner.ex +defmodule CartridgeProvisioner do + @moduledoc """ + Deploys cartridges to nodes. + """ + + def deploy(target, cartridge_id, scale: scale) do + # 1. Pull cartridge from registry + # 2. Configure node (BoJ Server or `panll`) + # 3. Start cartridge with health checks + # 4. Scale to `scale` instances + end +end +---- + +===== Panel Harness (Phoenix) + +[source,elixir] +---- +# lib/panel_harness_web/router.ex +scope "/api", PanelHarnessWeb do + post "/register", CartridgeController, :register + post "/expose", CartridgeController, :expose_to_panll +end +---- + +=== 8. Validation & Testing + +* *Unit Tests*: Test each component in isolation (e.g., config +validation, signature verification) +* *Integration Tests*: Deploy a test cartridge and verify end-to-end +flow +* *Chaos Engineering*: Kill nodes, simulate network partitions, and +verify recovery +* *Load Testing*: Use benchee or custom scripts to test scalability +* *Security Audits*: Regularly scan for vulnerabilities (e.g., :sobelow, +mix_audit) + +=== 9. Documentation & Onboarding + +* *Quickstart Guides*: +** "`Mint Your First Cartridge`" +** "`Deploy Cartridges to BoJ Server`" +** "`Configure Cartridges Dynamically`" +** "`Connect Cartridges to panll`" +* *Architecture Diagrams*: Use Mermaid/PlantUML to visualize data flow +* *API Reference*: Auto-generated from OpenAPI/Swagger +* *Example Cartridges*: Provide minimal cartridges for Elixir, Python, +and Rust + +=== 10. Open Questions & Further Research + +* How should multi-language cartridges (e.g., Python + Elixir) be +handled? +* What’s the best way to version A2ML Manifests alongside cartridges? +* Should the Panel Harness support WebAssembly (Wasm) for cartridges? +* How to audit cartridge usage for billing/metering? + +=== 11. Next Steps + +[arabic] +. Prototype the Cartridge Minter: Start with Docker + OCI packaging +. Build the Provisioner: Integrate with BoJ Server’s :libcluster +. Design the Panel Harness: Focus on A2ML/panll integration +. Implement the Configurator: Use :confex for dynamic config +. Test End-to-End: Deploy a test cartridge and verify all components +. Document & Publish: Add guides and examples to the repo + +Use this spec as a living document. Iterate based on feedback, new +requirements, or technological advancements. diff --git a/docs/specification/cartridge-tools/README.md b/docs/specification/cartridge-tools/README.md deleted file mode 100644 index 04c0a53f..00000000 --- a/docs/specification/cartridge-tools/README.md +++ /dev/null @@ -1,429 +0,0 @@ - -# Cartridge Tools Specification - -**Powerful Cartridge Minter, Provisioner, Configurator, and Panel Harness** - -For BoJ Server, BoJ Server + Elixir Multiplier, and panll Integration - -> For the cartridge specification itself (what a cartridge IS — the 2D matrix, lifecycle, HAT model, manifest schema, and ephemerality model), see [../cartridges/README.md](../cartridges/README.md). - -## 1. Core Philosophy & Design Tenets - -### Purpose - -- **Empower Global Collaboration**: Make it trivial for anyone to create, provision, configure, and connect cartridges to the BoJ Server ecosystem and panll. -- **Self-Service**: Users should be able to mint, deploy, and manage cartridges without deep technical knowledge. -- **Security & Trust**: Ensure authentication, authorization, and data integrity for all operations. -- **Scalability**: Support Amazon/Whatsapp-scale deployments, with parallelism, fault tolerance, and observability. -- **Versatility**: Work with BoJ Server, BoJ Server + Elixir Multiplier, and panll, both locally and in distributed environments. - -### Target Users - -- **Cartridge Developers**: Need tools to package, test, and distribute their cartridges. -- **System Administrators**: Require provisioning, configuration, and monitoring tools. -- **End Users**: Should be able to discover, install, and use cartridges with minimal friction. -- **AI Agents**: Must be able to interact programmatically with the tooling via APIs. - -## 2. Architecture Overview - -### 2.1. High-Level Component Map - -```mermaid -graph TD - A[Cartridge Minter] -->|Generates| B[Cartridge Artifacts] - B -->|Deploys| C[Cartridge Provisioner] - C -->|Configures| D[Cartridge Configurator] - D -->|Connects| E[Panel Harness] - E -->|Links| F[BoJ Server / BoJ Server + Elixir Multiplier] - E -->|Links| G[panll Framework] - F -->|Uses| B - G -->|Uses| B -``` - -### 2.2. Component Responsibilities - -| Component | Responsibility | Key Technologies | -|-----------|----------------|------------------| -| Cartridge Minter | Package, sign, and version cartridges | Docker, OCI, Sigstore, :mix for Elixir | -| Cartridge Provisioner | Deploy cartridges to servers/nodes | Terraform, Kubernetes, :libcluster, OTP | -| Cartridge Configurator | Apply runtime configuration to cartridges | JSON/YAML, :confex, :libcluster | -| Panel Harness | Bridge cartridges to BoJ Server/panll | A2ML Manifests, Phoenix Channels, gRPC | - -## 3. Detailed Component Specifications - -### 3.1. Cartridge Minter - -**Goal**: Make it easy to create, sign, and distribute cartridges that work with BoJ Server and panll. - -#### Functional Requirements - -| ID | Requirement | Description | Cross-References | -|----|-------------|-------------|------------------| -| CM-001 | Cartridge Packaging | Bundle code, config, and metadata into a distributable artifact | Docker, OCI, :mix archive | -| CM-002 | Versioning & Signing | Assign semantic versions and cryptographic signatures to cartridges | SemVer, Sigstore, :mix hex | -| CM-003 | Dependency Management | Resolve and bundle runtime dependencies | :mix deps, Docker multi-stage builds | -| CM-004 | Metadata Generation | Auto-generate A2ML Manifests and panll descriptors | JSON Schema, :jason | -| CM-005 | Local & Remote Registry | Publish to local cache or a global registry (e.g., GitHub Container Registry) | OCI, :mix hex | -| CM-006 | CLI & API | Provide a CLI (cartridge mint) and REST/gRPC API | escript, Absinthe/GraphQL | - -#### Non-Functional Requirements - -| ID | Requirement | Description | Cross-References | -|----|-------------|-------------|------------------| -| CM-NFR-001 | Reproducibility | Same input → same output every time | Docker layers, deterministic builds | -| CM-NFR-002 | Security | Cartridges are signed and verified before deployment | Sigstore, :mix hex signatures | -| CM-NFR-003 | Performance | Minting should be fast (<10s for most cartridges) | Parallel dependency resolution, caching | -| CM-NFR-004 | Extensibility | Support custom packagers (e.g., Guix, Bazel) | Plugin architecture | - -#### Example Workflow - -1. Developer writes a cartridge (e.g., a Python AI model wrapper) -2. Runs: `cartridge mint --name my-ai-cartridge --version 1.0.0 --sign` -3. Output: - - `my-ai-cartridge:1.0.0.sigstore` (signed OCI image) - - `my-ai-cartridge.a2ml.json` (A2ML Manifest) - - `my-ai-cartridge.panll.json` (panll descriptor) - -#### CLI/API Design - -```bash -# CLI -cartridge mint --name my-cartridge --language elixir --type ai-agent -cartridge publish --registry ghcr.io/myorg --sign - -# API (GraphQL example) -mutation { - mintCartridge( - input: { - name: "my-cartridge" - language: "elixir" - dependencies: ["libcluster", "telemetry"] - sign: true - } - ) { - artifactId - signature - manifest { - a2ml - panll - } - } -} -``` - -### 3.2. Cartridge Provisioner - -**Goal**: Deploy cartridges to BoJ Server, BoJ Server + Elixir Multiplier, or panll with zero downtime. - -#### Functional Requirements - -| ID | Requirement | Description | Cross-References | -|----|-------------|-------------|------------------| -| CP-001 | Multi-Target Deployment | Deploy to BoJ Server, BoJ Server + Multiplier, or panll | :libcluster, Kubernetes, Terraform | -| CP-002 | Auto-Scaling | Scale cartridges based on load | :telemetry, Prometheus | -| CP-003 | Health Checks | Monitor cartridge health and restart failed instances | :fuse, /health endpoints | -| CP-004 | Rollback & Recovery | Revert to previous versions on failure | Git-style versioning, checkpoints | -| CP-005 | Secret Management | Inject secrets securely (e.g., API keys, DB credentials) | Vault, Kubernetes Secrets | -| CP-006 | CLI & API | Provide a CLI (cartridge deploy) and REST/gRPC API | kubectl-like UX, Absinthe | - -#### Non-Functional Requirements - -| ID | Requirement | Description | Cross-References | -|----|-------------|-------------|------------------| -| CP-NFR-001 | Fault Tolerance | Survive node failures without data loss | Supervisors, :libcluster | -| CP-NFR-002 | Performance | Deploy <30s for most cartridges | Parallel node provisioning | -| CP-NFR-003 | Security | Only authorized users can deploy cartridges | RBAC, JWT | -| CP-NFR-004 | Observability | Log all provisioning events and cartridge metrics | :telemetry, Prometheus | - -#### Example Workflow - -1. User runs: `cartridge deploy --target boj-server --cartridge my-ai-cartridge:1.0.0` -2. Provisioner: - - Pulls the cartridge from the registry - - Configures the node (BoJ Server or panll) - - Starts the cartridge with health checks - - Publishes metrics to :telemetry - -#### CLI/API Design - -```bash -# CLI -cartridge deploy --target boj-server --cartridge my-cartridge:1.0.0 --scale 3 -cartridge status --cartridge my-cartridge - -# API (GraphQL example) -mutation { - deployCartridge( - input: { - target: "boj-server" - cartridgeId: "my-cartridge:1.0.0" - config: { env: { API_KEY: "***secret***" } } - scale: 3 - } - ) { - instanceId - status - healthCheck { - status - latency - } - } -} -``` - -### 3.3. Cartridge Configurator - -**Goal**: Apply runtime configuration to cartridges dynamically, without redeployment. - -#### Functional Requirements - -| ID | Requirement | Description | Cross-References | -|----|-------------|-------------|------------------| -| CC-001 | Dynamic Config | Update cartridge settings (e.g., API endpoints, rate limits) without restart | :confex, :libcluster | -| CC-002 | Multi-Environment | Support dev/staging/prod configurations | JSON/YAML, :mix env | -| CC-003 | Validation | Reject invalid configurations | JSON Schema, :jason | -| CC-004 | Secret Injection | Replace placeholders (e.g., `{{API_KEY}}`) with secrets | Vault, Kubernetes Secrets | -| CC-005 | CLI & API | Provide a CLI (cartridge config) and REST/gRPC API | kubectl configmap, Absinthe | - -#### Non-Functional Requirements - -| ID | Requirement | Description | Cross-References | -|----|-------------|-------------|------------------| -| CC-NFR-001 | Atomicity | Config changes are applied transactionally | :gen_statem, distributed locks | -| CC-NFR-002 | Consistency | All nodes see the same config | :libcluster, CRDTs | -| CC-NFR-003 | Performance | Apply config changes in <100ms | In-memory config, caching | - -#### Example Workflow - -1. User updates a config file: - ```yaml - # config/prod.yaml - api_endpoint: "https://api.example.com" - rate_limit: 1000 - ``` -2. Runs: `cartridge config apply --cartridge my-cartridge --file config/prod.yaml` -3. Configurator: - - Validates the config - - Pushes it to all nodes running my-cartridge - - Triggers a hot-reload if supported - -#### CLI/API Design - -```bash -# CLI -cartridge config apply --cartridge my-cartridge --file prod.yaml -cartridge config get --cartridge my-cartridge - -# API (GraphQL example) -mutation { - applyConfig( - input: { - cartridgeId: "my-cartridge" - config: { api_endpoint: "https://api.example.com", rate_limit: 1000 } - } - ) { - success - errors - } -} -``` - -### 3.4. Panel Harness - -**Goal**: Bridge cartridges to BoJ Server and panll, enabling seamless interaction. - -#### Functional Requirements - -| ID | Requirement | Description | Cross-References | -|----|-------------|-------------|------------------| -| PH-001 | BoJ Server Integration | Register cartridges as A2ML-compliant services | A2ML Manifests, :gen_server | -| PH-002 | panll Integration | Expose cartridges as panll modules | panll's plugin system | -| PH-003 | Protocol Translation | Convert between cartridge formats (e.g., REST ↔ gRPC ↔ GraphQL) | zig Triple Adapter, Phoenix Channels | -| PH-004 | Event Routing | Route events between cartridges and BoJ Server/panll | Phoenix PubSub, A2ML Events | -| PH-005 | CLI & API | Provide a CLI (panel harness) and REST/gRPC API | escript, Absinthe | - -#### Non-Functional Requirements - -| ID | Requirement | Description | Cross-References | -|----|-------------|-------------|------------------| -| PH-NFR-001 | Low Latency | Sub-100ms event routing | Phoenix Channels, WebSockets | -| PH-NFR-002 | Scalability | Handle 1M+ concurrent connections | :libcluster, :ranch | -| PH-NFR-003 | Security | Authenticate all panel ↔ cartridge ↔ BoJ/panll traffic | JWT, TLS 1.3 | -| PH-NFR-004 | Observability | Log all panel interactions | :telemetry, Jaeger | - -#### Example Workflow - -1. Cartridge my-ai-cartridge is deployed -2. Panel Harness: - - Registers it as an A2ML service in BoJ Server - - Exposes it as a panll module - - Routes events between my-ai-cartridge and BoJ/panll - -#### CLI/API Design - -```bash -# CLI -panel harness register --cartridge my-ai-cartridge --as a2ml-service -panel harness expose --cartridge my-ai-cartridge --to panll - -# API (GraphQL example) -mutation { - registerCartridge( - input: { - cartridgeId: "my-ai-cartridge" - as: "a2ml-service" - routes: [ - { protocol: "REST", port: 4000 }, - { protocol: "gRPC", port: 50051 } - ] - } - ) { - serviceId - endpoints - } -} -``` - -## 4. Cross-Component Integration - -### 4.1. Data Flow - -```mermaid -sequenceDiagram - Developer->>Cartridge Minter: mint --name my-cartridge - Cartridge Minter->>Cartridge Provisioner: deploys my-cartridge:1.0.0 - Cartridge Provisioner->>Cartridge Configurator: applies config - Cartridge Configurator->>Panel Harness: registers my-cartridge - Panel Harness->>BoJ Server: A2ML Manifest - Panel Harness->>panll: Plugin - BoJ Server->>my-cartridge: routes request - my-cartridge->>panll: interacts -``` - -### 4.2. Shared State & Configuration - -- **Cartridge Registry**: Central store of all cartridges (local + remote) - - Implemented as a Mnesia/Riak cluster or PostgreSQL logical replication -- **Configuration Store**: Versioned config for each cartridge - - Stored in Git (for dev) or Vault (for prod) -- **Authentication Service**: Manages JWT/OAuth2 tokens for all components - - Implemented as a Phoenix app or Elixir OTP app - -## 5. Security Model - -| Threat | Mitigation | Cross-References | -|--------|------------|------------------| -| Unauthorized cartridge deployment | RBAC + JWT | :rules, Absinthe middleware | -| Cartridge tampering | Sigstore signatures | :mix hex signatures, Sigstore | -| Secret leakage | Vault integration | :libcluster node encryption, Vault | -| Panel spoofing | Mutual TLS | TLS 1.3, :ssl | -| Data corruption | Checksums + versioning | Git-style hashes, :mix archive | - -## 6. Performance Targets - -| Metric | Target | Notes | -|--------|--------|-------| -| Cartridge minting | <10s | Parallel dependency resolution | -| Cartridge deployment | <30s | Parallel node provisioning | -| Config apply | <100ms | In-memory config, caching | -| Event routing | <50ms | Phoenix Channels, WebSockets | -| Scalability | 1M+ concurrent connections | :libcluster, :ranch | - -## 7. Implementation Guidance - -### 7.1. Tech Stack Recommendations - -| Component | Recommended Tech | Alternatives | -|-----------|------------------|--------------| -| Cartridge Minter | Elixir (:mix), Docker, OCI | Guix, Bazel | -| Cartridge Provisioner | Elixir (:libcluster, :telemetry), Kubernetes | Terraform, Nomad | -| Cartridge Configurator | Elixir (:confex, :libcluster) | Consul, etcd | -| Panel Harness | Phoenix Channels, Absinthe | gRPC, HTTP/2 | - -### 7.2. Boilerplate Code - -#### Cartridge Minter (Elixir) - -```elixir -# lib/cartridge_minter.ex -defmodule CartridgeMinter do - @moduledoc """ - Mints cartridges from code and config. - """ - - def mint(%{name: name, language: lang, dependencies: deps} = attrs) do - # 1. Package as Docker/OCI image - # 2. Generate A2ML Manifest - # 3. Sign the cartridge - # 4. Return artifact ID and signature - end -end -``` - -#### Cartridge Provisioner (Elixir) - -```elixir -# lib/cartridge_provisioner.ex -defmodule CartridgeProvisioner do - @moduledoc """ - Deploys cartridges to nodes. - """ - - def deploy(target, cartridge_id, scale: scale) do - # 1. Pull cartridge from registry - # 2. Configure node (BoJ Server or `panll`) - # 3. Start cartridge with health checks - # 4. Scale to `scale` instances - end -end -``` - -#### Panel Harness (Phoenix) - -```elixir -# lib/panel_harness_web/router.ex -scope "/api", PanelHarnessWeb do - post "/register", CartridgeController, :register - post "/expose", CartridgeController, :expose_to_panll -end -``` - -## 8. Validation & Testing - -- **Unit Tests**: Test each component in isolation (e.g., config validation, signature verification) -- **Integration Tests**: Deploy a test cartridge and verify end-to-end flow -- **Chaos Engineering**: Kill nodes, simulate network partitions, and verify recovery -- **Load Testing**: Use benchee or custom scripts to test scalability -- **Security Audits**: Regularly scan for vulnerabilities (e.g., :sobelow, mix_audit) - -## 9. Documentation & Onboarding - -- **Quickstart Guides**: - - "Mint Your First Cartridge" - - "Deploy Cartridges to BoJ Server" - - "Configure Cartridges Dynamically" - - "Connect Cartridges to panll" -- **Architecture Diagrams**: Use Mermaid/PlantUML to visualize data flow -- **API Reference**: Auto-generated from OpenAPI/Swagger -- **Example Cartridges**: Provide minimal cartridges for Elixir, Python, and Rust - -## 10. Open Questions & Further Research - -- How should multi-language cartridges (e.g., Python + Elixir) be handled? -- What's the best way to version A2ML Manifests alongside cartridges? -- Should the Panel Harness support WebAssembly (Wasm) for cartridges? -- How to audit cartridge usage for billing/metering? - -## 11. Next Steps - -1. Prototype the Cartridge Minter: Start with Docker + OCI packaging -2. Build the Provisioner: Integrate with BoJ Server's :libcluster -3. Design the Panel Harness: Focus on A2ML/panll integration -4. Implement the Configurator: Use :confex for dynamic config -5. Test End-to-End: Deploy a test cartridge and verify all components -6. Document & Publish: Add guides and examples to the repo - -Use this spec as a living document. Iterate based on feedback, new requirements, or technological advancements. \ No newline at end of file diff --git a/docs/specification/cartridges/README.adoc b/docs/specification/cartridges/README.adoc new file mode 100644 index 00000000..9c65ffef --- /dev/null +++ b/docs/specification/cartridges/README.adoc @@ -0,0 +1,872 @@ +== BoJ Cartridge Specification + +*Normative specification for BoJ cartridges — what a cartridge IS, how +it is structured, and what invariants it must satisfy.* + +For the cartridge specification itself (what a cartridge IS), see this +document. For the tooling that mints, provisions, configures, and +harnesses cartridges, see ../cartridge-tools/README.md. + +''''' + +=== Normative Language + +The key words *MUST*, *MUST NOT*, *REQUIRED*, *SHALL*, *SHALL NOT*, +*SHOULD*, *SHOULD NOT*, *RECOMMENDED*, *MAY*, and *OPTIONAL* in this +document are to be interpreted as described in RFC 2119. + +''''' + +=== Table of Contents + +[arabic] +. link:#1-preamble-and-scope[Preamble and Scope] +. link:++#2-axis-1--protocoltype-columns++[Axis 1 — ProtocolType +(columns)] +. link:++#3-axis-2--capabilitydomain-rows++[Axis 2 — CapabilityDomain +(rows)] +. link:#4-the-2d-capability-matrix[The 2D Capability Matrix] +. link:++#5-axis-3--hat-hardware-attached-on-top++[Axis 3 — HAT +(Hardware Attached on Top)] +. link:#6-cartridge-manifest-nickel[Cartridge Manifest (Nickel)] +. link:++#7-surface-ephemerality--three-axis-transport-model++[Surface +Ephemerality — Three-Axis Transport Model] +. link:#8-transaction-based-ephemerality-time-axis[Transaction-Based +Ephemerality (Time Axis)] +. link:#9-transport-preference-ordering-and-security-grades[Transport +Preference Ordering and Security Grades] +. link:#10-transport-provenance[Transport Provenance] +. link:#11-reference-implementation-pattern[Reference Implementation +Pattern] +. link:#12-relationship-to-cartridge-tools[Relationship to Cartridge +Tools] + +''''' + +=== 1. Preamble and Scope + +==== 1.1 What a Cartridge Is + +A *cartridge* is a formally verified, swappable capability module for +the BoJ (Bundle of Joy) server. Each cartridge occupies one or more +cells in a two-dimensional capability matrix whose axes are: + +* *ProtocolType* (the columns) — _how_ callers talk to the server. +* *CapabilityDomain* (the rows) — _what_ the server does. + +An optional third axis, *HAT* (Hardware Attached on Top), bridges a +verified cartridge to unverified external tools. HAT presence does not +alter the formal safety properties of the cartridge. + +The matrix is *sparse*: not every (ProtocolType, CapabilityDomain) cell +needs to be occupied. Unoccupied cells are simply absent from the +catalogue. + +==== 1.2 Formal Foundation + +The authoritative formal definition of a cartridge is in Idris 2: + +* `+src/abi/Boj/Catalogue.idr+` — the `+Cartridge+` record, lifecycle, +`+IsUnbreakable+` proof, and catalogue query functions. +* `+src/abi/Boj/Protocol.idr+` — the `+ProtocolType+` enumeration. +* `+src/abi/Boj/Domain.idr+` — the `+CapabilityDomain+` enumeration. + +This document is a *normative prose projection* of those definitions. If +there is ever a discrepancy between this document and the Idris 2 +source, the Idris 2 source is authoritative. + +==== 1.3 Scope of This Document + +This specification normatively defines: + +* The ProtocolType and CapabilityDomain axes and their allowed values. +* The 2D capability matrix structure, sparse-cell model, lifecycle, +proof requirement, menu tier, and hash attestation. +* The HAT third-dimension model and the four bridge types. +* The Nickel cartridge manifest schema. +* The three-axis surface ephemerality model. +* Transaction-based ephemerality. +* Transport preference ordering and security grades. +* Transport provenance tracking. +* The reference implementation pattern. + +This specification does *NOT* cover: + +* How cartridges are minted, provisioned, configured, or connected to +panll panels. Those are the responsibility of the cartridge-tools suite; +see ../cartridge-tools/README.md. +* The backend (third dimension for community extension) axis; see +../../EXTENSIBILITY.md. +* Federation, gossip protocol, or node topology. +* The BoJ REST/gRPC API surface. + +''''' + +=== 2. Axis 1 — ProtocolType (Columns) + +ProtocolType defines *how* a caller communicates with a cartridge. It +forms the *columns* of the 2D matrix. + +Every cartridge MUST declare at least one ProtocolType in its +`+protocols+` list. A cartridge MAY declare multiple protocols; each +(protocol, domain) pair constitutes one matrix cell. + +[width="100%",cols="17%,11%,29%,43%",options="header",] +|=== +|Value |Int |Description |Primary transport +|`+MCP+` |1 |Model Context Protocol — AI tool integration (stdio, SSE, +WebSocket) |stdio / SSE + +|`+LSP+` |2 |Language Server Protocol — editor/IDE integration |stdio / +TCP + +|`+DAP+` |3 |Debug Adapter Protocol — debugger integration |stdio / TCP + +|`+BSP+` |4 |Build Server Protocol — build system integration |stdio / +TCP + +|`+NeSy+` |5 |Neurosymbolic Protocol — proven-neurosym, Hypatia +integration |internal + +|`+Agentic+` |6 |Agentic Protocol — proven-agentic, OODA loop +orchestration |internal + +|`+Fleet+` |7 |Fleet Protocol — gitbot-fleet orchestration |internal + +|`+GRPC+` |8 |gRPC — high-performance binary RPC |TCP (TLS) + +|`+REST+` |9 |REST/HTTP — universal fallback |TCP (HTTP/1.1 or HTTP/2) +|=== + +The integer encoding is the C-ABI wire value used by the Zig FFI layer +(`+protocolToInt+` / `+intToProtocol+` in `+Boj.Protocol+`). + +''''' + +=== 3. Axis 2 — CapabilityDomain (Rows) + +CapabilityDomain defines *what* a cartridge does. It forms the *rows* of +the 2D matrix. + +Every cartridge MUST declare exactly one CapabilityDomain. + +[width="100%",cols="28%,20%,52%",options="header",] +|=== +|Value |Int |Description +|`+Cloud+` |1 |Cloud provider operations (AWS, GCP, Azure, Cloudflare, +etc.) + +|`+Container+` |2 |Container management — Podman, OCI image lifecycle + +|`+Database+` |3 |Database operations — SQL, NoSQL, VeriSimDB + +|`+K8s+` |4 |Kubernetes orchestration — workloads, namespaces, CRDs + +|`+Git+` |5 |Git/VCS operations — GitHub, GitLab, Bitbucket + +|`+Secrets+` |6 |Secret management — Vault, SOPS, sealed-secrets + +|`+Queues+` |7 |Message queues — NATS, RabbitMQ, Kafka + +|`+IaC+` |8 |Infrastructure as Code — Terraform, Pulumi, Guix, Guix + +|`+Observe+` |9 |Observability — metrics, logs, distributed traces + +|`+SSG+` |10 |Static site generation — Jekyll, Hugo, Zola + +|`+Proof+` |11 |Formal proof assistants — Idris2, Lean4, Coq/Rocq + +|`+FleetDom+` |12 |Gitbot fleet domain — rhodibot, echidnabot, +sustainabot, etc. + +|`+NeSyDom+` |13 |Neurosymbolic reasoning — Hypatia, ECHIDNA + +|`+Agent+` |14 |Autonomous AI agents — agentic-workflows, +autonomous-fleet + +|`+Lsp+` |15 |Language Server Protocol domain (complementary to the LSP +protocol axis) + +|`+Dap+` |16 |Debug Adapter Protocol domain + +|`+Bsp+` |17 |Build Server Protocol domain + +|`+CodeIntel+` |18 |Code intelligence — semantic search, knowledge +graph, Graph RAG +|=== + +The integer encoding is the C-ABI wire value used by the Zig FFI layer +(`+domainToInt+` / `+intToDomain+` in `+Boj.Domain+`). + +''''' + +=== 4. The 2D Capability Matrix + +==== 4.1 Sparse-Cell Model + +The matrix is indexed by `+(ProtocolType, CapabilityDomain)+`. A cell is +*occupied* if at least one cartridge in the catalogue claims that +(protocol, domain) pair. A cell is *empty* if no cartridge covers it; +empty cells carry no meaning and MUST NOT be treated as errors. + +.... + Protocol (column) + ┌─────┬─────┬─────┬─────┬─────┬─────┬─────┬──────┬──────┐ + Domain │ MCP │ LSP │ DAP │ BSP │NeSy │Agnt │Flt │ gRPC │ REST │ + (row) ├─────┼─────┼─────┼─────┼─────┼─────┼─────┼──────┼──────┤ + Cloud │ ██ │ │ │ │ │ │ │ ██ │ ██ │ + Container │ ██ │ │ │ │ │ │ │ ██ │ ██ │ + Database │ ██ │ │ │ │ │ │ │ ██ │ ██ │ + Git │ ██ │ │ │ │ │ │ ██ │ ██ │ ██ │ + ... │ │ │ │ │ │ │ │ │ │ + └─────┴─────┴─────┴─────┴─────┴─────┴─────┴──────┴──────┘ + (filled cells are illustrative; actual catalogue may differ) +.... + +A single cartridge with `+protocols = [MCP, GRPC, REST]+` occupies +*three cells* in the same domain row. + +==== 4.2 CartridgeStatus Lifecycle + +Every cartridge has a `+CartridgeStatus+` that governs whether it may be +mounted. + +.... + Development ──► Ready ──► Deprecated + │ + └──► Faulty +.... + +[width="100%",cols="23%,13%,29%,35%",options="header",] +|=== +|Status |Int |Mountable |Description +|`+Development+` |0 |No |Under construction; proofs incomplete + +|`+Ready+` |1 |Yes |Fully verified; safe to mount + +|`+Deprecated+` |2 |No |Scheduled for removal; MUST NOT be mounted by +new callers + +|`+Faulty+` |3 |No |Broken or compromised; MUST NOT be mounted +|=== + +A cartridge MUST be in `+Ready+` status before the Zig FFI layer will +activate it. This constraint is enforced by the `+IsUnbreakable+` proof +(§4.3). + +`+Deprecated+` cartridges MAY remain mounted for existing callers during +a migration window, but new `+lookupCell+` calls MUST NOT return them. + +==== 4.3 IsUnbreakable Proof + +The `+IsUnbreakable+` predicate is the core safety gate: + +[source,idris] +---- +data IsUnbreakable : Cartridge -> Type where + VerifiedReady : (c : Cartridge) -> + (status c = Ready) -> + IsUnbreakable c +---- + +A cartridge is *unbreakable* if and only if its status equals `+Ready+`. +The Zig FFI layer checks this predicate before mounting any cartridge. +No cartridge MAY be executed without a valid `+IsUnbreakable+` proof. + +The proof is *independent* of HAT presence (§5). A cartridge with a HAT +that is currently failing still holds its `+IsUnbreakable+` proof; the +HAT failure is isolated by the circuit breaker (§5.4), not by revoking +the proof. + +==== 4.4 MenuTier + +Every cartridge MUST declare a `+MenuTier+` that determines where it +appears in the Teranga navigation menu. + +[cols=",",options="header",] +|=== +|Tier |Description +|`+Teranga+` |Core cartridges maintained by the BoJ project +|`+Shield+` |Privacy and security cartridges (SDP, oDNS, zero-trust) +|`+Ayo+` |Community-contributed cartridges (joy of shared work) +|=== + +Community extensions MUST use the `+Ayo+` tier. + +==== 4.5 Hash Attestation + +Every cartridge MUST supply a `+binaryHash+` field containing the +SHA-256 hex digest of its compiled shared library (`+.so+` or platform +equivalent). Federation nodes MUST verify this hash before accepting a +remotely-supplied cartridge. + +An empty `+binaryHash+` is permissible only for `+Development+`-status +cartridges. A `+Ready+`-status cartridge with an empty `+binaryHash+` +MUST be rejected by the validator. + +''''' + +=== 5. Axis 3 — HAT (Hardware Attached on Top) + +==== 5.1 Motivation + +The `+IsUnbreakable+` proof ensures that only verified cartridges can be +mounted. This would be undermined if cartridges could invoke arbitrary +external tools directly — tools such as the Git CLI, Podman, Terraform, +or VeriSimDB that cannot themselves be formally verified. + +The *HAT* (Hardware Attached on Top) model resolves this tension. The +name is an analogy with the Raspberry Pi and BeagleBone hardware +extension ecosystem: small, well-defined add-on boards that sit _on top +of_ the verified platform without altering its guarantees. + +==== 5.2 What a HAT Is + +A HAT is a bridge script or module that translates a BoJ cartridge +invocation into a real-world tool call: + +.... + BoJ Cartridge (verified) ---> HAT Bridge ---> External Tool (unverified) + git-mcp git_hat.sh git CLI + container-mcp podman_hat.sh podman + ssg-mcp zola_hat.sh zola + database-mcp verisimdb_hat.so VeriSimDB (Library FFI) +.... + +The HAT is *outside* the BoJ safety perimeter. BoJ makes no formal +guarantees about HAT behaviour. Correctness of the external tool is +entirely the HAT author’s responsibility. + +A cartridge MAY have zero HATs (it operates entirely within the BoJ +perimeter), one HAT, or multiple HATs serving different external tools. + +==== 5.3 Four Bridge Types + +[width="100%",cols="22%,46%,32%",options="header",] +|=== +|Type |When to use |Example +|*CLI wrapper* |HAT invokes a command-line tool and parses its +stdout/stderr |`+git_hat.sh+` → `+git+` + +|*JSON-RPC stdio* |HAT speaks JSON-RPC 2.0 over stdin/stdout to an +existing MCP-compatible server; enables composition with the existing +ecosystem |wrapping a third-party MCP server + +|*HTTP API* |HAT calls a REST or GraphQL endpoint on a local or remote +service |cloud provider API calls + +|*Library FFI* |HAT links against a native shared library via C ABI; +highest performance, zero IPC overhead |`+verisimdb_hat.so+` → VeriSimDB +|=== + +HAT authors MUST document which bridge type their HAT uses. A single HAT +MUST use exactly one bridge type. + +==== 5.4 Safety Properties and Circuit-Breaker Isolation + +The HAT model preserves BoJ’s safety invariants: + +* *`+IsUnbreakable+` still holds*: the cartridge’s mounting proof is +independent of HAT presence or HAT health. A cartridge is unbreakable +because its _own_ status is `+Ready+` — not because its HAT is healthy. +* *Circuit-breaker isolation*: if a HAT call fails (non-zero exit code, +timeout, malformed output), the per-cartridge circuit breaker trips. The +cartridge enters a temporarily degraded state; other cartridges are +unaffected. The circuit breaker resets after a configurable back-off +period. +* *Thread safety preserved*: HAT invocations go through the same +mutex-protected FFI exports as internal operations. Concurrent HAT calls +for the same cartridge are serialised. +* *Attestation unaffected*: the HAT is not part of the attested binary. +Federation nodes verify only the BoJ core binary hash; HAT scripts are +treated as runtime configuration, not as part of the attested surface. + +A cartridge MUST NOT bypass the circuit breaker to retry a failing HAT +inline. Retry logic is the responsibility of the caller, not the +cartridge. + +''''' + +=== 6. Cartridge Manifest (Nickel) + +==== 6.1 Format Decision + +The authoritative cartridge manifest format is *Nickel* (`+.ncl+`). +Nickel manifests are type-checked at validation time, schema-enforced by +the cartridge validator, and consumed by the cartridge-tools suite. + +Existing `+cartridge.json+` files are *legacy* and MUST be migrated to +Nickel. Until migration is complete both formats coexist; the JSON +schema at `+https://boj.dev/schemas/cartridge/v1.json+` remains the +operative validator for JSON manifests. New cartridges MUST use Nickel. + +==== 6.2 Required Fields + +[width="100%",cols="27%,23%,50%",options="header",] +|=== +|Field |Type |Description +|`+name+` |`+String+` |Unique cartridge identifier (snake_case) + +|`+version+` |`+String+` |SemVer version string (e.g., `+"0.1.0"+`) + +|`+protocol_type+` |`+List ProtocolType+` |One or more protocol axis +values + +|`+capability_domain+` |`+CapabilityDomain+` |Single domain axis value + +|`+menu_tier+` |`+MenuTier+` |`+"Teranga"+` \| `+"Shield"+` \| `+"Ayo"+` + +|`+status+` |`+CartridgeStatus+` |`+"Development"+` \| `+"Ready"+` \| +`+"Deprecated"+` \| `+"Faulty"+` + +|`+hash_attestation+` |`+String+` |SHA-256 hex digest of the compiled +`+.so+` (empty string ONLY for `+Development+`) + +|`+supported_transports+` |`+List TransportEntry+` |At minimum one entry +per protocol_type declared +|=== + +A `+TransportEntry+` has the following shape: + +[width="100%",cols="36%,20%,44%",options="header",] +|=== +|Subfield |Type |Description +|`+name+` |`+String+` |Transport identifier (e.g., `+"stdio"+`, +`+"grpc-tls"+`, `+"rest-http2"+`) + +|`+security_grade+` |`+SecurityGrade+` |`+"A"+` \| `+"B"+` \| `+"C"+` \| +`+"D"+` — see §9 + +|`+provenance+` |`+String+` |Why this transport is present — see §10 +|=== + +==== 6.3 Example Nickel Manifest + +[source,nickel] +---- +# SPDX-License-Identifier: CC-BY-SA-4.0 +# git-mcp cartridge manifest (Nickel) +{ + name = "git-mcp", + version = "0.2.0", + protocol_type = ["MCP", "GRPC", "REST"], + capability_domain = "Git", + menu_tier = "Teranga", + status = "Ready", + hash_attestation = "a1b2c3d4e5f6...", # SHA-256 of libgit_mcp.so + + supported_transports = [ + { + name = "stdio", + security_grade = "A", + provenance = "Required by MCP protocol spec; default AI agent transport" + }, + { + name = "grpc-tls", + security_grade = "A", + provenance = "High-throughput CI pipeline integration; requested by fleet-bot" + }, + { + name = "rest-http2", + security_grade = "B", + provenance = "Universal fallback for HTTP clients" + } + ], + + # Proposed HAT block shape — no Nickel HAT schema is yet implemented. + # Fields: enabled (Bool), bridge_type (one of: "cli-wrapper" | "json-rpc-stdio" | + # "http-api" | "library-ffi" per §5.3), target (external tool name), plus one of: + # bridge_type = "cli-wrapper" → bridge_script (path to shell wrapper) + # bridge_type = "library-ffi" → library_path (path to .so) + # bridge_type = "http-api" → base_url (string) + # bridge_type = "json-rpc-stdio"→ command (array of argv) + # isolation_mechanism and trust_tier fields are OPTIONAL; if omitted, defaults + # are "circuit-breaker" and "external-trust" respectively. + # This shape is subject to change on first real deployment with a Nickel HAT schema. + hat = { + enabled = true, + bridge_type = "cli-wrapper", + target = "git", + bridge_script = "hats/git_hat.sh" + } +} +---- + +==== 6.4 Legacy JSON Shape (Reference) + +For migration reference, a legacy `+cartridge.json+` has this top-level +structure: + +[source,json] +---- +{ + "$schema": "https://boj.dev/schemas/cartridge/v1.json", + "spdx": "MPL-2.0", + "name": "aerie-mcp", + "version": "0.1.0", + "domain": "infrastructure", + "tier": "Ayo", + "protocols": ["MCP", "REST"], + "tools": [ ... ] +} +---- + +The Nickel schema extends this with explicit transport, security grade, +and provenance fields. The `+tools+` array is retained in the Nickel +manifest under an optional `+tools+` field; it does not affect the +matrix position of the cartridge. + +''''' + +=== 7. Surface Ephemerality — Three-Axis Transport Model + +Surface ephemerality is the property that the cartridge’s network attack +surface at any moment is the minimal set of transports actually in use — +nothing more. + +Three sub-axes together define this surface: + +==== 7.1 `+possible_transports+` (Manifest Axis) + +`+possible_transports+` is the set of transports a cartridge is *capable +of speaking at all*. It is derived from `+supported_transports+` in the +manifest. + +A transport that does not appear in `+possible_transports+` *MUST NOT* +ever be opened for this cartridge, regardless of caller demand. This is +a hard capability boundary, enforced at manifest validation time. + +==== 7.2 `+preferred_transports+` (Ordering Axis) + +`+preferred_transports+` is an *ordered list* of transports from +`+possible_transports+`, ranked from most preferred (index 0) to least +preferred. Security grade is the primary sort key (§9); the cartridge +author MAY override the ordering for operational reasons, but MUST +document the rationale in the transport’s `+provenance+` field. + +When multiple callers hold different active transports simultaneously, +BoJ honours `+preferred_transports+` when deciding which transport to +offer to a new caller that has not expressed a preference. + +==== 7.3 `+active_transports+` (Runtime Axis) + +`+active_transports+` is the *runtime set* of transports currently in +active use by at least one legitimate caller. + +*Default-locked semantics*: a transport that is not in +`+active_transports+` is *entirely locked down* — it is not listening on +any port or file descriptor, not registered in any routing table, not +reachable by any means. The absence is total, not merely rate-limited or +authenticated-only. + +*On-demand spin-up*: when a legitimate caller requests a transport for a +given (capability, cartridge) pair, BoJ performs an admission check: + +.... + 1. Is the transport in possible_transports? (manifest check) + 2. Does the caller's capability token authorise it? (capability check) + 3. Is the cartridge in Ready status? (IsUnbreakable check) +.... + +All three MUST pass. On success, BoJ opens the transport, adds it to +`+active_transports+`, and begins serving the caller. On failure, the +request is rejected with a typed error; no transport is opened. + +*Drain-down*: when all callers using a transport disconnect or their +capability tokens expire, BoJ MUST drain and close the transport, +removing it from `+active_transports+`. Drain is complete when no +in-flight requests remain on that transport. + +==== 7.4 Relationship to BoJ Design Philosophy + +Surface ephemerality and transaction-based ephemerality together +reconcile the design tension between "`many endpoints`" and "`reduced +attack surface`". A cartridge can declare nine transports in its +manifest while exposing zero of them at rest. Elixir/BEAM concurrency in +the BoJ multiplier layer handles many concurrent heterogeneous calls +without requiring transports to remain persistently open. + +''''' + +=== 8. Transaction-Based Ephemerality (Time Axis) + +==== 8.1 Ephemeral Capability Tokens + +Every capability invocation is scoped to a *single transaction*. A +transaction begins when a caller presents a capability token and issues +a request; it ends when the response is delivered (or the request errors +out). + +Capability tokens are *bound to one transaction* and are destroyed at +transaction end. They MUST NOT be reused across requests. This means: + +* No persistent session state survives between requests. Callers MUST +re-authenticate (present a fresh token) for every transaction. +* There is no concept of a "`logged-in session`" at the cartridge level. +The BoJ infrastructure layer MAY maintain a session abstraction for UX +purposes, but the cartridge itself sees only individual authenticated +transactions. +* Token destruction is synchronous: at transaction end, the token is +zeroed from memory before the response is written. The token MUST NOT +appear in logs, traces, or error payloads. + +==== 8.2 Long-Lived Streams + +Some protocols (WebSocket, SSE) maintain a persistent connection that +carries multiple logical messages. BoJ frames these as a sequence of +capability-gated messages: + +* Each *message* on a WebSocket or SSE stream is independently +capability-gated. +* A stream connection itself is opened with an initial capability token +that authorises stream establishment. +* Per-message capability gates are evaluated at message dispatch, not at +stream open. A capability token that was valid at stream-open MAY expire +during the stream; subsequent messages will be rejected. +* Callers SHOULD refresh their capability token while a stream is open +if they intend to send further messages. Streams MUST be closed +gracefully when the caller’s capability expires, not abruptly dropped. + +==== 8.3 Interaction with Surface Ephemerality + +Transaction-based ephemerality is the _time axis_; surface ephemerality +(§7) is the _space axis_. Together they ensure: + +* At any moment, only the transports demanded by current callers are +open (space). +* At any moment, only the capabilities granted by current tokens are +exercisable (time). + +''''' + +=== 9. Transport Preference Ordering and Security Grades + +==== 9.1 Security Grades + +Every transport entry in `+supported_transports+` MUST carry a security +grade. Grades rank the security properties of the transport channel +itself (not the application-level authentication). + +[width="100%",cols="28%,36%,36%",options="header",] +|=== +|Grade |Meaning |Examples +|`+A+` |Mutual TLS or equivalent forward-secrecy; no unauthenticated +transport |`+grpc-tls+` (mTLS), `+stdio+` (process-local, OS-enforced) + +|`+B+` |Server-authenticated TLS; client authentication at application +layer |`+rest-http2+` (TLS + JWT), `+wss+` (WSS + token) + +|`+C+` |Authenticated but in-transit not encrypted, or encrypted but +unauthenticated |`+rest-http1+` (plaintext + basic auth) + +|`+D+` |No transport-level security; suitable only for +loopback/localhost |`+rest-http-local+`, `+grpc-insecure-local+` +|=== + +Grade `+D+` transports MUST only be declared for localhost +(`+127.0.0.1+` / `+::1+`) endpoints. A grade `+D+` transport MUST NOT be +opened on a network-reachable address. + +____ +*Footnote (2026-04-17):* Security grades are derived from 2026 +TLS/transport-security consensus (RFC 8446 TLS 1.3, RFC 7540 HTTP/2, +mTLS best practices). They are not formally standardised by any SDO. The +grade ladder is internally consistent: mutual authentication + forward +secrecy = A; server-auth TLS + app-layer client auth = B; auth without +encryption or encryption without auth = C; no transport security at all += D. Grades may be revised as transport security standards evolve. +____ + +==== 9.2 Preference Ordering + +`+preferred_transports+` ranks entries from most preferred (highest +security, lowest latency) to least preferred. BoJ MUST honour this +ordering when selecting a transport on behalf of a caller that has not +expressed a preference. + +When multiple callers hold different active transports, BoJ: + +[arabic] +. Selects the highest-preference transport for the new caller. +. If that transport is not yet active, spins it up (§7.3). +. If the highest-preference transport is unavailable (HAT circuit +breaker open, port in use), BoJ falls back to the next entry in +`+preferred_transports+`. + +BoJ MUST NOT silently downgrade from a grade `+A+` or `+B+` transport to +a grade `+C+` or `+D+` transport without emitting a warning to the +caller. + +''''' + +=== 10. Transport Provenance + +Every entry in `+supported_transports+` MUST carry a `+provenance+` +string. Provenance records *why* the transport exists in the manifest — +which upstream requirement, plugin, or caller demanded it. + +==== 10.1 Purpose + +Provenance serves two purposes: + +[arabic] +. *Traceability*: auditors and operators can understand why each +transport is open, rather than finding unexplained network endpoints. +. *Clean-up signal*: if the upstream requirement that justified a +transport is removed, the provenance field identifies the transport as a +candidate for removal from `+supported_transports+`. + +==== 10.2 Interpretation Rules + +[width="100%",cols="54%,46%",options="header",] +|=== +|Provenance value |Interpretation +|`+"Required by spec"+` |The transport is mandated by the +protocol specification; it MUST be present if the protocol is declared + +|`+"Requested by "+` |A specific downstream consumer +requires this transport; it MAY be removed if that consumer is +decommissioned + +|`+"Fallback for "+` |Operational fallback; SHOULD be removed +if the primary transport achieves full coverage + +|`+"Loopback-only; for "+` |Grade-D localhost transport; MUST +be scoped to 127.0.0.1/::1 +|=== + +Provenance strings are free-form but MUST be human-readable. Empty +provenance is NOT PERMITTED for `+Ready+`-status cartridges. + +''''' + +=== 11. Reference Implementation Pattern + +The IDApTIK UMS (User Management System) cartridge is the *canonical +reference implementation* for how a cartridge crosses the Idris2 → Zig → +Rust boundary. + +==== 11.1 Layer Stack + +.... + ┌────────────────────────────────────┐ + │ Idris2 ABI (src/abi/) │ Dependent types, erased proof fields, + │ GuardsInZones, ZonesOrdered, │ compile-time invariants + │ PBXConsistent, DefenceTargets, │ + │ DevicesExist │ + └─────────────────┬──────────────────┘ + │ C-ABI integers / booleans + ┌─────────────────▼──────────────────┐ + │ Zig FFI (ffi/zig/src/) │ C-compatible implementation, + │ ValidationResult { bool, bool, │ mirrors every Idris2 type as a + │ bool, bool, bool } │ boolean struct for FFI export + └─────────────────┬──────────────────┘ + │ extern "C" + Tauri command wrappers + ┌─────────────────▼──────────────────┐ + │ Rust Tauri commands │ Named _cartridge_ + │ (IDApTIK/src-tauri/src/ │ so BoJ routes: + │ commands.rs) │ invoke("_cartridge_", …) + └─────────────────┬──────────────────┘ + │ shared filesystem bridge + ┌─────────────────▼──────────────────┐ + │ Bridge directory │ /tmp/panll/-bridge/ + │ /tmp/panll/ums-bridge/ │ for data exchange between + │ │ Tauri backend and BoJ cartridge + └────────────────────────────────────┘ +.... + +==== 11.2 Naming Convention + +Rust Tauri command wrappers MUST follow the naming scheme: + +.... + _cartridge_ +.... + +For example: + +[cols=",,",options="header",] +|=== +|Command name |Cartridge |Operation +|`+ums_cartridge_validate_guards+` |`+ums+` |`+validate_guards+` +|`+ums_cartridge_check_zones+` |`+ums+` |`+check_zones+` +|`+git_cartridge_clone+` |`+git+` |`+clone+` +|`+database_cartridge_query+` |`+database+` |`+query+` +|=== + +BoJ routes invocations using `+invoke("_cartridge_", …)+` +where `++` matches the `+name+` field in the cartridge manifest. +This convention MUST be followed for all cartridges that use the Tauri +command wrapper pattern. + +==== 11.3 Idris2 Erased Proof Fields + +Validation types in cartridge ABIs SHOULD use erased (runtime-zero-cost) +proof fields for their invariants, following the UMS pattern: + +[source,idris] +---- +-- Proof fields with multiplicity 0 (erased at runtime): +record ZoneConfig where + constructor MkZoneConfig + zones : List Zone + 0 zonesOk : ZonesOrdered zones -- erased; zero runtime cost + 0 pbxOk : PBXConsistent zones -- erased; zero runtime cost +---- + +The Zig FFI mirror exposes only the boolean results; the proofs +themselves are compile-time artefacts. + +==== 11.4 Canonical Reference Pattern + +The IDApTIK UMS cartridge (`+idaptik/idaptik-ums+`) is the canonical +reference for the Rust Tauri command pattern. The pattern is: + +* *Idris2 ABI* (`+src/abi/+`) declares the interface with dependent-type +proofs. +* *Zig FFI* (`+ffi/zig/src/+`) mirrors every Idris2 type as a +C-compatible boolean struct and exports it via `+extern "C"+`. +* *Rust Tauri commands* (`+src-tauri/src/commands.rs+` within the Tauri +app) wrap the Zig `+extern "C"+` symbols as `+#[tauri::command]+` +functions named `+_cartridge_+`. +* *Bridge directory* (`+/tmp/panll/-bridge/+`) is the shared +filesystem exchange point between the Tauri backend and the BoJ +cartridge. + +The naming convention, bridge directory structure, and extern +declarations in `+idaptik/idaptik-ums/src-tauri/src/commands.rs+` are +normative for all cartridges following this pattern. Read that file +directly when implementing a new cartridge of this type; do not +reproduce it here. + +____ +*Note (2026-04-17):* The specific file path +`+IDApTIK/src-tauri/src/commands.rs+` cited in earlier drafts refers to +the monorepo-root Tauri app path within the IDApTIK monorepo +(`+/var/mnt/eclipse/repos/idaptik/+`). Navigate to +`+idaptik/idaptik-ums/+` and check `+src-tauri/src/commands.rs+`. If +that path does not yet exist, the UMS cartridge is still in Development +status and the pattern described above is the authoritative description +until the file lands. +____ + +''''' + +=== 12. Relationship to Cartridge Tools + +The cartridge-tools suite (minter, provisioner, configurator, panel +harness) *consumes* this specification. Every requirement in +../cartridge-tools/README.md that refers to cartridge structure, +manifest fields, lifecycle states, or transport behaviour is grounded in +this document. + +If there is ever a conflict between the cartridge-tools spec and this +cartridge spec, this document takes precedence (subject to the Idris2 +source being authoritative over both). + +''''' + +_This specification was extracted and consolidated on 2026-04-17 from_ +_`+src/abi/Boj/Catalogue.idr+`, `+src/abi/Boj/Protocol.idr+`,_ +_`+src/abi/Boj/Domain.idr+`, and +`+docs/papers/boj-architecture-paper.md+` §§5–6._ diff --git a/docs/specification/cartridges/README.md b/docs/specification/cartridges/README.md deleted file mode 100644 index ade214c9..00000000 --- a/docs/specification/cartridges/README.md +++ /dev/null @@ -1,731 +0,0 @@ - - -# BoJ Cartridge Specification - -**Normative specification for BoJ cartridges — what a cartridge IS, how it is -structured, and what invariants it must satisfy.** - -For the cartridge specification itself (what a cartridge IS), see this document. -For the tooling that mints, provisions, configures, and harnesses cartridges, -see [../cartridge-tools/README.md](../cartridge-tools/README.md). - ---- - -## Normative Language - -The key words **MUST**, **MUST NOT**, **REQUIRED**, **SHALL**, **SHALL NOT**, -**SHOULD**, **SHOULD NOT**, **RECOMMENDED**, **MAY**, and **OPTIONAL** in this -document are to be interpreted as described in RFC 2119. - ---- - -## Table of Contents - -1. [Preamble and Scope](#1-preamble-and-scope) -2. [Axis 1 — ProtocolType (columns)](#2-axis-1--protocoltype-columns) -3. [Axis 2 — CapabilityDomain (rows)](#3-axis-2--capabilitydomain-rows) -4. [The 2D Capability Matrix](#4-the-2d-capability-matrix) -5. [Axis 3 — HAT (Hardware Attached on Top)](#5-axis-3--hat-hardware-attached-on-top) -6. [Cartridge Manifest (Nickel)](#6-cartridge-manifest-nickel) -7. [Surface Ephemerality — Three-Axis Transport Model](#7-surface-ephemerality--three-axis-transport-model) -8. [Transaction-Based Ephemerality (Time Axis)](#8-transaction-based-ephemerality-time-axis) -9. [Transport Preference Ordering and Security Grades](#9-transport-preference-ordering-and-security-grades) -10. [Transport Provenance](#10-transport-provenance) -11. [Reference Implementation Pattern](#11-reference-implementation-pattern) -12. [Relationship to Cartridge Tools](#12-relationship-to-cartridge-tools) - ---- - -## 1. Preamble and Scope - -### 1.1 What a Cartridge Is - -A **cartridge** is a formally verified, swappable capability module for the BoJ -(Bundle of Joy) server. Each cartridge occupies one or more cells in a -two-dimensional capability matrix whose axes are: - -- **ProtocolType** (the columns) — *how* callers talk to the server. -- **CapabilityDomain** (the rows) — *what* the server does. - -An optional third axis, **HAT** (Hardware Attached on Top), bridges a verified -cartridge to unverified external tools. HAT presence does not alter the formal -safety properties of the cartridge. - -The matrix is **sparse**: not every (ProtocolType, CapabilityDomain) cell needs -to be occupied. Unoccupied cells are simply absent from the catalogue. - -### 1.2 Formal Foundation - -The authoritative formal definition of a cartridge is in Idris 2: - -- `src/abi/Boj/Catalogue.idr` — the `Cartridge` record, lifecycle, `IsUnbreakable` - proof, and catalogue query functions. -- `src/abi/Boj/Protocol.idr` — the `ProtocolType` enumeration. -- `src/abi/Boj/Domain.idr` — the `CapabilityDomain` enumeration. - -This document is a **normative prose projection** of those definitions. If -there is ever a discrepancy between this document and the Idris 2 source, the -Idris 2 source is authoritative. - -### 1.3 Scope of This Document - -This specification normatively defines: - -- The ProtocolType and CapabilityDomain axes and their allowed values. -- The 2D capability matrix structure, sparse-cell model, lifecycle, proof - requirement, menu tier, and hash attestation. -- The HAT third-dimension model and the four bridge types. -- The Nickel cartridge manifest schema. -- The three-axis surface ephemerality model. -- Transaction-based ephemerality. -- Transport preference ordering and security grades. -- Transport provenance tracking. -- The reference implementation pattern. - -This specification does **NOT** cover: - -- How cartridges are minted, provisioned, configured, or connected to panll - panels. Those are the responsibility of the cartridge-tools suite; see - [../cartridge-tools/README.md](../cartridge-tools/README.md). -- The backend (third dimension for community extension) axis; see - [../../EXTENSIBILITY.md](../../EXTENSIBILITY.md). -- Federation, gossip protocol, or node topology. -- The BoJ REST/gRPC API surface. - ---- - -## 2. Axis 1 — ProtocolType (Columns) - -ProtocolType defines **how** a caller communicates with a cartridge. It forms -the **columns** of the 2D matrix. - -Every cartridge MUST declare at least one ProtocolType in its `protocols` list. -A cartridge MAY declare multiple protocols; each (protocol, domain) pair -constitutes one matrix cell. - -| Value | Int | Description | Primary transport | -|-------|-----|-------------|-------------------| -| `MCP` | 1 | Model Context Protocol — AI tool integration (stdio, SSE, WebSocket) | stdio / SSE | -| `LSP` | 2 | Language Server Protocol — editor/IDE integration | stdio / TCP | -| `DAP` | 3 | Debug Adapter Protocol — debugger integration | stdio / TCP | -| `BSP` | 4 | Build Server Protocol — build system integration | stdio / TCP | -| `NeSy` | 5 | Neurosymbolic Protocol — proven-neurosym, Hypatia integration | internal | -| `Agentic` | 6 | Agentic Protocol — proven-agentic, OODA loop orchestration | internal | -| `Fleet` | 7 | Fleet Protocol — gitbot-fleet orchestration | internal | -| `GRPC` | 8 | gRPC — high-performance binary RPC | TCP (TLS) | -| `REST` | 9 | REST/HTTP — universal fallback | TCP (HTTP/1.1 or HTTP/2) | - -The integer encoding is the C-ABI wire value used by the Zig FFI layer -(`protocolToInt` / `intToProtocol` in `Boj.Protocol`). - ---- - -## 3. Axis 2 — CapabilityDomain (Rows) - -CapabilityDomain defines **what** a cartridge does. It forms the **rows** of -the 2D matrix. - -Every cartridge MUST declare exactly one CapabilityDomain. - -| Value | Int | Description | -|-------|-----|-------------| -| `Cloud` | 1 | Cloud provider operations (AWS, GCP, Azure, Cloudflare, etc.) | -| `Container` | 2 | Container management — Podman, OCI image lifecycle | -| `Database` | 3 | Database operations — SQL, NoSQL, VeriSimDB | -| `K8s` | 4 | Kubernetes orchestration — workloads, namespaces, CRDs | -| `Git` | 5 | Git/VCS operations — GitHub, GitLab, Bitbucket | -| `Secrets` | 6 | Secret management — Vault, SOPS, sealed-secrets | -| `Queues` | 7 | Message queues — NATS, RabbitMQ, Kafka | -| `IaC` | 8 | Infrastructure as Code — Terraform, Pulumi, Guix, Guix | -| `Observe` | 9 | Observability — metrics, logs, distributed traces | -| `SSG` | 10 | Static site generation — Jekyll, Hugo, Zola | -| `Proof` | 11 | Formal proof assistants — Idris2, Lean4, Coq/Rocq | -| `FleetDom` | 12 | Gitbot fleet domain — rhodibot, echidnabot, sustainabot, etc. | -| `NeSyDom` | 13 | Neurosymbolic reasoning — Hypatia, ECHIDNA | -| `Agent` | 14 | Autonomous AI agents — agentic-workflows, autonomous-fleet | -| `Lsp` | 15 | Language Server Protocol domain (complementary to the LSP protocol axis) | -| `Dap` | 16 | Debug Adapter Protocol domain | -| `Bsp` | 17 | Build Server Protocol domain | -| `CodeIntel` | 18 | Code intelligence — semantic search, knowledge graph, Graph RAG | - -The integer encoding is the C-ABI wire value used by the Zig FFI layer -(`domainToInt` / `intToDomain` in `Boj.Domain`). - ---- - -## 4. The 2D Capability Matrix - -### 4.1 Sparse-Cell Model - -The matrix is indexed by `(ProtocolType, CapabilityDomain)`. A cell is -**occupied** if at least one cartridge in the catalogue claims that (protocol, -domain) pair. A cell is **empty** if no cartridge covers it; empty cells carry -no meaning and MUST NOT be treated as errors. - -``` - Protocol (column) - ┌─────┬─────┬─────┬─────┬─────┬─────┬─────┬──────┬──────┐ - Domain │ MCP │ LSP │ DAP │ BSP │NeSy │Agnt │Flt │ gRPC │ REST │ - (row) ├─────┼─────┼─────┼─────┼─────┼─────┼─────┼──────┼──────┤ - Cloud │ ██ │ │ │ │ │ │ │ ██ │ ██ │ - Container │ ██ │ │ │ │ │ │ │ ██ │ ██ │ - Database │ ██ │ │ │ │ │ │ │ ██ │ ██ │ - Git │ ██ │ │ │ │ │ │ ██ │ ██ │ ██ │ - ... │ │ │ │ │ │ │ │ │ │ - └─────┴─────┴─────┴─────┴─────┴─────┴─────┴──────┴──────┘ - (filled cells are illustrative; actual catalogue may differ) -``` - -A single cartridge with `protocols = [MCP, GRPC, REST]` occupies **three cells** -in the same domain row. - -### 4.2 CartridgeStatus Lifecycle - -Every cartridge has a `CartridgeStatus` that governs whether it may be mounted. - -``` - Development ──► Ready ──► Deprecated - │ - └──► Faulty -``` - -| Status | Int | Mountable | Description | -|--------|-----|-----------|-------------| -| `Development` | 0 | No | Under construction; proofs incomplete | -| `Ready` | 1 | Yes | Fully verified; safe to mount | -| `Deprecated` | 2 | No | Scheduled for removal; MUST NOT be mounted by new callers | -| `Faulty` | 3 | No | Broken or compromised; MUST NOT be mounted | - -A cartridge MUST be in `Ready` status before the Zig FFI layer will activate it. -This constraint is enforced by the `IsUnbreakable` proof (§4.3). - -`Deprecated` cartridges MAY remain mounted for existing callers during a -migration window, but new `lookupCell` calls MUST NOT return them. - -### 4.3 IsUnbreakable Proof - -The `IsUnbreakable` predicate is the core safety gate: - -```idris -data IsUnbreakable : Cartridge -> Type where - VerifiedReady : (c : Cartridge) -> - (status c = Ready) -> - IsUnbreakable c -``` - -A cartridge is **unbreakable** if and only if its status equals `Ready`. The Zig -FFI layer checks this predicate before mounting any cartridge. No cartridge MAY -be executed without a valid `IsUnbreakable` proof. - -The proof is **independent** of HAT presence (§5). A cartridge with a HAT -that is currently failing still holds its `IsUnbreakable` proof; the HAT -failure is isolated by the circuit breaker (§5.4), not by revoking the proof. - -### 4.4 MenuTier - -Every cartridge MUST declare a `MenuTier` that determines where it appears in -the Teranga navigation menu. - -| Tier | Description | -|------|-------------| -| `Teranga` | Core cartridges maintained by the BoJ project | -| `Shield` | Privacy and security cartridges (SDP, oDNS, zero-trust) | -| `Ayo` | Community-contributed cartridges (joy of shared work) | - -Community extensions MUST use the `Ayo` tier. - -### 4.5 Hash Attestation - -Every cartridge MUST supply a `binaryHash` field containing the SHA-256 hex -digest of its compiled shared library (`.so` or platform equivalent). Federation -nodes MUST verify this hash before accepting a remotely-supplied cartridge. - -An empty `binaryHash` is permissible only for `Development`-status cartridges. -A `Ready`-status cartridge with an empty `binaryHash` MUST be rejected by the -validator. - ---- - -## 5. Axis 3 — HAT (Hardware Attached on Top) - -### 5.1 Motivation - -The `IsUnbreakable` proof ensures that only verified cartridges can be mounted. -This would be undermined if cartridges could invoke arbitrary external tools -directly — tools such as the Git CLI, Podman, Terraform, or VeriSimDB that -cannot themselves be formally verified. - -The **HAT** (Hardware Attached on Top) model resolves this tension. The name is -an analogy with the Raspberry Pi and BeagleBone hardware extension ecosystem: -small, well-defined add-on boards that sit *on top of* the verified platform -without altering its guarantees. - -### 5.2 What a HAT Is - -A HAT is a bridge script or module that translates a BoJ cartridge invocation -into a real-world tool call: - -``` - BoJ Cartridge (verified) ---> HAT Bridge ---> External Tool (unverified) - git-mcp git_hat.sh git CLI - container-mcp podman_hat.sh podman - ssg-mcp zola_hat.sh zola - database-mcp verisimdb_hat.so VeriSimDB (Library FFI) -``` - -The HAT is **outside** the BoJ safety perimeter. BoJ makes no formal guarantees -about HAT behaviour. Correctness of the external tool is entirely the HAT -author's responsibility. - -A cartridge MAY have zero HATs (it operates entirely within the BoJ perimeter), -one HAT, or multiple HATs serving different external tools. - -### 5.3 Four Bridge Types - -| Type | When to use | Example | -|------|-------------|---------| -| **CLI wrapper** | HAT invokes a command-line tool and parses its stdout/stderr | `git_hat.sh` → `git` | -| **JSON-RPC stdio** | HAT speaks JSON-RPC 2.0 over stdin/stdout to an existing MCP-compatible server; enables composition with the existing ecosystem | wrapping a third-party MCP server | -| **HTTP API** | HAT calls a REST or GraphQL endpoint on a local or remote service | cloud provider API calls | -| **Library FFI** | HAT links against a native shared library via C ABI; highest performance, zero IPC overhead | `verisimdb_hat.so` → VeriSimDB | - -HAT authors MUST document which bridge type their HAT uses. A single HAT MUST -use exactly one bridge type. - -### 5.4 Safety Properties and Circuit-Breaker Isolation - -The HAT model preserves BoJ's safety invariants: - -- **`IsUnbreakable` still holds**: the cartridge's mounting proof is independent - of HAT presence or HAT health. A cartridge is unbreakable because its *own* - status is `Ready` — not because its HAT is healthy. -- **Circuit-breaker isolation**: if a HAT call fails (non-zero exit code, - timeout, malformed output), the per-cartridge circuit breaker trips. The - cartridge enters a temporarily degraded state; other cartridges are - unaffected. The circuit breaker resets after a configurable back-off period. -- **Thread safety preserved**: HAT invocations go through the same - mutex-protected FFI exports as internal operations. Concurrent HAT calls for - the same cartridge are serialised. -- **Attestation unaffected**: the HAT is not part of the attested binary. - Federation nodes verify only the BoJ core binary hash; HAT scripts are - treated as runtime configuration, not as part of the attested surface. - -A cartridge MUST NOT bypass the circuit breaker to retry a failing HAT -inline. Retry logic is the responsibility of the caller, not the cartridge. - ---- - -## 6. Cartridge Manifest (Nickel) - -### 6.1 Format Decision - -The authoritative cartridge manifest format is **Nickel** (`.ncl`). Nickel -manifests are type-checked at validation time, schema-enforced by the cartridge -validator, and consumed by the cartridge-tools suite. - -Existing `cartridge.json` files are **legacy** and MUST be migrated to Nickel. -Until migration is complete both formats coexist; the JSON schema at -`https://boj.dev/schemas/cartridge/v1.json` remains the operative validator for -JSON manifests. New cartridges MUST use Nickel. - -### 6.2 Required Fields - -| Field | Type | Description | -|-------|------|-------------| -| `name` | `String` | Unique cartridge identifier (snake_case) | -| `version` | `String` | SemVer version string (e.g., `"0.1.0"`) | -| `protocol_type` | `List ProtocolType` | One or more protocol axis values | -| `capability_domain` | `CapabilityDomain` | Single domain axis value | -| `menu_tier` | `MenuTier` | `"Teranga"` \| `"Shield"` \| `"Ayo"` | -| `status` | `CartridgeStatus` | `"Development"` \| `"Ready"` \| `"Deprecated"` \| `"Faulty"` | -| `hash_attestation` | `String` | SHA-256 hex digest of the compiled `.so` (empty string ONLY for `Development`) | -| `supported_transports` | `List TransportEntry` | At minimum one entry per protocol_type declared | - -A `TransportEntry` has the following shape: - -| Subfield | Type | Description | -|----------|------|-------------| -| `name` | `String` | Transport identifier (e.g., `"stdio"`, `"grpc-tls"`, `"rest-http2"`) | -| `security_grade` | `SecurityGrade` | `"A"` \| `"B"` \| `"C"` \| `"D"` — see §9 | -| `provenance` | `String` | Why this transport is present — see §10 | - -### 6.3 Example Nickel Manifest - -```nickel -# SPDX-License-Identifier: CC-BY-SA-4.0 -# git-mcp cartridge manifest (Nickel) -{ - name = "git-mcp", - version = "0.2.0", - protocol_type = ["MCP", "GRPC", "REST"], - capability_domain = "Git", - menu_tier = "Teranga", - status = "Ready", - hash_attestation = "a1b2c3d4e5f6...", # SHA-256 of libgit_mcp.so - - supported_transports = [ - { - name = "stdio", - security_grade = "A", - provenance = "Required by MCP protocol spec; default AI agent transport" - }, - { - name = "grpc-tls", - security_grade = "A", - provenance = "High-throughput CI pipeline integration; requested by fleet-bot" - }, - { - name = "rest-http2", - security_grade = "B", - provenance = "Universal fallback for HTTP clients" - } - ], - - # Proposed HAT block shape — no Nickel HAT schema is yet implemented. - # Fields: enabled (Bool), bridge_type (one of: "cli-wrapper" | "json-rpc-stdio" | - # "http-api" | "library-ffi" per §5.3), target (external tool name), plus one of: - # bridge_type = "cli-wrapper" → bridge_script (path to shell wrapper) - # bridge_type = "library-ffi" → library_path (path to .so) - # bridge_type = "http-api" → base_url (string) - # bridge_type = "json-rpc-stdio"→ command (array of argv) - # isolation_mechanism and trust_tier fields are OPTIONAL; if omitted, defaults - # are "circuit-breaker" and "external-trust" respectively. - # This shape is subject to change on first real deployment with a Nickel HAT schema. - hat = { - enabled = true, - bridge_type = "cli-wrapper", - target = "git", - bridge_script = "hats/git_hat.sh" - } -} -``` - -### 6.4 Legacy JSON Shape (Reference) - -For migration reference, a legacy `cartridge.json` has this top-level structure: - -```json -{ - "$schema": "https://boj.dev/schemas/cartridge/v1.json", - "spdx": "MPL-2.0", - "name": "aerie-mcp", - "version": "0.1.0", - "domain": "infrastructure", - "tier": "Ayo", - "protocols": ["MCP", "REST"], - "tools": [ ... ] -} -``` - -The Nickel schema extends this with explicit transport, security grade, and -provenance fields. The `tools` array is retained in the Nickel manifest under -an optional `tools` field; it does not affect the matrix position of the cartridge. - ---- - -## 7. Surface Ephemerality — Three-Axis Transport Model - -Surface ephemerality is the property that the cartridge's network attack surface -at any moment is the minimal set of transports actually in use — nothing more. - -Three sub-axes together define this surface: - -### 7.1 `possible_transports` (Manifest Axis) - -`possible_transports` is the set of transports a cartridge is **capable of -speaking at all**. It is derived from `supported_transports` in the manifest. - -A transport that does not appear in `possible_transports` **MUST NOT** ever be -opened for this cartridge, regardless of caller demand. This is a hard -capability boundary, enforced at manifest validation time. - -### 7.2 `preferred_transports` (Ordering Axis) - -`preferred_transports` is an **ordered list** of transports from -`possible_transports`, ranked from most preferred (index 0) to least preferred. -Security grade is the primary sort key (§9); the cartridge author MAY override -the ordering for operational reasons, but MUST document the rationale in the -transport's `provenance` field. - -When multiple callers hold different active transports simultaneously, BoJ -honours `preferred_transports` when deciding which transport to offer to a new -caller that has not expressed a preference. - -### 7.3 `active_transports` (Runtime Axis) - -`active_transports` is the **runtime set** of transports currently in active use -by at least one legitimate caller. - -**Default-locked semantics**: a transport that is not in `active_transports` is -**entirely locked down** — it is not listening on any port or file descriptor, -not registered in any routing table, not reachable by any means. The absence is -total, not merely rate-limited or authenticated-only. - -**On-demand spin-up**: when a legitimate caller requests a transport for a given -(capability, cartridge) pair, BoJ performs an admission check: - -``` - 1. Is the transport in possible_transports? (manifest check) - 2. Does the caller's capability token authorise it? (capability check) - 3. Is the cartridge in Ready status? (IsUnbreakable check) -``` - -All three MUST pass. On success, BoJ opens the transport, adds it to -`active_transports`, and begins serving the caller. On failure, the request is -rejected with a typed error; no transport is opened. - -**Drain-down**: when all callers using a transport disconnect or their -capability tokens expire, BoJ MUST drain and close the transport, removing it -from `active_transports`. Drain is complete when no in-flight requests remain -on that transport. - -### 7.4 Relationship to BoJ Design Philosophy - -Surface ephemerality and transaction-based ephemerality together reconcile the -design tension between "many endpoints" and "reduced attack surface". A cartridge -can declare nine transports in its manifest while exposing zero of them at rest. -Elixir/BEAM concurrency in the BoJ multiplier layer handles many concurrent -heterogeneous calls without requiring transports to remain persistently open. - ---- - -## 8. Transaction-Based Ephemerality (Time Axis) - -### 8.1 Ephemeral Capability Tokens - -Every capability invocation is scoped to a **single transaction**. A -transaction begins when a caller presents a capability token and issues a -request; it ends when the response is delivered (or the request errors out). - -Capability tokens are **bound to one transaction** and are destroyed at -transaction end. They MUST NOT be reused across requests. This means: - -- No persistent session state survives between requests. Callers MUST - re-authenticate (present a fresh token) for every transaction. -- There is no concept of a "logged-in session" at the cartridge level. The - BoJ infrastructure layer MAY maintain a session abstraction for UX purposes, - but the cartridge itself sees only individual authenticated transactions. -- Token destruction is synchronous: at transaction end, the token is zeroed - from memory before the response is written. The token MUST NOT appear in - logs, traces, or error payloads. - -### 8.2 Long-Lived Streams - -Some protocols (WebSocket, SSE) maintain a persistent connection that carries -multiple logical messages. BoJ frames these as a sequence of capability-gated -messages: - -- Each **message** on a WebSocket or SSE stream is independently capability-gated. -- A stream connection itself is opened with an initial capability token that - authorises stream establishment. -- Per-message capability gates are evaluated at message dispatch, not at stream - open. A capability token that was valid at stream-open MAY expire during the - stream; subsequent messages will be rejected. -- Callers SHOULD refresh their capability token while a stream is open if they - intend to send further messages. Streams MUST be closed gracefully when the - caller's capability expires, not abruptly dropped. - -### 8.3 Interaction with Surface Ephemerality - -Transaction-based ephemerality is the *time axis*; surface ephemerality (§7) -is the *space axis*. Together they ensure: - -- At any moment, only the transports demanded by current callers are open (space). -- At any moment, only the capabilities granted by current tokens are exercisable (time). - ---- - -## 9. Transport Preference Ordering and Security Grades - -### 9.1 Security Grades - -Every transport entry in `supported_transports` MUST carry a security grade. -Grades rank the security properties of the transport channel itself (not the -application-level authentication). - -| Grade | Meaning | Examples | -|-------|---------|---------| -| `A` | Mutual TLS or equivalent forward-secrecy; no unauthenticated transport | `grpc-tls` (mTLS), `stdio` (process-local, OS-enforced) | -| `B` | Server-authenticated TLS; client authentication at application layer | `rest-http2` (TLS + JWT), `wss` (WSS + token) | -| `C` | Authenticated but in-transit not encrypted, or encrypted but unauthenticated | `rest-http1` (plaintext + basic auth) | -| `D` | No transport-level security; suitable only for loopback/localhost | `rest-http-local`, `grpc-insecure-local` | - -Grade `D` transports MUST only be declared for localhost (`127.0.0.1` / `::1`) -endpoints. A grade `D` transport MUST NOT be opened on a network-reachable -address. - -> **Footnote (2026-04-17):** Security grades are derived from 2026 TLS/transport-security -> consensus (RFC 8446 TLS 1.3, RFC 7540 HTTP/2, mTLS best practices). They are not formally -> standardised by any SDO. The grade ladder is internally consistent: mutual authentication -> + forward secrecy = A; server-auth TLS + app-layer client auth = B; auth without encryption -> or encryption without auth = C; no transport security at all = D. Grades may be revised -> as transport security standards evolve. - -### 9.2 Preference Ordering - -`preferred_transports` ranks entries from most preferred (highest security, -lowest latency) to least preferred. BoJ MUST honour this ordering when selecting -a transport on behalf of a caller that has not expressed a preference. - -When multiple callers hold different active transports, BoJ: - -1. Selects the highest-preference transport for the new caller. -2. If that transport is not yet active, spins it up (§7.3). -3. If the highest-preference transport is unavailable (HAT circuit breaker open, - port in use), BoJ falls back to the next entry in `preferred_transports`. - -BoJ MUST NOT silently downgrade from a grade `A` or `B` transport to a grade -`C` or `D` transport without emitting a warning to the caller. - ---- - -## 10. Transport Provenance - -Every entry in `supported_transports` MUST carry a `provenance` string. -Provenance records **why** the transport exists in the manifest — which upstream -requirement, plugin, or caller demanded it. - -### 10.1 Purpose - -Provenance serves two purposes: - -1. **Traceability**: auditors and operators can understand why each transport is - open, rather than finding unexplained network endpoints. -2. **Clean-up signal**: if the upstream requirement that justified a transport is - removed, the provenance field identifies the transport as a candidate for - removal from `supported_transports`. - -### 10.2 Interpretation Rules - -| Provenance value | Interpretation | -|-----------------|---------------| -| `"Required by spec"` | The transport is mandated by the protocol specification; it MUST be present if the protocol is declared | -| `"Requested by "` | A specific downstream consumer requires this transport; it MAY be removed if that consumer is decommissioned | -| `"Fallback for "` | Operational fallback; SHOULD be removed if the primary transport achieves full coverage | -| `"Loopback-only; for "` | Grade-D localhost transport; MUST be scoped to 127.0.0.1/::1 | - -Provenance strings are free-form but MUST be human-readable. Empty provenance -is NOT PERMITTED for `Ready`-status cartridges. - ---- - -## 11. Reference Implementation Pattern - -The IDApTIK UMS (User Management System) cartridge is the **canonical reference -implementation** for how a cartridge crosses the Idris2 → Zig → Rust boundary. - -### 11.1 Layer Stack - -``` - ┌────────────────────────────────────┐ - │ Idris2 ABI (src/abi/) │ Dependent types, erased proof fields, - │ GuardsInZones, ZonesOrdered, │ compile-time invariants - │ PBXConsistent, DefenceTargets, │ - │ DevicesExist │ - └─────────────────┬──────────────────┘ - │ C-ABI integers / booleans - ┌─────────────────▼──────────────────┐ - │ Zig FFI (ffi/zig/src/) │ C-compatible implementation, - │ ValidationResult { bool, bool, │ mirrors every Idris2 type as a - │ bool, bool, bool } │ boolean struct for FFI export - └─────────────────┬──────────────────┘ - │ extern "C" + Tauri command wrappers - ┌─────────────────▼──────────────────┐ - │ Rust Tauri commands │ Named _cartridge_ - │ (IDApTIK/src-tauri/src/ │ so BoJ routes: - │ commands.rs) │ invoke("_cartridge_", …) - └─────────────────┬──────────────────┘ - │ shared filesystem bridge - ┌─────────────────▼──────────────────┐ - │ Bridge directory │ /tmp/panll/-bridge/ - │ /tmp/panll/ums-bridge/ │ for data exchange between - │ │ Tauri backend and BoJ cartridge - └────────────────────────────────────┘ -``` - -### 11.2 Naming Convention - -Rust Tauri command wrappers MUST follow the naming scheme: - -``` - _cartridge_ -``` - -For example: - -| Command name | Cartridge | Operation | -|-------------|-----------|-----------| -| `ums_cartridge_validate_guards` | `ums` | `validate_guards` | -| `ums_cartridge_check_zones` | `ums` | `check_zones` | -| `git_cartridge_clone` | `git` | `clone` | -| `database_cartridge_query` | `database` | `query` | - -BoJ routes invocations using `invoke("_cartridge_", …)` where `` -matches the `name` field in the cartridge manifest. This convention MUST be -followed for all cartridges that use the Tauri command wrapper pattern. - -### 11.3 Idris2 Erased Proof Fields - -Validation types in cartridge ABIs SHOULD use erased (runtime-zero-cost) proof -fields for their invariants, following the UMS pattern: - -```idris --- Proof fields with multiplicity 0 (erased at runtime): -record ZoneConfig where - constructor MkZoneConfig - zones : List Zone - 0 zonesOk : ZonesOrdered zones -- erased; zero runtime cost - 0 pbxOk : PBXConsistent zones -- erased; zero runtime cost -``` - -The Zig FFI mirror exposes only the boolean results; the proofs themselves are -compile-time artefacts. - -### 11.4 Canonical Reference Pattern - -The IDApTIK UMS cartridge (`idaptik/idaptik-ums`) is the canonical reference -for the Rust Tauri command pattern. The pattern is: - -- **Idris2 ABI** (`src/abi/`) declares the interface with dependent-type proofs. -- **Zig FFI** (`ffi/zig/src/`) mirrors every Idris2 type as a C-compatible - boolean struct and exports it via `extern "C"`. -- **Rust Tauri commands** (`src-tauri/src/commands.rs` within the Tauri app) - wrap the Zig `extern "C"` symbols as `#[tauri::command]` functions named - `_cartridge_`. -- **Bridge directory** (`/tmp/panll/-bridge/`) is the shared filesystem - exchange point between the Tauri backend and the BoJ cartridge. - -The naming convention, bridge directory structure, and extern declarations in -`idaptik/idaptik-ums/src-tauri/src/commands.rs` are normative for all cartridges -following this pattern. Read that file directly when implementing a new cartridge -of this type; do not reproduce it here. - -> **Note (2026-04-17):** The specific file path `IDApTIK/src-tauri/src/commands.rs` -> cited in earlier drafts refers to the monorepo-root Tauri app path within the -> IDApTIK monorepo (`/var/mnt/eclipse/repos/idaptik/`). Navigate to -> `idaptik/idaptik-ums/` and check `src-tauri/src/commands.rs`. If that path does -> not yet exist, the UMS cartridge is still in Development status and the pattern -> described above is the authoritative description until the file lands. - ---- - -## 12. Relationship to Cartridge Tools - -The cartridge-tools suite (minter, provisioner, configurator, panel harness) -**consumes** this specification. Every requirement in -[../cartridge-tools/README.md](../cartridge-tools/README.md) that refers to -cartridge structure, manifest fields, lifecycle states, or transport behaviour -is grounded in this document. - -If there is ever a conflict between the cartridge-tools spec and this cartridge -spec, this document takes precedence (subject to the Idris2 source being -authoritative over both). - ---- - -*This specification was extracted and consolidated on 2026-04-17 from* -*`src/abi/Boj/Catalogue.idr`, `src/abi/Boj/Protocol.idr`,* -*`src/abi/Boj/Domain.idr`, and `docs/papers/boj-architecture-paper.md` §§5–6.* diff --git a/docs/tech-debt-2026-05-26.adoc b/docs/tech-debt-2026-05-26.adoc new file mode 100644 index 00000000..aa96496a --- /dev/null +++ b/docs/tech-debt-2026-05-26.adoc @@ -0,0 +1,80 @@ +== Tech-Debt Audit — boj-server — 2026-05-26 + +*Source:* estate-wide automated scan 2026-05-26. *Companion:* +https://github.com/hyperpolymath/standards/tree/main/docs/audits[`+hyperpolymath/standards+` +2026-05-26-estate-*-debt audits]. *Combined severity:* `+LOW+`. + +This file records the _raw findings_ — it does not by itself fix the +debt. Each section ends with a '`Recommended next move`' line; closing +the debt is follow-up work. + +=== 1. Proof debt + +Scanner counted the following markers in proof-bearing files of this +repo: + +.... +files= 126 | Coq-Axm/Adm= 0 | Lean-srry/ax= 0 | Agda-pst= 0 | Idr-blv= 9 | Idr-prtl= 0 | Fstr-asm= 0 | TODO= 0 | Unsafe= 0 +.... + +*Total markers:* 9. *Severity:* `+>09+`. + +*Marker types* (any non-zero counts above): - Coq `+Axiom+`/`+Admitted+` +— unconditional proof escapes. - Lean `+sorry+`/`+axiom+` — Lean’s +equivalent. - Agda `+postulate+` — accepted axiomatically. - Idris2 +`+believe_me+`/`+assert_total+` — runtime-safe coercion / totality +assumption. - Idris2 top-level `+partial+` — totality-check waived. - F* +`+assume val+`/`+admit_p+` — F* admit. - `+TODO PROOF+` / `+OWED:+` — +self-documented debt markers. - `+unsafePerformIO+`/`+unsafeCoerce+` — +soundness-relevant escape hatches in Haskell/Rust source. + +*Recommended next move:* triage each finding into one of: (a) discharge +by proof, (b) cover with property-tests + a documented refutation +budget, or (c) annotate as a known/necessary axiom (e.g. `+funExt+`) in +`+docs/proof-debt.md+`. + +=== 2. Licence debt + +[cols=",",options="header",] +|=== +|Field |Value +|LICENSE file |`+LICENSE+` +|SPDX header |`+NONE+` +|Manifest licence |`+MPL-2.0+` +|Body classifier |`+MPL-2.0-pure+` +|Severity |`+ok+` +|=== + +*Recommended next move:* none for licence. + +=== 3. Documentation debt + +[cols=",",options="header",] +|=== +|Field |Value +|README lines |518 +|`+docs/+` files |89 +|`+docs/+` LoC |25917 +|CHANGELOG.md |Y +|CONTRIBUTING.md |Y +|CODE_OF_CONDUCT.md |Y +|SECURITY.md |Y +|Severity |`+OK+` +|=== + +*Recommended next move:* none for docs. + +=== Cross-references + +* Estate proof-debt audit: +`+hyperpolymath/standards/docs/audits/2026-05-26-estate-proof-debt.md+` +* Estate licence-debt audit: +`+hyperpolymath/standards/docs/audits/2026-05-26-estate-licence-debt.md+` +* Estate documentation-debt audit: +`+hyperpolymath/standards/docs/audits/2026-05-26-estate-documentation-debt.md+` + +''''' + +🤖 Generated by Claude Code estate-wide tech-debt scan (2026-05-26). +This file is informational — closing the debt is follow-up work owned by +the maintainer. diff --git a/docs/tech-debt-2026-05-26.md b/docs/tech-debt-2026-05-26.md deleted file mode 100644 index c6438cd9..00000000 --- a/docs/tech-debt-2026-05-26.md +++ /dev/null @@ -1,70 +0,0 @@ - -# Tech-Debt Audit — boj-server — 2026-05-26 - -**Source:** estate-wide automated scan 2026-05-26. -**Companion:** [`hyperpolymath/standards` 2026-05-26-estate-*-debt audits](https://github.com/hyperpolymath/standards/tree/main/docs/audits). -**Combined severity:** `LOW`. - -This file records the *raw findings* — it does not by itself fix the debt. Each section ends with a 'Recommended next move' line; closing the debt is follow-up work. - -## 1. Proof debt - -Scanner counted the following markers in proof-bearing files of this repo: - -``` -files= 126 | Coq-Axm/Adm= 0 | Lean-srry/ax= 0 | Agda-pst= 0 | Idr-blv= 9 | Idr-prtl= 0 | Fstr-asm= 0 | TODO= 0 | Unsafe= 0 -``` - -**Total markers:** 9. **Severity:** `>09`. - -**Marker types** (any non-zero counts above): -- Coq `Axiom`/`Admitted` — unconditional proof escapes. -- Lean `sorry`/`axiom` — Lean's equivalent. -- Agda `postulate` — accepted axiomatically. -- Idris2 `believe_me`/`assert_total` — runtime-safe coercion / totality assumption. -- Idris2 top-level `partial` — totality-check waived. -- F\* `assume val`/`admit_p` — F\* admit. -- `TODO PROOF` / `OWED:` — self-documented debt markers. -- `unsafePerformIO`/`unsafeCoerce` — soundness-relevant escape hatches in Haskell/Rust source. - -**Recommended next move:** triage each finding into one of: (a) discharge by proof, (b) cover with property-tests + a documented refutation budget, or (c) annotate as a known/necessary axiom (e.g. `funExt`) in `docs/proof-debt.md`. - -## 2. Licence debt - -| Field | Value | -|---|---| -| LICENSE file | `LICENSE` | -| SPDX header | `NONE` | -| Manifest licence | `MPL-2.0` | -| Body classifier | `MPL-2.0-pure` | -| Severity | `ok` | - -**Recommended next move:** none for licence. - -## 3. Documentation debt - -| Field | Value | -|---|---| -| README lines | 518 | -| `docs/` files | 89 | -| `docs/` LoC | 25917 | -| CHANGELOG.md | Y | -| CONTRIBUTING.md | Y | -| CODE_OF_CONDUCT.md | Y | -| SECURITY.md | Y | -| Severity | `OK` | - -**Recommended next move:** none for docs. - -## Cross-references - -- Estate proof-debt audit: `hyperpolymath/standards/docs/audits/2026-05-26-estate-proof-debt.md` -- Estate licence-debt audit: `hyperpolymath/standards/docs/audits/2026-05-26-estate-licence-debt.md` -- Estate documentation-debt audit: `hyperpolymath/standards/docs/audits/2026-05-26-estate-documentation-debt.md` - ---- - -🤖 Generated by Claude Code estate-wide tech-debt scan (2026-05-26). This file is informational — closing the debt is follow-up work owned by the maintainer. diff --git a/schemas/SCHEMA-MIRROR.adoc b/schemas/SCHEMA-MIRROR.adoc new file mode 100644 index 00000000..9d01a532 --- /dev/null +++ b/schemas/SCHEMA-MIRROR.adoc @@ -0,0 +1,46 @@ +== Cartridge schema mirror + +The file `+cartridge-v1.json+` in this directory is a *vendored mirror* +of the canonical schema living at: + +* *Canonical home:* +https://github.com/hyperpolymath/standards/blob/main/cartridges/cartridge-v1.json[`+hyperpolymath/standards+`] +(filed as PR https://github.com/hyperpolymath/standards/pull/200[#200] +on 2026-05-26). +* *Canonical URL:* +`+https://hyperpolymath.dev/standards/cartridges/cartridge-v1.json+` + +When the standards PR merges, this file should be SHA-pinned to the +merged content. Pinning ceremony: + +[arabic] +. After standards#200 merges, capture the commit SHA in +`+hyperpolymath/standards+`. +. Capture the SHA-256 of `+cartridges/cartridge-v1.json+` at that +commit. +. Update link:PINNED-SHA[`+PINNED-SHA+`] in this directory with both +values (commit SHA + file SHA-256). + +=== Why mirror + +[arabic] +. *Offline validation.* The Elixir BoJ catalog +(`+elixir/lib/boj_rest/catalog.ex+`) reads cartridges from disk at boot; +validation against the canonical URL would require network calls that +are out of scope for the catalog. +. *Reproducibility.* A given boj-server build must validate cartridges +against a deterministic schema version, so the bundled mirror is the +source of truth at runtime. + +=== What if local and canonical disagree? + +Standards wins. Local mirror is always advancing toward standards. When +standards moves to v2, this mirror will get a `+cartridge-v2.json+` and +the consumer code will accept both v1 and v2 manifests for a deprecation +period. + +See also: - standards +https://github.com/hyperpolymath/standards/blob/main/docs/decisions/ADR-002-cartridge-format-canonical-home.adoc[ADR-002 +— cartridge format canonical home] - boj-server-cartridges +https://github.com/hyperpolymath/boj-server-cartridges/blob/main/schemas/SCHEMA-MIRROR.md[`+schemas/SCHEMA-MIRROR.md+`] +(same content, different repo — keep both in sync) diff --git a/schemas/SCHEMA-MIRROR.md b/schemas/SCHEMA-MIRROR.md deleted file mode 100644 index a8d5df6f..00000000 --- a/schemas/SCHEMA-MIRROR.md +++ /dev/null @@ -1,31 +0,0 @@ - - - -# Cartridge schema mirror - -The file `cartridge-v1.json` in this directory is a **vendored mirror** of the canonical schema living at: - -- **Canonical home:** [`hyperpolymath/standards`](https://github.com/hyperpolymath/standards/blob/main/cartridges/cartridge-v1.json) (filed as PR [#200](https://github.com/hyperpolymath/standards/pull/200) on 2026-05-26). -- **Canonical URL:** `https://hyperpolymath.dev/standards/cartridges/cartridge-v1.json` - -When the standards PR merges, this file should be SHA-pinned to the merged content. Pinning ceremony: - -1. After standards#200 merges, capture the commit SHA in `hyperpolymath/standards`. -2. Capture the SHA-256 of `cartridges/cartridge-v1.json` at that commit. -3. Update [`PINNED-SHA`](PINNED-SHA) in this directory with both values (commit SHA + file SHA-256). - -## Why mirror - -1. **Offline validation.** The Elixir BoJ catalog (`elixir/lib/boj_rest/catalog.ex`) reads cartridges from disk at boot; validation against the canonical URL would require network calls that are out of scope for the catalog. -2. **Reproducibility.** A given boj-server build must validate cartridges against a deterministic schema version, so the bundled mirror is the source of truth at runtime. - -## What if local and canonical disagree? - -Standards wins. Local mirror is always advancing toward standards. When standards moves to v2, this mirror will get a `cartridge-v2.json` and the consumer code will accept both v1 and v2 manifests for a deprecation period. - -See also: -- standards [ADR-002 — cartridge format canonical home](https://github.com/hyperpolymath/standards/blob/main/docs/decisions/ADR-002-cartridge-format-canonical-home.adoc) -- boj-server-cartridges [`schemas/SCHEMA-MIRROR.md`](https://github.com/hyperpolymath/boj-server-cartridges/blob/main/schemas/SCHEMA-MIRROR.md) (same content, different repo — keep both in sync) diff --git a/setup-scripts/onboard-son.adoc b/setup-scripts/onboard-son.adoc new file mode 100644 index 00000000..cc6d6774 --- /dev/null +++ b/setup-scripts/onboard-son.adoc @@ -0,0 +1,268 @@ +== BoJ + PanLL Full Onboarding Script + +== SPDX-License-Identifier: CC-BY-SA-4.0 + +== + +== INSTRUCTIONS: Copy everything below the line into Claude Code as a single message. + +== Claude will walk you through each step interactively. + +== ───────────────────────────────────────────────────────────────────── + +I want you to set up my full development environment. Follow these +phases IN ORDER. Do not skip ahead. Ask me questions where indicated. Do +each phase completely before moving to the next. + +''''' + +=== PHASE 1 — BoJ Server (Bundle of Joy) + +BoJ is a cartridge-based MCP server that routes ALL tool integrations +through a single unified gateway. Every MCP, LSP, debugger, and AI agent +connects through BoJ cartridges instead of running as separate servers. + +[arabic] +. Clone the BoJ server: ++ +.... +git clone https://github.com/hyperpolymath/boj-server ~/Documents/hyperpolymath-repos/boj-server +cd ~/Documents/hyperpolymath-repos/boj-server +.... +. Check and install prerequisites (ask me before installing anything): +* *Deno* (runtime) — check `+deno --version+` +* *Zig* (FFI compilation) — check `+zig version+` +* *Rust* (if building from source) — check `+rustc --version+` +* *just* (task runner) — check `+just --version+` +. Build the server: ++ +.... +just deps +just build +.... +. Start the MCP bridge and verify health: ++ +.... +deno run --allow-net --allow-env mcp-bridge/main.js +.... ++ +Then in another terminal: `+curl http://localhost:7700/health+` +. Register BoJ as my Claude Code MCP server. Create/update +`+~/.config/claude/mcp_servers.json+`: ++ +[source,json] +---- +{ + "boj-server": { + "command": "npx", + "args": ["-y", "@hyperpolymath/boj-server@latest"], + "env": { "BOJ_URL": "http://localhost:7700" } + } +} +---- +. Verify BoJ is working by calling `+boj_health+` and `+boj_menu+` +tools. + +''''' + +=== PHASE 2 — Cartridge Inventory + +BoJ ships with 22 built-in cartridges. List them with `+boj_menu+` and +show me what’s available in a table: + +[cols=",,",options="header",] +|=== +|Cartridge |Protocols |Purpose +|=== + +The 22 built-in cartridges cover: database, container, git, cloud +(Verpex/Cloudflare/Vercel), comms (Gmail/Calendar), ML (HuggingFace), +research, LSP, DAP (debug), BSP (build), NeSy (neurosymbolic), agentic, +fleet (bot orchestration), proof (Idris2), secrets, observability, IaC, +queues, model-router, SSG, UMS, lang, feedback. + +For any service I’m currently using that does NOT have a cartridge, use +the *cartridge minter* to create one: + +*Cartridge Minting Wizard* (for each new cartridge): 1. Ask: "`What +service/tool do you want to connect?`" 2. Ask: "`What protocols does it +need?`" (MCP, LSP, DAP, BSP, NeSy, Agentic, gRPC, REST) 3. Ask: "`What +domain?`" (Database, Container, K8s, Git, Cloud, Comms, ML, Research, +etc.) 4. Generate the three-layer cartridge scaffold: +`+cartridges/-mcp/ ├── abi/Mcp/.idr # Idris2 ABI definition ├── ffi/build.zig # Zig FFI implementation ├── ffi/_ffi.zig # Zig FFI source └── adapter/_adapter.v # zig REST/gRPC/GraphQL adapter+` +5. Configure and provision the cartridge via `+boj_cartridge_invoke+` + +''''' + +=== PHASE 3 — Firefox Setup + +Set up Firefox as the browser bridge for Claude Code: + +[arabic] +. Check if Firefox is installed (`+which firefox+`) +. If not installed, ask me which package manager to use +. Create a health-check hook at `+~/.claude/hooks/boj-health-check.sh+` +that: +* Checks if BoJ server is running on startup, restarts if not +* Checks if Firefox is running — if YES, do nothing; if NO, launch it +*minimised* +* Logs diagnostics to `+~/.claude/hooks/boj-diagnostics.log+` +. Make the hook executable: +`+chmod +x ~/.claude/hooks/boj-health-check.sh+` + +''''' + +=== PHASE 4 — Code Editor + +*ASK ME:* "`Which code editor do you want to use? (e.g., VS Code, +Neovim, Helix, Zed, Lapce, or something else)`" + +Based on my answer: - Configure the LSP cartridge (`+lsp-mcp+`) to +connect to my editor’s LSP - Configure the DAP cartridge (`+dap-mcp+`) +for debugging in my editor - Set up any editor-specific integrations +(extensions, plugins, config files) - If my editor supports it, +configure the BSP cartridge for build server integration + +''''' + +=== PHASE 5 — PanLL (Optional) + +*ASK ME:* "`Would you like to install PanLL?`" + +*Explain PanLL like this:* > PanLL is an *eNSAID* — an Environment for +NeSy-Agentic Integrated Development. > > Think of it as a mission +control dashboard for coding. Instead of just a code editor, PanLL gives +you three co-working panels: > - *Panel-L* (Symbolic) — formal logic, +type checking, proof verification > - *Panel-N* (Neural) — AI reasoning, +suggestions, and an advisor called ECHIDNA > - *Panel-W* (World) — +results, dashboards, databases, security tools > > You and the AI work +together as equals — neither is "`the assistant.`" PanLL monitors +cognitive load (a "`Vexometer`" tracks frustration), adjusts information +density, and manages 79+ specialist overlay panels you can summon for +different tasks. > > It runs as a lightweight Tauri app (5 MB, not an +Electron bloat-fest) built in AffineScript + Rust. + +*If YES:* + +[arabic] +. Clone PanLL: ++ +.... +git clone https://github.com/hyperpolymath/panll ~/Documents/hyperpolymath-repos/panll +cd ~/Documents/hyperpolymath-repos/panll +.... +. Install PanLL prerequisites: +* AffineScript compiler: `+npm install affinescript+` (exception to npm +ban — AffineScript requires it) +* Deno (already installed from Phase 1) +* Rust + Tauri 2.0: check `+cargo tauri --version+` +* Tailwind CSS: handled by deno tasks +. Build PanLL: ++ +.... +just build +.... ++ +Or for development: ++ +.... +just dev +.... +. Use the *Panel Minter* to verify it works: +* Open PanLL → click Minter in the panel bar +* Or via CLI: the minter generates an 8-file scaffold per panel + +*If NO:* Skip to Phase 7. + +''''' + +=== PHASE 6 — PanLL Panels for IDApTIK & Game Dev + +*ASK ME:* "`Would you like to set up the existing game development and +IDApTIK panels?`" + +*If YES*, provision these panel sets using the Panel Provisioner: + +==== IDApTIK eNSAID Panels (11): + +[cols=",",options="header",] +|=== +|Panel |Purpose +|Valence Shell |Embedded terminal with session recording +|Game Preview |Live IDApTIK preview with hot-reload +|VM Inspector |Reversible debugger (step forward/backward) +|Network Topology |Force-directed in-game network graph +|Level Architect |Visual level design with validation +|Coprocessors |Backend coprocessor heatmap (10 types) +|Multiplayer Monitor |Real-time multiplayer session tracking +|DLC Workshop |Puzzle pack creation + testing +|Editor Bridge |Connects PanLL to your code editor +|Build Dashboard |Build status across all targets +|Release Manager |Release pipeline and versioning +|=== + +==== Game Dev Testing Panels (10): + +Unit Test Runner, Functional Tester, Regression Guard, Performance +Profiler, Load Tester, Soak Monitor, Compatibility Matrix, Exploratory +Workbench, Beta Feedback Hub, Balance Analyser + +==== Game Dev Bridge Panels (8): + +Typing Bridge, Neurosym Bridge, Agentic Bridge, Automation Bridge, +Database Bridge, Protocol Bridge, Proofs Bridge, Scripting Bridge + +==== Game Dev Specific Panels (6): + +Generator Mode, Architect Mode, Guard AI Tuner, Device Network Designer, +Asset Manager, Playtest Recorder + +For each panel set, use the *Panel Provisioner* to select isolation +tier: - *Native* (in-process, fastest) — recommended for trusted panels +- *Standard Pod* (Alpine + Podman) — community panels - *Hardened Pod* +(Stapeln + Chainguard) — untrusted panels + +Then use the *Panel Configurator* (Workspace panel) to arrange them into +workspace modes. + +''''' + +=== PHASE 7 — Custom Cartridges & Panels + +*ASK ME:* "`Would you like to create any custom BoJ cartridges or PanLL +panels for your development work?`" + +*If YES for cartridges*, run the *Cartridge Minting Wizard* for each +one: 1. "`What’s the name of this cartridge?`" 2. "`What does it do? +(one sentence)`" 3. "`What protocols?`" — show checklist: ☐ MCP ☐ LSP ☐ +DAP ☐ BSP ☐ NeSy ☐ Agentic ☐ gRPC ☐ REST 4. "`What domain?`" — show +checklist: ☐ Database ☐ Container ☐ Git ☐ Cloud ☐ Comms ☐ ML ☐ Research +☐ Other 5. "`Any external APIs or services it connects to?`" 6. Generate +the scaffold, then ask: "`Want me to implement the adapter logic now, or +leave it as a scaffold?`" + +*If YES for panels* (and PanLL is installed), run the *Panel Minting +Wizard* for each one: 1. "`What’s the name of this panel?`" 2. "`What +clade kind?`" — explain each: directive (controls), scanner (monitors), +builder (creates), viewer (displays), ai (neural), bridge (connects +external tools) 3. "`What does it do? (one sentence)`" 4. "`Should it +connect to any BoJ cartridges?`" 5. "`What isolation tier?`" — Native / +Standard Pod / Hardened Pod 6. Generate the 8-file scaffold using the +Minter 7. Ask: "`Want me to implement the panel logic now, or leave it +as a scaffold?`" + +*Keep asking* "`Any more?`" until I say I’m done. + +''''' + +=== PHASE 8 — Final Verification + +Run a full system check: 1. `+curl http://localhost:7700/health+` — BoJ +healthy 2. `+boj_menu+` — all cartridges listed 3. Firefox running +(check `+pgrep firefox+`) 4. Editor integration working 5. If PanLL +installed: `+just test+` in panll directory — all tests pass 6. Show me +a summary table of everything installed + +*Done!* Tell me: "`Your environment is ready. You have [N] BoJ +cartridges and [M] PanLL panels configured. Type `+boj_menu+` any time +to see your cartridges, or open PanLL to access your panels.`" diff --git a/setup-scripts/onboard-son.md b/setup-scripts/onboard-son.md deleted file mode 100644 index de0524b7..00000000 --- a/setup-scripts/onboard-son.md +++ /dev/null @@ -1,233 +0,0 @@ - -# BoJ + PanLL Full Onboarding Script -# SPDX-License-Identifier: CC-BY-SA-4.0 -# -# INSTRUCTIONS: Copy everything below the line into Claude Code as a single message. -# Claude will walk you through each step interactively. -# ───────────────────────────────────────────────────────────────────── - -I want you to set up my full development environment. Follow these phases IN ORDER. Do not skip ahead. Ask me questions where indicated. Do each phase completely before moving to the next. - ---- - -## PHASE 1 — BoJ Server (Bundle of Joy) - -BoJ is a cartridge-based MCP server that routes ALL tool integrations through a single unified gateway. Every MCP, LSP, debugger, and AI agent connects through BoJ cartridges instead of running as separate servers. - -1. Clone the BoJ server: - ``` - git clone https://github.com/hyperpolymath/boj-server ~/Documents/hyperpolymath-repos/boj-server - cd ~/Documents/hyperpolymath-repos/boj-server - ``` - -2. Check and install prerequisites (ask me before installing anything): - - **Deno** (runtime) — check `deno --version` - - **Zig** (FFI compilation) — check `zig version` - - **Rust** (if building from source) — check `rustc --version` - - **just** (task runner) — check `just --version` - -3. Build the server: - ``` - just deps - just build - ``` - -4. Start the MCP bridge and verify health: - ``` - deno run --allow-net --allow-env mcp-bridge/main.js - ``` - Then in another terminal: `curl http://localhost:7700/health` - -5. Register BoJ as my Claude Code MCP server. Create/update `~/.config/claude/mcp_servers.json`: - ```json - { - "boj-server": { - "command": "npx", - "args": ["-y", "@hyperpolymath/boj-server@latest"], - "env": { "BOJ_URL": "http://localhost:7700" } - } - } - ``` - -6. Verify BoJ is working by calling `boj_health` and `boj_menu` tools. - ---- - -## PHASE 2 — Cartridge Inventory - -BoJ ships with 22 built-in cartridges. List them with `boj_menu` and show me what's available in a table: - -| Cartridge | Protocols | Purpose | -|-----------|-----------|---------| - -The 22 built-in cartridges cover: database, container, git, cloud (Verpex/Cloudflare/Vercel), comms (Gmail/Calendar), ML (HuggingFace), research, LSP, DAP (debug), BSP (build), NeSy (neurosymbolic), agentic, fleet (bot orchestration), proof (Idris2), secrets, observability, IaC, queues, model-router, SSG, UMS, lang, feedback. - -For any service I'm currently using that does NOT have a cartridge, use the **cartridge minter** to create one: - -**Cartridge Minting Wizard** (for each new cartridge): -1. Ask: "What service/tool do you want to connect?" -2. Ask: "What protocols does it need?" (MCP, LSP, DAP, BSP, NeSy, Agentic, gRPC, REST) -3. Ask: "What domain?" (Database, Container, K8s, Git, Cloud, Comms, ML, Research, etc.) -4. Generate the three-layer cartridge scaffold: - ``` - cartridges/-mcp/ - ├── abi/Mcp/.idr # Idris2 ABI definition - ├── ffi/build.zig # Zig FFI implementation - ├── ffi/_ffi.zig # Zig FFI source - └── adapter/_adapter.v # zig REST/gRPC/GraphQL adapter - ``` -5. Configure and provision the cartridge via `boj_cartridge_invoke` - ---- - -## PHASE 3 — Firefox Setup - -Set up Firefox as the browser bridge for Claude Code: - -1. Check if Firefox is installed (`which firefox`) -2. If not installed, ask me which package manager to use -3. Create a health-check hook at `~/.claude/hooks/boj-health-check.sh` that: - - Checks if BoJ server is running on startup, restarts if not - - Checks if Firefox is running — if YES, do nothing; if NO, launch it **minimised** - - Logs diagnostics to `~/.claude/hooks/boj-diagnostics.log` -4. Make the hook executable: `chmod +x ~/.claude/hooks/boj-health-check.sh` - ---- - -## PHASE 4 — Code Editor - -**ASK ME:** "Which code editor do you want to use? (e.g., VS Code, Neovim, Helix, Zed, Lapce, or something else)" - -Based on my answer: -- Configure the LSP cartridge (`lsp-mcp`) to connect to my editor's LSP -- Configure the DAP cartridge (`dap-mcp`) for debugging in my editor -- Set up any editor-specific integrations (extensions, plugins, config files) -- If my editor supports it, configure the BSP cartridge for build server integration - ---- - -## PHASE 5 — PanLL (Optional) - -**ASK ME:** "Would you like to install PanLL?" - -**Explain PanLL like this:** -> PanLL is an **eNSAID** — an Environment for NeSy-Agentic Integrated Development. -> -> Think of it as a mission control dashboard for coding. Instead of just a code editor, PanLL gives you three co-working panels: -> - **Panel-L** (Symbolic) — formal logic, type checking, proof verification -> - **Panel-N** (Neural) — AI reasoning, suggestions, and an advisor called ECHIDNA -> - **Panel-W** (World) — results, dashboards, databases, security tools -> -> You and the AI work together as equals — neither is "the assistant." PanLL monitors cognitive load (a "Vexometer" tracks frustration), adjusts information density, and manages 79+ specialist overlay panels you can summon for different tasks. -> -> It runs as a lightweight Tauri app (5 MB, not an Electron bloat-fest) built in AffineScript + Rust. - -**If YES:** - -1. Clone PanLL: - ``` - git clone https://github.com/hyperpolymath/panll ~/Documents/hyperpolymath-repos/panll - cd ~/Documents/hyperpolymath-repos/panll - ``` - -2. Install PanLL prerequisites: - - AffineScript compiler: `npm install affinescript` (exception to npm ban — AffineScript requires it) - - Deno (already installed from Phase 1) - - Rust + Tauri 2.0: check `cargo tauri --version` - - Tailwind CSS: handled by deno tasks - -3. Build PanLL: - ``` - just build - ``` - Or for development: - ``` - just dev - ``` - -4. Use the **Panel Minter** to verify it works: - - Open PanLL → click Minter in the panel bar - - Or via CLI: the minter generates an 8-file scaffold per panel - -**If NO:** Skip to Phase 7. - ---- - -## PHASE 6 — PanLL Panels for IDApTIK & Game Dev - -**ASK ME:** "Would you like to set up the existing game development and IDApTIK panels?" - -**If YES**, provision these panel sets using the Panel Provisioner: - -### IDApTIK eNSAID Panels (11): -| Panel | Purpose | -|-------|---------| -| Valence Shell | Embedded terminal with session recording | -| Game Preview | Live IDApTIK preview with hot-reload | -| VM Inspector | Reversible debugger (step forward/backward) | -| Network Topology | Force-directed in-game network graph | -| Level Architect | Visual level design with validation | -| Coprocessors | Backend coprocessor heatmap (10 types) | -| Multiplayer Monitor | Real-time multiplayer session tracking | -| DLC Workshop | Puzzle pack creation + testing | -| Editor Bridge | Connects PanLL to your code editor | -| Build Dashboard | Build status across all targets | -| Release Manager | Release pipeline and versioning | - -### Game Dev Testing Panels (10): -Unit Test Runner, Functional Tester, Regression Guard, Performance Profiler, Load Tester, Soak Monitor, Compatibility Matrix, Exploratory Workbench, Beta Feedback Hub, Balance Analyser - -### Game Dev Bridge Panels (8): -Typing Bridge, Neurosym Bridge, Agentic Bridge, Automation Bridge, Database Bridge, Protocol Bridge, Proofs Bridge, Scripting Bridge - -### Game Dev Specific Panels (6): -Generator Mode, Architect Mode, Guard AI Tuner, Device Network Designer, Asset Manager, Playtest Recorder - -For each panel set, use the **Panel Provisioner** to select isolation tier: -- **Native** (in-process, fastest) — recommended for trusted panels -- **Standard Pod** (Alpine + Podman) — community panels -- **Hardened Pod** (Stapeln + Chainguard) — untrusted panels - -Then use the **Panel Configurator** (Workspace panel) to arrange them into workspace modes. - ---- - -## PHASE 7 — Custom Cartridges & Panels - -**ASK ME:** "Would you like to create any custom BoJ cartridges or PanLL panels for your development work?" - -**If YES for cartridges**, run the **Cartridge Minting Wizard** for each one: -1. "What's the name of this cartridge?" -2. "What does it do? (one sentence)" -3. "What protocols?" — show checklist: ☐ MCP ☐ LSP ☐ DAP ☐ BSP ☐ NeSy ☐ Agentic ☐ gRPC ☐ REST -4. "What domain?" — show checklist: ☐ Database ☐ Container ☐ Git ☐ Cloud ☐ Comms ☐ ML ☐ Research ☐ Other -5. "Any external APIs or services it connects to?" -6. Generate the scaffold, then ask: "Want me to implement the adapter logic now, or leave it as a scaffold?" - -**If YES for panels** (and PanLL is installed), run the **Panel Minting Wizard** for each one: -1. "What's the name of this panel?" -2. "What clade kind?" — explain each: directive (controls), scanner (monitors), builder (creates), viewer (displays), ai (neural), bridge (connects external tools) -3. "What does it do? (one sentence)" -4. "Should it connect to any BoJ cartridges?" -5. "What isolation tier?" — Native / Standard Pod / Hardened Pod -6. Generate the 8-file scaffold using the Minter -7. Ask: "Want me to implement the panel logic now, or leave it as a scaffold?" - -**Keep asking** "Any more?" until I say I'm done. - ---- - -## PHASE 8 — Final Verification - -Run a full system check: -1. `curl http://localhost:7700/health` — BoJ healthy -2. `boj_menu` — all cartridges listed -3. Firefox running (check `pgrep firefox`) -4. Editor integration working -5. If PanLL installed: `just test` in panll directory — all tests pass -6. Show me a summary table of everything installed - -**Done!** Tell me: "Your environment is ready. You have [N] BoJ cartridges and [M] PanLL panels configured. Type `boj_menu` any time to see your cartridges, or open PanLL to access your panels." diff --git a/templates/cartridge-template/README.md b/templates/cartridge-template/README.adoc similarity index 73% rename from templates/cartridge-template/README.md rename to templates/cartridge-template/README.adoc index 2fd66580..c0f0ed7d 100644 --- a/templates/cartridge-template/README.md +++ b/templates/cartridge-template/README.adoc @@ -1,14 +1,11 @@ - -# Cartridge Template +== Cartridge Template -This template provides a starting point for creating new cartridges for the BoJ server. +This template provides a starting point for creating new cartridges for +the BoJ server. -## Structure +=== Structure -``` +.... cartridge-template/ ├── README.md # This file ├── cartridge.json # Cartridge metadata @@ -20,43 +17,44 @@ cartridge-template/ │ └── cartridge_ffi.zig # FFI implementation └── adapter/ └── README.adoc # Adapter layer documentation -``` - -## Getting Started - -1. **Copy the Template**: - ```bash - cp -r templates/cartridge-template cartridges/my-new-cartridge - ``` - -2. **Update Metadata**: - - Edit `cartridge.json` to reflect your cartridge's metadata. - - Update the `name`, `version`, `description`, and other fields. - -3. **Implement the ABI Layer**: - - Define the abstract interfaces and types in the `abi/` directory. - - Use Idris2 for type safety and correctness. - -4. **Implement the FFI Layer**: - - Implement the foreign function interface in the `ffi/` directory. - - Use Zig for high-performance bindings. - -5. **Implement the Adapter Layer**: - - Implement the actual functionality in the `adapter/` directory. - - Use Zig for the adapter layer. - -6. **Add Tests**: - - Add tests to the `ffi/` directory to ensure your cartridge works as expected. - - Use Zig's testing framework. - -7. **Update Documentation**: - - Update the `README.adoc` files in each directory to reflect your cartridge's purpose, boundaries, invariants, and execution surfaces. - -## Cartridge Metadata - -The `cartridge.json` file contains metadata about your cartridge. Here's an example: - -```json +.... + +=== Getting Started + +[arabic] +. *Copy the Template*: ++ +[source,bash] +---- +cp -r templates/cartridge-template cartridges/my-new-cartridge +---- +. *Update Metadata*: +* Edit `+cartridge.json+` to reflect your cartridge’s metadata. +* Update the `+name+`, `+version+`, `+description+`, and other fields. +. *Implement the ABI Layer*: +* Define the abstract interfaces and types in the `+abi/+` directory. +* Use Idris2 for type safety and correctness. +. *Implement the FFI Layer*: +* Implement the foreign function interface in the `+ffi/+` directory. +* Use Zig for high-performance bindings. +. *Implement the Adapter Layer*: +* Implement the actual functionality in the `+adapter/+` directory. +* Use Zig for the adapter layer. +. *Add Tests*: +* Add tests to the `+ffi/+` directory to ensure your cartridge works as +expected. +* Use Zig’s testing framework. +. *Update Documentation*: +* Update the `+README.adoc+` files in each directory to reflect your +cartridge’s purpose, boundaries, invariants, and execution surfaces. + +=== Cartridge Metadata + +The `+cartridge.json+` file contains metadata about your cartridge. +Here’s an example: + +[source,json] +---- { "name": "my-new-cartridge", "version": "0.1.0", @@ -83,13 +81,15 @@ The `cartridge.json` file contains metadata about your cartridge. Here's an exam } ] } -``` +---- -## FFI Implementation +=== FFI Implementation -The `ffi/cartridge_ffi.zig` file contains the FFI implementation. Here's a basic template: +The `+ffi/cartridge_ffi.zig+` file contains the FFI implementation. +Here’s a basic template: -```zig +[source,zig] +---- // SPDX-License-Identifier: CC-BY-SA-4.0 // Copyright (c) 2026 Your Name @@ -168,13 +168,15 @@ test "invoke with too-small buffer returns -3 and sets required length" { try std.testing.expectEqual(@as(i32, -3), rc); try std.testing.expect(len > 4); } -``` +---- -## Build Configuration +=== Build Configuration -The `ffi/build.zig` file contains the build configuration. Here's a basic template: +The `+ffi/build.zig+` file contains the build configuration. Here’s a +basic template: -```zig +[source,zig] +---- // SPDX-License-Identifier: CC-BY-SA-4.0 const std = @import("std"); @@ -203,13 +205,16 @@ pub fn build(b: *std.Build) void { const test_step = b.step("test", "Run FFI tests"); test_step.dependOn(&run_tests.step); } -``` +---- -## Documentation +=== Documentation -Update the `README.adoc` files in each directory to reflect your cartridge's purpose, boundaries, invariants, and execution surfaces. Here's an example for the `ffi/` directory: +Update the `+README.adoc+` files in each directory to reflect your +cartridge’s purpose, boundaries, invariants, and execution surfaces. +Here’s an example for the `+ffi/+` directory: -```adoc +[source,adoc] +---- = My New Cartridge FFI Layer == Purpose @@ -233,24 +238,27 @@ This layer provides the foreign function interface for the My New Cartridge cart - **Entry Points**: `boj_cartridge_init`, `boj_cartridge_deinit`, `boj_cartridge_name`, `boj_cartridge_version`, `boj_cartridge_invoke`. - **Error Handling**: Returns appropriate error codes for invalid inputs or failed operations. - **Testing**: Unit tests for each function. -``` +---- -## Testing +=== Testing Run the tests using the following command: -```bash +[source,bash] +---- cd ffi && zig build test -``` +---- -## Building +=== Building Build the shared library using the following command: -```bash +[source,bash] +---- cd ffi && zig build -``` +---- -## License +=== License -This cartridge is licensed under the MPL-2.0 license. See the LICENSE file for more information. +This cartridge is licensed under the MPL-2.0 license. See the LICENSE +file for more information. diff --git a/tests/backend-assurance/README.adoc b/tests/backend-assurance/README.adoc new file mode 100644 index 00000000..de120abb --- /dev/null +++ b/tests/backend-assurance/README.adoc @@ -0,0 +1,33 @@ +== `+tests/backend-assurance/+` — pointer + +The runnable property-test harness for the backend-assurance campaign +lives under *`+elixir/test/backend_assurance/+`* (BEAM-native), where +`+mix test+` picks it up automatically and `+stream_data+` is already a +declared test-only dep. + +This directory is intentionally a pointer rather than a parallel test +home: keeping the tests under `+elixir/test/+` means one test runner, +one dep set, and one set of fixtures. + +The prose-side trusted-extraction validation lives under +*`+docs/backend-assurance/+`*. + +=== Run + +.... +cd elixir +mix deps.get +mix test --only backend_assurance +.... + +=== CI + +`+.github/workflows/backend-assurance.yml+` runs the same command on PRs +that touch `+src/abi/Boj/SafetyLemmas.idr+` or +`+elixir/test/backend_assurance/**+`. + +=== See also + +* `+docs/backend-assurance/README.md+` — campaign overview + coverage +table. +* `+PROOF-NEEDS.md+` — axiom audit + why this campaign exists. diff --git a/tests/backend-assurance/README.md b/tests/backend-assurance/README.md deleted file mode 100644 index 15c5fd9b..00000000 --- a/tests/backend-assurance/README.md +++ /dev/null @@ -1,36 +0,0 @@ - - - -# `tests/backend-assurance/` — pointer - -The runnable property-test harness for the backend-assurance campaign -lives under **`elixir/test/backend_assurance/`** (BEAM-native), where -`mix test` picks it up automatically and `stream_data` is already a -declared test-only dep. - -This directory is intentionally a pointer rather than a parallel test -home: keeping the tests under `elixir/test/` means one test runner, -one dep set, and one set of fixtures. - -The prose-side trusted-extraction validation lives under -**`docs/backend-assurance/`**. - -## Run - - cd elixir - mix deps.get - mix test --only backend_assurance - -## CI - -`.github/workflows/backend-assurance.yml` runs the same command on PRs -that touch `src/abi/Boj/SafetyLemmas.idr` or -`elixir/test/backend_assurance/**`. - -## See also - -- `docs/backend-assurance/README.md` — campaign overview + coverage table. -- `PROOF-NEEDS.md` — axiom audit + why this campaign exists. diff --git a/tools/cartridge-configurator/README.adoc b/tools/cartridge-configurator/README.adoc new file mode 100644 index 00000000..79db41c2 --- /dev/null +++ b/tools/cartridge-configurator/README.adoc @@ -0,0 +1,67 @@ +== Cartridge Configurator + +The Cartridge Configurator is a tool for applying runtime configuration +to cartridges dynamically. + +=== Features + +* Apply runtime configuration to cartridges +* Validate configuration against a schema +* Support multi-environment configurations +* Trigger hot-reload if supported + +=== Installation + +[source,bash] +---- +cd tools/cartridge-configurator +npm install +---- + +=== Usage + +==== Command Line + +[source,bash] +---- +node configurator.js [--no-validate] +---- + +==== Options + +* `++`: Path to the cartridge directory containing +`+cartridge.json+` +* `++`: Path to the configuration file for the cartridge +* `+--no-validate+`: Skip configuration validation + +==== Example + +[source,bash] +---- +node configurator.js ../../cartridges/my-cartridge config.json +---- + +==== Programmatic Usage + +[source,javascript] +---- +import { applyConfig } from './configurator.js'; + +const result = await applyConfig('path/to/cartridge', { + api_endpoint: 'https://api.example.com', + rate_limit: 1000, +}, { validate: true }); + +console.log('Configuration applied:', result); +---- + +=== Output + +The Cartridge Configurator applies the configuration to the cartridge +and writes it to a `+config.json+` file in the cartridge directory. If +hot-reload is supported, it triggers a hot-reload of the cartridge. + +=== License + +This tool is licensed under the MPL-2.0 license. See the LICENSE file +for more information. diff --git a/tools/cartridge-configurator/README.md b/tools/cartridge-configurator/README.md deleted file mode 100644 index 62c5491e..00000000 --- a/tools/cartridge-configurator/README.md +++ /dev/null @@ -1,62 +0,0 @@ - -# Cartridge Configurator - -The Cartridge Configurator is a tool for applying runtime configuration to cartridges dynamically. - -## Features - -- Apply runtime configuration to cartridges -- Validate configuration against a schema -- Support multi-environment configurations -- Trigger hot-reload if supported - -## Installation - -```bash -cd tools/cartridge-configurator -npm install -``` - -## Usage - -### Command Line - -```bash -node configurator.js [--no-validate] -``` - -### Options - -- ``: Path to the cartridge directory containing `cartridge.json` -- ``: Path to the configuration file for the cartridge -- `--no-validate`: Skip configuration validation - -### Example - -```bash -node configurator.js ../../cartridges/my-cartridge config.json -``` - -### Programmatic Usage - -```javascript -import { applyConfig } from './configurator.js'; - -const result = await applyConfig('path/to/cartridge', { - api_endpoint: 'https://api.example.com', - rate_limit: 1000, -}, { validate: true }); - -console.log('Configuration applied:', result); -``` - -## Output - -The Cartridge Configurator applies the configuration to the cartridge and writes it to a `config.json` file in the cartridge directory. If hot-reload is supported, it triggers a hot-reload of the cartridge. - -## License - -This tool is licensed under the MPL-2.0 license. See the LICENSE file for more information. diff --git a/tools/cartridge-provisioner/README.adoc b/tools/cartridge-provisioner/README.adoc new file mode 100644 index 00000000..d38e5a07 --- /dev/null +++ b/tools/cartridge-provisioner/README.adoc @@ -0,0 +1,75 @@ +== Cartridge Provisioner + +The Cartridge Provisioner is a tool for deploying cartridges to BoJ +Server, BoJ Server + Elixir Multiplier, or panll. + +=== Features + +* Deploy cartridges to BoJ Server, BoJ Server + Elixir Multiplier, or +panll +* Scale cartridges based on load +* Apply runtime configuration to cartridges +* Support health checks and rollback + +=== Installation + +[source,bash] +---- +cd tools/cartridge-provisioner +npm install +---- + +=== Usage + +==== Command Line + +[source,bash] +---- +node provisioner.js [--scale ] [--config ] +---- + +==== Options + +* `++`: Path to the cartridge directory containing +`+cartridge.json+` +* `++`: Target deployment environment (`+boj-server+`, +`+boj-server-elixir+`, or `+panll+`) +* `+--scale +`: Number of instances to scale the cartridge to +(default: `+1+`) +* `+--config +`: Path to the configuration file for the +cartridge + +==== Example + +[source,bash] +---- +node provisioner.js ../../cartridges/my-cartridge boj-server --scale 3 --config config.json +---- + +==== Programmatic Usage + +[source,javascript] +---- +import { deployCartridge } from './provisioner.js'; + +const result = await deployCartridge('path/to/cartridge', 'boj-server', { + scale: 3, + config: { + api_key: 'your-api-key', + rate_limit: 1000, + }, +}); + +console.log('Cartridge deployed:', result); +---- + +=== Output + +The Cartridge Provisioner deploys the cartridge to the specified target +and applies the configuration. The cartridge files are copied to the +target directory, and the configuration is applied. + +=== License + +This tool is licensed under the MPL-2.0 license. See the LICENSE file +for more information. diff --git a/tools/cartridge-provisioner/README.md b/tools/cartridge-provisioner/README.md deleted file mode 100644 index 8b842a26..00000000 --- a/tools/cartridge-provisioner/README.md +++ /dev/null @@ -1,66 +0,0 @@ - -# Cartridge Provisioner - -The Cartridge Provisioner is a tool for deploying cartridges to BoJ Server, BoJ Server + Elixir Multiplier, or panll. - -## Features - -- Deploy cartridges to BoJ Server, BoJ Server + Elixir Multiplier, or panll -- Scale cartridges based on load -- Apply runtime configuration to cartridges -- Support health checks and rollback - -## Installation - -```bash -cd tools/cartridge-provisioner -npm install -``` - -## Usage - -### Command Line - -```bash -node provisioner.js [--scale ] [--config ] -``` - -### Options - -- ``: Path to the cartridge directory containing `cartridge.json` -- ``: Target deployment environment (`boj-server`, `boj-server-elixir`, or `panll`) -- `--scale `: Number of instances to scale the cartridge to (default: `1`) -- `--config `: Path to the configuration file for the cartridge - -### Example - -```bash -node provisioner.js ../../cartridges/my-cartridge boj-server --scale 3 --config config.json -``` - -### Programmatic Usage - -```javascript -import { deployCartridge } from './provisioner.js'; - -const result = await deployCartridge('path/to/cartridge', 'boj-server', { - scale: 3, - config: { - api_key: 'your-api-key', - rate_limit: 1000, - }, -}); - -console.log('Cartridge deployed:', result); -``` - -## Output - -The Cartridge Provisioner deploys the cartridge to the specified target and applies the configuration. The cartridge files are copied to the target directory, and the configuration is applied. - -## License - -This tool is licensed under the MPL-2.0 license. See the LICENSE file for more information. diff --git a/tools/panel-harness/README.adoc b/tools/panel-harness/README.adoc new file mode 100644 index 00000000..4c500b8a --- /dev/null +++ b/tools/panel-harness/README.adoc @@ -0,0 +1,70 @@ +== Panel Harness + +The Panel Harness is a tool for bridging cartridges to BoJ Server and +panll. + +=== Features + +* Register cartridges with BoJ Server or panll +* Support protocol translation (REST ↔ gRPC ↔ GraphQL) +* Route events between cartridges and BoJ Server/panll +* Expose cartridges as A2ML-compliant services or panll modules + +=== Installation + +[source,bash] +---- +cd tools/panel-harness +npm install +---- + +=== Usage + +==== Command Line + +[source,bash] +---- +node harness.js [--routes ] +---- + +==== Options + +* `++`: Path to the cartridge directory containing +`+cartridge.json+` +* `++`: Target environment (`+boj-server+` or `+panll+`) +* `+--routes +`: Path to the routes configuration file for +the cartridge + +==== Example + +[source,bash] +---- +node harness.js ../../cartridges/my-cartridge boj-server --routes routes.json +---- + +==== Programmatic Usage + +[source,javascript] +---- +import { registerCartridge } from './harness.js'; + +const result = await registerCartridge('path/to/cartridge', 'boj-server', { + routes: [ + { protocol: 'REST', port: 4000 }, + { protocol: 'gRPC', port: 50051 }, + ], +}); + +console.log('Cartridge registered:', result); +---- + +=== Output + +The Panel Harness registers the cartridge with the specified target and +writes the registration to a `+registration.json+` file in the harness +directory. + +=== License + +This tool is licensed under the MPL-2.0 license. See the LICENSE file +for more information. diff --git a/tools/panel-harness/README.md b/tools/panel-harness/README.md deleted file mode 100644 index 346dbd34..00000000 --- a/tools/panel-harness/README.md +++ /dev/null @@ -1,64 +0,0 @@ - -# Panel Harness - -The Panel Harness is a tool for bridging cartridges to BoJ Server and panll. - -## Features - -- Register cartridges with BoJ Server or panll -- Support protocol translation (REST ↔ gRPC ↔ GraphQL) -- Route events between cartridges and BoJ Server/panll -- Expose cartridges as A2ML-compliant services or panll modules - -## Installation - -```bash -cd tools/panel-harness -npm install -``` - -## Usage - -### Command Line - -```bash -node harness.js [--routes ] -``` - -### Options - -- ``: Path to the cartridge directory containing `cartridge.json` -- ``: Target environment (`boj-server` or `panll`) -- `--routes `: Path to the routes configuration file for the cartridge - -### Example - -```bash -node harness.js ../../cartridges/my-cartridge boj-server --routes routes.json -``` - -### Programmatic Usage - -```javascript -import { registerCartridge } from './harness.js'; - -const result = await registerCartridge('path/to/cartridge', 'boj-server', { - routes: [ - { protocol: 'REST', port: 4000 }, - { protocol: 'gRPC', port: 50051 }, - ], -}); - -console.log('Cartridge registered:', result); -``` - -## Output - -The Panel Harness registers the cartridge with the specified target and writes the registration to a `registration.json` file in the harness directory. - -## License - -This tool is licensed under the MPL-2.0 license. See the LICENSE file for more information.