From 1f432c32893706ae26793c380ba716b3722cc735 Mon Sep 17 00:00:00 2001 From: Codex Date: Mon, 28 Sep 2026 13:40:41 +0800 Subject: [PATCH 1/4] feat(runtime): ship open reference CLI and installable SDK --- .github/workflows/ci.yml | 10 +- .github/workflows/release.yml | 25 +- .gitignore | 3 + ARCHITECTURE.md | 6 + CHANGELOG.md | 10 + CLAUDE.md | 12 +- CMakeLists.txt | 61 +- CONTRIBUTING.md | 6 + README.md | 761 ++---------------- VERSION.json | 4 +- backends/http_model.cpp | 55 ++ cli/CMakeLists.txt | 17 +- cli/include/sparx_pipeline_harness.h | 7 +- cli/include/sparx_speculative.h | 10 +- cli/src/reference_main.cpp | 94 +++ cli/src/sparx_cloud_backend.cpp | 37 +- cli/src/sparx_pipeline_harness.cpp | 177 ++-- cli/src/sparx_speculative.cpp | 11 +- cmake/MasterAgentConfig.cmake.in | 3 + cmake/ReferenceTests.cmake | 27 + cmake/TestConsumer.cmake | 15 + config/harness.yaml | 2 +- core/reference_runtime.cpp | 177 ++++ docs/reference_runtime.md | 85 ++ eval/run_all.sh | 18 +- examples/reference_agent/CMakeLists.txt | 7 + examples/reference_agent/main.cpp | 13 + include/master_agent/runtime/http_model.h | 14 + .../master_agent/runtime/reference_runtime.h | 76 ++ scripts/package.sh | 26 + tests/check.h | 9 + tests/check_failure.cpp | 2 + tests/consumer/CMakeLists.txt | 12 + tests/consumer/main.cpp | 19 + tests/http_model_probe.cpp | 11 + tests/test_embedding.cpp | 18 +- tests/test_eval_runner.py | 38 + tests/test_harness.cpp | 54 ++ tests/test_http_model.py | 71 ++ tests/test_integration_speculation.cpp | 21 +- tests/test_merkle.cpp | 22 +- tests/test_orset.cpp | 28 +- tests/test_reference_runtime.cpp | 58 ++ 43 files changed, 1241 insertions(+), 891 deletions(-) create mode 100644 backends/http_model.cpp create mode 100644 cli/src/reference_main.cpp create mode 100644 cmake/ReferenceTests.cmake create mode 100644 cmake/TestConsumer.cmake create mode 100644 core/reference_runtime.cpp create mode 100644 docs/reference_runtime.md create mode 100644 examples/reference_agent/CMakeLists.txt create mode 100644 examples/reference_agent/main.cpp create mode 100644 include/master_agent/runtime/http_model.h create mode 100644 include/master_agent/runtime/reference_runtime.h create mode 100755 scripts/package.sh create mode 100644 tests/check.h create mode 100644 tests/check_failure.cpp create mode 100644 tests/consumer/CMakeLists.txt create mode 100644 tests/consumer/main.cpp create mode 100644 tests/http_model_probe.cpp create mode 100644 tests/test_eval_runner.py create mode 100644 tests/test_http_model.py create mode 100644 tests/test_reference_runtime.cpp diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e128289..3f33073 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -2,7 +2,7 @@ name: CI on: push: - branches: [main, develop, 'feat/**', 'fix/**'] + branches: [main, develop, 'feat/**', 'fix/**', 'codex/**'] pull_request: branches: [main, develop] @@ -48,6 +48,12 @@ jobs: working-directory: build run: ctest --output-on-failure --timeout 120 + - name: Run all synthetic evaluations + run: bash eval/run_all.sh + + - name: Verify SDK and CLI archive + run: bash scripts/package.sh build dist/sparx-sdk.tar.gz + - name: Upload build artifacts if: matrix.os == 'ubuntu-latest' uses: actions/upload-artifact@v4 @@ -55,7 +61,7 @@ jobs: name: sparx-linux-x64 path: build/cli/sparx retention-days: 7 - if-no-files-found: ignore + if-no-files-found: error lint: runs-on: ubuntu-latest diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index eb92cbd..da31c62 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -47,26 +47,21 @@ jobs: cmake -B build \ -DCMAKE_BUILD_TYPE=${{ env.BUILD_TYPE }} \ -DSPARX_VERSION=${{ steps.version.outputs.version }} \ - -DMASTER_AGENT_BUILD_TESTS=OFF \ + -DMASTER_AGENT_BUILD_TESTS=ON \ -DMASTER_AGENT_BUILD_CLI=ON \ ${{ matrix.cmake_flags || '' }} - name: Build run: cmake --build build --config ${{ env.BUILD_TYPE }} -j$(nproc 2>/dev/null || sysctl -n hw.logicalcpu) - - name: Package - run: | - mkdir -p dist - # The sparx CLI requires the proprietary kernel; in OSS builds it is - # absent. Package whatever binaries the build produced. - if [ -f build/cli/sparx ]; then - cp build/cli/sparx dist/ - fi - for bin in build/bench_strategic build/eval_*; do - [ -f "$bin" ] && cp "$bin" dist/ || true - done - cp README.md LICENSE dist/ - cd dist && tar czf ../sparx-${{ matrix.target }}.tar.gz . + - name: Test + run: ctest --test-dir build --output-on-failure --timeout 120 + + - name: Evaluate + run: bash eval/run_all.sh + + - name: Package and verify installed SDK + run: bash scripts/package.sh build sparx-${{ matrix.target }}.tar.gz - name: Checksum run: shasum -a 256 sparx-${{ matrix.target }}.tar.gz > sparx-${{ matrix.target }}.tar.gz.sha256 @@ -98,4 +93,4 @@ jobs: artifacts/**/*.sha256 generate_release_notes: true draft: false - prerelease: ${{ contains(github.ref, '-rc') || contains(github.ref, '-beta') }} + prerelease: ${{ contains(github.ref, '-') }} diff --git a/.gitignore b/.gitignore index ed8b812..a1cf9fc 100644 --- a/.gitignore +++ b/.gitignore @@ -56,3 +56,6 @@ dist/ # Claude .claude/ + +/build-*/ +/install-*/ diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 5647167..3f47e55 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -1,3 +1,9 @@ +> Public 0.4.0-alpha: use the open reference runtime in `core/`, +> `include/master_agent/runtime/`, `backends/`, and `cli/src/reference_main.cpp`. +> The current runnable contract and commands are in README.md and +> docs/reference_runtime.md. Legacy kernel descriptions below require private +> source; experimental modules are not automatically integrated into this CLI. + # Architecture & Code Map ## Directory Structure diff --git a/CHANGELOG.md b/CHANGELOG.md index e7cf599..5646cde 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,13 @@ +# 0.4.0-alpha — open reference runtime + +- Add an independently runnable reference CLI and installable C++ Core/Http SDK. +- Validate tool arguments; scope in-memory request replay and history by session. +- Keep test expectations active in Release and verify installed/relocated consumers. +- Fix eval executable paths, failure exit status, and Bash counter portability. +- Bound edge/cloud inference waits and preserve backend lifetime after timeouts. +- Initialize intent metadata and isolate speculation test persistence. +- Require verified CLI + SDK archives in CI/release; document experimental limits. + # Changelog All notable changes to OAK (Open Agent Kernel) will be documented in this file. diff --git a/CLAUDE.md b/CLAUDE.md index b7551f3..a74e08c 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1,3 +1,9 @@ +> Public 0.4.0-alpha: use the open reference runtime in `core/`, +> `include/master_agent/runtime/`, `backends/`, and `cli/src/reference_main.cpp`. +> The current runnable contract and commands are in README.md and +> docs/reference_runtime.md. Legacy kernel descriptions below require private +> source; experimental modules are not automatically integrated into this CLI. + # OAK (Open Agent Kernel) — CLAUDE.md ## Project Overview @@ -13,8 +19,8 @@ cmake --build build -j$(nproc) ctest --test-dir build --output-on-failure ``` -OSS build produces: test binaries + eval harnesses + bench_strategic. -Full CLI (`sparx`) requires proprietary kernel source in `src/`. +OSS build produces the reference sparx CLI, installable Core/Http SDK, tests and optional eval binaries. +The legacy full CLI still requires private kernel source. See docs/reference_runtime.md. ## Code Conventions @@ -54,4 +60,4 @@ Full CLI (`sparx`) requires proprietary kernel source in `src/`. ## Version -0.3.0-alpha. Versions follow semver. Don't inflate beyond actual stability. +0.4.0-alpha. Versions follow semver. Don't inflate beyond actual stability. diff --git a/CMakeLists.txt b/CMakeLists.txt index 821f8a0..5895b23 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1,9 +1,12 @@ cmake_minimum_required(VERSION 3.18) -project(MasterAgent VERSION 0.3.0 LANGUAGES CXX) +project(MasterAgent VERSION 0.4.0 LANGUAGES CXX) include(GNUInstallDirs) include(CMakePackageConfigHelpers) +find_package(Threads REQUIRED) +option(MASTER_AGENT_ENABLE_HTTP "Build the HTTP model adapter (libcurl)" ON) +option(BUILD_EVAL "Build synthetic evaluation programs" ON) set(CMAKE_CXX_STANDARD 17) set(CMAKE_CXX_STANDARD_REQUIRED ON) @@ -45,10 +48,17 @@ if(MINGW) endfunction() endif() +# Legacy kernel files are optional; the public build uses core/reference_runtime.cpp. +set(_core_source_present OFF) +if(EXISTS "${CMAKE_CURRENT_SOURCE_DIR}/src/agent_dispatch/atomic_lineage.cpp") + set(_core_source_present ON) +endif() + set(MEMORY_SHORT_TERM_SOURCE_ROOT "${CMAKE_SOURCE_DIR}/third_party/memory_short_term" CACHE PATH "Short-term memory source root") +if(_core_source_present) if(NOT EXISTS "${MEMORY_SHORT_TERM_SOURCE_ROOT}/include/vehicle_memory/conversation_journal.h") message(FATAL_ERROR @@ -72,14 +82,7 @@ if(WIN32) target_link_libraries(memory_short_term_core PRIVATE bcrypt) endif() -# Check if the core source tree is present. -# The full kernel source (src/) is proprietary and not included in the OSS -# distribution. When absent, we build a stub library so that downstream -# targets (cli, tests, eval) can still link. -set(_core_source_present OFF) -if(EXISTS "${CMAKE_SOURCE_DIR}/src/agent_dispatch/atomic_lineage.cpp") - set(_core_source_present ON) -endif() +endif() # legacy short-term memory library if(_core_source_present) @@ -214,21 +217,29 @@ set_target_properties(master_agent_core PROPERTIES EXPORT_NAME Core) add_library(MasterAgent::Core ALIAS master_agent_core) else() -# ─── OSS stub: core source not present ─────────────────────────────────── -# Build an INTERFACE library so cli/ and tests/ can still compile and link. -# The cli implements all Agent OS functionality directly; it only needs -# headers from include/ for type definitions. -message(STATUS "Core source not present (OSS build) — using stub library") -add_library(master_agent_core INTERFACE) -target_compile_features(master_agent_core INTERFACE cxx_std_17) -target_include_directories(master_agent_core INTERFACE - "$" - "$" +# The public reference runtime is a real, independently usable library. +message(STATUS "Building the open reference runtime") +add_library(master_agent_core STATIC core/reference_runtime.cpp) +target_compile_features(master_agent_core PUBLIC cxx_std_17) +target_include_directories(master_agent_core PUBLIC + "$" + "$" "$") -target_link_libraries(master_agent_core INTERFACE memory_short_term_core) +target_link_libraries(master_agent_core PUBLIC Threads::Threads) +set_target_properties(master_agent_core PROPERTIES EXPORT_NAME Core POSITION_INDEPENDENT_CODE ON) add_library(MasterAgent::Core ALIAS master_agent_core) endif() +if(MASTER_AGENT_ENABLE_HTTP) + find_package(CURL REQUIRED) + add_library(master_agent_http STATIC backends/http_model.cpp) + target_link_libraries(master_agent_http PUBLIC MasterAgent::Core PRIVATE CURL::libcurl) + set_target_properties(master_agent_http PROPERTIES EXPORT_NAME Http) + add_library(MasterAgent::Http ALIAS master_agent_http) + install(TARGETS master_agent_http EXPORT MasterAgentTargets + ARCHIVE DESTINATION ${CMAKE_INSTALL_LIBDIR}) +endif() + # Prompt and Skill use runtime-managed JSON/text configuration. Mirror the # source configuration into the build tree so local executables and CTest share # the same relative paths as a source-tree launch. @@ -372,7 +383,13 @@ elseif(MASTER_AGENT_BUILD_TESTS) endif() endif() -add_subdirectory(eval) +if(BUILD_EVAL) + add_subdirectory(eval) +endif() + +if(MASTER_AGENT_BUILD_TESTS AND NOT _core_source_present) + include(cmake/ReferenceTests.cmake) +endif() if(_core_source_present) install(TARGETS master_agent_core memory_short_term_core @@ -380,7 +397,7 @@ install(TARGETS master_agent_core memory_short_term_core ARCHIVE DESTINATION ${CMAKE_INSTALL_LIBDIR} LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR}) else() -install(TARGETS memory_short_term_core +install(TARGETS master_agent_core EXPORT MasterAgentTargets ARCHIVE DESTINATION ${CMAKE_INSTALL_LIBDIR} LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR}) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index ebc2753..f232df8 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1,3 +1,9 @@ +> Public 0.4.0-alpha: use the open reference runtime in `core/`, +> `include/master_agent/runtime/`, `backends/`, and `cli/src/reference_main.cpp`. +> The current runnable contract and commands are in README.md and +> docs/reference_runtime.md. Legacy kernel descriptions below require private +> source; experimental modules are not automatically integrated into this CLI. + # Contributing to OAK (Open Agent Kernel) Thank you for your interest in contributing! OAK is an open-source project diff --git a/README.md b/README.md index 76d6e15..97d090d 100644 --- a/README.md +++ b/README.md @@ -1,724 +1,133 @@ -
+# OAK / MasterAgent -OAK +An embeddable, local-first C++17 agent runtime: turn input into validated tool +execution, with deterministic skills before model inference. -# 🌳 OAK — Open Agent Kernel +**0.4.0-alpha — open reference runtime.** Public source now builds a working +`sparx` CLI and an installable C++ SDK. This is a small synchronous runtime, not +the proprietary durable kernel described by the legacy interfaces. -**The Linux kernel for AI agents.**
-Build agents that run 100% on-device. No cloud. No latency. No data leaks. +中文:开源版现可独立构建、运行,并作为 C++ SDK 集成。先匹配确定性技能,再按需调用模型; +工具名称和参数必须通过校验。当前会话、历史和请求去重仅保存在内存中,不承诺崩溃恢复。 -构建 100% 端侧运行的 AI Agent。无云端依赖,无网络延迟,无数据泄露。 +## Quick start -[![License](https://img.shields.io/badge/license-Apache%202.0-blue.svg)](LICENSE) -[![CI](https://github.com/OpenSparX/MasterAgent/workflows/CI/badge.svg)](https://github.com/OpenSparX/MasterAgent/actions) -[![Platform](https://img.shields.io/badge/platform-CPU%20%7C%20Qualcomm%20NPU-green.svg)](#-supported-hardware) - -> ⚠️ **Status: Alpha** — Core kernel is functional. APIs are unstable. Contributions welcome. - -```bash -git clone https://github.com/OpenSparX/MasterAgent.git && cd MasterAgent -cmake -B build -DCMAKE_BUILD_TYPE=Release && cmake --build build -j$(nproc) -``` - -[Quick Start](#-quick-start) · [Why OAK?](#-why-oak) · [Docs](#-documentation) · [中文文档](#中文) - -
- ---- - -## ⚡ 30-Second Demo - -```bash -$ cmake -B build && cmake --build build -j$(nproc) && ctest --test-dir build - -[100%] Built target bench_strategic -Test project /home/you/MasterAgent/build - Start 1: test_integration_speculation -1/5 Test #1: test_integration_speculation ..... Passed 0.8 sec - Start 2: test_orset -2/5 Test #2: test_orset ....................... Passed 0.4 sec - Start 3: test_merkle -3/5 Test #3: test_merkle ...................... Passed 0.4 sec - Start 4: test_embedding -4/5 Test #4: test_embedding ................... Passed 0.4 sec - Start 5: bench_strategic -5/5 Test #5: bench_strategic .................. Passed 2.9 sec - -100% tests passed, 0 tests failed out of 5 -``` - -The OSS build compiles and tests the strategic features: speculative execution, -CRDT mesh sync, formal verification, and embedding search. The full CLI -(with model inference) requires the kernel runtime — see [Architecture](#-architecture). - ---- - -## 🧠 Why OAK? - - - - - - - -
- -### ⚡ Sub-100ms -No network round-trip. 80% of requests resolve via pattern matching in **microseconds**. The other 20% run local LLM inference. - - - -### 🔒 Private by Default -Data never leaves the device. No telemetry. No cloud calls. Encrypted-at-rest storage with device-bound keys. - - - -### 🔋 NPU-Optimized -Develop on CPU anywhere. Deploy to Qualcomm NPU for **14× speedup** at **3.5× less power**. Same code, different backend. - -
- -### How OAK compares - -| | OAK | LangChain | AutoGPT | Apple Intelligence | -|:---|:---:|:---:|:---:|:---:| -| Runs 100% on-device | ✅ | ❌ | ❌ | ✅ | -| Open source | ✅ | ✅ | ✅ | ❌ | -| Crash recovery (WAL) | ✅ | ❌ | ❌ | ❌ | -| Formal verification | ✅ | ❌ | ❌ | ❌ | -| Multi-device mesh | ✅ | ❌ | ❌ | ❌ | -| Speculative execution | ✅ | ❌ | ❌ | ❌ | -| On-device learning | ✅ | ❌ | ❌ | ❌ | -| NPU acceleration | ✅ | ❌ | ❌ | ✅ | -| Latency (typical) | **87ms** | 2-5s | 3-10s | ~200ms | - ---- - -## 🚀 Quick Start - -### Build from Source +Requires CMake 3.18+, C++17, and libcurl development files for the optional HTTP +adapter. Tests require Python 3. The release CI targets Linux and macOS. ```bash -# Prerequisites: CMake 3.18+, C++17 compiler (GCC 9+, Clang 11+) git clone https://github.com/OpenSparX/MasterAgent.git cd MasterAgent -cmake -B build -DCMAKE_BUILD_TYPE=Release \ - -DMASTER_AGENT_BUILD_CLI=ON \ - -DMASTER_AGENT_BUILD_TESTS=ON -cmake --build build -j$(nproc) - -# Run tests +# Ubuntu HTTP dependency: sudo apt-get install libcurl4-openssl-dev +cmake -S . -B build -DCMAKE_BUILD_TYPE=Release +cmake --build build --parallel 4 ctest --test-dir build --output-on-failure -``` - -### Run with a Local Model (llama.cpp) - -```bash -# The sparx CLI connects to any llama-server compatible endpoint. -# Start llama-server (install separately: https://github.com/ggml-org/llama.cpp) -llama-server -m your-model.gguf --port 8080 - -# Run the CLI -./build/cli/sparx run --endpoint 127.0.0.1:8080 -``` - -### Run in Deterministic-Only Mode (No Model Needed) - -```bash -# Deterministic skills respond without any model loaded ./build/cli/sparx demo automotive ``` -> **💡** Most intent routing works without a model — only open-ended queries need LLM inference. - ---- - -## 📦 What's Open Source - -OAK uses an **open-core** model. This repository contains: - -| Component | Status | LOC | -|:---|:---:|---:| -| Speculative Execution (LSTM + HNSW) | ✅ Full source | 2,565 | -| Formal Plan Verification (CDCL SAT) | ✅ Full source | 3,656 | -| Agent Mesh (mDNS + CRDT + Merkle) | ✅ Full source | 4,875 | -| On-Device Learning (DP-SGD) | ✅ Full source | 1,800+ | -| Constrained Decoding (GBNF) | ✅ Full source | 1,200+ | -| llama.cpp Model Runtime | ✅ Full source | 527 | -| Agent Scheduler | ✅ Full source | 600+ | -| Kernel Interfaces (headers) | ✅ Public API | — | -| Kernel Runtime (orchestrator, WAL, dispatch) | ❌ Proprietary | — | - -The proprietary kernel runtime handles task orchestration, WAL recovery, and agent dispatch. The strategic feature modules (the algorithmic innovations) are fully open and independently testable. - -We're working toward open-sourcing the kernel runtime. Track progress in [#1](https://github.com/OpenSparX/MasterAgent/issues/1). - ---- - -## 🏗️ Architecture - -``` -┌─────────────────────────────────────────────────────────────────┐ -│ User Input │ -└──────────────────────────────┬──────────────────────────────────┘ - ▼ -┌──────────────────────────────────────────────────────────────────┐ -│ Preprocessing: UTF-8 normalize → parameter extract → memory │ -└──────────────────────────────┬───────────────────────────────────┘ - ▼ - ┌─────────────────────┐ - │ Route Decision │ - │ (80% deterministic │ - │ 20% inference) │ - └────┬──────────┬─────┘ - │ │ - ┌──────────▼──┐ ┌───▼────────────┐ - │ Skill Engine │ │ LLM Inference │ - │ (0.02ms) │ │ (87ms NPU / │ - │ │ │ 1200ms CPU) │ - └──────────┬───┘ └───┬────────────┘ - │ │ - ▼ ▼ - ┌────────────────────────────────────┐ - │ Task Orchestrator (DAG execution) │ - │ + WAL Recovery + MCP Services │ - └────────────────────────────────────┘ - ▼ - ┌────────────────────────────────────┐ - │ Response (sub-100ms typical) │ - └────────────────────────────────────┘ -``` - - - -
-📊 Full architecture diagram -

- Architecture -

-
- -**Design principles:** -- **Deterministic first** — pattern matching handles 80% of requests at sub-ms latency -- **Crash-safe** — WAL (Write-Ahead Log) with three terminal states: `COMMITTED`, `FAILED`, `UNKNOWN` -- **Hardware-agnostic** — same code runs on CPU (dev) and NPU (production) -- **Speculate ahead** — predict user's next intent and pre-compute during idle time - ---- - -## 💎 Key Features - -### 🔮 Speculative Execution - -OAK predicts what you'll ask next and pre-computes the answer during idle NPU time. - -``` -You: "navigate to office" ← observed - ↓ predictor: P("play music") = 0.83 - ↓ pre-computes playlist response during idle -You: "play my commute mix" ← cache HIT, 0.11μs response -``` - -| Metric | Value | -|:---|---:| -| Prediction (top-3) | 0.27 μs | -| Cache hit (exact) | 0.11 μs | -| Embedding similarity | 8.79 μs | -| Cold-start threshold | 10 interactions | - -### 🛡️ Formal Plan Verification - -Plans are verified for safety **before** execution using CTL* model checking: - -```bash -$ sparx plan verify plans/payment-flow.yaml - -Plan Verification Report -═══════════════════════════ - ✓ PASS auth-before-destructive (12μs) - ✓ PASS no-resource-deadlock (8μs) - ✓ PASS all-nodes-terminate (15μs) - ✓ PASS data-flow-integrity (11μs) - ✗ FAIL no-conflicting-destructive (23μs) - → Node "charge" and "refund" conflict on resource "wallet" - -✗ Plan should NOT be executed. Fix conflicts first. -``` - -- CTL* temporal logic (AG, AF, AX, AU, EF, EX) -- Partial-order reduction: **60% state-space reduction** on typical plans -- Counterexample traces pinpoint the exact violation path -- Runtime monitor for online verification during execution - -### 🌐 Agent Mesh Protocol - -Zero-config multi-device collaboration. Your phone, laptop, and car share agent memory and route work to the most capable device: - -```bash -$ sparx mesh status - -Mesh: oak-home (3 peers, healthy) -┌────────────────┬──────────┬───────┬────────┬─────────┐ -│ Device │ NPU │ RAM │ Idle │ Score │ -├────────────────┼──────────┼───────┼────────┼─────────┤ -│ 🚗 Car (local) │ 45 TOPS │ 16GB │ yes │ 0.92 │ -│ 📱 Phone │ 12 TOPS │ 8GB │ no │ 0.45 │ -│ 💻 Laptop │ — │ 32GB │ yes │ 0.38 │ -└────────────────┴──────────┴───────┴────────┴─────────┘ - -CRDT sync: 142 keys, last sync 2s ago -Merkle: roots match (no divergence) -``` - -- **mDNS/DNS-SD** zero-config discovery (`_sparx-mesh._tcp.local.`) -- **CRDT state sync**: GCounter, PNCounter, GSet, ORSet (add-wins), LWW-Register -- **Merkle anti-entropy**: O(log K) divergence detection, not O(K) full scan -- **Capability routing**: intent → best device by NPU TOPS, model, idle state -- **Split inference**: partition large models across multiple NPU devices - -### 🧱 Crash Recovery (UNKNOWN State) - -**Industry first.** When an agent crashes mid-operation, the only honest answer is "I don't know if it succeeded." - -``` -┌──────────┐ ┌──────────┐ ┌──────────────┐ -│ COMMITTED│ │ FAILED │ │ UNKNOWN │ -│ (success)│ │ (error) │ │ (crashed │ -│ │ │ │ │ mid-flight) │ -└──────────┘ └──────────┘ └──────────────┘ - │ - ▼ - Manual reconciliation - required (sparx reconcile) -``` - -Other frameworks retry (duplicate charges) or ignore (lost money). OAK is honest. - -### 🧬 On-Device Continual Learning - -Your agent gets smarter with every correction — entirely on-device, with mathematical privacy guarantees. - -```bash -$ sparx learn correct -# Last response was wrong? Record a correction: -# Original: "Setting AC to 22°C" → turned on heat -# Correct: "Setting AC to 22°C" → ac.setCooling(22) - -$ sparx learn status - -Learning Status -═══════════════ - Adapter: v3 (merged 2 hours ago) - Corrections: 47 recorded, 38 trained - Privacy: ε = 2.1 / budget 8.0 (73% remaining) - Quality: perplexity 12.3 → 11.1 (↓9.7%) - Next train: idle + charging + cool (estimated 3:00 AM) - -$ sparx learn train -# ⚙️ QLoRA fine-tuning with DP-SGD... -# ├─ Batch: 38 corrections -# ├─ Privacy: Rényi DP, ε = 0.4 this round -# ├─ Validation: perplexity 12.3 → 11.1 ✓ (improved) -# └─ Adapter merged: v3 → v4 -``` - -**Why this matters:** -- **No cloud training** — corrections never leave the device -- **Differential privacy** — DP-SGD with configurable ε budget, mathematically bounded information leakage -- **Quality guard** — perplexity validation before/after; auto-rollback on degradation -- **Idle scheduling** — trains only when NPU idle + charging + thermally cool -- **Progressive merge** — weighted adapter averaging prevents catastrophic forgetting - -The more you use it, the better it gets. Your data stays yours. - -### 📚 More Features - -| Feature | Description | -|:---|:---| -| **Constrained Decoding** | GBNF grammar forces valid JSON — zero hallucinated tool calls | -| **DAG Orchestrator** | Multi-step plan execution with dependency resolution | -| **Deterministic Skills** | YAML-defined pattern matching, no model needed | -| **NPU Acceleration** | Qualcomm QNN backend, 14× faster than CPU at 3.5× less power | - ---- - -## Evaluation Results - -| Feature | Key Metric | Value | Baseline | Improvement | -|---------|-----------|-------|----------|-------------| -| Speculative Execution | Cache Hit Rate | **73.2%** | 0% (no speculation) | 3.71× latency speedup | -| Agent Mesh | Convergence Rounds | **1–2 rounds** | Full-sync every round | 88% bandwidth savings | -| Formal Verification | Unsafe Plan Detection | **71.4%** | No verification (100% escape) | 0% false positives | -| On-Device Learning | Personalization Accuracy | **66.8%** | 5% (static model) | +61.8pp lift | -| Constrained Decoding | Valid Output Rate | **100%** | 16.7% (unconstrained) | 83.3pp improvement | - -> Run `./eval/run_all.sh` to reproduce these results. - ---- - -## Technical Report - -- [Technical Report](docs/TECHNICAL_REPORT.md) — detailed evaluation methodology, results analysis, and system design decisions -- [Why On-Device?](docs/WHY_ON_DEVICE.md) — rationale for on-device agent execution over cloud-based alternatives - ---- - -## Reproducing Results - -```bash -# Build evaluation suite -cd build && cmake .. -DBUILD_EVAL=ON && make -j$(nproc) -# Run all evaluations -./eval/run_all.sh -# Results appear in eval/results/ -``` - ---- - -## 📦 Examples +The demo sets an **in-memory simulated** AC temperature, then reads it back. +It needs no model, credentials, network, or vehicle hardware. Each output is JSON. ```bash -git clone https://github.com/OpenSparX/MasterAgent.git -cd MasterAgent -``` - -| Example | Path | Description | -|:---|:---|:---| -| 🚗 **Automotive** | `examples/automotive_assistant/` | Voice commands → vehicle control | -| 🏠 **Smart Home** | `examples/smart_home/` | Multi-room device orchestration | -| 📡 **IoT Edge** | `examples/iot_edge/` | Battery-optimized sensor agent | - -```bash -cd examples/automotive_assistant && sparx run - -# "Turn on AC, set to 22°C" → 87ms -# "Navigate to nearest charger" → 1.2s (inference) -# "What's my tire pressure?" → 0.03ms (deterministic) -``` - ---- - -## 🔌 Supported Hardware - -Develop on **any machine** (CPU). Deploy to NPU for production: - -| Platform | Backend | Latency | Power | Status | -|:---|:---|---:|---:|:---:| -| Mac / Linux / Windows | llama.cpp (CPU) | ~1,200ms | 8.1W | ✅ | -| SA8155P / SA8295P | Qualcomm QNN (NPU) | **87ms** | **2.3W** | ✅ | -| SA8650P / SA8775P | Qualcomm QNN (NPU) | ~70ms | ~2.0W | ✅ | -| Snapdragon 8 Gen 3+ | Qualcomm QNN (NPU) | TBD | TBD | 🔄 Q4 2026 | - ---- - -## 📐 Project Structure - +./build/cli/sparx run --input '把空调调到22度' +./build/cli/sparx run --input 'vehicle status' ``` -MasterAgent/ -├── cli/ # Sparx CLI (commands + strategic features) -│ ├── include/ # Public headers -│ │ ├── sparx_speculative.h # Speculative execution -│ │ ├── sparx_formal_verify.h # CTL* model checker -│ │ ├── sparx_mesh.h # Agent mesh protocol -│ │ ├── sparx_learning.h # Continual learning -│ │ └── sparx_constrained_decode.h -│ └── src/ # Implementations (~5,500 LOC strategic features) -├── include/master_agent/ # Core kernel API -│ ├── orchestrator/ # DAG task execution -│ ├── inference/ # Model runtime abstraction -│ ├── atomic_service/ # MCP tool integration + WAL -│ ├── intent/ # Intent recognition engine -│ ├── skill/ # Deterministic skill engine -│ ├── memory/ # Short-term context -│ └── transport/ipc/ # Inter-process communication -├── src/ # Core kernel implementation (~40,000 LOC) -├── tests/ # 19 test suites + 5 strategic feature tests -├── examples/ # Ready-to-run example agents -├── docs/ # Architecture docs + ROADMAP -└── .github/workflows/ # CI/CD (8-platform release) -``` - ---- - -## 🗺️ Roadmap - -See [docs/ROADMAP_v3.md](docs/ROADMAP_v3.md) for the full plan. - -| Version | Target | Key Features | -|:---|:---|:---| -| ~~v2.0~~ | ~~2025~~ | ✅ Core kernel, WAL, MCP, NPU | -| ~~v2.1~~ | ~~Aug 2026~~ | ✅ Speculation, Verification, Mesh, Learning | -| **v3.0** | Q4 2026 | Neural predictor (LSTM), CEGAR, BLE mesh | -| **v3.1** | Q1 2027 | Intent-aware speculation, causal broadcast | -| **v3.2** | Q2 2027 | mTLS mesh, adaptive Merkle, observability | -| **v3.3** | Q3 2027 | WAN relay, federated learning, heterogeneous compute | - ---- -## 📚 Documentation +Each process starts a new demo state. To preserve state between requests in a +single process, run `sparx run` and enter one input per line until EOF. +An unmatched phrase returns `NO_ROUTE` unless a model is explicitly configured. -| Doc | Description | -|:---|:---| -| [System Overview](docs/SYSTEM_OVERVIEW.md) | Architecture deep-dive | -| [Build & Test](docs/BUILD_AND_TEST.md) | Compilation from source | -| [WAL Recovery](docs/WAL_RECOVERY.md) | Crash recovery mechanism | -| [MCP Services](docs/MCP_SERVICES.md) | Adding custom tool capabilities | -| [Qualcomm NPU](docs/QUALCOMM_NPU.md) | QNN SDK integration | -| [v3.x Roadmap](docs/ROADMAP_v3.md) | Future direction | +## Use an existing local model server ---- - -## 🤝 Contributing - -We welcome contributions! See [CONTRIBUTING.md](CONTRIBUTING.md) for guidelines. +Start a compatible llama-server separately with your chosen GGUF model, then: ```bash -# Clone and build -git clone https://github.com/OpenSparX/MasterAgent.git -cd MasterAgent -cmake -B build -DCMAKE_BUILD_TYPE=Release -cmake --build build -j$(nproc) - -# Run tests -ctest --test-dir build --output-on-failure +./build/cli/sparx run --endpoint 127.0.0.1:8080 --model local \ + --input 'Set the AC to 24 degrees' ``` -**Good first issues:** [GitHub Issues](https://github.com/OpenSparX/MasterAgent/labels/good%20first%20issue) - ---- - -## ❓ FAQ - -
-Do I need Qualcomm hardware? -No. Develop with CPU inference (llama.cpp) on any machine. NPU is optional for production. -
- -
-What models work? -Any GGUF model: Qwen2/3, Llama 3, Mistral, Phi, etc. For NPU: models need QNN conversion. -
- -
-Is this production-ready? -Yes. 19 test suites, WAL crash recovery, formal verification. Deployed on SA8295P vehicles. -
- -
-How is this different from LangChain? -LangChain orchestrates cloud API calls. OAK runs the entire agent (model + tools + memory) on-device with crash safety guarantees that cloud frameworks cannot provide. -
- -
-Can I use it for non-automotive apps? -Yes — smart home, robotics, IoT, medical devices, industrial automation. The automotive demo is just the showcase. -
+The adapter sends non-streaming `/v1/chat/completions` requests. The runtime asks +for a JSON response or a single tool call, rejects unknown tools and invalid +arguments, and returns the actual tool result. Model compliance depends on the +model; malformed output produces an error, never a guessed tool call. No model +is downloaded or started by this CLI. A full HTTP(S) endpoint URL may also be +provided explicitly; no cloud fallback happens automatically. ---- - -## 📄 License - -Apache 2.0 — see [LICENSE](LICENSE) - ---- - -## 💬 Community - -- [GitHub Discussions](https://github.com/OpenSparX/MasterAgent/discussions) -- [GitHub Issues](https://github.com/OpenSparX/MasterAgent/issues) -- Email: dev@opensparc.com - ---- - -
- -**Ready to build?** +For an SDK and deterministic CLI without libcurl: ```bash -npm install -g @sparx/cli && sparx init my-agent +cmake -S . -B build-offline -DMASTER_AGENT_ENABLE_HTTP=OFF -DBUILD_EVAL=OFF +cmake --build build-offline --parallel 4 ``` -[⭐ Star this repo](https://github.com/OpenSparX/MasterAgent) · [📖 Read the docs](docs/) · [💬 Join the discussion](https://github.com/OpenSparX/MasterAgent/discussions) - -
- ---- ---- - - - -
- -# 🌳 OAK — 开放智能体内核 - -**AI Agent 的 Linux 内核。**
-构建 100% 端侧运行的智能体。无云端,无延迟,无数据泄露。 +## Embed the SDK ```bash -npm install -g @sparx/cli && sparx demo automotive -``` - -[快速开始](#-快速开始-1) · [为什么选 OAK](#-为什么选-oak) · [English](#-quick-start) - -
- ---- - -## 🧠 为什么选 OAK? - -| | OAK | LangChain | AutoGPT | -|:---|:---:|:---:|:---:| -| 100% 端侧运行 | ✅ | ❌ | ❌ | -| 崩溃恢复 (WAL) | ✅ | ❌ | ❌ | -| 形式化验证 | ✅ | ❌ | ❌ | -| 多设备 Mesh | ✅ | ❌ | ❌ | -| 投机执行 | ✅ | ❌ | ❌ | -| 端侧自学习 | ✅ | ❌ | ❌ | -| 典型延迟 | **87ms** | 2-5s | 3-10s | - -**核心理念:** OAK 之于 Agent OS,如同 Linux 内核之于 Android/Ubuntu。我们不做完整操作系统 — 我们提供开源内核层,车企、手机厂商、机器人公司基于 OAK 自研专属 Agent OS。 - ---- - -## ⚡ 快速开始 - -```bash -# 安装 -npm install -g @sparx/cli - -# 初始化项目 -sparx init my-agent && cd my-agent - -# 下载模型(530MB,1-2 分钟) -sparx pull qwen2.5-0.5b-instruct - -# 运行 -sparx run +cmake --install build --prefix "$PWD/install-sdk" +cmake -S examples/reference_agent -B build-consumer \ + -DCMAKE_PREFIX_PATH="$PWD/install-sdk" +cmake --build build-consumer +ctest --test-dir build-consumer --output-on-failure ``` -``` -> 你好 -✓ route=deterministic skill=hello 0.02ms (未调用模型) +Consumer CMake: -> 法国的首都是哪里? -✓ route=inference ttft=142ms total=1830ms tokens=28 - 法国的首都是巴黎。 +```cmake +find_package(MasterAgent CONFIG REQUIRED) +target_link_libraries(your_app PRIVATE MasterAgent::Core) +# Optional HTTP adapter: +# target_link_libraries(your_app PRIVATE MasterAgent::Http) ``` -> **💡** 不装模型也能用 — 确定性技能照常工作,只有开放问题需要模型。 - ---- - -## 💎 核心特性 - -### 🔮 投机执行 — 预测你的下一步 - -预测用户意图,NPU 空闲时预计算结果。命中缓存时 **0.11 μs** 响应。 - -### 🛡️ 形式化验证 — 执行前证明安全 - -CTL* 模型检查 + 偏序归约,在执行前验证计划不会死锁、不会越权、不会超时。 - -### 🌐 Agent Mesh — 零配置多设备协作 - -mDNS 发现 + CRDT 状态同步 + Merkle 反熵。你的手机、车机、电脑自动组网,将任务路由到最强设备。 - -### 🧱 UNKNOWN 终态 — 业界首创 - -Agent 崩溃时不盲目重试(重复扣费),不静默忽略(钱丢了)。进入 UNKNOWN 状态,要求显式对账。 +The public reference API is in +[`reference_runtime.h`](include/master_agent/runtime/reference_runtime.h). +Register a tool with a supported parameter schema and callback, register exact +phrase skills, optionally set a model callback, then call `Runtime::run()`. +See the [integration contract](docs/reference_runtime.md) for errors, schema +support, sessions, request replay, and concurrency limitations. -### 🧬 端侧自学习 — 越用越聪明 +## Capability status -每次纠正都让 Agent 变强,完全在设备上完成,数学保证隐私: +| Capability | Public reference release | +|---|---| +| Deterministic exact-phrase skills | Available, no model required | +| Host-defined tools and argument validation | Available; primitive object schema subset | +| Local model HTTP adapter | Available when built with libcurl | +| Sessions and request deduplication | Available in memory; bounded; no restart persistence | +| CLI and relocatable CMake SDK package | Built and exercised by CI | +| Speculation, mesh, verification, learning, decoding | Experimental modules and synthetic evaluations; not wired into the reference CLI | +| Edge/cloud harness | Experimental; explicit opt-in; bounded outstanding inference workers | +| Durable DAG/WAL recovery and legacy kernel factories | Not supplied by this reference runtime | +| Qualcomm QNN/Genie integration | Platform integration required; not verified by this release | +| Production encryption / DP learning guarantees | Not claimed for this release | -- **QLoRA 微调** — 纠正 → 训练 → adapter 合并,全流程端侧 -- **差分隐私** — DP-SGD + Rényi 隐私预算,信息泄露有数学上界 -- **质量守门** — 训练前后验证困惑度,退步自动回滚 -- **空闲调度** — 仅在 NPU 空闲 + 充电 + 温控正常时训练 -- **渐进合并** — 加权平均防止灾难性遗忘 +Other headers under `include/master_agent/` describe legacy kernel contracts; +including them does not mean their factories have public implementations. Use +`master_agent::reference` for the supported open runtime entry point. -你的数据永远不离开设备。用得越多,越懂你。 - -### 📚 更多特性 - -| 特性 | 说明 | -|:---|:---| -| 约束解码 | GBNF 语法强制有效 JSON,零幻觉工具调用 | -| DAG 编排 | 多步计划执行,带依赖解析 | -| 确定性路由 | 80% 请求不过模型,微秒级响应 | - ---- - -## 🔌 支持平台 - -| 平台 | 后端 | 延迟 | 功耗 | 状态 | -|:---|:---|---:|---:|:---:| -| Mac / Linux / Windows | llama.cpp (CPU) | ~1,200ms | 8.1W | ✅ | -| SA8155P / SA8295P / SA8650P | Qualcomm QNN | **87ms** | **2.3W** | ✅ | -| Snapdragon 8 Gen 3+ | Qualcomm QNN | 待测 | 待测 | 🔄 2026 Q4 | - ---- - -## 📦 示例 - -| 示例 | 路径 | 说明 | -|:---|:---|:---| -| 🚗 车载助手 | `examples/automotive_assistant/` | 语音 → 车控 | -| 🏠 智能家居 | `examples/smart_home/` | 多房间设备编排 | -| 📡 IoT 边缘 | `examples/iot_edge/` | 电池优化传感器 Agent | - ---- - -## 🗺️ 路线图 - -| 版本 | 时间 | 关键特性 | -|:---|:---|:---| -| **v0.3** | 2026.8 | ✅ 内核、WAL、Agent 调度、llama.cpp 集成 | -| **v0.4** | 2026 Q4 | 投机执行稳定化、端到端验证覆盖 | -| **v0.5** | 2027 Q1 | Agent Mesh (mDNS + CRDT)、NPU 加速 | -| **v1.0** | 2027 Q2 | API 稳定、npm CLI 发布、完整文档 | - -详见 [docs/ROADMAP_v3.md](docs/ROADMAP_v3.md) - ---- - -## 📚 文档 - -- [系统概述](docs/01_系统概述.md) -- [构建和测试](docs/10_构建运行与测试.md) -- [WAL 恢复机制](docs/WAL_RECOVERY_zh-CN.md) -- [MCP 服务开发](docs/MCP_SERVICES_zh-CN.md) -- [Qualcomm NPU 集成](docs/QUALCOMM_NPU_zh-CN.md) - ---- - -## 🤝 贡献 - -欢迎贡献!详见 [CONTRIBUTING_zh-CN.md](CONTRIBUTING_zh-CN.md) +## Tests, evaluations, and releases ```bash -git clone https://github.com/OpenSparX/MasterAgent.git -cd MasterAgent -cmake -B build -DCMAKE_BUILD_TYPE=Release -cmake --build build -j$(nproc) ctest --test-dir build --output-on-failure +bash eval/run_all.sh +bash scripts/package.sh build dist/sparx-sdk.tar.gz ``` ---- - -## 📄 许可证 - -Apache 2.0 — 见 [LICENSE](LICENSE) +Evaluations use synthetic workloads (learning includes a simulated model), not +end-to-end evidence for real-model latency, personalization, or NPU performance. +Results are written to `build/eval-results/`. Missing or failed evaluations +return a nonzero status. `BUILD_EVAL=OFF` excludes them from an SDK-only build. ---- +The package script installs the SDK and CLI, unpacks the archive in a new +location, runs the demo, and builds an external consumer before publishing the +archive. The package uses platform system dependencies, including libcurl when +enabled; it is not a universally static binary. -
+## Contributing -**立即开始 ↓** - -```bash -git clone https://github.com/OpenSparX/MasterAgent.git && cd MasterAgent -cmake -B build && cmake --build build -j$(nproc) -``` +See [CONTRIBUTING.md](CONTRIBUTING.md). Changes must keep Release assertions +active and pass the CLI, HTTP protocol fixture, installed consumer, and runtime +failure tests. Hardware and actual model validation should record the model, +quantization, device, workload, and latency distribution separately. -
+Apache-2.0. See [LICENSE](LICENSE). diff --git a/VERSION.json b/VERSION.json index b9833e3..e21e68d 100644 --- a/VERSION.json +++ b/VERSION.json @@ -1,9 +1,9 @@ { "product": "OAK", - "version": "0.3.0", + "version": "0.4.0-alpha", "api_namespace": "master_agent", "cmake_target": "MasterAgent::Core", - "model_backends": ["llama_cpp", "mock"], + "model_backends": ["http_chat_completions", "custom_callback"], "default_topology": "single-process", "ipc_default": false } diff --git a/backends/http_model.cpp b/backends/http_model.cpp new file mode 100644 index 0000000..60ee91b --- /dev/null +++ b/backends/http_model.cpp @@ -0,0 +1,55 @@ +#include "master_agent/runtime/http_model.h" +#include +#include + +namespace master_agent::reference { +ModelHandler makeHttpModel(HttpModelConfig config) { + return [config = std::move(config)](const Json& messages) -> Result { + auto fail = [](const std::string& code, const std::string& message) { + return Result::failure({code, message, "", 502}); + }; + if (config.timeout_ms <= 0 || (config.endpoint.rfind("http://", 0) != 0 && config.endpoint.rfind("https://", 0) != 0) || + config.api_key.find_first_of("\r\n") != std::string::npos) + return fail("INVALID_CONFIG", "Expected HTTP(S) endpoint, positive timeout and valid credentials"); + static const auto initialized = curl_global_init(CURL_GLOBAL_DEFAULT); + if (initialized != CURLE_OK) return fail("HTTP_ERROR", "curl initialization failed"); + std::unique_ptr curl(curl_easy_init(), curl_easy_cleanup); + if (!curl) return fail("HTTP_ERROR", "Cannot allocate HTTP client"); + curl_slist* raw_headers = curl_slist_append(nullptr, "Content-Type: application/json"); + if (!config.api_key.empty()) raw_headers = curl_slist_append(raw_headers, ("Authorization: Bearer " + config.api_key).c_str()); + std::unique_ptr headers(raw_headers, curl_slist_free_all); + const auto body = Json{{"model", config.model}, {"messages", messages}, {"stream", false}, {"temperature", 0}, {"max_tokens", 512}}.dump(); + std::string response; + curl_easy_setopt(curl.get(), CURLOPT_URL, config.endpoint.c_str()); + curl_easy_setopt(curl.get(), CURLOPT_HTTPHEADER, headers.get()); + curl_easy_setopt(curl.get(), CURLOPT_POSTFIELDS, body.c_str()); + curl_easy_setopt(curl.get(), CURLOPT_POSTFIELDSIZE, static_cast(body.size())); + curl_easy_setopt(curl.get(), CURLOPT_TIMEOUT_MS, config.timeout_ms); + curl_easy_setopt(curl.get(), CURLOPT_CONNECTTIMEOUT_MS, config.timeout_ms); + curl_easy_setopt(curl.get(), CURLOPT_NOSIGNAL, 1L); + curl_easy_setopt(curl.get(), CURLOPT_FOLLOWLOCATION, 0L); + curl_easy_setopt(curl.get(), CURLOPT_WRITEFUNCTION, + +[](char* data, size_t size, size_t count, void* output) -> size_t { + auto& text = *static_cast(output); + const auto bytes = size * count; + if (bytes > 1024 * 1024 - text.size()) return 0; + try { text.append(data, bytes); } catch (...) { return 0; } + return bytes; + }); + curl_easy_setopt(curl.get(), CURLOPT_WRITEDATA, &response); + auto status = curl_easy_perform(curl.get()); + if (status != CURLE_OK) return fail(status == CURLE_OPERATION_TIMEDOUT ? "TIMEOUT" : "HTTP_ERROR", curl_easy_strerror(status)); + long http_status = 0; + curl_easy_getinfo(curl.get(), CURLINFO_RESPONSE_CODE, &http_status); + if (http_status != 200) return fail("HTTP_ERROR", "Model endpoint returned HTTP " + std::to_string(http_status)); + try { + auto json = Json::parse(response); + auto content = json.at("choices").at(0).at("message").at("content"); + if (!content.is_string() || content.get_ref().empty()) return fail("INVALID_RESPONSE", "Missing model content"); + return Result::success(content.get()); + } catch (const Json::exception&) { + return fail("INVALID_RESPONSE", "Expected choices[0].message.content string"); + } + }; +} +} // namespace master_agent::reference diff --git a/cli/CMakeLists.txt b/cli/CMakeLists.txt index 6ddf98e..c5bdd11 100644 --- a/cli/CMakeLists.txt +++ b/cli/CMakeLists.txt @@ -43,9 +43,8 @@ set(SPARX_SOURCES # "NPU unavailable" at runtime on a host without the Qualcomm stack. That is # what lets one binary serve both a dev laptop and a device. # -# In OSS builds (core source not present), the runtime adapters and sparx -# executable are skipped — only the strategic feature sources compile (used by -# eval/ and tests/). +# The legacy CLI/adapters require the private kernel. The OSS branch below +# builds a separate thin CLI against the public reference runtime. if(_core_source_present) add_library(sparx_runtimes STATIC @@ -127,4 +126,16 @@ endif() install(TARGETS sparx RUNTIME DESTINATION bin) +else() + add_executable(sparx src/reference_main.cpp) + target_link_libraries(sparx PRIVATE MasterAgent::Core) + if(MASTER_AGENT_ENABLE_HTTP) + target_link_libraries(sparx PRIVATE MasterAgent::Http) + target_compile_definitions(sparx PRIVATE MASTER_AGENT_HAS_HTTP=1) + endif() + if(NOT DEFINED SPARX_VERSION) + set(SPARX_VERSION "${PROJECT_VERSION}-alpha") + endif() + target_compile_definitions(sparx PRIVATE SPARX_VERSION="${SPARX_VERSION}") + install(TARGETS sparx RUNTIME DESTINATION ${CMAKE_INSTALL_BINDIR}) endif() # _core_source_present diff --git a/cli/include/sparx_pipeline_harness.h b/cli/include/sparx_pipeline_harness.h index 95e73ef..e3b98c7 100644 --- a/cli/include/sparx_pipeline_harness.h +++ b/cli/include/sparx_pipeline_harness.h @@ -64,7 +64,7 @@ struct HarnessConfig { ArbiterConfig arbiter_config; /// Master enable switch for cloud path. - bool cloud_enabled = true; + bool cloud_enabled = false; /// Trace/log every arbitration decision. bool trace_decisions = false; @@ -157,7 +157,7 @@ class PipelineHarness { void applyConfig(const HarnessConfig& config); /// Get current config (read-only). - const HarnessConfig& config() const { return config_; } + HarnessConfig config() const { std::lock_guard lock(mutex_); return config_; } // ─── Execution ─── @@ -185,6 +185,8 @@ class PipelineHarness { bool isReady() const; private: + std::shared_ptr> local_busy_ = std::make_shared>(false); + std::shared_ptr> cloud_busy_ = std::make_shared>(false); HarnessConfig config_; mutable std::mutex mutex_; @@ -205,7 +207,6 @@ class PipelineHarness { // Internal helpers void resolveActiveComponents(); PreScoreSignals buildPreScoreSignals(const PipelineRequest& request) const; - bool shouldFireCloud(const ConfidenceScore& pre_score) const; }; // ─── YAML Config Parser ───────────────────────────────────────────────────── diff --git a/cli/include/sparx_speculative.h b/cli/include/sparx_speculative.h index 01ab7dd..9e0844e 100644 --- a/cli/include/sparx_speculative.h +++ b/cli/include/sparx_speculative.h @@ -75,11 +75,11 @@ namespace sparx::speculation { struct IntentRecord { std::string intent_name; // skill or inference category std::string raw_input; // user's actual text - std::int64_t timestamp_utc; // when it happened - std::uint8_t hour_of_day; // 0-23, local time - std::uint8_t day_of_week; // 0-6, Mon=0 - bool was_deterministic; // resolved without model? - std::uint32_t latency_ms; // actual response time + std::int64_t timestamp_utc = 0; // when it happened + std::uint8_t hour_of_day = 0; // 0-23, local time + std::uint8_t day_of_week = 0; // 0-6, Mon=0 + bool was_deterministic = false; // resolved without model? + std::uint32_t latency_ms = 0; // actual response time }; /// A prediction of what the user will likely ask next. diff --git a/cli/src/reference_main.cpp b/cli/src/reference_main.cpp new file mode 100644 index 0000000..758b862 --- /dev/null +++ b/cli/src/reference_main.cpp @@ -0,0 +1,94 @@ +#include "master_agent/runtime/reference_runtime.h" +#ifdef MASTER_AGENT_HAS_HTTP +#include "master_agent/runtime/http_model.h" +#endif +#include +#include + +using namespace master_agent; +using namespace master_agent::reference; +namespace { +void help() { + std::cout << "sparx — open reference runtime\n" + " sparx version\n" + " sparx demo automotive\n" + " sparx run [--input TEXT] [--endpoint URL] [--model NAME]\n" + " [--session ID] [--request-id ID]\n" + "Without --input, reads one request per line until EOF.\n" + "Without --endpoint, runs deterministic skills only.\n" + "Example skills: set AC to 22 degrees; 把空调调到22度; vehicle status\n" + "The automotive tools control in-memory demo state, not a vehicle.\n"; +} +Json encode(const Result& result) { + if (!result) { + const auto e = result.error.value_or(StructuredError{"INTERNAL", "Missing result", "", 500}); + return {{"ok", false}, {"error", {{"code", e.code}, {"message", e.message}}}}; + } + const auto& r = *result; + return {{"ok", true}, {"session_id", r.session_id}, {"request_id", r.request_id}, + {"route", r.route}, {"tool", r.tool}, {"output", r.output}, {"replayed", r.replayed}}; +} +} +int main(int argc, char** argv) { + if (argc < 2 || std::string(argv[1]) == "--help" || std::string(argv[1]) == "help") { help(); return 0; } + const std::string command = argv[1]; + if (command == "version" && argc == 2) { std::cout << SPARX_VERSION << " (open reference runtime)\n"; return 0; } + const bool demo = command == "demo" && argc == 3 && std::string(argv[2]) == "automotive"; + if (command != "run" && !demo) { std::cerr << "Unsupported command. Use sparx --help.\n"; return 2; } + std::string input, endpoint, model = "local", session = "cli", request; + bool single = false; + if (!demo) for (int i = 2; i < argc; ++i) { + const std::string flag = argv[i]; + if (flag == "--help") { help(); return 0; } + if (i + 1 >= argc) { std::cerr << "Missing option value\n"; return 2; } + const std::string value = argv[++i]; + if (flag == "--input") { input = value; single = true; } + else if (flag == "--endpoint") endpoint = value; + else if (flag == "--model") model = value; + else if (flag == "--session") session = value; + else if (flag == "--request-id") request = value; + else { std::cerr << "Unknown option: " << flag << '\n'; return 2; } + } + if (!single && !request.empty()) { std::cerr << "--request-id requires --input\n"; return 2; } + Runtime runtime; + int temperature = 20; + const Json empty_schema{{"type", "object"}, {"properties", Json::object()}, {"additionalProperties", false}}; + const Json temperature_schema{{"type", "object"}, {"properties", {{"temperature", {{"type", "integer"}, {"minimum", 16}, {"maximum", 30}}}}}, {"required", {"temperature"}}, {"additionalProperties", false}}; + if (!runtime.registerTool({"ac.set_temperature", "Set demo AC temperature in Celsius", temperature_schema, + [&temperature](const Json& args) { temperature = args.at("temperature").get(); return Result::success({{"temperature", temperature}, {"simulated", true}}); }}) || + !runtime.registerTool({"vehicle.status", "Read demo vehicle status", empty_schema, + [&temperature](const Json&) { return Result::success({{"temperature", temperature}, {"simulated", true}}); }}) || + !runtime.registerSkill("set AC to 22 degrees", "ac.set_temperature", {{"temperature", 22}}) || + !runtime.registerSkill("把空调调到22度", "ac.set_temperature", {{"temperature", 22}}) || + !runtime.registerSkill("vehicle status", "vehicle.status", Json::object())) { + std::cerr << "Cannot initialize reference example\n"; return 1; + } + if (!endpoint.empty()) { +#ifdef MASTER_AGENT_HAS_HTTP + if (endpoint.find("://") == std::string::npos) endpoint = "http://" + endpoint; + if (endpoint.find('/', endpoint.find("://") + 3) == std::string::npos) endpoint += "/v1/chat/completions"; + HttpModelConfig config; config.endpoint = endpoint; config.model = model; + runtime.setModel(makeHttpModel(std::move(config))); +#else + std::cerr << "HTTP support is disabled; rebuild with MASTER_AGENT_ENABLE_HTTP=ON\n"; return 2; +#endif + } + if (demo) { + auto set = runtime.run({"demo", "1", "set AC to 22 degrees"}); + auto read = runtime.run({"demo", "2", "vehicle status"}); + std::cout << encode(set).dump() << '\n' << encode(read).dump() << '\n'; + return set && read ? 0 : 1; + } + if (single) { + auto result = runtime.run({session, request.empty() ? "1" : request, input}); + std::cout << encode(result).dump() << '\n'; return result ? 0 : 1; + } + unsigned sequence = 0; + bool success = true; + while (std::getline(std::cin, input)) { + auto result = runtime.run({session, std::to_string(++sequence), input}); + std::cout << encode(result).dump() << std::endl; + success = success && result.ok(); + } + return success ? 0 : 1; +} diff --git a/cli/src/sparx_cloud_backend.cpp b/cli/src/sparx_cloud_backend.cpp index 597cba1..2381ac2 100644 --- a/cli/src/sparx_cloud_backend.cpp +++ b/cli/src/sparx_cloud_backend.cpp @@ -284,15 +284,11 @@ std::future OpenAICompatBackend::inferAsync( const std::string& user_prompt, const std::string& system_prompt) const { - // Capture by value for thread safety - auto config = config_; - auto api_key = resolved_api_key_; - auto endpoint = config_.endpoint; - auto body = buildRequestBody(user_prompt, system_prompt); - - return std::async(std::launch::async, [this, body]() { - return doHttpPost(body); - }); + auto task = std::make_shared>( + [backend = *this, user_prompt, system_prompt] { return backend.infer(user_prompt, system_prompt); }); + auto future = task->get_future(); + std::thread([task] { (*task)(); }).detach(); + return future; } void OpenAICompatBackend::inferWithCallback( @@ -300,12 +296,9 @@ void OpenAICompatBackend::inferWithCallback( const std::string& system_prompt, CloudCallback callback) const { - auto body = buildRequestBody(user_prompt, system_prompt); - - // Fire in a detached thread (harness manages lifetime via deadline) - std::thread([this, body, callback]() { - auto result = doHttpPost(body); - if (callback) callback(std::move(result)); + std::thread([backend = *this, user_prompt, system_prompt, callback = std::move(callback)] { + try { auto result = backend.infer(user_prompt, system_prompt); if (callback) callback(std::move(result)); } + catch (...) { /* A callback exception must not terminate the host process. */ } }).detach(); } @@ -336,9 +329,11 @@ std::future MockCloudBackend::inferAsync( const std::string& user_prompt, const std::string& system_prompt) const { - return std::async(std::launch::async, [this, user_prompt, system_prompt]() { - return infer(user_prompt, system_prompt); - }); + auto task = std::make_shared>( + [backend = *this, user_prompt, system_prompt] { return backend.infer(user_prompt, system_prompt); }); + auto future = task->get_future(); + std::thread([task] { (*task)(); }).detach(); + return future; } void MockCloudBackend::inferWithCallback( @@ -346,9 +341,9 @@ void MockCloudBackend::inferWithCallback( const std::string& system_prompt, CloudCallback callback) const { - std::thread([this, user_prompt, system_prompt, callback]() { - auto result = infer(user_prompt, system_prompt); - if (callback) callback(std::move(result)); + std::thread([backend = *this, user_prompt, system_prompt, callback]() { + try { auto result = backend.infer(user_prompt, system_prompt); if (callback) callback(std::move(result)); } + catch (...) { } }).detach(); } diff --git a/cli/src/sparx_pipeline_harness.cpp b/cli/src/sparx_pipeline_harness.cpp index 9883156..aaaa87c 100644 --- a/cli/src/sparx_pipeline_harness.cpp +++ b/cli/src/sparx_pipeline_harness.cpp @@ -25,6 +25,7 @@ #include #include #include +#include namespace sparx { namespace harness { @@ -82,6 +83,11 @@ void PipelineHarness::loadConfig(const std::string& yaml_path) { } void PipelineHarness::resolveActiveComponents() { + active_prompt_engine_.reset(); + active_cloud_backend_.reset(); + active_arbiter_.reset(); + active_scorer_.reset(); + active_local_.reset(); // Resolve prompt engine if (auto it = prompt_engines_.find(config_.prompt_engine); it != prompt_engines_.end()) { active_prompt_engine_ = it->second; @@ -116,6 +122,7 @@ void PipelineHarness::setCloudEnabled(bool enabled) { } bool PipelineHarness::isCloudEnabled() const { + std::lock_guard lock(mutex_); return config_.cloud_enabled; } @@ -138,6 +145,7 @@ void PipelineHarness::setActiveArbiter(const std::string& name) { } bool PipelineHarness::isReady() const { + std::lock_guard lock(mutex_); return active_prompt_engine_ != nullptr && active_arbiter_ != nullptr && active_scorer_ != nullptr && @@ -168,118 +176,112 @@ PreScoreSignals PipelineHarness::buildPreScoreSignals( return signals; } -bool PipelineHarness::shouldFireCloud(const ConfidenceScore& pre_score) const { - if (!config_.cloud_enabled) return false; - if (!active_cloud_backend_ || !active_cloud_backend_->isReady()) return false; - // High confidence → skip cloud - if (pre_score.overall >= config_.confidence_thresholds.high) return false; - - // Below high threshold → fire cloud - return true; +namespace { +// Promise futures do not join a worker during destruction. Each worker owns its +// captured backends and one shared slot, never a pointer to the harness. At most +// one local and one cloud call can remain outstanding per harness, even when an +// adapter ignores its timeout. Late results are discarded, not re-arbitrated. +template +std::future launchBounded(const std::shared_ptr>& busy, F work) { + if (busy->exchange(true)) return {}; + auto promise = std::make_shared>(); + auto future = promise->get_future(); + try { + std::thread([busy, promise, work = std::move(work)]() mutable { + try { promise->set_value(work()); } + catch (...) { promise->set_exception(std::current_exception()); } + busy->store(false); + }).detach(); + } catch (...) { busy->store(false); throw; } + return future; +} +template +std::optional completed(std::future& future, std::chrono::steady_clock::time_point deadline) { + if (!future.valid() || future.wait_until(deadline) != std::future_status::ready) return {}; + try { return future.get(); } catch (...) { return {}; } +} } PipelineResponse PipelineHarness::execute(const PipelineRequest& request) { + const auto start = std::chrono::steady_clock::now(); PipelineResponse response; - auto pipeline_start = std::chrono::steady_clock::now(); - - if (!isReady()) { + std::shared_ptr prompt; + std::shared_ptr cloud; + std::shared_ptr arbiter; + std::shared_ptr scorer; + std::shared_ptr local; + HarnessConfig config; + { + std::lock_guard lock(mutex_); + prompt = active_prompt_engine_; cloud = active_cloud_backend_; + arbiter = active_arbiter_; scorer = active_scorer_; local = active_local_; + config = config_; + } + if (!prompt || !arbiter || !scorer || !local) { response.result.content = "Pipeline not initialized"; response.result.source = ArbiterOutput::Source::Fallback; response.result.reason = "harness not ready"; return response; } - - // ── Step 1: Pre-score confidence ── - auto pre_signals = buildPreScoreSignals(request); - auto pre_score = active_scorer_->preScore(pre_signals); - response.confidence = pre_score; - - // ── Step 2: Decide whether to fire cloud ── - bool fire_cloud = shouldFireCloud(pre_score); - response.cloud_fired = fire_cloud; - response.prompt_engine_used = active_prompt_engine_->name(); - - // ── Step 3: If firing cloud, compress prompt and launch async ── + const auto deadline = start + std::chrono::milliseconds(std::max(0, arbiter->getDeadline(request.intent_type))); + const auto signals = buildPreScoreSignals(request); + const auto pre = scorer->preScore(signals); + response.confidence = pre; + response.prompt_engine_used = prompt->name(); std::future cloud_future; - if (fire_cloud) { - auto compressed = active_prompt_engine_->compress( - request.user_input, request.history, request.context_vars); - response.cloud_input_tokens = compressed.estimated_tokens; - - cloud_future = active_cloud_backend_->inferAsync( - compressed.user_prompt, compressed.system_prompt); + if (config.cloud_enabled && config.arbiter_config.strategy != ArbiterStrategy::LocalOnly && + cloud && cloud->isReady() && pre.overall < config.confidence_thresholds.high) { + const auto compressed = prompt->compress(request.user_input, request.history, request.context_vars); + cloud_future = launchBounded(cloud_busy_, [cloud, compressed] { + return cloud->infer(compressed.user_prompt, compressed.system_prompt); + }); + response.cloud_fired = cloud_future.valid(); + if (response.cloud_fired) response.cloud_input_tokens = compressed.estimated_tokens; } - - // ── Step 4: Run local inference (blocking) ── - std::string local_prompt = active_prompt_engine_->renderLocal( - request.user_input, request.history, request.context_vars); - - auto local_start = std::chrono::steady_clock::now(); - auto local_result = active_local_->infer(local_prompt); - auto local_elapsed = std::chrono::steady_clock::now() - local_start; - local_result.latency_ms = static_cast( - std::chrono::duration_cast(local_elapsed).count()); - response.local_latency_ms = local_result.latency_ms; - - // ── Step 5: Post-score local result ── - if (local_result.success && active_scorer_) { - auto post_signals = active_local_->getLastPostSignals(); - auto post_score = active_scorer_->postScore(pre_signals, post_signals); - local_result.confidence = post_score; - response.confidence = post_score; + const auto local_prompt = prompt->renderLocal(request.user_input, request.history, request.context_vars); + auto local_future = launchBounded(local_busy_, [local, scorer, signals, local_prompt] { + const auto begin = std::chrono::steady_clock::now(); + auto result = local->infer(local_prompt); + result.latency_ms = static_cast(std::chrono::duration_cast(std::chrono::steady_clock::now() - begin).count()); + if (result.success) result.confidence = scorer->postScore(signals, local->getLastPostSignals()); + return result; + }); + const auto local_result = completed(local_future, deadline); + const auto cloud_result = completed(cloud_future, deadline); + if (local_result) { + response.local_latency_ms = local_result->latency_ms; + response.confidence = local_result->confidence; } - - // ── Step 6: Wait for cloud result (bounded by deadline) ── - std::optional cloud_result; - if (fire_cloud && cloud_future.valid()) { - int deadline = active_arbiter_->getDeadline(request.intent_type); - - // Subtract time already spent on local inference - int remaining_ms = deadline - local_result.latency_ms; - remaining_ms = std::max(remaining_ms, 0); - - auto status = cloud_future.wait_for(std::chrono::milliseconds(remaining_ms)); - if (status == std::future_status::ready) { - cloud_result = cloud_future.get(); - response.cloud_latency_ms = cloud_result->latency_ms; - response.cloud_output_tokens = cloud_result->output_tokens; - } - // If timeout: cloud_result stays nullopt → arbiter uses local only + if (cloud_result) { + response.cloud_latency_ms = cloud_result->latency_ms; + response.cloud_output_tokens = cloud_result->output_tokens; } - - // ── Step 7: Arbiter picks final output ── - std::optional local_opt; - if (local_result.success || !local_result.content.empty()) { - local_opt = local_result; - } - - response.result = active_arbiter_->arbitrate(local_opt, cloud_result, request.intent_type); - - auto pipeline_elapsed = std::chrono::steady_clock::now() - pipeline_start; - response.total_latency_ms = static_cast( - std::chrono::duration_cast(pipeline_elapsed).count()); + response.result = arbiter->arbitrate(local_result, cloud_result, request.intent_type); + response.total_latency_ms = static_cast(std::chrono::duration_cast(std::chrono::steady_clock::now() - start).count()); response.result.total_latency_ms = response.total_latency_ms; - return response; } PipelineResponse PipelineHarness::executeCloudOnly(const PipelineRequest& request) { + std::shared_ptr prompt; + std::shared_ptr cloud; + { std::lock_guard lock(mutex_); prompt = active_prompt_engine_; cloud = active_cloud_backend_; } PipelineResponse response; auto start = std::chrono::steady_clock::now(); - if (!active_prompt_engine_ || !active_cloud_backend_) { + if (!prompt || !cloud) { response.result.content = "Cloud path not configured"; response.result.source = ArbiterOutput::Source::Fallback; return response; } - auto compressed = active_prompt_engine_->compress( + auto compressed = prompt->compress( request.user_input, request.history, request.context_vars); response.cloud_input_tokens = compressed.estimated_tokens; - response.prompt_engine_used = active_prompt_engine_->name(); + response.prompt_engine_used = prompt->name(); - auto cloud_result = active_cloud_backend_->infer( + auto cloud_result = cloud->infer( compressed.user_prompt, compressed.system_prompt); response.cloud_latency_ms = cloud_result.latency_ms; @@ -303,20 +305,23 @@ PipelineResponse PipelineHarness::executeCloudOnly(const PipelineRequest& reques } PipelineResponse PipelineHarness::executeLocalOnly(const PipelineRequest& request) { + std::shared_ptr prompt; + std::shared_ptr local; + { std::lock_guard lock(mutex_); prompt = active_prompt_engine_; local = active_local_; } PipelineResponse response; auto start = std::chrono::steady_clock::now(); - if (!active_prompt_engine_ || !active_local_) { + if (!prompt || !local) { response.result.content = "Local path not configured"; response.result.source = ArbiterOutput::Source::Fallback; return response; } - std::string prompt = active_prompt_engine_->renderLocal( + std::string local_prompt = prompt->renderLocal( request.user_input, request.history, request.context_vars); - response.prompt_engine_used = active_prompt_engine_->name(); + response.prompt_engine_used = prompt->name(); - auto local_result = active_local_->infer(prompt); + auto local_result = local->infer(local_prompt); response.local_latency_ms = local_result.latency_ms; response.cloud_fired = false; diff --git a/cli/src/sparx_speculative.cpp b/cli/src/sparx_speculative.cpp index 093cb0b..9b84888 100644 --- a/cli/src/sparx_speculative.cpp +++ b/cli/src/sparx_speculative.cpp @@ -1355,8 +1355,10 @@ void IntentPredictor::observe(const IntentRecord& record) { const auto& prev = recent_history_.back().intent_name; // Compute age-weighted increment - int64_t age_seconds = now - recent_history_.back().timestamp_utc; - double age_days = static_cast(age_seconds) / 86400.0; + // Missing or out-of-order timestamps must not amplify counts or overflow. + double age_seconds = (now > 0 && recent_history_.back().timestamp_utc > 0) + ? std::max(0.0, static_cast(now) - static_cast(recent_history_.back().timestamp_utc)) : 0.0; + double age_days = age_seconds / 86400.0; float weight = static_cast(std::exp(-decay_lambda * age_days)); // Store as float counts (fractional weights) @@ -1412,9 +1414,8 @@ std::vector IntentPredictor::predict( const auto& current = recent_history_.back(); const auto current_intent = current.intent_name; - auto now_hour = static_cast( - std::chrono::duration_cast( - std::chrono::system_clock::now().time_since_epoch()).count() % 24); + // The observation carries local context; the host UTC clock may differ. + const auto now_hour = current.hour_of_day; // Collect candidates with scores from n-gram models (Level 1 & 2) std::map ngram_scores; diff --git a/cmake/MasterAgentConfig.cmake.in b/cmake/MasterAgentConfig.cmake.in index 3d1640c..f416692 100644 --- a/cmake/MasterAgentConfig.cmake.in +++ b/cmake/MasterAgentConfig.cmake.in @@ -2,6 +2,9 @@ include(CMakeFindDependencyMacro) find_dependency(Threads) +if(@MASTER_AGENT_ENABLE_HTTP@) + find_dependency(CURL) +endif() include("${CMAKE_CURRENT_LIST_DIR}/MasterAgentTargets.cmake") diff --git a/cmake/ReferenceTests.cmake b/cmake/ReferenceTests.cmake new file mode 100644 index 0000000..31ef3cf --- /dev/null +++ b/cmake/ReferenceTests.cmake @@ -0,0 +1,27 @@ +add_executable(test_reference_runtime tests/test_reference_runtime.cpp) +target_link_libraries(test_reference_runtime PRIVATE MasterAgent::Core) +add_test(NAME test_reference_runtime COMMAND test_reference_runtime) +if(MASTER_AGENT_BUILD_CLI) + add_test(NAME reference_cli_demo COMMAND sparx demo automotive) + set_tests_properties(reference_cli_demo PROPERTIES PASS_REGULAR_EXPRESSION "\"temperature\":22") + add_test(NAME reference_cli_unknown COMMAND sparx unsupported-command) + set_tests_properties(reference_cli_unknown PROPERTIES WILL_FAIL TRUE) +endif() +add_test(NAME installed_consumer + COMMAND ${CMAKE_COMMAND} + -DSOURCE_DIR=${CMAKE_CURRENT_SOURCE_DIR} + -DBUILD_DIR=${CMAKE_CURRENT_BINARY_DIR} + -DTEST_CONFIG=$ + -P ${CMAKE_CURRENT_SOURCE_DIR}/cmake/TestConsumer.cmake) +find_package(Python3 COMPONENTS Interpreter REQUIRED) +if(MASTER_AGENT_ENABLE_HTTP AND MASTER_AGENT_BUILD_CLI) + add_executable(http_model_probe tests/http_model_probe.cpp) + target_link_libraries(http_model_probe PRIVATE MasterAgent::Http) + add_test(NAME http_model_contract + COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/tests/test_http_model.py $ $) +endif() +add_executable(test_check_failure tests/check_failure.cpp) +add_test(NAME release_assertions_are_active COMMAND test_check_failure) +set_tests_properties(release_assertions_are_active PROPERTIES WILL_FAIL TRUE) +add_test(NAME evaluation_script_contract + COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/tests/test_eval_runner.py ${CMAKE_CURRENT_SOURCE_DIR}/eval/run_all.sh) diff --git a/cmake/TestConsumer.cmake b/cmake/TestConsumer.cmake new file mode 100644 index 0000000..6845201 --- /dev/null +++ b/cmake/TestConsumer.cmake @@ -0,0 +1,15 @@ +function(checked) + execute_process(COMMAND ${ARGV} RESULT_VARIABLE result) + if(NOT result EQUAL 0) + message(FATAL_ERROR "Consumer check failed: ${ARGV}") + endif() +endfunction() +set(stage "${BUILD_DIR}/consumer-stage") +set(relocated "${BUILD_DIR}/consumer-relocated") +file(REMOVE_RECURSE "${stage}" "${relocated}" "${BUILD_DIR}/consumer-build") +checked("${CMAKE_COMMAND}" --install "${BUILD_DIR}" --prefix "${stage}" --config "${TEST_CONFIG}") +file(RENAME "${stage}" "${relocated}") +checked("${CMAKE_COMMAND}" -S "${SOURCE_DIR}/tests/consumer" -B "${BUILD_DIR}/consumer-build" + "-DCMAKE_PREFIX_PATH=${relocated}") +checked("${CMAKE_COMMAND}" --build "${BUILD_DIR}/consumer-build" --config "${TEST_CONFIG}") +checked("${CMAKE_CTEST_COMMAND}" --test-dir "${BUILD_DIR}/consumer-build" -C "${TEST_CONFIG}" --output-on-failure) diff --git a/config/harness.yaml b/config/harness.yaml index c5fbe2e..d2b7d81 100644 --- a/config/harness.yaml +++ b/config/harness.yaml @@ -14,7 +14,7 @@ harness: cloud_backend: "openai_compatible" # openai_compatible | mock arbiter: "cloud_prefer" # cloud_prefer | latency_first | confidence | local_only confidence_scorer: "heuristic" # heuristic - cloud_enabled: true # Master switch for cloud path + cloud_enabled: false # Master switch for cloud path trace_decisions: false # Log every routing/arbitration decision # ─── Cloud Backend Configuration ───────────────────────────────────────────── diff --git a/core/reference_runtime.cpp b/core/reference_runtime.cpp new file mode 100644 index 0000000..ba28c81 --- /dev/null +++ b/core/reference_runtime.cpp @@ -0,0 +1,177 @@ +#include "master_agent/runtime/reference_runtime.h" +#include +#include +#include + +namespace master_agent::reference { +namespace { +template Result fail(const std::string& code, const std::string& text) { + return Result::failure({code, text, "", 400}); +} +bool primitive(const Json& value, const std::string& type) { + if (type == "string") return value.is_string(); + if (type == "boolean") return value.is_boolean(); + if (type == "integer") return value.is_number_integer(); + if (type == "number") return value.is_number() && std::isfinite(value.get()); + return false; +} +Status schemaValid(const Json& schema) { + if (!schema.is_object() || schema.value("type", "") != "object" || + !schema.contains("properties") || !schema["properties"].is_object() || + !schema.contains("additionalProperties") || schema["additionalProperties"] != false) + return Status::Error("INVALID_SCHEMA", "Expected object schema with additionalProperties:false"); + const std::set root_keys{"type", "properties", "required", "additionalProperties"}; + for (const auto& item : schema.items()) + if (!root_keys.count(item.key())) return Status::Error("INVALID_SCHEMA", "Unsupported schema keyword: " + item.key()); + for (const auto& item : schema["properties"].items()) { + const auto& field = item.value(); + if (!field.is_object()) return Status::Error("INVALID_SCHEMA", "Property must be an object"); + const auto type = field.value("type", ""); + if (type != "string" && type != "boolean" && type != "integer" && type != "number") + return Status::Error("INVALID_SCHEMA", "Only primitive property types are supported"); + for (const auto& keyword : field.items()) { + const auto& k = keyword.key(); + if (k != "type" && k != "description" && k != "enum" && k != "minimum" && k != "maximum") + return Status::Error("INVALID_SCHEMA", "Unsupported property keyword: " + k); + } + if (field.contains("description") && !field["description"].is_string()) + return Status::Error("INVALID_SCHEMA", "description must be a string"); + for (const auto* bound : {"minimum", "maximum"}) + if (field.contains(bound) && ((type != "number" && type != "integer") || !primitive(field[bound], "number"))) + return Status::Error("INVALID_SCHEMA", "Invalid numeric bound"); + if (field.contains("minimum") && field.contains("maximum") && field["minimum"] > field["maximum"]) + return Status::Error("INVALID_SCHEMA", "minimum exceeds maximum"); + if (field.contains("enum")) { + if (!field["enum"].is_array() || field["enum"].empty()) return Status::Error("INVALID_SCHEMA", "enum must be a nonempty array"); + for (const auto& value : field["enum"]) + if (!primitive(value, type)) return Status::Error("INVALID_SCHEMA", "enum type mismatch"); + } + } + if (schema.contains("required")) { + if (!schema["required"].is_array()) return Status::Error("INVALID_SCHEMA", "required must be an array"); + for (const auto& name : schema["required"]) + if (!name.is_string() || !schema["properties"].contains(name.get())) + return Status::Error("INVALID_SCHEMA", "Unknown required property"); + } + return Status::Ok(); +} +Status validate(const Json& args, const Json& schema) { + if (!args.is_object()) return Status::Error("INVALID_ARGUMENTS", "Arguments must be an object"); + for (const auto& key : schema.value("required", Json::array())) + if (!args.contains(key.get())) return Status::Error("INVALID_ARGUMENTS", "Missing field: " + key.get()); + for (const auto& arg : args.items()) { + if (!schema["properties"].contains(arg.key())) return Status::Error("INVALID_ARGUMENTS", "Unknown field: " + arg.key()); + const auto& field = schema["properties"][arg.key()]; + if (!primitive(arg.value(), field["type"].get())) return Status::Error("INVALID_ARGUMENTS", "Wrong type: " + arg.key()); + if (field.contains("enum") && std::find(field["enum"].begin(), field["enum"].end(), arg.value()) == field["enum"].end()) + return Status::Error("INVALID_ARGUMENTS", "Value outside enum: " + arg.key()); + if ((field.contains("minimum") && arg.value() < field["minimum"]) || + (field.contains("maximum") && arg.value() > field["maximum"])) + return Status::Error("INVALID_ARGUMENTS", "Value outside range: " + arg.key()); + } + return Status::Ok(); +} +} // namespace + +Runtime::Runtime(Limits limits) : limits_(limits) {} +Status Runtime::registerTool(Tool tool) { + std::unique_lock lock(mutex_, std::try_to_lock); + if (!lock) return Status::Error("BUSY"); + if (tool.name.empty() || !tool.execute || tools_.count(tool.name)) return Status::Error("INVALID_TOOL", "Tool requires a unique name and handler"); + try { auto status = schemaValid(tool.parameters); if (!status) return status; } + catch (const Json::exception& e) { return Status::Error("INVALID_SCHEMA", e.what()); } + tools_.emplace(tool.name, std::move(tool)); + return Status::Ok(); +} +Status Runtime::registerSkill(std::string phrase, std::string tool, Json arguments) { + std::unique_lock lock(mutex_, std::try_to_lock); + if (!lock) return Status::Error("BUSY"); + if (phrase.empty() || skills_.count(phrase) || !tools_.count(tool)) return Status::Error("INVALID_SKILL", "Skill needs a unique phrase and registered tool"); + auto status = validate(arguments, tools_.at(tool).parameters); + if (!status) return status; + skills_.emplace(std::move(phrase), Skill{std::move(tool), std::move(arguments)}); + return Status::Ok(); +} +Status Runtime::setModel(ModelHandler model) { + std::unique_lock lock(mutex_, std::try_to_lock); + if (!lock) return Status::Error("BUSY"); + model_ = std::move(model); + return Status::Ok(); +} +Status Runtime::clearSession(const std::string& id) { + std::unique_lock lock(mutex_, std::try_to_lock); + if (!lock) return Status::Error("BUSY"); + sessions_.erase(id); + return Status::Ok(); +} +Result Runtime::run(const Turn& turn) { + std::unique_lock lock(mutex_, std::try_to_lock); + if (!lock) return fail("BUSY", "Runtime is executing another operation"); + if (turn.session_id.empty() || turn.request_id.empty() || turn.input.empty() || turn.input.size() > 65536 || turn.session_id.size() > 256 || turn.request_id.size() > 256) + return fail("INVALID_REQUEST", "Nonempty bounded session, request and input are required"); + if (!sessions_.count(turn.session_id) && sessions_.size() >= limits_.sessions) return fail("RESOURCE_LIMIT", "Session limit reached"); + auto& session = sessions_[turn.session_id]; + if (auto it = session.requests.find(turn.request_id); it != session.requests.end()) { + if (it->second.input != turn.input) return fail("REQUEST_CONFLICT", "Request ID was already used for different input"); + auto result = it->second.result; + if (result) result.value->replayed = true; + return result; + } + if (session.requests.size() >= limits_.requests_per_session) return fail("RESOURCE_LIMIT", "Request limit reached; start a new session"); + auto result = execute(turn, session); + // Cache failures too: an UNKNOWN tool outcome must never trigger an automatic retry. + session.requests.emplace(turn.request_id, Cached{turn.input, result}); + if (result) { + session.history.push_back({{"role", "user"}, {"content", turn.input}}); + session.history.push_back({{"role", "assistant"}, {"content", result.value->output.dump()}}); + while (session.history.size() / 2 > limits_.history_turns) session.history.erase(session.history.begin(), session.history.begin() + 2); + } + return result; +} +Result Runtime::execute(const Turn& turn, Session& session) { + Reply reply{turn.session_id, turn.request_id, "skill", "", Json{}, false}; + Json arguments; + if (auto it = skills_.find(turn.input); it != skills_.end()) { + reply.tool = it->second.tool; + arguments = it->second.arguments; + } else { + if (!model_) return fail("NO_ROUTE", "No matching skill and no model configured"); + Json definitions = Json::array(); + for (const auto& entry : tools_) definitions.push_back({{"name", entry.first}, {"description", entry.second.description}, {"parameters", entry.second.parameters}}); + Json messages = Json::array({{{"role", "system"}, {"content", "Return only JSON: {\"response\":\"text\"} or {\"tool\":\"name\",\"arguments\":{...}}. Use only these tools: " + definitions.dump()}}}); + for (const auto& message : session.history) messages.push_back(message); + messages.push_back({{"role", "user"}, {"content", turn.input}}); + Result model_result; + try { model_result = model_(messages); } + catch (const std::exception& e) { return fail("MODEL_ERROR", e.what()); } + catch (...) { return fail("MODEL_ERROR", "Model callback threw"); } + if (!model_result) return Result::failure(model_result.error.value_or(StructuredError{"MODEL_ERROR", "Model returned no result", "", 500})); + if (model_result.value->size() > 65536) return fail("INVALID_MODEL_OUTPUT", "Model output exceeds 64 KiB"); + auto decision = Json::parse(*model_result, nullptr, false); + if (!decision.is_object()) return fail("INVALID_MODEL_OUTPUT", "Expected a JSON decision object"); + if (decision.size() == 1 && decision.contains("response") && decision["response"].is_string()) { + reply.route = "model"; + reply.output = decision["response"]; + return Result::success(std::move(reply)); + } + if (decision.size() != 2 || !decision.contains("tool") || !decision["tool"].is_string() || !decision.contains("arguments")) + return fail("INVALID_MODEL_OUTPUT", "Expected response or tool/arguments"); + reply.route = "model_tool"; + reply.tool = decision["tool"].get(); + arguments = decision["arguments"]; + } + auto tool = tools_.find(reply.tool); + if (tool == tools_.end()) return fail("UNKNOWN_TOOL", "Model selected an unregistered tool"); + auto valid = validate(arguments, tool->second.parameters); + if (!valid) return fail(valid.error_code, valid.error_message); + try { + auto result = tool->second.execute(arguments); + if (!result) return Result::failure(result.error.value_or(StructuredError{"TOOL_ERROR", "Tool returned no result", "", 500})); + reply.output = *result; + } catch (...) { + // The handler may have performed its side effect before throwing. + return fail("UNKNOWN", "Tool threw; reconcile its outcome before issuing a new request"); + } + return Result::success(std::move(reply)); +} +} // namespace master_agent::reference diff --git a/docs/reference_runtime.md b/docs/reference_runtime.md new file mode 100644 index 0000000..070a7c3 --- /dev/null +++ b/docs/reference_runtime.md @@ -0,0 +1,85 @@ +# Public reference runtime contract + +The supported public entry point is `master_agent::reference::Runtime`, linked +through `MasterAgent::Core`. This alpha API does not implement legacy private +kernel factories. The HTTP adapter is a separate `MasterAgent::Http` target. + +## Execution + +1. Validate bounded, nonempty session ID, request ID, and input. +2. Return a cached outcome for a repeated request ID with identical input. +3. Match an exact phrase skill before considering the model. +4. Otherwise call the supplied model callback with system/tool descriptions and + bounded conversation history; require a single JSON decision. +5. Resolve the registered tool and validate every argument before invoking it. +6. Return the actual result and record the outcome in the in-memory session. + +A model can return either `{"response":"text"}` or +`{"tool":"registered.name","arguments":{...}}`. Mixed, malformed, unknown-tool, +and invalid-argument output is rejected. This release performs at most one tool +call per turn; it does not generate or execute arbitrary code, multi-step DAGs, +or a model-driven retry loop. + +## Tools and schemas + +Tool callbacks are trusted native application code. They are not sandboxed and +must implement domain authorization before performing a side effect. Register +only tools appropriate to the caller/session. This runtime does not supply a +multi-tenant authorization system or MCP server transport. + +Schemas must use `type:object`, `properties`, and +`additionalProperties:false`. `required` is optional. Property types are +`string`, `boolean`, `integer`, and `number`; optional constraints are `enum`, +`minimum`, and `maximum`, plus descriptive strings. Unsupported keywords, +nested objects/arrays, malformed schemas, and unknown required fields are +rejected at registration. Do not assume full JSON Schema conformance. + +## Sessions, replay, and failure + +Defaults: 32 sessions, 256 recorded requests per session, 8 successful turns of +model history, 64 KiB input/model text, and 256-byte IDs. A full session returns +`RESOURCE_LIMIT`; it never silently evicts request IDs and repeats a tool. +`clearSession()` explicitly removes both history and replay protection. + +Request IDs are scoped to the session. Reusing an ID with different input +returns `REQUEST_CONFLICT`. Both success and failure are cached. For a deliberate +retry, the caller must inspect the outcome and provide a new ID. All state is +memory-only: deduplication is not guaranteed across process restarts or separate +runtime instances. Do not use it as a durable exactly-once execution guarantee. + +A handler exception returns `UNKNOWN`, because its side effect may already have +occurred. Repeating that ID returns the same error without invoking the handler +again. Tool handlers should return explicit `Result` failures when the outcome +is known. Durable recovery, reconciliation storage, and journal encryption are +future work, not hidden behavior of this release. + +## Concurrency and timeouts + +Runtime methods use a nonblocking instance lock. Concurrent or reentrant calls +return `BUSY`; they do not race or deadlock on a recursive tool call. Tools and +model callbacks run synchronously. The host must bound their runtime; native +callbacks cannot be safely preempted by this library. + +The optional libcurl model callback has a finite HTTP timeout (30 seconds by +default), disables redirects, verifies TLS using libcurl defaults, and limits +responses to 1 MiB. Credentials are supplied through the SDK config, not a CLI +argument. Endpoints are explicit; the reference CLI defaults to no model. + +The separate experimental edge/cloud harness uses a common inference deadline +and retains shared backend ownership for late workers. It permits at most one +outstanding local and one outstanding cloud inference per instance; a timed-out +worker must finish before that lane admits another task. It does not forcibly +cancel third-party synchronous callbacks. Prompt compression, scoring and +arbitration callbacks must be fast and thread-safe. Direct `executeLocalOnly` +and `executeCloudOnly` methods remain synchronous diagnostic calls. + +## Verification boundaries + +CI tests the real HTTP client against a loopback HTTP protocol fixture, including +Unicode, malformed payloads, unknown tools, invalid arguments and HTTP failures. +That proves transport and dispatch behavior, not model quality. A real GGUF model +and supported hardware are required for model/NPU performance claims. + +The legacy learning store uses XOR obfuscation and the bundled legacy journal +contains incomplete persistence methods. Neither is used by this reference +runtime. They must not be used as production encryption or crash recovery. diff --git a/eval/run_all.sh b/eval/run_all.sh index 87e894d..730d1bf 100755 --- a/eval/run_all.sh +++ b/eval/run_all.sh @@ -13,8 +13,8 @@ set -euo pipefail SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" PROJECT_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd)" -BUILD_DIR="${PROJECT_ROOT}/build" -RESULTS_DIR="${SCRIPT_DIR}/results" +BUILD_DIR="${SPARX_BUILD_DIR:-${PROJECT_ROOT}/build}" +RESULTS_DIR="${SPARX_RESULTS_DIR:-${BUILD_DIR}/eval-results}" VERBOSE_FLAG="" for arg in "$@"; do @@ -27,12 +27,11 @@ done # --- Build --- echo "=== Building evaluation binaries ===" mkdir -p "${BUILD_DIR}" -cmake -S "${PROJECT_ROOT}" -B "${BUILD_DIR}" -DCMAKE_BUILD_TYPE=Release 2>&1 | tail -5 +cmake -S "${PROJECT_ROOT}" -B "${BUILD_DIR}" -DCMAKE_BUILD_TYPE=Release -DBUILD_EVAL=ON 2>&1 | tail -5 cmake --build "${BUILD_DIR}" --target eval_speculation eval_mesh eval_formal eval_learning eval_constrained -j"$(nproc 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || echo 4)" echo "" # --- Prepare results directory --- -rm -rf "${RESULTS_DIR}" mkdir -p "${RESULTS_DIR}" EVALS=( @@ -49,7 +48,7 @@ declare -a STATUSES=() # --- Run each evaluation --- for eval_name in "${EVALS[@]}"; do - BINARY="${BUILD_DIR}/${eval_name}" + BINARY="${BUILD_DIR}/eval/${eval_name}" OUTPUT_FILE="${RESULTS_DIR}/${eval_name}.txt" echo "--- Running ${eval_name} ---" @@ -57,16 +56,16 @@ for eval_name in "${EVALS[@]}"; do if "${BINARY}" ${VERBOSE_FLAG} > "${OUTPUT_FILE}" 2>&1; then echo " PASSED (output: ${OUTPUT_FILE})" STATUSES+=("PASS") - ((PASS_COUNT++)) + PASS_COUNT=$((PASS_COUNT + 1)) else echo " FAILED (exit code $?, output: ${OUTPUT_FILE})" STATUSES+=("FAIL") - ((FAIL_COUNT++)) + FAIL_COUNT=$((FAIL_COUNT + 1)) fi else echo " SKIPPED (binary not found: ${BINARY})" STATUSES+=("SKIP") - ((FAIL_COUNT++)) + FAIL_COUNT=$((FAIL_COUNT + 1)) fi done @@ -123,3 +122,6 @@ for eval_name in "${EVALS[@]}"; do done echo "Summary written to: ${SUMMARY}" + +# A skipped or failed evaluation must fail automation. +[[ "$FAIL_COUNT" -eq 0 ]] diff --git a/examples/reference_agent/CMakeLists.txt b/examples/reference_agent/CMakeLists.txt new file mode 100644 index 0000000..ad06767 --- /dev/null +++ b/examples/reference_agent/CMakeLists.txt @@ -0,0 +1,7 @@ +cmake_minimum_required(VERSION 3.18) +project(MasterAgentConsumer LANGUAGES CXX) +find_package(MasterAgent CONFIG REQUIRED) +add_executable(consumer main.cpp) +target_link_libraries(consumer PRIVATE MasterAgent::Core) +enable_testing() +add_test(NAME run_consumer COMMAND consumer) diff --git a/examples/reference_agent/main.cpp b/examples/reference_agent/main.cpp new file mode 100644 index 0000000..cc6e8ba --- /dev/null +++ b/examples/reference_agent/main.cpp @@ -0,0 +1,13 @@ +#include +using namespace master_agent; +using namespace master_agent::reference; +int main() { + Runtime runtime; + int calls = 0; + Json schema{{"type", "object"}, {"properties", Json::object()}, {"additionalProperties", false}}; + if (!runtime.registerTool({"ping", "Consumer integration", schema, + [&](const Json&) { ++calls; return Result::success("pong"); }})) return 1; + if (!runtime.registerSkill("ping", "ping", Json::object())) return 2; + auto result = runtime.run({"consumer", "1", "ping"}); + return result && result.value->output == "pong" && calls == 1 ? 0 : 3; +} diff --git a/include/master_agent/runtime/http_model.h b/include/master_agent/runtime/http_model.h new file mode 100644 index 0000000..703f78c --- /dev/null +++ b/include/master_agent/runtime/http_model.h @@ -0,0 +1,14 @@ +#pragma once +#include "master_agent/runtime/reference_runtime.h" + +namespace master_agent::reference { +struct HttpModelConfig { + std::string endpoint = "http://127.0.0.1:8080/v1/chat/completions"; + std::string model = "local"; + std::string api_key; + long timeout_ms = 30000; +}; +// Connects to an existing server; never starts processes or downloads models. +// Explicit endpoint configuration is required to contact a remote server. +ModelHandler makeHttpModel(HttpModelConfig config); +} // namespace master_agent::reference diff --git a/include/master_agent/runtime/reference_runtime.h b/include/master_agent/runtime/reference_runtime.h new file mode 100644 index 0000000..fabf388 --- /dev/null +++ b/include/master_agent/runtime/reference_runtime.h @@ -0,0 +1,76 @@ +#pragma once + +#include "master_agent/common/types.h" +#include +#include +#include +#include +#include +#include + +namespace master_agent::reference { +using Json = nlohmann::json; +using ToolHandler = std::function(const Json&)>; +// Return JSON text: {"response":"..."} or {"tool":"name","arguments":{...}}. +// Callbacks run synchronously; the adapter owns timeout/cancellation semantics. +using ModelHandler = std::function(const Json& messages)>; + +struct Tool { + std::string name; + std::string description; + // Supported schema: object, primitive properties, required, enum, + // numeric minimum/maximum, additionalProperties:false. Others are rejected. + Json parameters; + ToolHandler execute; +}; + +struct Turn { + std::string session_id; + std::string request_id; + std::string input; +}; + +struct Reply { + std::string session_id; + std::string request_id; + std::string route; // skill, model, model_tool + std::string tool; + Json output; + bool replayed = false; +}; + +struct Limits { + std::size_t sessions = 32; + std::size_t requests_per_session = 256; + std::size_t history_turns = 8; +}; + +// A synchronous, in-process reference runtime. Concurrent/reentrant calls return +// BUSY. Tool callbacks are trusted host code, not sandboxed. Session history and +// request replay are memory-only; this is not the proprietary durable kernel. +class Runtime { +public: + explicit Runtime(Limits limits = {}); + Status registerTool(Tool tool); + Status registerSkill(std::string phrase, std::string tool, Json arguments); + Status setModel(ModelHandler model); + Result run(const Turn& turn); + // Explicitly discards history AND request deduplication for this session. + Status clearSession(const std::string& session_id); + +private: + struct Skill { std::string tool; Json arguments; }; + struct Cached { std::string input; Result result; }; + struct Session { + std::map requests; + std::vector history; + }; + Result execute(const Turn& turn, Session& session); + Limits limits_; + std::mutex mutex_; + std::map tools_; + std::map skills_; + std::map sessions_; + ModelHandler model_; +}; +} // namespace master_agent::reference diff --git a/scripts/package.sh b/scripts/package.sh new file mode 100755 index 0000000..c23b758 --- /dev/null +++ b/scripts/package.sh @@ -0,0 +1,26 @@ +#!/usr/bin/env bash +# Package the installed SDK + CLI and verify the relocated archive. +set -euo pipefail +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +BUILD_DIR="${1:-${ROOT}/build}" +ARCHIVE="${2:-${ROOT}/dist/sparx-sdk.tar.gz}" +mkdir -p "$(dirname "${ARCHIVE}")" +ARCHIVE="$(cd "$(dirname "${ARCHIVE}")" && pwd)/$(basename "${ARCHIVE}")" +STAGING="$(mktemp -d "${TMPDIR:-/tmp}/sparx-package.XXXXXX")" +trap 'rm -rf "${STAGING}"' EXIT +cmake --install "${BUILD_DIR}" --prefix "${STAGING}/sdk" --config Release +test -x "${STAGING}/sdk/bin/sparx" +cp "${ROOT}/LICENSE" "${ROOT}/README.md" "${STAGING}/sdk/" +cp -R "${ROOT}/examples/reference_agent" "${STAGING}/sdk/example" +tar -czf "${STAGING}/candidate.tar.gz" -C "${STAGING}/sdk" . +mkdir "${STAGING}/unpacked" +tar -xzf "${STAGING}/candidate.tar.gz" -C "${STAGING}/unpacked" +"${STAGING}/unpacked/bin/sparx" version +"${STAGING}/unpacked/bin/sparx" demo automotive +cmake -S "${STAGING}/unpacked/example" -B "${STAGING}/consumer" \ + -DCMAKE_PREFIX_PATH="${STAGING}/unpacked" +cmake --build "${STAGING}/consumer" --config Release +ctest --test-dir "${STAGING}/consumer" -C Release --output-on-failure +# Publish the file only after all verification has succeeded. +cp "${STAGING}/candidate.tar.gz" "${ARCHIVE}" +printf 'Verified archive: %s\n' "${ARCHIVE}" diff --git a/tests/check.h b/tests/check.h new file mode 100644 index 0000000..0537cc9 --- /dev/null +++ b/tests/check.h @@ -0,0 +1,9 @@ +#pragma once +#include +#include + +// Unlike assert(), test expectations must execute in Release builds too. +#define CHECK(...) do { if (!(__VA_ARGS__)) { \ + std::cerr << __FILE__ << ':' << __LINE__ << ": " << #__VA_ARGS__ << '\n'; \ + std::exit(EXIT_FAILURE); \ +} } while (false) diff --git a/tests/check_failure.cpp b/tests/check_failure.cpp new file mode 100644 index 0000000..02bd4da --- /dev/null +++ b/tests/check_failure.cpp @@ -0,0 +1,2 @@ +#include "check.h" +int main() { CHECK(false); } diff --git a/tests/consumer/CMakeLists.txt b/tests/consumer/CMakeLists.txt new file mode 100644 index 0000000..0139b8f --- /dev/null +++ b/tests/consumer/CMakeLists.txt @@ -0,0 +1,12 @@ +cmake_minimum_required(VERSION 3.18) +project(MasterAgentConsumer LANGUAGES CXX) +find_package(MasterAgent CONFIG REQUIRED) +add_executable(consumer main.cpp) +target_link_libraries(consumer PRIVATE MasterAgent::Core) +enable_testing() +add_test(NAME run_consumer COMMAND consumer) + +if(TARGET MasterAgent::Http) + target_link_libraries(consumer PRIVATE MasterAgent::Http) + target_compile_definitions(consumer PRIVATE TEST_HTTP=1) +endif() diff --git a/tests/consumer/main.cpp b/tests/consumer/main.cpp new file mode 100644 index 0000000..0cc1714 --- /dev/null +++ b/tests/consumer/main.cpp @@ -0,0 +1,19 @@ +#include +#ifdef TEST_HTTP +#include +#endif +using namespace master_agent; +using namespace master_agent::reference; +int main() { + Runtime runtime; +#ifdef TEST_HTTP + if (!runtime.setModel(makeHttpModel({}))) return 4; +#endif + int calls = 0; + Json schema{{"type", "object"}, {"properties", Json::object()}, {"additionalProperties", false}}; + if (!runtime.registerTool({"ping", "Consumer integration", schema, + [&](const Json&) { ++calls; return Result::success("pong"); }})) return 1; + if (!runtime.registerSkill("ping", "ping", Json::object())) return 2; + auto result = runtime.run({"consumer", "1", "ping"}); + return result && result.value->output == "pong" && calls == 1 ? 0 : 3; +} diff --git a/tests/http_model_probe.cpp b/tests/http_model_probe.cpp new file mode 100644 index 0000000..c8a7e00 --- /dev/null +++ b/tests/http_model_probe.cpp @@ -0,0 +1,11 @@ +#include "master_agent/runtime/http_model.h" +#include +using namespace master_agent::reference; +int main(int argc, char** argv) { + if (argc != 3) return 2; + HttpModelConfig config; config.endpoint = argv[1]; config.timeout_ms = 50; + auto result = makeHttpModel(config)(Json::array({{{"role", "user"}, {"content", argv[2]}}})); + if (!result) { std::cout << result.error->code << '\n'; return 1; } + std::cout << *result << '\n'; + return 0; +} diff --git a/tests/test_embedding.cpp b/tests/test_embedding.cpp index 126ecb5..14fb36b 100644 --- a/tests/test_embedding.cpp +++ b/tests/test_embedding.cpp @@ -1,5 +1,5 @@ #include "../cli/include/sparx_speculative.h" -#include +#include "check.h" #include #include @@ -12,27 +12,27 @@ int main() { auto v1 = idx.embed("show me the weather"); auto v2 = idx.embed("show me the weather"); float sim = EmbeddingIndex::cosineSimilarity(v1, v2); - assert(std::abs(sim - 1.0f) < 0.001f); + CHECK(std::abs(sim - 1.0f) < 0.001f); std::cout << "PASS: identical text → cosine=1.0\n"; // Test 2: Similar text → high similarity auto v3 = idx.embed("show me weather"); sim = EmbeddingIndex::cosineSimilarity(v1, v3); std::cout << " similar text similarity: " << sim << "\n"; - assert(sim > 0.7f); + CHECK(sim > 0.7f); std::cout << "PASS: similar text → high cosine (" << sim << ")\n"; // Test 3: Different text → low similarity auto v4 = idx.embed("delete all files from system"); sim = EmbeddingIndex::cosineSimilarity(v1, v4); std::cout << " different text similarity: " << sim << "\n"; - assert(sim < 0.5f); + CHECK(sim < 0.5f); std::cout << "PASS: different text → low cosine (" << sim << ")\n"; // Test 4: Case insensitivity auto v5 = idx.embed("Show Me The Weather"); sim = EmbeddingIndex::cosineSimilarity(v1, v5); - assert(std::abs(sim - 1.0f) < 0.001f); + CHECK(std::abs(sim - 1.0f) < 0.001f); std::cout << "PASS: case insensitive\n"; // Test 5: Nearest-neighbor search @@ -40,15 +40,15 @@ int main() { idx.insert("delete:xyz789", v4); auto nearest = idx.findNearest(v3, 0.7f); - assert(nearest.has_value()); - assert(nearest->cache_key == "weather:abc123"); + CHECK(nearest.has_value()); + CHECK(nearest->cache_key == "weather:abc123"); std::cout << "PASS: nearest neighbor finds weather (sim=" << nearest->similarity << ")\n"; // Test 6: Threshold filtering auto too_far = idx.findNearest(v4, 0.99f); // very strict threshold // v4 itself is in the index, so it should match itself auto exact_self = idx.findNearest(v4, 0.99f); - assert(exact_self.has_value()); + CHECK(exact_self.has_value()); std::cout << "PASS: threshold works (self-match at " << exact_self->similarity << ")\n"; // Test 7: Paraphrase similarity @@ -56,7 +56,7 @@ int main() { auto qb = idx.embed("what's the temperature outdoors"); sim = EmbeddingIndex::cosineSimilarity(qa, qb); std::cout << " paraphrase similarity: " << sim << "\n"; - assert(sim > 0.6f); + CHECK(sim > 0.6f); std::cout << "PASS: paraphrases have high similarity (" << sim << ")\n"; std::cout << "\nAll embedding tests passed!\n"; diff --git a/tests/test_eval_runner.py b/tests/test_eval_runner.py new file mode 100644 index 0000000..7f4aef2 --- /dev/null +++ b/tests/test_eval_runner.py @@ -0,0 +1,38 @@ +"""Verify success, failed evaluations and missing binaries with isolated fixtures.""" +import os +from pathlib import Path +import shutil +import subprocess +import sys +import tempfile + +with tempfile.TemporaryDirectory(prefix='sparx eval ') as temp: + root = Path(temp) + (root / 'eval').mkdir() + shutil.copyfile(sys.argv[1], root / 'eval/run_all.sh') + (root / 'bin').mkdir() + cmake = root / 'bin/cmake' + cmake.write_text('#!/bin/sh\nexit 0\n') + cmake.chmod(0o755) + binaries = root / 'build/eval' + binaries.mkdir(parents=True) + names = ['speculation', 'mesh', 'formal', 'learning', 'constrained'] + for name in names: + p = binaries / ('eval_' + name) + p.write_text('#!/bin/sh\necho fixture\nexit 0\n') + p.chmod(0o755) + env = dict(os.environ, PATH=str(root / 'bin') + os.pathsep + os.environ['PATH']) + env.pop('SPARX_BUILD_DIR', None) + env.pop('SPARX_RESULTS_DIR', None) + def run(): + return subprocess.run(['bash', str(root / 'eval/run_all.sh')], env=env, capture_output=True, text=True) + good = run() + assert good.returncode == 0 and '5 passed, 0 failed' in good.stdout, good + (binaries / 'eval_mesh').write_text('#!/bin/sh\nexit 7\n') + bad = run() + assert bad.returncode != 0 and '4 passed, 1 failed' in bad.stdout, bad + (binaries / 'eval_formal').unlink() + missing = run() + assert missing.returncode != 0 and '3 passed, 2 failed' in missing.stdout, missing + summary = (root / 'build/eval-results/SUMMARY.md').read_text() + assert '| eval_mesh | FAIL |' in summary and '| eval_formal | SKIP |' in summary diff --git a/tests/test_harness.cpp b/tests/test_harness.cpp index 04c7815..9487f95 100644 --- a/tests/test_harness.cpp +++ b/tests/test_harness.cpp @@ -393,6 +393,60 @@ TEST(harness_local_only_mode) { // Main // ═══════════════════════════════════════════════════════════════════════════════ + +static void configureSlowHarness(PipelineHarness& harness, int local_ms, int cloud_ms) { + harness.registerPromptEngine("compressed", std::make_shared(PromptEngineConfig{})); + harness.registerLocalInference("mock", std::make_shared("local", local_ms)); + harness.registerCloudBackend("mock", std::make_shared("cloud", cloud_ms)); + harness.registerConfidenceScorer("heuristic", std::make_shared()); + ArbiterConfig ac; ac.deadline_ms = 50; + harness.registerArbiter("cloud_prefer", std::make_shared(ac)); + HarnessConfig config; config.cloud_enabled = true; config.cloud_backend = "mock"; + config.confidence_thresholds.high = 2.0f; + harness.applyConfig(config); +} + +TEST(harness_deadline_includes_return_and_owns_late_workers) { + for (bool slow_local : {false, true}) { + auto begin = std::chrono::steady_clock::now(); + int reported = 0; + { + PipelineHarness harness; + configureSlowHarness(harness, slow_local ? 600 : 10, slow_local ? 10 : 600); + PipelineRequest request; request.user_input = "general question"; + auto result = harness.execute(request); + reported = result.total_latency_ms; + ASSERT_EQ(result.result.source, slow_local ? ArbiterOutput::Source::Cloud : ArbiterOutput::Source::Local); + // Retry while the slow worker is outstanding: no second slow job. + auto next = harness.execute(request); + if (!slow_local) ASSERT_TRUE(!next.cloud_fired); + } + auto wall = std::chrono::duration_cast(std::chrono::steady_clock::now() - begin).count(); + ASSERT_LT(wall, 400); // Old implementation waited for the 600ms future. + ASSERT_LT(reported, 250); + // Exercise late completion after both the harness and registry die. + std::this_thread::sleep_for(std::chrono::milliseconds(650)); + } +} + +TEST(cloud_future_survives_backend_destruction) { + std::future result; + { + MockCloudBackend backend("owned snapshot", 50); + result = backend.inferAsync("question"); + } + ASSERT_EQ(result.get().content, std::string("owned snapshot")); +} + +TEST(harness_bad_config_does_not_keep_previous_backend) { + PipelineHarness harness; + configureSlowHarness(harness, 1, 1); + ASSERT_TRUE(harness.isReady()); + auto config = harness.config(); config.prompt_engine = "missing"; + harness.applyConfig(config); + ASSERT_TRUE(!harness.isReady()); +} + int main() { std::cout << "\n=== Edge-Cloud Harness Tests ===\n\n"; diff --git a/tests/test_http_model.py b/tests/test_http_model.py new file mode 100644 index 0000000..37332cb --- /dev/null +++ b/tests/test_http_model.py @@ -0,0 +1,71 @@ +"""Exercise the real HTTP adapter/CLI against a loopback protocol fixture, not an LLM.""" +import http.server +import json +import subprocess +import sys +import threading +import time + + +class Server(http.server.BaseHTTPRequestHandler): + def log_message(self, *_): + pass + + def do_POST(self): + request = json.loads(self.rfile.read(int(self.headers['Content-Length']))) + if self.path != '/v1/chat/completions' or request['stream'] is not False: + self.send_error(400) + return + text = request['messages'][-1]['content'] + decisions = { + 'set via model': {'tool': 'ac.set_temperature', 'arguments': {'temperature': 24}}, + 'bad argument': {'tool': 'ac.set_temperature', 'arguments': {'temperature': 100}}, + 'unknown tool': {'tool': 'shell.exec', 'arguments': {}}, + 'hello': {'response': '你好 🌳'}, + } + if text == 'slow': + time.sleep(2) + payload = {'choices': [{'message': {'content': json.dumps(decisions.get(text, {'response': 'ok'}), ensure_ascii=True)}}]} + if text == 'large': + payload['padding'] = 'x' * (1024 * 1024 + 1) + if text == 'malformed': + payload = {'choices': [{'message': {'content': None}}]} + self.send_response(503 if text == 'unavailable' else 200) + body = json.dumps(payload).encode() + self.send_header('Content-Length', str(len(body))) + self.end_headers() + try: + self.wfile.write(body) + except (BrokenPipeError, ConnectionResetError): + pass + + +server = http.server.ThreadingHTTPServer(('127.0.0.1', 0), Server) +thread = threading.Thread(target=server.serve_forever, daemon=True) +thread.start() +try: + for text, expected in [('set via model', None), ('hello', None), ('bad argument', 'INVALID_ARGUMENTS'), + ('unknown tool', 'UNKNOWN_TOOL'), ('malformed', 'INVALID_RESPONSE'), ('unavailable', 'HTTP_ERROR')]: + result = subprocess.run([sys.argv[1], 'run', '--endpoint', f'127.0.0.1:{server.server_port}', '--input', text], + capture_output=True, text=True, timeout=10) + data = json.loads(result.stdout) + if expected: + assert result.returncode != 0 and data['error']['code'] == expected, (text, result, data) + else: + assert result.returncode == 0 and data['ok'], (text, result, data) + if text == 'hello': + assert data['output'] == '你好 🌳' + else: + assert data['route'] == 'model_tool' and data['output']['temperature'] == 24 + for text, code in [('slow', 'TIMEOUT'), ('large', 'HTTP_ERROR')]: + begin = time.monotonic() + probe = subprocess.run([sys.argv[2], f'http://127.0.0.1:{server.server_port}/v1/chat/completions', text], + capture_output=True, text=True, timeout=5) + assert probe.returncode == 1 and probe.stdout.strip() == code, probe + assert time.monotonic() - begin < 1.5, 'HTTP timeout exceeded' + offline = subprocess.run([sys.argv[1], 'run', '--input', 'set AC to 22 degrees'], capture_output=True, text=True) + assert offline.returncode == 0 and json.loads(offline.stdout)['route'] == 'skill' +finally: + server.shutdown() + server.server_close() + thread.join() diff --git a/tests/test_integration_speculation.cpp b/tests/test_integration_speculation.cpp index cbf19a8..e42630b 100644 --- a/tests/test_integration_speculation.cpp +++ b/tests/test_integration_speculation.cpp @@ -1,8 +1,9 @@ #include "../cli/include/sparx_speculative.h" -#include +#include "check.h" #include #include #include +#include using namespace sparx::speculation; @@ -16,9 +17,19 @@ static std::optional mockInference( int main() { std::cout << "=== Integration Test: Speculation → Cache → Hit ===\n\n"; + // Isolate model persistence from the developer's real ~/.sparx data. + struct Scratch { + std::filesystem::path path = std::filesystem::temp_directory_path() / + ("sparx-speculation-" + std::to_string(std::chrono::steady_clock::now().time_since_epoch().count())); + Scratch() { CHECK(std::filesystem::create_directory(path)); } + ~Scratch() { std::error_code ec; std::filesystem::remove_all(path, ec); } + } scratch; // Setup PredictionConfig pred_config; pred_config.cold_start_threshold = 3; + pred_config.history_path = (scratch.path / "history.json").string(); + pred_config.lstm_weights_path = (scratch.path / "lstm.bin").string(); + pred_config.mamba_weights_path = (scratch.path / "mamba.bin").string(); IntentPredictor predictor(pred_config); CacheConfig cache_config; @@ -53,8 +64,8 @@ int main() { std::this_thread::sleep_for(std::chrono::milliseconds(200)); std::cout << " Predictor observations: " << predictor.observationCount() << "\n"; - assert(predictor.observationCount() == 10); - assert(predictor.isWarmedUp()); + CHECK(predictor.observationCount() == 10); + CHECK(predictor.isWarmedUp()); std::cout << " Predictor warmed up: yes\n"; // Check predictions @@ -64,9 +75,9 @@ int main() { std::cout << " " << p.predicted_intent << " (conf=" << p.confidence << ")\n"; } - assert(!predictions.empty()); + CHECK(!predictions.empty()); // After repeated weather→forecast pattern, "weather" should be predicted - assert(predictions[0].predicted_intent == "weather"); + CHECK(predictions[0].predicted_intent == "weather"); std::cout << " PASS: predictor learned weather→forecast→weather pattern\n\n"; // Phase 2: Check cache for speculation results diff --git a/tests/test_merkle.cpp b/tests/test_merkle.cpp index 981d79b..088df6f 100644 --- a/tests/test_merkle.cpp +++ b/tests/test_merkle.cpp @@ -1,5 +1,5 @@ #include "../cli/include/sparx_mesh.h" -#include +#include "check.h" #include using namespace sparx::mesh; @@ -41,14 +41,14 @@ int main() { std::cout << "Root A: " << digestA.root_hash << "\n"; std::cout << "Root B: " << digestB.root_hash << "\n"; std::cout << "Keys A: " << digestA.key_count << ", B: " << digestB.key_count << "\n"; - assert(digestA.key_count == 3); - assert(digestB.key_count == 3); + CHECK(digestA.key_count == 3); + CHECK(digestB.key_count == 3); std::cout << "PASS: both trees have 3 keys\n"; // Test 2: Compare with itself — no divergence auto selfDiff = merkleA.compare(digestA); - assert(selfDiff.divergent_keys.empty()); - assert(selfDiff.nodes_matched == 1); // root matched, skip all + CHECK(selfDiff.divergent_keys.empty()); + CHECK(selfDiff.nodes_matched == 1); // root matched, skip all std::cout << "PASS: self-compare → no divergence\n"; // Test 3: Add a key to A only — causes divergence @@ -58,7 +58,7 @@ int main() { merkleA.rebuild(mapA); auto newDigestA = merkleA.digest(); - assert(newDigestA.key_count == 4); + CHECK(newDigestA.key_count == 4); // Compare: B's digest vs A's tree → should find divergence auto diff = merkleA.compare(digestB); @@ -68,7 +68,7 @@ int main() { std::cout << "Sync efficiency: " << diff.sync_efficiency() << "\n"; // The new key should show up in divergent keys // (may also include other keys in same bucket) - assert(diff.nodes_compared > 0); + CHECK(diff.nodes_compared > 0); std::cout << "PASS: divergence detected after mutation\n"; // Test 4: Efficiency — with many keys, only divergent bucket is flagged @@ -84,16 +84,16 @@ int main() { MerkleAntiEntropy merkleBig; merkleBig.rebuild(bigMap); auto bigDigest = merkleBig.digest(); - assert(bigDigest.key_count == 100); + CHECK(bigDigest.key_count == 100); // Self-compare: O(1) auto bigSelf = merkleBig.compare(bigDigest); - assert(bigSelf.divergent_keys.empty()); - assert(bigSelf.nodes_compared == 1); + CHECK(bigSelf.divergent_keys.empty()); + CHECK(bigSelf.nodes_compared == 1); std::cout << "PASS: 100-key self-compare is O(1)\n"; // Test 5: Bucket count is branching_factor^depth - assert(merkleBig.bucketCount() == 256); // 16^2 + CHECK(merkleBig.bucketCount() == 256); // 16^2 std::cout << "PASS: bucket count = " << merkleBig.bucketCount() << "\n"; std::cout << "\nAll Merkle anti-entropy tests passed!\n"; diff --git a/tests/test_orset.cpp b/tests/test_orset.cpp index 2582449..a9e486e 100644 --- a/tests/test_orset.cpp +++ b/tests/test_orset.cpp @@ -1,5 +1,5 @@ #include "../cli/include/sparx_mesh.h" -#include +#include "check.h" #include using namespace sparx::mesh; @@ -17,9 +17,9 @@ int main() { nodeB.merge(opA); auto stateA = nodeA.get("fruits"); - assert(stateA.has_value()); + CHECK(stateA.has_value()); // Both tags present — apple appears in the alive section - assert(stateA->value.find("apple") != std::string::npos); + CHECK(stateA->value.find("apple") != std::string::npos); std::cout << "PASS: concurrent add - element survives\n"; // Test 2: Multiple elements across nodes @@ -30,9 +30,9 @@ int main() { nodeB.merge(opCherry); stateA = nodeA.get("fruits"); - assert(stateA->value.find("apple") != std::string::npos); - assert(stateA->value.find("banana") != std::string::npos); - assert(stateA->value.find("cherry") != std::string::npos); + CHECK(stateA->value.find("apple") != std::string::npos); + CHECK(stateA->value.find("banana") != std::string::npos); + CHECK(stateA->value.find("cherry") != std::string::npos); std::cout << "PASS: multi-element ORSet\n"; // Test 3: Tag uniqueness — same element added twice gets different tags @@ -41,8 +41,8 @@ int main() { node.mutate("items", CrdtType::ORSet, "x"); auto state = node.get("items"); // Should have two tags for "x" (solo#1 and solo#2) - assert(state->value.find("solo#1") != std::string::npos); - assert(state->value.find("solo#2") != std::string::npos); + CHECK(state->value.find("solo#1") != std::string::npos); + CHECK(state->value.find("solo#2") != std::string::npos); std::cout << "PASS: multiple adds generate unique tags\n"; // Test 4: single-element propagation @@ -51,8 +51,8 @@ int main() { auto op1 = empty1.mutate("k", CrdtType::ORSet, "v"); empty2.merge(op1); auto s = empty2.get("k"); - assert(s.has_value()); - assert(s->value.find("v") != std::string::npos); + CHECK(s.has_value()); + CHECK(s->value.find("v") != std::string::npos); std::cout << "PASS: single-element propagation\n"; // Test 5: Remove element — observed-remove semantics @@ -71,8 +71,8 @@ int main() { auto s1 = n1.get("set"); auto s2 = n2.get("set"); // The element's tag is tombstoned — "item" shouldn't appear in alive section - assert(s1->value.find("item\x1f") == std::string::npos); - assert(s2->value.find("item\x1f") == std::string::npos); + CHECK(s1->value.find("item\x1f") == std::string::npos); + CHECK(s2->value.find("item\x1f") == std::string::npos); std::cout << "PASS: remove propagates and element disappears\n"; } @@ -95,8 +95,8 @@ int main() { // The re-add's tag is NOT tombstoned → item survives (add-wins) auto s1 = n1.get("set"); auto s2 = n2.get("set"); - assert(s1->value.find("item") != std::string::npos); - assert(s2->value.find("item") != std::string::npos); + CHECK(s1->value.find("item") != std::string::npos); + CHECK(s2->value.find("item") != std::string::npos); std::cout << "PASS: add-wins — concurrent re-add survives remove\n"; } diff --git a/tests/test_reference_runtime.cpp b/tests/test_reference_runtime.cpp new file mode 100644 index 0000000..a92422d --- /dev/null +++ b/tests/test_reference_runtime.cpp @@ -0,0 +1,58 @@ +#include "master_agent/runtime/reference_runtime.h" +#include "check.h" +#include + +using namespace master_agent; +using namespace master_agent::reference; +int main() { + Runtime runtime; + int calls = 0; + Json schema{{"type", "object"}, {"properties", {{"value", {{"type", "integer"}, {"minimum", 0}, {"maximum", 10}}}}}, {"required", {"value"}}, {"additionalProperties", false}}; + CHECK(runtime.registerTool({"set", "Set a value", schema, [&](const Json& args) { + ++calls; + CHECK(!runtime.run({"recursive", "1", "set five"})); + return Result::success(args); + }})); + CHECK(!runtime.registerTool({"set", "duplicate", schema, [](const Json&) { return Result::success(0); }})); + CHECK(!runtime.registerSkill("bad", "set", {{"value", 11}})); + CHECK(runtime.registerSkill("set five", "set", {{"value", 5}})); + auto first = runtime.run({"a", "1", "set five"}); + CHECK(first && first.value->route == "skill" && first.value->output["value"] == 5 && calls == 1); + auto replay = runtime.run({"a", "1", "set five"}); + CHECK(replay && replay.value->replayed && calls == 1); + CHECK(runtime.run({"a", "1", "different"}).error->code == "REQUEST_CONFLICT"); + CHECK(runtime.run({"b", "1", "set five"}) && calls == 2); + CHECK(runtime.run({"a", "2", "unmatched"}).error->code == "NO_ROUTE"); + std::string decision = R"({"tool":"set","arguments":{"value":8}})"; + CHECK(runtime.setModel([&](const Json& messages) { + CHECK(messages.front()["role"] == "system"); + return Result::success(decision); + })); + CHECK(runtime.run({"a", "3", "model"}).value->route == "model_tool" && calls == 3); + int sequence = 3; + for (auto output : {R"({"tool":"set","arguments":{"value":11}})", + R"({"tool":"set","arguments":{"value":"8"}})", + R"({"tool":"set","arguments":{}})", + R"({"tool":"set","arguments":{"value":1,"extra":true}})", + R"({"tool":"unknown","arguments":{}})", + R"({"response":"text","tool":"set","arguments":{"value":1}})", + "not json"}) { + decision = output; + CHECK(!runtime.run({"a", std::to_string(++sequence), "model"})); + CHECK(calls == 3); + } + decision = R"({"response":"你好"})"; + CHECK(runtime.run({"a", std::to_string(++sequence), "hello"}).value->output == "你好"); + CHECK(runtime.registerTool({"uncertain", "Throws after side effect", schema, [&](const Json&) -> Result { ++calls; throw std::runtime_error("after effect"); }})); + CHECK(runtime.registerSkill("uncertain", "uncertain", {{"value", 1}})); + CHECK(runtime.run({"c", "1", "uncertain"}).error->code == "UNKNOWN"); + CHECK(runtime.run({"c", "1", "uncertain"}).error->code == "UNKNOWN" && calls == 4); + CHECK(runtime.clearSession("a")); + Runtime limited({1, 1, 0}); + CHECK(limited.setModel([](const Json&) { return Result::success(R"({"response":"ok"})"); })); + CHECK(limited.run({"s", "1", "a"})); + CHECK(limited.run({"s", "2", "a"}).error->code == "RESOURCE_LIMIT"); + CHECK(limited.run({"t", "1", "a"}).error->code == "RESOURCE_LIMIT"); + auto unsupported = schema; unsupported["properties"]["value"]["pattern"] = "x"; + CHECK(!runtime.registerTool({"invalid", "invalid", unsupported, [](const Json&) { return Result::success(0); }})); +} From 87def465bf076e1a023f23232c6c287ae9374b3c Mon Sep 17 00:00:00 2001 From: Codex Date: Mon, 28 Sep 2026 14:06:48 +0800 Subject: [PATCH 2/4] feat(runtime): persist receipts and reconcile interrupted tool execution --- .github/workflows/ci.yml | 9 +- .github/workflows/release.yml | 2 +- ARCHITECTURE.md | 2 +- CHANGELOG.md | 10 + CLAUDE.md | 4 +- CMakeLists.txt | 11 +- CONTRIBUTING.md | 2 +- README.md | 42 ++- VERSION.json | 2 +- cli/CMakeLists.txt | 2 +- cli/src/reference_main.cpp | 85 +++++- cmake/MasterAgentConfig.cmake.in | 5 + cmake/ReferenceTests.cmake | 13 + core/durable_store.cpp | 250 ++++++++++++++++++ core/durable_store.h | 32 +++ core/reference_runtime.cpp | 120 ++++++++- docs/durable_recovery.md | 94 +++++++ docs/reference_runtime.md | 17 +- examples/reference_agent/CMakeLists.txt | 10 + examples/reference_agent/main.cpp | 23 +- .../master_agent/runtime/reference_runtime.h | 37 ++- scripts/build_manifest.py | 30 +++ scripts/package.sh | 8 + tests/consumer/CMakeLists.txt | 5 + tests/consumer/main.cpp | 17 +- tests/durable_probe.cpp | 42 +++ tests/test_crash_recovery.py | 84 ++++++ tests/test_durable_runtime.cpp | 69 +++++ 28 files changed, 980 insertions(+), 47 deletions(-) create mode 100644 core/durable_store.cpp create mode 100644 core/durable_store.h create mode 100644 docs/durable_recovery.md create mode 100644 scripts/build_manifest.py create mode 100644 tests/durable_probe.cpp create mode 100644 tests/test_crash_recovery.py create mode 100644 tests/test_durable_runtime.cpp diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 3f33073..02dac85 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -31,7 +31,7 @@ jobs: if: runner.os == 'Linux' run: | sudo apt-get update - sudo apt-get install -y cmake ninja-build libcurl4-openssl-dev + sudo apt-get install -y cmake ninja-build libcurl4-openssl-dev libsqlite3-dev - name: Configure CMake run: | @@ -55,11 +55,12 @@ jobs: run: bash scripts/package.sh build dist/sparx-sdk.tar.gz - name: Upload build artifacts - if: matrix.os == 'ubuntu-latest' uses: actions/upload-artifact@v4 with: - name: sparx-linux-x64 - path: build/cli/sparx + name: sparx-sdk-${{ runner.os }}-${{ runner.arch }} + path: | + dist/*.tar.gz + dist/*.sha256 retention-days: 7 if-no-files-found: error diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index da31c62..6d696e0 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -40,7 +40,7 @@ jobs: if: runner.os == 'Linux' run: | sudo apt-get update - sudo apt-get install -y cmake ninja-build libcurl4-openssl-dev + sudo apt-get install -y cmake ninja-build libcurl4-openssl-dev libsqlite3-dev - name: Configure run: | diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 3f47e55..45cec33 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -1,4 +1,4 @@ -> Public 0.4.0-alpha: use the open reference runtime in `core/`, +> Public 0.4.0-alpha.2: use the open reference runtime in `core/`, > `include/master_agent/runtime/`, `backends/`, and `cli/src/reference_main.cpp`. > The current runnable contract and commands are in README.md and > docs/reference_runtime.md. Legacy kernel descriptions below require private diff --git a/CHANGELOG.md b/CHANGELOG.md index 5646cde..06b6aa6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,13 @@ +# 0.4.0-alpha.2 — durable receipts and recovery + +- Add an opt-in SQLite store with WAL/FULL synchronization, exclusive ownership, + persistent request outcomes/history, conservative UNKNOWN recovery and audited reconciliation. +- Fail closed on storage errors; write validated dispatch information before tool callbacks. +- Add CLI state, JSONL, recovery, reconciliation, HTTP timeout and credential-env options. +- Test real process termination after an external fsync, storage failure injection, + restart replay/history, locking, corrupted stores and schema versions. +- Package provenance metadata/checksums and validate persistent installed consumers. + # 0.4.0-alpha — open reference runtime - Add an independently runnable reference CLI and installable C++ Core/Http SDK. diff --git a/CLAUDE.md b/CLAUDE.md index a74e08c..1b7f201 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1,4 +1,4 @@ -> Public 0.4.0-alpha: use the open reference runtime in `core/`, +> Public 0.4.0-alpha.2: use the open reference runtime in `core/`, > `include/master_agent/runtime/`, `backends/`, and `cli/src/reference_main.cpp`. > The current runnable contract and commands are in README.md and > docs/reference_runtime.md. Legacy kernel descriptions below require private @@ -60,4 +60,4 @@ The legacy full CLI still requires private kernel source. See docs/reference_run ## Version -0.4.0-alpha. Versions follow semver. Don't inflate beyond actual stability. +0.4.0-alpha.2. Versions follow semver. Don't inflate beyond actual stability. diff --git a/CMakeLists.txt b/CMakeLists.txt index 5895b23..d7d4a98 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -5,6 +5,7 @@ project(MasterAgent VERSION 0.4.0 LANGUAGES CXX) include(GNUInstallDirs) include(CMakePackageConfigHelpers) find_package(Threads REQUIRED) +option(MASTER_AGENT_ENABLE_STORAGE "Enable durable SQLite request storage (Linux/macOS)" ON) option(MASTER_AGENT_ENABLE_HTTP "Build the HTTP model adapter (libcurl)" ON) option(BUILD_EVAL "Build synthetic evaluation programs" ON) @@ -219,13 +220,21 @@ add_library(MasterAgent::Core ALIAS master_agent_core) else() # The public reference runtime is a real, independently usable library. message(STATUS "Building the open reference runtime") -add_library(master_agent_core STATIC core/reference_runtime.cpp) +add_library(master_agent_core STATIC core/reference_runtime.cpp core/durable_store.cpp) target_compile_features(master_agent_core PUBLIC cxx_std_17) target_include_directories(master_agent_core PUBLIC "$" "$" "$") target_link_libraries(master_agent_core PUBLIC Threads::Threads) +if(MASTER_AGENT_ENABLE_STORAGE) + if(NOT UNIX) + message(FATAL_ERROR "Durable store currently supports Linux/macOS; set MASTER_AGENT_ENABLE_STORAGE=OFF") + endif() + find_package(SQLite3 REQUIRED) + target_link_libraries(master_agent_core PRIVATE SQLite::SQLite3) + target_compile_definitions(master_agent_core PRIVATE MASTER_AGENT_HAS_SQLITE=1) +endif() set_target_properties(master_agent_core PROPERTIES EXPORT_NAME Core POSITION_INDEPENDENT_CODE ON) add_library(MasterAgent::Core ALIAS master_agent_core) endif() diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index f232df8..314488f 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1,4 +1,4 @@ -> Public 0.4.0-alpha: use the open reference runtime in `core/`, +> Public 0.4.0-alpha.2: use the open reference runtime in `core/`, > `include/master_agent/runtime/`, `backends/`, and `cli/src/reference_main.cpp`. > The current runnable contract and commands are in README.md and > docs/reference_runtime.md. Legacy kernel descriptions below require private diff --git a/README.md b/README.md index 97d090d..71b55e7 100644 --- a/README.md +++ b/README.md @@ -3,22 +3,22 @@ An embeddable, local-first C++17 agent runtime: turn input into validated tool execution, with deterministic skills before model inference. -**0.4.0-alpha — open reference runtime.** Public source now builds a working +**0.4.0-alpha.2 — open reference runtime with durable receipts.** Public source now builds a working `sparx` CLI and an installable C++ SDK. This is a small synchronous runtime, not the proprietary durable kernel described by the legacy interfaces. 中文:开源版现可独立构建、运行,并作为 C++ SDK 集成。先匹配确定性技能,再按需调用模型; -工具名称和参数必须通过校验。当前会话、历史和请求去重仅保存在内存中,不承诺崩溃恢复。 +工具名称和参数必须通过校验。默认使用内存;显式启用 SQLite 后,支持跨重启请求去重、历史恢复、UNKNOWN 故障恢复和人工对账。 ## Quick start Requires CMake 3.18+, C++17, and libcurl development files for the optional HTTP -adapter. Tests require Python 3. The release CI targets Linux and macOS. +adapter, plus SQLite development files for durable storage. Tests/packaging require Python 3. The release CI targets Linux and macOS. ```bash git clone https://github.com/OpenSparX/MasterAgent.git cd MasterAgent -# Ubuntu HTTP dependency: sudo apt-get install libcurl4-openssl-dev +# Ubuntu HTTP dependency: sudo apt-get install libcurl4-openssl-dev libsqlite3-dev cmake -S . -B build -DCMAKE_BUILD_TYPE=Release cmake --build build --parallel 4 ctest --test-dir build --output-on-failure @@ -53,6 +53,10 @@ model; malformed output produces an error, never a guessed tool call. No model is downloaded or started by this CLI. A full HTTP(S) endpoint URL may also be provided explicitly; no cloud fallback happens automatically. +A request timeout can be set with `--timeout-ms 3000`. To use an authenticated +endpoint, pass the environment-variable **name** via `--api-key-env MODEL_API_KEY`; +credentials are not supplied as CLI arguments. + For an SDK and deterministic CLI without libcurl: ```bash @@ -60,6 +64,27 @@ cmake -S . -B build-offline -DMASTER_AGENT_ENABLE_HTTP=OFF -DBUILD_EVAL=OFF cmake --build build-offline --parallel 4 ``` +## Persist requests and recover interrupted tools + +```bash +./build/cli/sparx run --state ./state/agent.db --session alice --request-id ac-001 \ + --input 'set AC to 22 degrees' +./build/cli/sparx recover --state ./state/agent.db +``` + +Repeating the same request after restarting returns the recorded outcome without +calling the tool again. Unfinished requests become `UNKNOWN`, require external +verification, and can be explicitly reconciled. A database failure blocks further +execution instead of silently switching to memory. See the [recovery guide](docs/durable_recovery.md) +for JSONL input, ownership locks, audit notes, backups and failure behavior. + +Only execution receipts/history are persisted; the AC simulator itself still resets +on process restart. There is no transaction spanning a real device and SQLite, so this +is not a distributed exactly-once guarantee. Storage is not encrypted. + +To omit both optional dependencies, build with `MASTER_AGENT_ENABLE_HTTP=OFF` and +`MASTER_AGENT_ENABLE_STORAGE=OFF`. + ## Embed the SDK ```bash @@ -93,11 +118,12 @@ support, sessions, request replay, and concurrency limitations. | Deterministic exact-phrase skills | Available, no model required | | Host-defined tools and argument validation | Available; primitive object schema subset | | Local model HTTP adapter | Available when built with libcurl | -| Sessions and request deduplication | Available in memory; bounded; no restart persistence | +| Sessions and request deduplication | In memory, or persistent SQLite receipts/history with explicit opt-in | +| Crash recovery and reconciliation | Interrupted single requests become UNKNOWN; explicit evidence-based reconciliation | | CLI and relocatable CMake SDK package | Built and exercised by CI | | Speculation, mesh, verification, learning, decoding | Experimental modules and synthetic evaluations; not wired into the reference CLI | | Edge/cloud harness | Experimental; explicit opt-in; bounded outstanding inference workers | -| Durable DAG/WAL recovery and legacy kernel factories | Not supplied by this reference runtime | +| Multi-step durable DAG and legacy kernel factories | Not supplied by this reference runtime | | Qualcomm QNN/Genie integration | Platform integration required; not verified by this release | | Production encryption / DP learning guarantees | Not claimed for this release | @@ -120,8 +146,8 @@ return a nonzero status. `BUILD_EVAL=OFF` excludes them from an SDK-only build. The package script installs the SDK and CLI, unpacks the archive in a new location, runs the demo, and builds an external consumer before publishing the -archive. The package uses platform system dependencies, including libcurl when -enabled; it is not a universally static binary. +archive. The archive includes `BUILD_INFO.json` and a SHA-256 sidecar. The package uses +platform system dependencies, including libcurl and SQLite when enabled; it is not a universally static binary. ## Contributing diff --git a/VERSION.json b/VERSION.json index e21e68d..433ce2a 100644 --- a/VERSION.json +++ b/VERSION.json @@ -1,6 +1,6 @@ { "product": "OAK", - "version": "0.4.0-alpha", + "version": "0.4.0-alpha.2", "api_namespace": "master_agent", "cmake_target": "MasterAgent::Core", "model_backends": ["http_chat_completions", "custom_callback"], diff --git a/cli/CMakeLists.txt b/cli/CMakeLists.txt index c5bdd11..a04d6a9 100644 --- a/cli/CMakeLists.txt +++ b/cli/CMakeLists.txt @@ -134,7 +134,7 @@ else() target_compile_definitions(sparx PRIVATE MASTER_AGENT_HAS_HTTP=1) endif() if(NOT DEFINED SPARX_VERSION) - set(SPARX_VERSION "${PROJECT_VERSION}-alpha") + set(SPARX_VERSION "${PROJECT_VERSION}-alpha.2") endif() target_compile_definitions(sparx PRIVATE SPARX_VERSION="${SPARX_VERSION}") install(TARGETS sparx RUNTIME DESTINATION ${CMAKE_INSTALL_BINDIR}) diff --git a/cli/src/reference_main.cpp b/cli/src/reference_main.cpp index 758b862..3ebcf8e 100644 --- a/cli/src/reference_main.cpp +++ b/cli/src/reference_main.cpp @@ -3,6 +3,8 @@ #include "master_agent/runtime/http_model.h" #endif #include +#include +#include #include using namespace master_agent; @@ -13,7 +15,12 @@ void help() { " sparx version\n" " sparx demo automotive\n" " sparx run [--input TEXT] [--endpoint URL] [--model NAME]\n" - " [--session ID] [--request-id ID]\n" + " [--session ID] [--request-id ID] [--state FILE] [--jsonl]\n" + " [--timeout-ms N] [--api-key-env NAME]\n" + " sparx recover --state FILE\n" + " sparx reconcile --state FILE --session ID --request-id ID\n" + " --outcome committed|failed --note EVIDENCE [--output JSON]\n" + "Durable runs require --request-id with --input, or explicit IDs in JSONL.\n" "Without --input, reads one request per line until EOF.\n" "Without --endpoint, runs deterministic skills only.\n" "Example skills: set AC to 22 degrees; 把空调调到22度; vehicle status\n" @@ -34,12 +41,17 @@ int main(int argc, char** argv) { const std::string command = argv[1]; if (command == "version" && argc == 2) { std::cout << SPARX_VERSION << " (open reference runtime)\n"; return 0; } const bool demo = command == "demo" && argc == 3 && std::string(argv[2]) == "automotive"; - if (command != "run" && !demo) { std::cerr << "Unsupported command. Use sparx --help.\n"; return 2; } + if (command != "run" && command != "recover" && command != "reconcile" && !demo) { std::cerr << "Unsupported command. Use sparx --help.\n"; return 2; } std::string input, endpoint, model = "local", session = "cli", request; - bool single = false; + std::string state, outcome, note, output = "null", key_env; + int timeout_ms = 30000; + bool single = false, jsonl = false; + std::vector options; if (!demo) for (int i = 2; i < argc; ++i) { const std::string flag = argv[i]; if (flag == "--help") { help(); return 0; } + options.push_back(flag); + if (flag == "--jsonl") { jsonl = true; continue; } if (i + 1 >= argc) { std::cerr << "Missing option value\n"; return 2; } const std::string value = argv[++i]; if (flag == "--input") { input = value; single = true; } @@ -47,10 +59,57 @@ int main(int argc, char** argv) { else if (flag == "--model") model = value; else if (flag == "--session") session = value; else if (flag == "--request-id") request = value; + else if (flag == "--state") { + if (value.empty()) { std::cerr << "--state must not be empty\n"; return 2; } + state = value; + } + else if (flag == "--outcome") outcome = value; + else if (flag == "--note") note = value; + else if (flag == "--output") output = value; + else if (flag == "--api-key-env") key_env = value; + else if (flag == "--timeout-ms") { + auto parsed = std::from_chars(value.data(), value.data() + value.size(), timeout_ms); + if (parsed.ec != std::errc{} || parsed.ptr != value.data() + value.size() || timeout_ms <= 0 || timeout_ms > 300000) { + std::cerr << "--timeout-ms must be 1..300000\n"; return 2; + } + } else { std::cerr << "Unknown option: " << flag << '\n'; return 2; } } - if (!single && !request.empty()) { std::cerr << "--request-id requires --input\n"; return 2; } + if (command == "run" && !single && !request.empty()) { std::cerr << "--request-id requires --input\n"; return 2; } + for (const auto& flag : options) { + const bool recovery_flag = flag == "--state" || (command == "reconcile" && + (flag == "--session" || flag == "--request-id" || flag == "--outcome" || flag == "--note" || flag == "--output")); + if (((command == "recover" || command == "reconcile") && !recovery_flag) || + (command == "run" && (flag == "--outcome" || flag == "--note" || flag == "--output"))) { + std::cerr << "Option does not apply to this command: " << flag << '\n'; return 2; + } + } + if ((jsonl && single) || (command == "run" && !state.empty() && !jsonl && (!single || request.empty()))) { + std::cerr << "Use --state with --input and --request-id, or --jsonl with explicit IDs\n"; return 2; + } + if ((command == "recover" || command == "reconcile") && state.empty()) { std::cerr << "--state is required\n"; return 2; } Runtime runtime; + if (!state.empty()) { + auto status = runtime.openStore(state); + if (!status) { std::cout << encode(Result::failure({status.error_code, status.error_message, "", 500})).dump() << '\n'; return 1; } + } + if (command == "recover") { + auto records = runtime.unresolved(); + if (!records) return 1; + Json pending = Json::array(); + for (const auto& item : *records) pending.push_back({{"session_id", item.turn.session_id}, {"request_id", item.turn.request_id}, + {"input", item.turn.input}, {"tool", item.tool}, {"arguments", item.arguments}, {"state", "UNKNOWN"}}); + std::cout << Json{{"ok", true}, {"unresolved", pending}}.dump() << '\n'; return 0; + } + if (command == "reconcile") { + auto parsed = Json::parse(output, nullptr, false); + if (request.empty() || session.empty() || note.empty() || parsed.is_discarded() || (outcome != "committed" && outcome != "failed")) { + std::cerr << "Reconciliation requires IDs, committed|failed outcome, evidence note and valid JSON output\n"; return 2; + } + auto status = runtime.reconcile(session, request, {outcome == "committed", parsed, note}); + std::cout << Json{{"ok", status.ok}, {"error", {{"code", status.error_code}, {"message", status.error_message}}}}.dump() << '\n'; + return status ? 0 : 1; + } int temperature = 20; const Json empty_schema{{"type", "object"}, {"properties", Json::object()}, {"additionalProperties", false}}; const Json temperature_schema{{"type", "object"}, {"properties", {{"temperature", {{"type", "integer"}, {"minimum", 16}, {"maximum", 30}}}}}, {"required", {"temperature"}}, {"additionalProperties", false}}; @@ -67,7 +126,12 @@ int main(int argc, char** argv) { #ifdef MASTER_AGENT_HAS_HTTP if (endpoint.find("://") == std::string::npos) endpoint = "http://" + endpoint; if (endpoint.find('/', endpoint.find("://") + 3) == std::string::npos) endpoint += "/v1/chat/completions"; - HttpModelConfig config; config.endpoint = endpoint; config.model = model; + HttpModelConfig config; config.endpoint = endpoint; config.model = model; config.timeout_ms = timeout_ms; + if (!key_env.empty()) { + const char* key = std::getenv(key_env.c_str()); + if (!key || !*key) { std::cerr << "Requested credential environment variable is empty\n"; return 2; } + config.api_key = key; + } runtime.setModel(makeHttpModel(std::move(config))); #else std::cerr << "HTTP support is disabled; rebuild with MASTER_AGENT_ENABLE_HTTP=ON\n"; return 2; @@ -86,7 +150,16 @@ int main(int argc, char** argv) { unsigned sequence = 0; bool success = true; while (std::getline(std::cin, input)) { - auto result = runtime.run({session, std::to_string(++sequence), input}); + Result result; + if (jsonl) { + try { + const auto item = Json::parse(input); + if (!item.is_object() || item.size() != 3) throw std::runtime_error("Expected session_id, request_id and input"); + result = runtime.run({item.at("session_id").get(), item.at("request_id").get(), item.at("input").get()}); + } catch (const std::exception&) { + result = Result::failure({"INVALID_REQUEST", "Expected JSON object with session_id, request_id and input strings", "", 400}); + } + } else result = runtime.run({session, std::to_string(++sequence), input}); std::cout << encode(result).dump() << std::endl; success = success && result.ok(); } diff --git a/cmake/MasterAgentConfig.cmake.in b/cmake/MasterAgentConfig.cmake.in index f416692..b004eae 100644 --- a/cmake/MasterAgentConfig.cmake.in +++ b/cmake/MasterAgentConfig.cmake.in @@ -2,6 +2,11 @@ include(CMakeFindDependencyMacro) find_dependency(Threads) +set(MasterAgent_WITH_STORAGE OFF) +if(@MASTER_AGENT_ENABLE_STORAGE@ AND NOT @_core_source_present@) + set(MasterAgent_WITH_STORAGE ON) + find_dependency(SQLite3) +endif() if(@MASTER_AGENT_ENABLE_HTTP@) find_dependency(CURL) endif() diff --git a/cmake/ReferenceTests.cmake b/cmake/ReferenceTests.cmake index 31ef3cf..458c50e 100644 --- a/cmake/ReferenceTests.cmake +++ b/cmake/ReferenceTests.cmake @@ -25,3 +25,16 @@ add_test(NAME release_assertions_are_active COMMAND test_check_failure) set_tests_properties(release_assertions_are_active PROPERTIES WILL_FAIL TRUE) add_test(NAME evaluation_script_contract COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/tests/test_eval_runner.py ${CMAKE_CURRENT_SOURCE_DIR}/eval/run_all.sh) +if(MASTER_AGENT_ENABLE_STORAGE) + add_executable(test_durable_runtime tests/test_durable_runtime.cpp) + target_link_libraries(test_durable_runtime PRIVATE MasterAgent::Core) + add_test(NAME test_durable_runtime COMMAND test_durable_runtime) + if(MASTER_AGENT_BUILD_CLI) + add_executable(durable_probe tests/durable_probe.cpp) + target_link_libraries(durable_probe PRIVATE MasterAgent::Core) + add_test(NAME crash_recovery_contract + COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/tests/test_crash_recovery.py + $ $) + set_tests_properties(crash_recovery_contract PROPERTIES TIMEOUT 60) + endif() +endif() diff --git a/core/durable_store.cpp b/core/durable_store.cpp new file mode 100644 index 0000000..604b9d9 --- /dev/null +++ b/core/durable_store.cpp @@ -0,0 +1,250 @@ +#include "durable_store.h" +#include +#include +#ifdef MASTER_AGENT_HAS_SQLITE +#include +#include +#include +#include +#include +#endif + +namespace master_agent::reference::detail { +#ifdef MASTER_AGENT_HAS_SQLITE +namespace { +constexpr int application_id = 0x4D415254; // MART, distinct from unrelated SQLite databases. +struct Statement { + sqlite3_stmt* stmt = nullptr; + sqlite3* db; + Statement(sqlite3* connection, const char* sql) : db(connection) { + if (sqlite3_prepare_v2(db, sql, -1, &stmt, nullptr) != SQLITE_OK) + throw std::runtime_error(sqlite3_errmsg(db)); + } + ~Statement() { sqlite3_finalize(stmt); } + void bind(int index, const std::string& value) { + if (sqlite3_bind_text(stmt, index, value.data(), static_cast(value.size()), SQLITE_TRANSIENT) != SQLITE_OK) + throw std::runtime_error(sqlite3_errmsg(db)); + } + bool next() { + int code = sqlite3_step(stmt); + if (code == SQLITE_ROW) return true; + if (code != SQLITE_DONE) throw std::runtime_error(sqlite3_errmsg(db)); + return false; + } + std::string text(int column) const { + const auto* bytes = sqlite3_column_text(stmt, column); + if (!bytes) throw std::runtime_error("Unexpected NULL in store"); + return {reinterpret_cast(bytes), static_cast(sqlite3_column_bytes(stmt, column))}; + } +}; +void sql(sqlite3* db, const char* statement) { + if (sqlite3_exec(db, statement, nullptr, nullptr, nullptr) != SQLITE_OK) + throw std::runtime_error(sqlite3_errmsg(db)); +} +struct Transaction { + sqlite3* db; + bool committed = false; + explicit Transaction(sqlite3* connection) : db(connection) { sql(db, "BEGIN IMMEDIATE"); } + ~Transaction() { if (!committed) sqlite3_exec(db, "ROLLBACK", nullptr, nullptr, nullptr); } + void commit() { sql(db, "COMMIT"); committed = true; } +}; +int scalar(sqlite3* db, const char* query) { + Statement statement(db, query); + if (!statement.next()) throw std::runtime_error("Missing store metadata"); + return sqlite3_column_int(statement.stmt, 0); +} +Json encode(const Result& result) { + if (result) { + const auto& r = *result; + return {{"ok", true}, {"session", r.session_id}, {"request", r.request_id}, + {"route", r.route}, {"tool", r.tool}, {"output", r.output}}; + } + const auto e = result.error.value_or(StructuredError{"INTERNAL", "Missing result", "", 500}); + return {{"ok", false}, {"code", e.code}, {"message", e.message}, {"detail", e.detail}, {"http_status", e.http_status}}; +} +Result decode(const Json& value) { + if (value.at("ok").get()) + return Result::success({value.at("session").get(), value.at("request").get(), + value.at("route").get(), value.at("tool").get(), value.at("output"), false}); + return Result::failure({value.at("code").get(), value.at("message").get(), + value.at("detail").get(), value.at("http_status").get()}); +} +Result unknown() { + return Result::failure({"UNKNOWN", "Interrupted request; reconcile the external outcome before retrying", "", 409}); +} +std::string stateFor(const Result& result) { + if (result) return "COMMITTED"; + return result.error && result.error->code == "UNKNOWN" ? "UNKNOWN" : "FAILED"; +} +void event(sqlite3* db, const Turn& turn, const std::string& kind, const std::string& note = "") { + Statement statement(db, "INSERT INTO events(session_id,request_id,event,note) VALUES(?,?,?,?)"); + statement.bind(1, turn.session_id); statement.bind(2, turn.request_id); + statement.bind(3, kind); statement.bind(4, note); statement.next(); +} +Status error(const std::exception& e) { return Status::Error("STORAGE_ERROR", e.what()); } +} +struct DurableStore::Impl { + sqlite3* db = nullptr; + int lock_fd = -1; + ~Impl() { + if (db) sqlite3_close(db); + if (lock_fd >= 0) { flock(lock_fd, LOCK_UN); close(lock_fd); } + } +}; +DurableStore::DurableStore() : impl_(std::make_unique()) {} +DurableStore::~DurableStore() = default; +Status DurableStore::open(const std::string& path) { + try { + if (path.empty()) return Status::Error("INVALID_STORE", "Store path must not be empty"); + auto parent = std::filesystem::path(path).parent_path(); + if (!parent.empty()) std::filesystem::create_directories(parent); + // Keep an independent lock file: Apple's SQLite VFS can use flock on + // the database itself. Canonical parents and rejected aliases keep the + // ownership lock stable across relative paths and symlinked parents. + const auto normalized = (std::filesystem::weakly_canonical(parent.empty() ? std::filesystem::path(".") : parent) / + std::filesystem::path(path).filename()).string(); + impl_->lock_fd = ::open((normalized + ".lock").c_str(), O_RDWR | O_CREAT | O_CLOEXEC | O_NOFOLLOW, 0600); + if (impl_->lock_fd < 0) return Status::Error("STORAGE_ERROR", "Cannot open store ownership lock"); + struct stat info{}; + if (fstat(impl_->lock_fd, &info) != 0 || !S_ISREG(info.st_mode) || info.st_nlink != 1) + return Status::Error("INVALID_STORE", "Ownership lock must be a regular unaliased file"); + if (flock(impl_->lock_fd, LOCK_EX | LOCK_NB) != 0) + return Status::Error("STORAGE_BUSY", "Another runtime owns this store"); + const int db_fd = ::open(normalized.c_str(), O_RDWR | O_CREAT | O_CLOEXEC | O_NOFOLLOW, 0600); + if (db_fd < 0) return Status::Error("STORAGE_ERROR", "Cannot open regular store file"); + const bool regular = fstat(db_fd, &info) == 0 && S_ISREG(info.st_mode) && info.st_nlink == 1; + close(db_fd); + if (!regular) return Status::Error("INVALID_STORE", "Store must be a regular file without hard links"); + if (sqlite3_open_v2(normalized.c_str(), &impl_->db, SQLITE_OPEN_READWRITE | SQLITE_OPEN_FULLMUTEX, nullptr) != SQLITE_OK) + return Status::Error("STORAGE_ERROR", "Cannot open SQLite store"); + sqlite3_busy_timeout(impl_->db, 1000); + const int id = scalar(impl_->db, "PRAGMA application_id"); + const int version = scalar(impl_->db, "PRAGMA user_version"); + if (id != application_id && !(id == 0 && version == 0 && scalar(impl_->db, "SELECT count(*) FROM sqlite_master WHERE name NOT LIKE 'sqlite_%'") == 0)) + return Status::Error("INVALID_STORE", "Not a MasterAgent request store"); + if ((id == application_id && version != 1) || (id == 0 && version != 0)) + return Status::Error("STORE_VERSION", "Unsupported store schema version"); + Statement check(impl_->db, "PRAGMA quick_check"); + if (!check.next() || check.text(0) != "ok") return Status::Error("STORAGE_ERROR", "Store integrity check failed"); + // Finish the read before changing journal mode. + while (check.next()) {} + if (fchmod(impl_->lock_fd, 0600) != 0 || chmod(normalized.c_str(), 0600) != 0) return Status::Error("STORAGE_ERROR", "Cannot restrict store permissions"); + sql(impl_->db, "PRAGMA journal_mode=WAL"); + sql(impl_->db, "PRAGMA synchronous=FULL"); + sql(impl_->db, "PRAGMA foreign_keys=ON"); + if (id == 0) { + Transaction transaction(impl_->db); + sql(impl_->db, "CREATE TABLE sessions(session_id TEXT PRIMARY KEY,history TEXT NOT NULL)"); + sql(impl_->db, "CREATE TABLE requests(session_id TEXT NOT NULL REFERENCES sessions(session_id) ON DELETE CASCADE,request_id TEXT NOT NULL,input TEXT NOT NULL,state TEXT NOT NULL,result TEXT NOT NULL,tool TEXT NOT NULL DEFAULT '',arguments TEXT NOT NULL DEFAULT '{}',PRIMARY KEY(session_id,request_id))"); + sql(impl_->db, "CREATE TABLE events(sequence INTEGER PRIMARY KEY,session_id TEXT NOT NULL,request_id TEXT NOT NULL,event TEXT NOT NULL,note TEXT NOT NULL,created_at TEXT NOT NULL DEFAULT CURRENT_TIMESTAMP)"); + sql(impl_->db, "PRAGMA application_id=1296126548"); + sql(impl_->db, "PRAGMA user_version=1"); + transaction.commit(); + } + return Status::Ok(); + } catch (const std::exception& e) { return error(e); } +} +Result DurableStore::load(const Limits& limits) { + try { + Snapshot snapshot; + Transaction transaction(impl_->db); + Statement sessions(impl_->db, "SELECT session_id,history FROM sessions"); + while (sessions.next()) { + if (snapshot.histories.size() >= limits.sessions) throw std::runtime_error("Stored sessions exceed configured limit"); + auto id = sessions.text(0); + auto history = Json::parse(sessions.text(1)); + if (id.empty() || id.size() > 256 || !history.is_array() || history.size() % 2 != 0) + throw std::runtime_error("Invalid stored session"); + for (const auto& message : history) + if (!message.is_object() || !message.contains("role") || !message["role"].is_string() || + !message.contains("content") || !message["content"].is_string() || + (message["role"] != "user" && message["role"] != "assistant")) + throw std::runtime_error("Invalid stored history"); + auto& target = snapshot.histories[id]; + for (const auto& message : history) target.push_back(message); + while (target.size() / 2 > limits.history_turns) target.erase(target.begin(), target.begin() + 2); + } + std::map counts; + Statement requests(impl_->db, "SELECT session_id,request_id,input,state,result,tool,arguments FROM requests ORDER BY rowid"); + while (requests.next()) { + StoredRequest r; + r.turn = {requests.text(0), requests.text(1), requests.text(2)}; + r.state = requests.text(3); r.tool = requests.text(5); r.arguments = Json::parse(requests.text(6)); + if (!snapshot.histories.count(r.turn.session_id) || r.turn.request_id.empty() || r.turn.request_id.size() > 256 || + r.turn.input.empty() || r.turn.input.size() > 65536 || !r.arguments.is_object() || + ++counts[r.turn.session_id] > limits.requests_per_session) + throw std::runtime_error("Invalid stored request or exceeded request limit"); + if (r.state == "STARTED" || r.state == "DISPATCHED") r.result = unknown(); + else { + r.result = decode(Json::parse(requests.text(4))); + if (stateFor(r.result) != r.state || (r.result && (r.result.value->session_id != r.turn.session_id || r.result.value->request_id != r.turn.request_id))) + throw std::runtime_error("Inconsistent stored outcome"); + } + snapshot.requests.push_back(std::move(r)); + } + for (auto& r : snapshot.requests) { + if (r.state != "STARTED" && r.state != "DISPATCHED") continue; + Statement update(impl_->db, "UPDATE requests SET state='UNKNOWN',result=? WHERE session_id=? AND request_id=?"); + update.bind(1, encode(r.result).dump()); update.bind(2, r.turn.session_id); update.bind(3, r.turn.request_id); update.next(); + event(impl_->db, r.turn, "RECOVERED_UNKNOWN"); r.state = "UNKNOWN"; + } + transaction.commit(); + return Result::success(std::move(snapshot)); + } catch (const std::exception& e) { return Result::failure({"STORAGE_ERROR", e.what(), "", 500}); } +} +Status DurableStore::begin(const Turn& turn) { + try { + Transaction transaction(impl_->db); + Statement session(impl_->db, "INSERT OR IGNORE INTO sessions(session_id,history) VALUES(?,'[]')"); + session.bind(1, turn.session_id); session.next(); + Statement request(impl_->db, "INSERT INTO requests(session_id,request_id,input,state,result) VALUES(?,?,?,'STARTED','{}')"); + request.bind(1, turn.session_id); request.bind(2, turn.request_id); request.bind(3, turn.input); request.next(); + event(impl_->db, turn, "STARTED"); transaction.commit(); return Status::Ok(); + } catch (const std::exception& e) { return error(e); } +} +Status DurableStore::dispatch(const Turn& turn, const std::string& tool, const Json& arguments) { + try { + Transaction transaction(impl_->db); + Statement update(impl_->db, "UPDATE requests SET state='DISPATCHED',tool=?,arguments=? WHERE session_id=? AND request_id=? AND state='STARTED'"); + update.bind(1, tool); update.bind(2, arguments.dump()); update.bind(3, turn.session_id); update.bind(4, turn.request_id); update.next(); + if (sqlite3_changes(impl_->db) != 1) throw std::runtime_error("Invalid dispatch transition"); + event(impl_->db, turn, "DISPATCHED"); transaction.commit(); return Status::Ok(); + } catch (const std::exception& e) { return error(e); } +} +Status DurableStore::complete(const Turn& turn, const Result& result, + const std::vector& history, const std::string& note) { + try { + Transaction transaction(impl_->db); + Statement update(impl_->db, "UPDATE requests SET state=?,result=? WHERE session_id=? AND request_id=? AND state IN ('STARTED','DISPATCHED','UNKNOWN')"); + update.bind(1, stateFor(result)); update.bind(2, encode(result).dump()); + update.bind(3, turn.session_id); update.bind(4, turn.request_id); update.next(); + if (sqlite3_changes(impl_->db) != 1) throw std::runtime_error("Invalid completion transition"); + Statement session(impl_->db, "UPDATE sessions SET history=? WHERE session_id=?"); + session.bind(1, Json(history).dump()); session.bind(2, turn.session_id); session.next(); + event(impl_->db, turn, note.empty() ? stateFor(result) : "RECONCILED_" + stateFor(result), note); + transaction.commit(); return Status::Ok(); + } catch (const std::exception& e) { return error(e); } +} +Status DurableStore::clear(const std::string& id) { + try { + Transaction transaction(impl_->db); + Statement pending(impl_->db, "SELECT 1 FROM requests WHERE session_id=? AND state IN ('STARTED','DISPATCHED','UNKNOWN')"); + pending.bind(1, id); + if (pending.next()) return Status::Error("UNRESOLVED_REQUESTS", "Reconcile unknown outcomes before clearing the session"); + Statement remove(impl_->db, "DELETE FROM sessions WHERE session_id=?"); remove.bind(1, id); remove.next(); + event(impl_->db, {id, "", ""}, "SESSION_CLEARED"); transaction.commit(); return Status::Ok(); + } catch (const std::exception& e) { return error(e); } +} +#else +struct DurableStore::Impl {}; +DurableStore::DurableStore() : impl_(std::make_unique()) {} +DurableStore::~DurableStore() = default; +static Status disabled() { return Status::Error("STORAGE_UNAVAILABLE", "Build with MASTER_AGENT_ENABLE_STORAGE=ON"); } +Status DurableStore::open(const std::string&) { return disabled(); } +Result DurableStore::load(const Limits&) { return Result::failure({"STORAGE_UNAVAILABLE", "Storage disabled", "", 500}); } +Status DurableStore::begin(const Turn&) { return disabled(); } +Status DurableStore::dispatch(const Turn&, const std::string&, const Json&) { return disabled(); } +Status DurableStore::complete(const Turn&, const Result&, const std::vector&, const std::string&) { return disabled(); } +Status DurableStore::clear(const std::string&) { return disabled(); } +#endif +} // namespace master_agent::reference::detail diff --git a/core/durable_store.h b/core/durable_store.h new file mode 100644 index 0000000..77afde2 --- /dev/null +++ b/core/durable_store.h @@ -0,0 +1,32 @@ +#pragma once +#include "master_agent/runtime/reference_runtime.h" + +namespace master_agent::reference::detail { +struct StoredRequest { + Turn turn; + std::string state; + std::string tool; + Json arguments; + Result result; +}; +struct Snapshot { + std::vector requests; + std::map> histories; +}; + +class DurableStore { +public: + DurableStore(); + ~DurableStore(); + Status open(const std::string& path); + Result load(const Limits& limits); + Status begin(const Turn& turn); + Status dispatch(const Turn& turn, const std::string& tool, const Json& arguments); + Status complete(const Turn& turn, const Result& result, + const std::vector& history, const std::string& note = ""); + Status clear(const std::string& session_id); +private: + struct Impl; + std::unique_ptr impl_; +}; +} // namespace master_agent::reference::detail diff --git a/core/reference_runtime.cpp b/core/reference_runtime.cpp index ba28c81..fb9bf04 100644 --- a/core/reference_runtime.cpp +++ b/core/reference_runtime.cpp @@ -1,4 +1,5 @@ #include "master_agent/runtime/reference_runtime.h" +#include "durable_store.h" #include #include #include @@ -74,6 +75,79 @@ Status validate(const Json& args, const Json& schema) { } // namespace Runtime::Runtime(Limits limits) : limits_(limits) {} +Runtime::~Runtime() = default; + +Status Runtime::openStore(const std::string& path) { + std::unique_lock lock(mutex_, std::try_to_lock); + if (!lock) return Status::Error("BUSY"); + if (store_ || !sessions_.empty()) return Status::Error("STORE_ALREADY_ACTIVE", "Open a store before creating sessions"); + auto candidate = std::make_unique(); + storage_status_ = candidate->open(path); + if (!storage_status_) return storage_status_; + auto snapshot = candidate->load(limits_); + if (!snapshot) { + storage_status_ = Status::Error("STORAGE_ERROR", snapshot.error->message); + return storage_status_; + } + std::map recovered; + for (auto& entry : snapshot.value->histories) recovered[entry.first].history = std::move(entry.second); + for (auto& entry : snapshot.value->requests) { + recovered[entry.turn.session_id].requests.emplace(entry.turn.request_id, + Cached{entry.turn.input, std::move(entry.result), entry.tool, std::move(entry.arguments)}); + } + sessions_ = std::move(recovered); + store_ = std::move(candidate); + return Status::Ok(); +} + +Result> Runtime::unresolved() { + std::unique_lock lock(mutex_, std::try_to_lock); + if (!lock) return fail>("BUSY", "Runtime is executing"); + std::vector records; + for (const auto& session : sessions_) for (const auto& request : session.second.requests) { + const auto& cached = request.second; + if (cached.result.error && cached.result.error->code == "UNKNOWN") + records.push_back({{session.first, request.first, cached.input}, cached.tool, cached.arguments}); + } + return Result>::success(std::move(records)); +} + +void Runtime::remember(Session& session, const Turn& turn, const Reply& reply) { + Json user{{"role", "user"}, {"content", turn.input}}; + Json assistant{{"role", "assistant"}, {"content", reply.output.dump()}}; + session.history.reserve(session.history.size() + 2); + session.history.push_back(std::move(user)); + session.history.push_back(std::move(assistant)); + while (session.history.size() / 2 > limits_.history_turns) + session.history.erase(session.history.begin(), session.history.begin() + 2); +} + +Status Runtime::reconcile(const std::string& session_id, const std::string& request_id, const Resolution& resolution) { + std::unique_lock lock(mutex_, std::try_to_lock); + if (!lock) return Status::Error("BUSY"); + if (!storage_status_) return storage_status_; + if (resolution.note.empty() || resolution.note.size() > 4096) return Status::Error("INVALID_RESOLUTION", "A bounded evidence note is required"); + auto session = sessions_.find(session_id); + if (session == sessions_.end() || !session->second.requests.count(request_id)) return Status::Error("NOT_FOUND", "No such request"); + auto& cached = session->second.requests.at(request_id); + if (!cached.result.error || cached.result.error->code != "UNKNOWN") return Status::Error("NOT_UNKNOWN", "Only UNKNOWN requests can be reconciled"); + try { + if (resolution.output.dump().size() > limits_.output_bytes) return Status::Error("INVALID_RESOLUTION", "Output exceeds limit"); + Json(resolution.note).dump(); // Validate UTF-8 before any store mutation. + auto result = resolution.committed + ? Result::success({session_id, request_id, "reconciled", cached.tool, resolution.output, false}) + : fail("RECONCILED_FAILED", resolution.note); + auto previous_history = session->second.history; + if (result) remember(session->second, {session_id, request_id, cached.input}, *result); + if (store_) { + storage_status_ = store_->complete({session_id, request_id, cached.input}, result, session->second.history, resolution.note); + if (!storage_status_) { session->second.history = std::move(previous_history); return storage_status_; } + } + cached.result = std::move(result); + return Status::Ok(); + } catch (const Json::exception& e) { return Status::Error("INVALID_RESOLUTION", e.what()); } +} + Status Runtime::registerTool(Tool tool) { std::unique_lock lock(mutex_, std::try_to_lock); if (!lock) return Status::Error("BUSY"); @@ -101,14 +175,23 @@ Status Runtime::setModel(ModelHandler model) { Status Runtime::clearSession(const std::string& id) { std::unique_lock lock(mutex_, std::try_to_lock); if (!lock) return Status::Error("BUSY"); + if (!storage_status_) return storage_status_; + if (auto it = sessions_.find(id); it != sessions_.end()) + for (const auto& request : it->second.requests) + if (request.second.result.error && request.second.result.error->code == "UNKNOWN") + return Status::Error("UNRESOLVED_REQUESTS", "Reconcile unknown outcomes before clearing the session"); + if (store_) { storage_status_ = store_->clear(id); if (!storage_status_) return storage_status_; } sessions_.erase(id); return Status::Ok(); } Result Runtime::run(const Turn& turn) { std::unique_lock lock(mutex_, std::try_to_lock); if (!lock) return fail("BUSY", "Runtime is executing another operation"); + if (!storage_status_) return fail(storage_status_.error_code, storage_status_.error_message); if (turn.session_id.empty() || turn.request_id.empty() || turn.input.empty() || turn.input.size() > 65536 || turn.session_id.size() > 256 || turn.request_id.size() > 256) return fail("INVALID_REQUEST", "Nonempty bounded session, request and input are required"); + try { Json{{"session", turn.session_id}, {"request", turn.request_id}, {"input", turn.input}}.dump(); } + catch (const Json::exception&) { return fail("INVALID_REQUEST", "Request text must be valid UTF-8"); } if (!sessions_.count(turn.session_id) && sessions_.size() >= limits_.sessions) return fail("RESOURCE_LIMIT", "Session limit reached"); auto& session = sessions_[turn.session_id]; if (auto it = session.requests.find(turn.request_id); it != session.requests.end()) { @@ -118,14 +201,30 @@ Result Runtime::run(const Turn& turn) { return result; } if (session.requests.size() >= limits_.requests_per_session) return fail("RESOURCE_LIMIT", "Request limit reached; start a new session"); - auto result = execute(turn, session); - // Cache failures too: an UNKNOWN tool outcome must never trigger an automatic retry. - session.requests.emplace(turn.request_id, Cached{turn.input, result}); - if (result) { - session.history.push_back({{"role", "user"}, {"content", turn.input}}); - session.history.push_back({{"role", "assistant"}, {"content", result.value->output.dump()}}); - while (session.history.size() / 2 > limits_.history_turns) session.history.erase(session.history.begin(), session.history.begin() + 2); + if (store_) { + storage_status_ = store_->begin(turn); + if (!storage_status_) return fail(storage_status_.error_code, storage_status_.error_message); + } + auto& cached = session.requests.emplace(turn.request_id, Cached{turn.input, + fail("UNKNOWN", "Request started; outcome not recorded"), "", Json::object()}).first->second; + auto previous_history = session.history; + Result result; + try { + result = execute(turn, session); + if (result) { + if (result.value->output.dump().size() > limits_.output_bytes) + result = fail(cached.tool.empty() ? "OUTPUT_LIMIT" : "UNKNOWN", "Output exceeds limit; inspect any tool side effect"); + else remember(session, turn, *result); + } + } catch (const std::exception&) { + result = fail(cached.tool.empty() ? "INTERNAL" : "UNKNOWN", "Unable to encode the outcome; inspect any tool side effect"); + } + if (store_ && storage_status_) storage_status_ = store_->complete(turn, result, session.history); + if (!storage_status_) { + session.history = std::move(previous_history); + result = fail("UNKNOWN", "Could not durably record the outcome; reopen the store and reconcile before retrying"); } + cached.result = result; return result; } Result Runtime::execute(const Turn& turn, Session& session) { @@ -164,6 +263,13 @@ Result Runtime::execute(const Turn& turn, Session& session) { if (tool == tools_.end()) return fail("UNKNOWN_TOOL", "Model selected an unregistered tool"); auto valid = validate(arguments, tool->second.parameters); if (!valid) return fail(valid.error_code, valid.error_message); + auto& receipt = session.requests.at(turn.request_id); + receipt.tool = reply.tool; + receipt.arguments = arguments; + if (store_) { + storage_status_ = store_->dispatch(turn, reply.tool, arguments); + if (!storage_status_) return fail("STORAGE_ERROR", "Dispatch was not persisted; tool was not called"); + } try { auto result = tool->second.execute(arguments); if (!result) return Result::failure(result.error.value_or(StructuredError{"TOOL_ERROR", "Tool returned no result", "", 500})); diff --git a/docs/durable_recovery.md b/docs/durable_recovery.md new file mode 100644 index 0000000..76c2ae1 --- /dev/null +++ b/docs/durable_recovery.md @@ -0,0 +1,94 @@ +# Durable requests and crash recovery + +The public reference runtime can persist request receipts and successful conversation +history in SQLite. Enable `MASTER_AGENT_ENABLE_STORAGE` (default ON for the public +build), then call `Runtime::openStore(path)` before creating sessions. Linux/macOS +are supported. A failed open blocks subsequent execution; there is no silent fallback +to memory. An instance owns the store until its destructor runs. + +## State transitions + +`STARTED` is committed before routing/model execution. After a tool name and arguments +pass validation, `DISPATCHED` is committed **before invoking the callback**. Completion +atomically commits the receipt, bounded session history and an audit event as +`COMMITTED`, `FAILED` or `UNKNOWN`. + +SQLite uses WAL and `synchronous=FULL`. File permissions are restricted to the owner. +A separate ownership lock prevents a second cooperating runtime/process from opening +the same store while work is active. The lock file must remain present. Database +symlinks/hard links are rejected; use a local filesystem, not NFS or shared storage. + +When reopening an exclusively owned store, unfinished `STARTED`/`DISPATCHED` requests +become `UNKNOWN`. This is deliberately conservative: the external system may have +performed a side effect before the process stopped. Repeating the same session/request +ID returns the stored outcome without rerunning the model or tool. A different input +with that ID returns `REQUEST_CONFLICT`. + +This is **not a distributed exactly-once guarantee**. There is no atomic transaction +spanning a device API and SQLite. Where a tool supports an idempotency key, the host +should pass the request identity through to that system as well. This release handles +one tool per turn; it is not a durable multi-node DAG scheduler. + +## CLI + +```bash +sparx run --state ./state/agent.db --session alice --request-id command-001 \ + --input 'set AC to 22 degrees' +# The same invocation after a restart returns replayed:true. +sparx recover --state ./state/agent.db +``` + +The AC example remains an in-memory **simulator**. A durable receipt records what a +previous invocation returned; it does not restore physical/device or simulator state. +Use the installed C++ example and crash tests to study the runtime/tool boundary. + +Durable CLI requests require explicit IDs. For a long-running process: + +```bash +printf '%s\n' '{"session_id":"alice","request_id":"command-002","input":"vehicle status"}' \ + | sparx run --state ./state/agent.db --jsonl +``` + +`recover` reports the input, selected tool and validated arguments for unresolved +requests. Inspect the external system before choosing an outcome: + +```bash +sparx reconcile --state ./state/agent.db --session alice --request-id command-001 \ + --outcome committed --output '{"temperature":22}' \ + --note 'Verified the device reports the requested state' +# Or: --outcome failed --note 'Device audit confirms no operation occurred' +``` + +Reconciliation never invokes a tool. It records the operator's conclusion and evidence +note; the runtime cannot establish that evidence itself. Only `UNKNOWN` receipts can +be reconciled. Notes must be nonempty. Final outcomes cannot be rewritten through this +API. `clearSession()` refuses unresolved requests and explicitly discards the replay +protection of resolved ones; audit events remain in the database. + +## Failures and storage maintenance + +- A failed write before dispatch prevents the callback from running. +- Failure to commit a completion returns `UNKNOWN` and blocks further execution on + that instance. Destroy it, fix the storage problem, reopen, and reconcile. +- Corrupt, unrelated, or future-version databases fail closed. Back up and inspect + them; do not delete the database to make an ambiguous request run again. +- A second owner gets `STORAGE_BUSY`; a failed open never executes in memory. +- Session/request limits are preserved across restarts; receipts are never silently + evicted. Operators own retention and storage capacity planning. + +The database includes prompt text, arguments, outputs and notes. It is **not encrypted**. +Use operating-system storage protection as required by the deployment. Do not edit, +replace, copy, or delete a live database or its `.lock`/`-wal`/`-shm` files. For a simple +backup, stop the runtime and copy the closed database. Restoring an older backup also +restores older replay knowledge; reconcile external operations since that backup before +resuming side-effecting tools. A SQLite-aware online backup requires application-level +coordination beyond this reference release. + +## Tests + +`test_durable_runtime` covers reopen/replay, history restoration, explicit reconciliation, +limits and exclusive ownership. `crash_recovery_contract` starts a subprocess, fsyncs an +external effect and exits before the receipt commits. A fresh process must report +`UNKNOWN` and must not produce a second effect. It also injects database failures before +start, before dispatch and on completion, and checks corruption/version rejection. +These are process-crash tests, not destructive hardware power-loss tests. diff --git a/docs/reference_runtime.md b/docs/reference_runtime.md index 070a7c3..e15f465 100644 --- a/docs/reference_runtime.md +++ b/docs/reference_runtime.md @@ -43,15 +43,16 @@ model history, 64 KiB input/model text, and 256-byte IDs. A full session returns Request IDs are scoped to the session. Reusing an ID with different input returns `REQUEST_CONFLICT`. Both success and failure are cached. For a deliberate -retry, the caller must inspect the outcome and provide a new ID. All state is -memory-only: deduplication is not guaranteed across process restarts or separate -runtime instances. Do not use it as a durable exactly-once execution guarantee. - +retry, the caller must inspect the outcome and provide a new ID. With no store, state is memory-only and deduplication does not survive restart. +Call `openStore(path)` before creating sessions to persist receipts and history in +SQLite. Recovery is conservative: interrupted requests become UNKNOWN and cannot +be automatically retried. See [durable recovery](durable_recovery.md). A handler exception returns `UNKNOWN`, because its side effect may already have occurred. Repeating that ID returns the same error without invoking the handler again. Tool handlers should return explicit `Result` failures when the outcome -is known. Durable recovery, reconciliation storage, and journal encryption are -future work, not hidden behavior of this release. +is known. Use `unresolved()` to inspect ambiguous operations and `reconcile()` to record an +operator-verified outcome and evidence note. `clearSession()` refuses unknown +outcomes. The store does not encrypt data and does not implement a multi-step DAG. ## Concurrency and timeouts @@ -62,8 +63,8 @@ callbacks cannot be safely preempted by this library. The optional libcurl model callback has a finite HTTP timeout (30 seconds by default), disables redirects, verifies TLS using libcurl defaults, and limits -responses to 1 MiB. Credentials are supplied through the SDK config, not a CLI -argument. Endpoints are explicit; the reference CLI defaults to no model. +responses to 1 MiB. Credentials are supplied through the SDK config or a CLI-named environment +variable, never a CLI credential argument. Endpoints are explicit; the reference CLI defaults to no model. The separate experimental edge/cloud harness uses a common inference deadline and retains shared backend ownership for late workers. It permits at most one diff --git a/examples/reference_agent/CMakeLists.txt b/examples/reference_agent/CMakeLists.txt index ad06767..ffec51a 100644 --- a/examples/reference_agent/CMakeLists.txt +++ b/examples/reference_agent/CMakeLists.txt @@ -5,3 +5,13 @@ add_executable(consumer main.cpp) target_link_libraries(consumer PRIVATE MasterAgent::Core) enable_testing() add_test(NAME run_consumer COMMAND consumer) + +if(TARGET MasterAgent::Http) + target_link_libraries(consumer PRIVATE MasterAgent::Http) + target_compile_definitions(consumer PRIVATE TEST_HTTP=1) +endif() + +if(MasterAgent_WITH_STORAGE) + target_compile_definitions(consumer PRIVATE TEST_STORAGE=1) + add_test(NAME installed_persistent_replay COMMAND consumer "${CMAKE_CURRENT_BINARY_DIR}/consumer.db") +endif() diff --git a/examples/reference_agent/main.cpp b/examples/reference_agent/main.cpp index cc6e8ba..529e92f 100644 --- a/examples/reference_agent/main.cpp +++ b/examples/reference_agent/main.cpp @@ -1,8 +1,29 @@ #include +#ifdef TEST_HTTP +#include +#endif using namespace master_agent; using namespace master_agent::reference; -int main() { +int main(int argc, char** argv) { +#ifdef TEST_STORAGE + if (argc == 2) { + { + Runtime durable; + if (!durable.openStore(argv[1])) return 5; + if (!durable.setModel([](const Json&) { return Result::success(R"({"response":"stored"})"); })) return 6; + if (!durable.run({"installed", "1", "persist this"})) return 7; + } + Runtime reopened; + if (!reopened.openStore(argv[1])) return 8; + auto receipt = reopened.run({"installed", "1", "persist this"}); + if (!receipt || !receipt.value->replayed || receipt.value->output != "stored") return 9; + return reopened.clearSession("installed") ? 0 : 10; + } +#endif Runtime runtime; +#ifdef TEST_HTTP + if (!runtime.setModel(makeHttpModel({}))) return 4; +#endif int calls = 0; Json schema{{"type", "object"}, {"properties", Json::object()}, {"additionalProperties", false}}; if (!runtime.registerTool({"ping", "Consumer integration", schema, diff --git a/include/master_agent/runtime/reference_runtime.h b/include/master_agent/runtime/reference_runtime.h index fabf388..ec8f4b1 100644 --- a/include/master_agent/runtime/reference_runtime.h +++ b/include/master_agent/runtime/reference_runtime.h @@ -4,6 +4,7 @@ #include #include #include +#include #include #include #include @@ -43,34 +44,62 @@ struct Limits { std::size_t sessions = 32; std::size_t requests_per_session = 256; std::size_t history_turns = 8; + std::size_t output_bytes = 65536; }; +struct RecoveryRecord { + Turn turn; + std::string tool; + Json arguments; +}; + +struct Resolution { + bool committed = false; + Json output; + std::string note; // Required explanation/evidence supplied by the operator. +}; + +namespace detail { class DurableStore; } + // A synchronous, in-process reference runtime. Concurrent/reentrant calls return -// BUSY. Tool callbacks are trusted host code, not sandboxed. Session history and -// request replay are memory-only; this is not the proprietary durable kernel. +// BUSY. Tool callbacks are trusted host code, not sandboxed. Call openStore() +// before run() for persistent receipts/history and conservative crash recovery. class Runtime { public: explicit Runtime(Limits limits = {}); + ~Runtime(); + Status openStore(const std::string& path); + Result> unresolved(); + Status reconcile(const std::string& session_id, const std::string& request_id, + const Resolution& resolution); Status registerTool(Tool tool); Status registerSkill(std::string phrase, std::string tool, Json arguments); Status setModel(ModelHandler model); Result run(const Turn& turn); - // Explicitly discards history AND request deduplication for this session. + // Discards history AND deduplication. Refuses sessions with UNKNOWN outcomes. Status clearSession(const std::string& session_id); private: struct Skill { std::string tool; Json arguments; }; - struct Cached { std::string input; Result result; }; + struct Cached { + std::string input; + Result result; + std::string tool; + Json arguments = Json::object(); + }; struct Session { std::map requests; std::vector history; }; Result execute(const Turn& turn, Session& session); + void remember(Session& session, const Turn& turn, const Reply& reply); Limits limits_; std::mutex mutex_; std::map tools_; std::map skills_; std::map sessions_; ModelHandler model_; + std::unique_ptr store_; + Status storage_status_ = Status::Ok(); }; } // namespace master_agent::reference diff --git a/scripts/build_manifest.py b/scripts/build_manifest.py new file mode 100644 index 0000000..46df047 --- /dev/null +++ b/scripts/build_manifest.py @@ -0,0 +1,30 @@ +#!/usr/bin/env python3 +"""Write non-secret provenance for a packaged build; never dump the full CMake cache.""" +import json +from pathlib import Path +import platform +import subprocess +import sys + +root, build, output = map(Path, sys.argv[1:]) +def git(*args): + result = subprocess.run(['git', '-C', str(root), *args], capture_output=True, text=True) + return result.stdout.strip() if result.returncode == 0 else None + +cache = {} +for line in (build / 'CMakeCache.txt').read_text().splitlines(): + if line and not line.startswith(('#', '//')) and ':' in line and '=' in line: + key, value = line.split('=', 1) + cache[key.split(':', 1)[0]] = value + +manifest = { + 'schema_version': 1, + 'version': json.loads((root / 'VERSION.json').read_text())['version'], + 'source_commit': git('rev-parse', 'HEAD'), + 'working_tree_dirty': bool(git('status', '--porcelain')), + 'platform': platform.system(), + 'architecture': platform.machine(), + 'build_type': cache.get('CMAKE_BUILD_TYPE'), + 'features': {key: cache.get(key) == 'ON' for key in ('MASTER_AGENT_ENABLE_HTTP', 'MASTER_AGENT_ENABLE_STORAGE')}, +} +output.write_text(json.dumps(manifest, indent=2) + '\n') diff --git a/scripts/package.sh b/scripts/package.sh index c23b758..0a22fcf 100755 --- a/scripts/package.sh +++ b/scripts/package.sh @@ -12,6 +12,7 @@ cmake --install "${BUILD_DIR}" --prefix "${STAGING}/sdk" --config Release test -x "${STAGING}/sdk/bin/sparx" cp "${ROOT}/LICENSE" "${ROOT}/README.md" "${STAGING}/sdk/" cp -R "${ROOT}/examples/reference_agent" "${STAGING}/sdk/example" +python3 "${ROOT}/scripts/build_manifest.py" "${ROOT}" "${BUILD_DIR}" "${STAGING}/sdk/BUILD_INFO.json" tar -czf "${STAGING}/candidate.tar.gz" -C "${STAGING}/sdk" . mkdir "${STAGING}/unpacked" tar -xzf "${STAGING}/candidate.tar.gz" -C "${STAGING}/unpacked" @@ -23,4 +24,11 @@ cmake --build "${STAGING}/consumer" --config Release ctest --test-dir "${STAGING}/consumer" -C Release --output-on-failure # Publish the file only after all verification has succeeded. cp "${STAGING}/candidate.tar.gz" "${ARCHIVE}" +python3 - "${ARCHIVE}" <<'PY_CHECKSUM' +import hashlib +from pathlib import Path +import sys +p = Path(sys.argv[1]) +p.with_name(p.name + '.sha256').write_text(hashlib.sha256(p.read_bytes()).hexdigest() + ' ' + p.name + '\n') +PY_CHECKSUM printf 'Verified archive: %s\n' "${ARCHIVE}" diff --git a/tests/consumer/CMakeLists.txt b/tests/consumer/CMakeLists.txt index 0139b8f..ffec51a 100644 --- a/tests/consumer/CMakeLists.txt +++ b/tests/consumer/CMakeLists.txt @@ -10,3 +10,8 @@ if(TARGET MasterAgent::Http) target_link_libraries(consumer PRIVATE MasterAgent::Http) target_compile_definitions(consumer PRIVATE TEST_HTTP=1) endif() + +if(MasterAgent_WITH_STORAGE) + target_compile_definitions(consumer PRIVATE TEST_STORAGE=1) + add_test(NAME installed_persistent_replay COMMAND consumer "${CMAKE_CURRENT_BINARY_DIR}/consumer.db") +endif() diff --git a/tests/consumer/main.cpp b/tests/consumer/main.cpp index 0cc1714..529e92f 100644 --- a/tests/consumer/main.cpp +++ b/tests/consumer/main.cpp @@ -4,7 +4,22 @@ #endif using namespace master_agent; using namespace master_agent::reference; -int main() { +int main(int argc, char** argv) { +#ifdef TEST_STORAGE + if (argc == 2) { + { + Runtime durable; + if (!durable.openStore(argv[1])) return 5; + if (!durable.setModel([](const Json&) { return Result::success(R"({"response":"stored"})"); })) return 6; + if (!durable.run({"installed", "1", "persist this"})) return 7; + } + Runtime reopened; + if (!reopened.openStore(argv[1])) return 8; + auto receipt = reopened.run({"installed", "1", "persist this"}); + if (!receipt || !receipt.value->replayed || receipt.value->output != "stored") return 9; + return reopened.clearSession("installed") ? 0 : 10; + } +#endif Runtime runtime; #ifdef TEST_HTTP if (!runtime.setModel(makeHttpModel({}))) return 4; diff --git a/tests/durable_probe.cpp b/tests/durable_probe.cpp new file mode 100644 index 0000000..83d27c2 --- /dev/null +++ b/tests/durable_probe.cpp @@ -0,0 +1,42 @@ +#include "master_agent/runtime/reference_runtime.h" +#include +#include +#include +#include +using namespace master_agent; +using namespace master_agent::reference; +int main(int argc, char** argv) { + if (argc != 4) return 2; + const std::string mode = argv[1], path = argv[2], effect = argv[3]; + Runtime runtime; + auto status = runtime.openStore(path); + if (!status) { + // Ignoring openStore failure must not silently run in memory mode. + auto guarded = runtime.run({"s", "1", "increment"}); + std::cout << Json{{"error", status.error_code}, {"run_blocked", !guarded}}.dump() << std::endl; + return 1; + } + if (mode == "init") return 0; + if (mode == "hold") { std::cout << "ready" << std::endl; std::string line; std::getline(std::cin, line); return 0; } + Json schema{{"type", "object"}, {"properties", Json::object()}, {"additionalProperties", false}}; + status = runtime.registerTool({"counter.increment", "Append a durable external effect", schema, + [&](const Json&) -> Result { + int fd = open(effect.c_str(), O_WRONLY | O_APPEND | O_CREAT, 0600); + if (fd < 0) throw std::runtime_error("Cannot open effect"); + const auto count = write(fd, "effect\n", 7); + const auto sync = fsync(fd); close(fd); + if (count != 7 || sync != 0) throw std::runtime_error("Cannot persist effect"); + if (mode == "crash") _exit(86); // No destructors or SQLite close. + if (mode == "throw") throw std::runtime_error("after effect"); + return Result::success({{"count", 1}}); + }}); + if (!status || !runtime.registerSkill("increment", "counter.increment", Json::object())) return 3; + auto result = runtime.run({"s", "1", "increment"}); + Json output{{"ok", result.ok()}}; + if (result) { output["replayed"] = result.value->replayed; output["output"] = result.value->output; } + else output["error"] = result.error->code; + auto pending = runtime.unresolved(); + output["unresolved"] = pending ? pending.value->size() : 0; + std::cout << output.dump() << std::endl; + return result ? 0 : 1; +} diff --git a/tests/test_crash_recovery.py b/tests/test_crash_recovery.py new file mode 100644 index 0000000..5a7bd7e --- /dev/null +++ b/tests/test_crash_recovery.py @@ -0,0 +1,84 @@ +"""Real process death between external fsync and receipt commit, plus store faults.""" +import json +from pathlib import Path +import sqlite3 +import subprocess +import sys +import tempfile + +probe, cli = sys.argv[1:] + + +def run(args, code=0, stdin=None): + result = subprocess.run(args, input=stdin, text=True, capture_output=True, timeout=15) + assert result.returncode == code, (args, result.returncode, result.stdout, result.stderr) + return json.loads(result.stdout) if result.stdout.strip() else None + + +with tempfile.TemporaryDirectory(prefix='sparx crash ') as temp: + root = Path(temp) + for mode in ['normal', 'crash', 'throw']: + db, effect = str(root / (mode + '.db')), root / (mode + '.effect') + run([probe, mode, db, str(effect)], 86 if mode == 'crash' else 1 if mode == 'throw' else 0) + assert effect.read_text() == 'effect\n' + replay = run([probe, 'normal', db, str(effect)], 0 if mode == 'normal' else 1) + assert effect.read_text() == 'effect\n', 'repeated external effect' + if mode == 'normal': + assert replay['replayed'] + continue + assert replay['error'] == 'UNKNOWN' + recovered = run([cli, 'recover', '--state', db])['unresolved'] + assert len(recovered) == 1 and recovered[0]['tool'] == 'counter.increment' + resolution = 'committed' if mode == 'crash' else 'failed' + run([cli, 'reconcile', '--state', db, '--session', 's', '--request-id', '1', + '--outcome', resolution, '--output', '{"count":1}', '--note', 'Operator checked external ledger']) + assert run([cli, 'recover', '--state', db])['unresolved'] == [] + final = run([probe, 'normal', db, str(effect)], 0 if resolution == 'committed' else 1) + assert (final.get('replayed') if resolution == 'committed' else final['error'] == 'RECONCILED_FAILED') + assert effect.read_text() == 'effect\n' + with sqlite3.connect(db) as connection: + audit = connection.execute("SELECT event,note FROM events WHERE event LIKE 'RECONCILED_%'").fetchall() + assert audit == [('RECONCILED_' + ('COMMITTED' if resolution == 'committed' else 'FAILED'), 'Operator checked external ledger')] + + db, effect = str(root / 'locked.db'), str(root / 'locked.effect') + owner = subprocess.Popen([probe, 'hold', db, effect], stdin=subprocess.PIPE, stdout=subprocess.PIPE, text=True) + try: + assert owner.stdout.readline().strip() == 'ready' + assert run([probe, 'normal', db, effect], 1) == {'error': 'STORAGE_BUSY', 'run_blocked': True} + finally: + owner.communicate('\n', timeout=10) + + for stage in ['begin', 'dispatch', 'finish']: + db, effect = str(root / (stage + '.db')), root / (stage + '.effect') + run([probe, 'init', db, str(effect)]) + trigger = {'begin': 'BEFORE INSERT ON requests', + 'dispatch': "BEFORE UPDATE ON requests WHEN NEW.state='DISPATCHED'", + 'finish': "BEFORE UPDATE ON requests WHEN NEW.state='COMMITTED'"}[stage] + with sqlite3.connect(db) as connection: + connection.execute(f"CREATE TRIGGER injected_failure {trigger} BEGIN SELECT RAISE(ABORT,'injected write failure'); END") + fault = run([probe, 'normal', db, str(effect)], 1) + assert fault['error'] == ('STORAGE_ERROR' if stage == 'begin' else 'UNKNOWN') + assert effect.exists() == (stage == 'finish'), 'tool ran without a durable dispatch record' + if stage == 'finish': + assert run([cli, 'recover', '--state', db])['unresolved'][0]['request_id'] == '1' + + db = root / 'corrupt.db' + db.write_bytes(b'not a database') + assert run([probe, 'normal', str(db), str(root / 'never.effect')], 1)['run_blocked'] + assert not (root / 'never.effect').exists() + db = root / 'future.db' + run([probe, 'init', str(db), str(root / 'unused')]) + with sqlite3.connect(db) as connection: + connection.execute('PRAGMA user_version=999') + assert run([probe, 'normal', str(db), str(root / 'never.effect')], 1)['error'] == 'STORE_VERSION' + + # Explicit request IDs survive separate CLI processes and JSONL sessions. + db = str(root / 'cli.db') + command = [cli, 'run', '--state', db, '--session', 'cli-test', '--request-id', '1', '--input', 'set AC to 22 degrees'] + assert not run(command)['replayed'] + assert run(command)['replayed'] + result = run([cli, 'run', '--state', db, '--jsonl'], stdin=json.dumps({'session_id': 'cli-test', 'request_id': '2', 'input': 'vehicle status'}) + '\n') + assert result['ok'] + run([cli, 'run', '--state', db, '--input', 'vehicle status'], 2) + + run([cli, 'run', '--state', '', '--input', 'vehicle status'], 2) diff --git a/tests/test_durable_runtime.cpp b/tests/test_durable_runtime.cpp new file mode 100644 index 0000000..33d77c3 --- /dev/null +++ b/tests/test_durable_runtime.cpp @@ -0,0 +1,69 @@ +#include "master_agent/runtime/reference_runtime.h" +#include "check.h" +#include +#include +#include +using namespace master_agent; +using namespace master_agent::reference; +int main() { + struct Scratch { + std::filesystem::path path = std::filesystem::temp_directory_path() / + ("sparx-durable-" + std::to_string(std::chrono::steady_clock::now().time_since_epoch().count())); + Scratch() { CHECK(std::filesystem::create_directory(path)); } + ~Scratch() { std::error_code error; std::filesystem::remove_all(path, error); } + } scratch; + const auto db = (scratch.path / "requests.db").string(); + int calls = 0; + Json schema{{"type", "object"}, {"properties", Json::object()}, {"additionalProperties", false}}; + { + Runtime runtime; + CHECK(runtime.openStore(db)); + Runtime other; + CHECK(other.openStore(db).error_code == "STORAGE_BUSY"); + CHECK(!other.run({"x", "1", "ignored-open-failure"})); + CHECK(runtime.registerTool({"effect", "test", schema, [&](const Json&) { ++calls; return Result::success(1); }})); + CHECK(runtime.registerSkill("effect", "effect", Json::object())); + CHECK(runtime.run({"s", "1", "effect"})); + CHECK(runtime.run({"s", "1", "effect"}).value->replayed && calls == 1); + CHECK(runtime.registerTool({"throws", "test", schema, [&](const Json&) -> Result { ++calls; throw std::runtime_error("uncertain"); }})); + CHECK(runtime.registerSkill("throws", "throws", Json::object())); + CHECK(runtime.run({"s", "2", "throws"}).error->code == "UNKNOWN"); + CHECK(runtime.clearSession("s").error_code == "UNRESOLVED_REQUESTS"); + } + { + Runtime runtime; + CHECK(runtime.openStore(db)); + // No tools are registered: replay and recovery must be independent of them. + CHECK(runtime.run({"s", "1", "effect"}).value->replayed && calls == 2); + CHECK(runtime.run({"s", "1", "different"}).error->code == "REQUEST_CONFLICT"); + auto pending = runtime.unresolved(); + CHECK(pending && pending.value->size() == 1 && pending.value->front().tool == "throws"); + CHECK(!runtime.reconcile("s", "2", {true, 2, ""})); + CHECK(runtime.reconcile("s", "2", {true, 2, "Verified external effect #2"})); + CHECK(runtime.unresolved().value->empty()); + CHECK(runtime.run({"s", "2", "throws"}).value->route == "reconciled"); + CHECK(!runtime.reconcile("s", "2", {true, 3, "cannot rewrite a final outcome"})); + CHECK(runtime.setModel([](const Json& messages) { + CHECK(messages.size() == 6); // system + two recovered turns + current input + return Result::success(R"({"response":"history recovered"})"); + })); + CHECK(runtime.run({"s", "3", "question"})); + } + { + Runtime runtime; + CHECK(runtime.openStore(db)); + CHECK(runtime.run({"s", "2", "throws"}).value->output == 2); + CHECK(runtime.clearSession("s")); + } + { + Runtime runtime; + CHECK(runtime.openStore(db)); + CHECK(runtime.run({"s", "1", "effect"}).error->code == "NO_ROUTE"); + } + Runtime memory; + CHECK(memory.run({"s", "1", std::string(1, static_cast(0xff))}).error->code == "INVALID_REQUEST"); + Runtime small({1, 10, 1, 4}); + CHECK(small.registerTool({"large", "test", schema, [](const Json&) { return Result::success("too long"); }})); + CHECK(small.registerSkill("large", "large", Json::object())); + CHECK(small.run({"s", "1", "large"}).error->code == "UNKNOWN"); +} From 0c76097f42cf1ff69d231d070bbed42aa1f69139 Mon Sep 17 00:00:00 2001 From: Codex Date: Mon, 28 Sep 2026 14:50:42 +0800 Subject: [PATCH 3/4] feat(runtime): add contextual tool outcomes and HTTP recovery example --- .github/workflows/ci.yml | 55 ++++++--- ARCHITECTURE.md | 2 +- CHANGELOG.md | 11 ++ CLAUDE.md | 4 +- CONTRIBUTING.md | 2 +- README.md | 20 +++- VERSION.json | 2 +- backends/http_model.cpp | 7 +- cli/CMakeLists.txt | 2 +- cmake/ReferenceTests.cmake | 13 +++ core/reference_runtime.cpp | 69 ++++++++--- docs/reference_runtime.md | 54 +++++++++ examples/inventory_agent/CMakeLists.txt | 9 ++ examples/inventory_agent/README.md | 72 ++++++++++++ examples/inventory_agent/main.cpp | 104 +++++++++++++++++ examples/inventory_agent/service.py | 107 ++++++++++++++++++ examples/reference_agent/main.cpp | 6 +- .../master_agent/runtime/reference_runtime.h | 58 +++++++++- scripts/package.sh | 15 +++ tests/consumer/main.cpp | 6 +- tests/test_execution_context.cpp | 79 +++++++++++++ tests/test_inventory_contract.py | 93 +++++++++++++++ 22 files changed, 744 insertions(+), 46 deletions(-) create mode 100644 examples/inventory_agent/CMakeLists.txt create mode 100644 examples/inventory_agent/README.md create mode 100644 examples/inventory_agent/main.cpp create mode 100644 examples/inventory_agent/service.py create mode 100644 tests/test_execution_context.cpp create mode 100644 tests/test_inventory_contract.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 02dac85..5758f89 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -68,25 +68,46 @@ jobs: runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 - - - name: Install clang-tidy + - name: Install analysis dependencies + run: sudo apt-get update && sudo apt-get install -y clang-tidy libcurl4-openssl-dev libsqlite3-dev + - name: Configure compilation database + run: cmake -S . -B build-lint -DBUILD_EVAL=OFF -DCMAKE_BUILD_TYPE=Debug + - name: Analyze public runtime (blocking) run: | - sudo apt-get update - sudo apt-get install -y clang-tidy - - - name: Run clang-tidy on Agent OS modules + clang-tidy -p build-lint \ + --checks='-*,clang-analyzer-*,bugprone-use-after-move,bugprone-sizeof-expression' \ + --warnings-as-errors='*' \ + --header-filter='/(core|backends|include/master_agent/runtime|examples/inventory_agent)/' \ + core/reference_runtime.cpp core/durable_store.cpp backends/http_model.cpp \ + cli/src/reference_main.cpp examples/inventory_agent/main.cpp + + sanitizer: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - name: Install dependencies + run: sudo apt-get update && sudo apt-get install -y clang libcurl4-openssl-dev libsqlite3-dev + - name: Build with ASan and UBSan + env: + CC: clang + CXX: clang++ + run: | + cmake -S . -B build-sanitize -DBUILD_EVAL=OFF -DCMAKE_BUILD_TYPE=Debug \ + -DCMAKE_CXX_FLAGS='-fsanitize=address,undefined -fno-omit-frame-pointer' \ + -DCMAKE_EXE_LINKER_FLAGS='-fsanitize=address,undefined' + cmake --build build-sanitize --parallel 2 --target test_reference_runtime test_execution_context \ + test_durable_runtime durable_probe sparx http_model_probe inventory_agent test_harness + - name: Runtime, network and recovery sanitizer contracts + env: + ASAN_OPTIONS: detect_leaks=1:halt_on_error=1 + UBSAN_OPTIONS: halt_on_error=1:print_stacktrace=1 + run: ctest --test-dir build-sanitize --output-on-failure --timeout 90 -R 'test_reference_runtime|test_execution_context|test_durable_runtime|crash_recovery_contract|http_model_contract|inventory_service_contract|test_harness' + - name: Build and test without optional dependencies run: | - clang-tidy \ - --checks='-*,bugprone-*,performance-*,modernize-*,-modernize-use-trailing-return-type' \ - cli/src/sparx_agent_scheduler.cpp \ - cli/src/sparx_context_manager.cpp \ - cli/src/sparx_memory_manager.cpp \ - cli/src/sparx_access_control.cpp \ - cli/src/sparx_tool_registry.cpp \ - -- -std=c++17 -Icli/include -Iinclude \ - -Ithird_party/memory_short_term/include \ - -Ithird_party \ - || true # Non-blocking for now + cmake -S . -B build-minimal -DCMAKE_BUILD_TYPE=Release -DBUILD_EVAL=OFF \ + -DMASTER_AGENT_ENABLE_HTTP=OFF -DMASTER_AGENT_ENABLE_STORAGE=OFF + cmake --build build-minimal --parallel 2 + ctest --test-dir build-minimal --output-on-failure --timeout 90 security-check: runs-on: ubuntu-latest diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 45cec33..b143163 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -1,4 +1,4 @@ -> Public 0.4.0-alpha.2: use the open reference runtime in `core/`, +> Public 0.4.0-alpha.3: use the open reference runtime in `core/`, > `include/master_agent/runtime/`, `backends/`, and `cli/src/reference_main.cpp`. > The current runnable contract and commands are in README.md and > docs/reference_runtime.md. Legacy kernel descriptions below require private diff --git a/CHANGELOG.md b/CHANGELOG.md index 06b6aa6..a6d1e36 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,14 @@ +# 0.4.0-alpha.3 — execution contracts and HTTP business integration + +- Add contextual tools with session/request identity, stable scoped idempotency + keys, monotonic deadlines and shared cooperative cancellation signals. +- Introduce typed committed/failed/unknown outcomes; preserve the legacy callback API. +- Check cancellation before dispatch; preserve authoritative results returned after cancellation. +- Add an independent HTTP inventory service with atomic stock/idempotency receipts, + lost-response injection, external reconciliation and service/runtime restart tests. +- Make public-runtime static analysis blocking; run ASan/UBSan and dependency-free + builds in CI. Verify contextual consumers and the HTTP example from SDK archives. + # 0.4.0-alpha.2 — durable receipts and recovery - Add an opt-in SQLite store with WAL/FULL synchronization, exclusive ownership, diff --git a/CLAUDE.md b/CLAUDE.md index 1b7f201..150ab2a 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1,4 +1,4 @@ -> Public 0.4.0-alpha.2: use the open reference runtime in `core/`, +> Public 0.4.0-alpha.3: use the open reference runtime in `core/`, > `include/master_agent/runtime/`, `backends/`, and `cli/src/reference_main.cpp`. > The current runnable contract and commands are in README.md and > docs/reference_runtime.md. Legacy kernel descriptions below require private @@ -60,4 +60,4 @@ The legacy full CLI still requires private kernel source. See docs/reference_run ## Version -0.4.0-alpha.2. Versions follow semver. Don't inflate beyond actual stability. +0.4.0-alpha.3. Versions follow semver. Don't inflate beyond actual stability. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 314488f..5cfec2c 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1,4 +1,4 @@ -> Public 0.4.0-alpha.2: use the open reference runtime in `core/`, +> Public 0.4.0-alpha.3: use the open reference runtime in `core/`, > `include/master_agent/runtime/`, `backends/`, and `cli/src/reference_main.cpp`. > The current runnable contract and commands are in README.md and > docs/reference_runtime.md. Legacy kernel descriptions below require private diff --git a/README.md b/README.md index 71b55e7..57a4f38 100644 --- a/README.md +++ b/README.md @@ -3,7 +3,7 @@ An embeddable, local-first C++17 agent runtime: turn input into validated tool execution, with deterministic skills before model inference. -**0.4.0-alpha.2 — open reference runtime with durable receipts.** Public source now builds a working +**0.4.0-alpha.3 — open reference runtime with contextual tools and durable receipts.** Public source now builds a working `sparx` CLI and an installable C++ SDK. This is a small synchronous runtime, not the proprietary durable kernel described by the legacy interfaces. @@ -111,12 +111,28 @@ phrase skills, optionally set a model callback, then call `Runtime::run()`. See the [integration contract](docs/reference_runtime.md) for errors, schema support, sessions, request replay, and concurrency limitations. +## Connect a business tool + +Context-aware tools receive the session/request identity, a stable scoped +idempotency key, deadline and cooperative cancellation signal. They return an +explicit committed, failed or unknown outcome. The existing `registerTool()` API +remains available. See the [execution contract](docs/reference_runtime.md). + +The [inventory example](examples/inventory_agent/README.md) talks to a separate +HTTP service that commits inventory and an idempotency receipt in its own SQLite +database. Its integration test deliberately drops a response after commit, checks +UNKNOWN recovery, reconciles against the service receipt and proves no second +stock decrement after either process restarts. This is a local business-service +integration example, not a real-model or third-party production validation. + ## Capability status | Capability | Public reference release | |---|---| | Deterministic exact-phrase skills | Available, no model required | | Host-defined tools and argument validation | Available; primitive object schema subset | +| Execution context and typed tool outcomes | Available; identity, deadline, cooperative cancellation, committed/failed/unknown | +| Independent HTTP business example | Inventory service with atomic idempotency and lost-response recovery | | Local model HTTP adapter | Available when built with libcurl | | Sessions and request deduplication | In memory, or persistent SQLite receipts/history with explicit opt-in | | Crash recovery and reconciliation | Interrupted single requests become UNKNOWN; explicit evidence-based reconciliation | @@ -146,7 +162,7 @@ return a nonzero status. `BUILD_EVAL=OFF` excludes them from an SDK-only build. The package script installs the SDK and CLI, unpacks the archive in a new location, runs the demo, and builds an external consumer before publishing the -archive. The archive includes `BUILD_INFO.json` and a SHA-256 sidecar. The package uses +archive. The archive also includes the inventory example; HTTP/storage-enabled packages build and exercise it after unpacking. The archive includes `BUILD_INFO.json` and a SHA-256 sidecar. The package uses platform system dependencies, including libcurl and SQLite when enabled; it is not a universally static binary. ## Contributing diff --git a/VERSION.json b/VERSION.json index 433ce2a..e6c104e 100644 --- a/VERSION.json +++ b/VERSION.json @@ -1,6 +1,6 @@ { "product": "OAK", - "version": "0.4.0-alpha.2", + "version": "0.4.0-alpha.3", "api_namespace": "master_agent", "cmake_target": "MasterAgent::Core", "model_backends": ["http_chat_completions", "custom_callback"], diff --git a/backends/http_model.cpp b/backends/http_model.cpp index 60ee91b..1d3d1e7 100644 --- a/backends/http_model.cpp +++ b/backends/http_model.cpp @@ -16,7 +16,12 @@ ModelHandler makeHttpModel(HttpModelConfig config) { std::unique_ptr curl(curl_easy_init(), curl_easy_cleanup); if (!curl) return fail("HTTP_ERROR", "Cannot allocate HTTP client"); curl_slist* raw_headers = curl_slist_append(nullptr, "Content-Type: application/json"); - if (!config.api_key.empty()) raw_headers = curl_slist_append(raw_headers, ("Authorization: Bearer " + config.api_key).c_str()); + if (!raw_headers) return fail("HTTP_ERROR", "Cannot allocate HTTP headers"); + if (!config.api_key.empty()) { + auto* extended = curl_slist_append(raw_headers, ("Authorization: Bearer " + config.api_key).c_str()); + if (!extended) { curl_slist_free_all(raw_headers); return fail("HTTP_ERROR", "Cannot allocate credential header"); } + raw_headers = extended; + } std::unique_ptr headers(raw_headers, curl_slist_free_all); const auto body = Json{{"model", config.model}, {"messages", messages}, {"stream", false}, {"temperature", 0}, {"max_tokens", 512}}.dump(); std::string response; diff --git a/cli/CMakeLists.txt b/cli/CMakeLists.txt index a04d6a9..2ec88a5 100644 --- a/cli/CMakeLists.txt +++ b/cli/CMakeLists.txt @@ -134,7 +134,7 @@ else() target_compile_definitions(sparx PRIVATE MASTER_AGENT_HAS_HTTP=1) endif() if(NOT DEFINED SPARX_VERSION) - set(SPARX_VERSION "${PROJECT_VERSION}-alpha.2") + set(SPARX_VERSION "${PROJECT_VERSION}-alpha.3") endif() target_compile_definitions(sparx PRIVATE SPARX_VERSION="${SPARX_VERSION}") install(TARGETS sparx RUNTIME DESTINATION ${CMAKE_INSTALL_BINDIR}) diff --git a/cmake/ReferenceTests.cmake b/cmake/ReferenceTests.cmake index 458c50e..490c1bd 100644 --- a/cmake/ReferenceTests.cmake +++ b/cmake/ReferenceTests.cmake @@ -38,3 +38,16 @@ if(MASTER_AGENT_ENABLE_STORAGE) set_tests_properties(crash_recovery_contract PROPERTIES TIMEOUT 60) endif() endif() + +add_executable(test_execution_context tests/test_execution_context.cpp) +target_link_libraries(test_execution_context PRIVATE MasterAgent::Core) +add_test(NAME test_execution_context COMMAND test_execution_context) +set_tests_properties(test_execution_context PROPERTIES TIMEOUT 15) +if(MASTER_AGENT_ENABLE_HTTP AND MASTER_AGENT_ENABLE_STORAGE AND MASTER_AGENT_BUILD_CLI) + add_subdirectory(examples/inventory_agent) + add_test(NAME inventory_service_contract + COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/tests/test_inventory_contract.py + $ $ + ${CMAKE_CURRENT_SOURCE_DIR}/examples/inventory_agent/service.py) + set_tests_properties(inventory_service_contract PROPERTIES TIMEOUT 45) +endif() diff --git a/core/reference_runtime.cpp b/core/reference_runtime.cpp index fb9bf04..b28dc0c 100644 --- a/core/reference_runtime.cpp +++ b/core/reference_runtime.cpp @@ -72,8 +72,28 @@ Status validate(const Json& args, const Json& schema) { } return Status::Ok(); } +Status stopped(const ExecutionContext& context) { + if (context.options.cancellation.cancelled()) return Status::Error("CANCELLED", "Execution cancelled before tool invocation"); + if (context.options.deadline && std::chrono::steady_clock::now() >= *context.options.deadline) + return Status::Error("DEADLINE_EXCEEDED", "Execution deadline elapsed before tool invocation"); + return Status::Ok(); +} } // namespace +bool ExecutionContext::stopRequested() const { return !stopped(*this); } +std::string ExecutionContext::idempotencyKey() const { + constexpr char hex[] = "0123456789abcdef"; + std::string key; + for (const auto* part : {&session_id, &request_id}) { + if (!key.empty()) key += '.'; + for (unsigned char byte : *part) { key += hex[byte >> 4]; key += hex[byte & 15]; } + } + return key; +} +ToolOutcome ToolOutcome::committed(Json output) { return {ToolState::Committed, std::move(output), {}}; } +ToolOutcome ToolOutcome::failed(std::string message) { return {ToolState::Failed, {}, {"TOOL_FAILED", std::move(message), "", 422}}; } +ToolOutcome ToolOutcome::unknown(std::string message) { return {ToolState::Unknown, {}, {"UNKNOWN", std::move(message), "", 409}}; } + Runtime::Runtime(Limits limits) : limits_(limits) {} Runtime::~Runtime() = default; @@ -149,6 +169,17 @@ Status Runtime::reconcile(const std::string& session_id, const std::string& requ } Status Runtime::registerTool(Tool tool) { + if (!tool.execute) return Status::Error("INVALID_TOOL", "Tool requires a handler"); + return registerContextTool({std::move(tool.name), std::move(tool.description), std::move(tool.parameters), + [handler = std::move(tool.execute)](const ExecutionContext&, const Json& args) { + auto result = handler(args); + if (result) return ToolOutcome::committed(std::move(*result)); + auto error = result.error.value_or(StructuredError{"TOOL_ERROR", "Tool returned no result", "", 500}); + if (error.code == "UNKNOWN") return ToolOutcome::unknown(error.message); + return ToolOutcome(ToolState::Failed, {}, std::move(error)); + }}); +} +Status Runtime::registerContextTool(ContextTool tool) { std::unique_lock lock(mutex_, std::try_to_lock); if (!lock) return Status::Error("BUSY"); if (tool.name.empty() || !tool.execute || tools_.count(tool.name)) return Status::Error("INVALID_TOOL", "Tool requires a unique name and handler"); @@ -184,7 +215,7 @@ Status Runtime::clearSession(const std::string& id) { sessions_.erase(id); return Status::Ok(); } -Result Runtime::run(const Turn& turn) { +Result Runtime::run(const Turn& turn, const RunOptions& options) { std::unique_lock lock(mutex_, std::try_to_lock); if (!lock) return fail("BUSY", "Runtime is executing another operation"); if (!storage_status_) return fail(storage_status_.error_code, storage_status_.error_message); @@ -192,14 +223,21 @@ Result Runtime::run(const Turn& turn) { return fail("INVALID_REQUEST", "Nonempty bounded session, request and input are required"); try { Json{{"session", turn.session_id}, {"request", turn.request_id}, {"input", turn.input}}.dump(); } catch (const Json::exception&) { return fail("INVALID_REQUEST", "Request text must be valid UTF-8"); } - if (!sessions_.count(turn.session_id) && sessions_.size() >= limits_.sessions) return fail("RESOURCE_LIMIT", "Session limit reached"); - auto& session = sessions_[turn.session_id]; - if (auto it = session.requests.find(turn.request_id); it != session.requests.end()) { - if (it->second.input != turn.input) return fail("REQUEST_CONFLICT", "Request ID was already used for different input"); - auto result = it->second.result; - if (result) result.value->replayed = true; - return result; + auto existing = sessions_.find(turn.session_id); + if (existing != sessions_.end()) { + auto it = existing->second.requests.find(turn.request_id); + if (it != existing->second.requests.end()) { + if (it->second.input != turn.input) return fail("REQUEST_CONFLICT", "Request ID was already used for different input"); + auto result = it->second.result; + if (result) result.value->replayed = true; + return result; + } } + const ExecutionContext context{turn.session_id, turn.request_id, options}; + auto ready = stopped(context); + if (!ready) return fail(ready.error_code, ready.error_message); + if (existing == sessions_.end() && sessions_.size() >= limits_.sessions) return fail("RESOURCE_LIMIT", "Session limit reached"); + auto& session = sessions_[turn.session_id]; if (session.requests.size() >= limits_.requests_per_session) return fail("RESOURCE_LIMIT", "Request limit reached; start a new session"); if (store_) { storage_status_ = store_->begin(turn); @@ -210,7 +248,7 @@ Result Runtime::run(const Turn& turn) { auto previous_history = session.history; Result result; try { - result = execute(turn, session); + result = execute(turn, session, context); if (result) { if (result.value->output.dump().size() > limits_.output_bytes) result = fail(cached.tool.empty() ? "OUTPUT_LIMIT" : "UNKNOWN", "Output exceeds limit; inspect any tool side effect"); @@ -227,7 +265,7 @@ Result Runtime::run(const Turn& turn) { cached.result = result; return result; } -Result Runtime::execute(const Turn& turn, Session& session) { +Result Runtime::execute(const Turn& turn, Session& session, const ExecutionContext& context) { Reply reply{turn.session_id, turn.request_id, "skill", "", Json{}, false}; Json arguments; if (auto it = skills_.find(turn.input); it != skills_.end()) { @@ -263,6 +301,8 @@ Result Runtime::execute(const Turn& turn, Session& session) { if (tool == tools_.end()) return fail("UNKNOWN_TOOL", "Model selected an unregistered tool"); auto valid = validate(arguments, tool->second.parameters); if (!valid) return fail(valid.error_code, valid.error_message); + auto ready = stopped(context); + if (!ready) return fail(ready.error_code, ready.error_message); auto& receipt = session.requests.at(turn.request_id); receipt.tool = reply.tool; receipt.arguments = arguments; @@ -270,10 +310,13 @@ Result Runtime::execute(const Turn& turn, Session& session) { storage_status_ = store_->dispatch(turn, reply.tool, arguments); if (!storage_status_) return fail("STORAGE_ERROR", "Dispatch was not persisted; tool was not called"); } + ready = stopped(context); + if (!ready) return fail(ready.error_code, ready.error_message); try { - auto result = tool->second.execute(arguments); - if (!result) return Result::failure(result.error.value_or(StructuredError{"TOOL_ERROR", "Tool returned no result", "", 500})); - reply.output = *result; + auto result = tool->second.execute(context, arguments); + if (result.state_ != ToolState::Committed) return Result::failure(std::move(result.error_)); + // A confirmed effect remains committed even if cancellation arrived during it. + reply.output = std::move(result.output_); } catch (...) { // The handler may have performed its side effect before throwing. return fail("UNKNOWN", "Tool threw; reconcile its outcome before issuing a new request"); diff --git a/docs/reference_runtime.md b/docs/reference_runtime.md index e15f465..3e87844 100644 --- a/docs/reference_runtime.md +++ b/docs/reference_runtime.md @@ -34,6 +34,60 @@ Schemas must use `type:object`, `properties`, and nested objects/arrays, malformed schemas, and unknown required fields are rejected at registration. Do not assume full JSON Schema conformance. +## Context-aware tools and outcomes + +`registerContextTool(ContextTool)` is the preferred integration API. The handler +receives `const ExecutionContext&` and validated arguments and returns a +`ToolOutcome`: + +```cpp +runtime.registerContextTool({"reserve", "Reserve stock", schema, + [](const ExecutionContext& context, const Json& arguments) { + // Pass context.idempotencyKey() to a service that supports atomic deduplication. + // Check context.stopRequested() before initiating an operation. + return ToolOutcome::committed(Json{{"reserved", true}}); + }}); +``` + +The snippet shows the API shape only; the [inventory example](../examples/inventory_agent/README.md) +implements the external operation and lost-response handling. + +- `committed(output)`: the effect/result is confirmed. A cancellation or deadline + arriving during the callback does not erase that confirmation. +- `failed(message)`: definitely unsuccessful; becomes `TOOL_FAILED`. Use only when + the service establishes failure or the tool knows it performed no effect. +- `unknown(message)`: the external outcome cannot be established; becomes `UNKNOWN` + and requires reconciliation. A timeout after sending a mutation is not proof of failure. + +This alpha release preserves source compatibility for legacy callback registration; +rebuild applications against the updated SDK. Binary ABI stability is not promised. + +Legacy `registerTool(Tool)` callbacks still receive arguments only. Their successful +`Result` maps to committed; explicit errors map to failed, except `UNKNOWN`, which +remains unknown. Exceptions from either handler become UNKNOWN. Prefer the typed API +for new adapters so a generic transport failure is not mistaken for a definite failure. + +`run(turn, options)` accepts `RunOptions` with an optional `steady_clock` deadline +and a copyable `CancellationToken`. Copies share an atomic flag; another thread may +call `options.cancellation.cancel()` while the synchronous run is in progress. +A tool should poll `context.stopRequested()` or integrate the signal with its I/O. +Native code is not forcibly interrupted, and cancellation cannot undo a remote effect. +Custom model handlers still own their I/O timeout; the runtime checks again before +invoking any selected tool. + +For an unseen request, cancellation/deadline is checked before recording STARTED, +then before dispatch and invocation. A pre-start rejection does not consume a session +or request slot; a stop after STARTED is recorded as a terminal failure. Replaying an +existing receipt returns that authoritative result even if the new options are expired +or cancelled. Deliberately retrying a recorded failure needs a new request ID. + +`idempotencyKey()` encodes UTF-8 session/request bytes separately in hex with a dot +separator. It is stable across restarts and collision-free for the bounded IDs accepted +by `run()`. Its namespace is one host application/store; independent applications must +provide distinct external scopes. Session/request IDs are not credentials. Authorization +and tenant isolation remain the host's responsibility. Do not retain references to the +context/arguments after the synchronous callback returns; copy needed values explicitly. + ## Sessions, replay, and failure Defaults: 32 sessions, 256 recorded requests per session, 8 successful turns of diff --git a/examples/inventory_agent/CMakeLists.txt b/examples/inventory_agent/CMakeLists.txt new file mode 100644 index 0000000..101ae41 --- /dev/null +++ b/examples/inventory_agent/CMakeLists.txt @@ -0,0 +1,9 @@ +cmake_minimum_required(VERSION 3.18) +project(MasterAgentInventoryExample LANGUAGES CXX) +if(NOT TARGET MasterAgent::Core) + find_package(MasterAgent CONFIG REQUIRED) +endif() +find_package(CURL REQUIRED) +add_executable(inventory_agent main.cpp) +target_compile_features(inventory_agent PRIVATE cxx_std_17) +target_link_libraries(inventory_agent PRIVATE MasterAgent::Core CURL::libcurl) diff --git a/examples/inventory_agent/README.md b/examples/inventory_agent/README.md new file mode 100644 index 0000000..4645053 --- /dev/null +++ b/examples/inventory_agent/README.md @@ -0,0 +1,72 @@ +# Inventory reservation over HTTP + +This is a runnable local integration example, not a mock callback: the agent +contacts an independent Python service which atomically updates stock and a +receipt in its own SQLite database. It starts with 10 units of `widget`. +No model, external account or credentials are required. Python 3.9+, SQLite and +libcurl are required. The service is unauthenticated and binds only to loopback; +do not expose it publicly. It is not a production inventory server. + +After installing the SDK (see the repository README): + +```bash +cmake -S examples/inventory_agent -B build-inventory -DCMAKE_PREFIX_PATH="$PWD/install-sdk" +cmake --build build-inventory +python3 examples/inventory_agent/service.py --database inventory.db --port 8765 +``` + +In another terminal: + +```bash +./build-inventory/inventory_agent http://127.0.0.1:8765 agent.db shop order-1 widget 2 +# Repeat: replayed:true; inventory is not decremented twice. +curl http://127.0.0.1:8765/stock/widget +``` + +The CLI takes `BASE_URL STATE_DB SESSION REQUEST SKU QUANTITY [TIMEOUT_MS]`. +A request ID identifies one immutable reservation in a session. Do not reuse it +for a different item/quantity. Both sides must agree on the application/store +namespace: the sample key encodes the session and request bytes as hex separated +by a dot. Independent tenants must use distinct scopes/stores. These IDs are not +authentication or authorization credentials. + +## Reproduce a lost response + +Start a fresh service database with `--drop-first-response`. Submit a new request. +The service commits the stock change and closes the socket before replying. The +agent returns `UNKNOWN`; submitting the same IDs again, including after restart, +returns the same uncertainty without sending another reservation. + +```bash +sparx recover --state agent.db +# For session=shop and request=order-1: +curl http://127.0.0.1:8765/receipts/73686f70.6f726465722d31 +``` + +Inspect the authoritative receipt and compare its key, SKU and quantity with the +unresolved request. Only then reconcile using the receipt's `output`: + +```bash +sparx reconcile --state agent.db --session shop --request-id order-1 \ + --outcome committed --output '{"sku":"widget","quantity":2,"remaining":8}' \ + --note 'Verified the matching committed inventory service receipt' +``` + +Use the actual receipt, not the example output, when reconciling. A missing receipt +alone does not prove failure while a remote operation could still be in flight. +The automated `inventory_service_contract` verifies this sequence, rejected stock +requests, conflicting idempotency keys and independent service/runtime restarts. + +## What the adapter demonstrates + +- `registerContextTool()` receives stable request identity, deadline and a shared + cancellation token. The idempotency key is sent to the service unchanged. +- The service commits its deduplication record and stock update in one transaction. +- `ToolOutcome::committed` requires a matching service receipt. A confirmed stock + rejection is `failed`; transfer errors, malformed receipts and key conflicts are + `unknown`. There is no automatic retry of uncertain effects. +- libcurl observes a bounded timeout and cooperative cancellation. Cancellation + after sending does not prove the remote service stopped or rolled back. +- Neither this example nor the runtime creates a transaction spanning the two + databases. Real services need their own authentication, idempotency guarantees, + retention policy and receipt-query interface. diff --git a/examples/inventory_agent/main.cpp b/examples/inventory_agent/main.cpp new file mode 100644 index 0000000..7d07486 --- /dev/null +++ b/examples/inventory_agent/main.cpp @@ -0,0 +1,104 @@ +#include +#include +#include +#include +#include +#include +using namespace master_agent; +using namespace master_agent::reference; +namespace { +ToolOutcome reserve(const std::string& endpoint, const ExecutionContext& context, const Json& args) { + if (context.stopRequested()) return ToolOutcome::failed("Cancelled before sending"); + static const auto initialized = curl_global_init(CURL_GLOBAL_DEFAULT); + if (initialized != CURLE_OK) return ToolOutcome::failed("HTTP initialization failed before sending"); + std::unique_ptr curl(curl_easy_init(), curl_easy_cleanup); + if (!curl) return ToolOutcome::failed("HTTP allocation failed before sending"); + const auto key = context.idempotencyKey(); + const std::string header = "Idempotency-Key: " + key; + curl_slist* raw = curl_slist_append(nullptr, "Content-Type: application/json"); + if (!raw) return ToolOutcome::failed("Header allocation failed before sending"); + auto* with_key = curl_slist_append(raw, header.c_str()); + if (!with_key) { curl_slist_free_all(raw); return ToolOutcome::failed("Header allocation failed before sending"); } + std::unique_ptr headers(with_key, curl_slist_free_all); + const auto body = args.dump(); + std::string response; + long timeout = 30000; + if (context.options.deadline) { + auto remaining = std::chrono::duration_cast(*context.options.deadline - std::chrono::steady_clock::now()).count(); + if (remaining <= 0) return ToolOutcome::failed("Deadline elapsed before sending"); + timeout = static_cast(std::min(remaining, timeout)); + } + const auto url = endpoint + "/reservations"; + curl_easy_setopt(curl.get(), CURLOPT_URL, url.c_str()); + curl_easy_setopt(curl.get(), CURLOPT_HTTPHEADER, headers.get()); + curl_easy_setopt(curl.get(), CURLOPT_POSTFIELDS, body.c_str()); + curl_easy_setopt(curl.get(), CURLOPT_POSTFIELDSIZE, static_cast(body.size())); + curl_easy_setopt(curl.get(), CURLOPT_TIMEOUT_MS, timeout); + curl_easy_setopt(curl.get(), CURLOPT_NOSIGNAL, 1L); + curl_easy_setopt(curl.get(), CURLOPT_FOLLOWLOCATION, 0L); + curl_easy_setopt(curl.get(), CURLOPT_WRITEFUNCTION, + +[](char* bytes, size_t size, size_t count, void* output) -> size_t { + auto& text = *static_cast(output); + const auto length = size * count; + if (length > 16384 - text.size()) return 0; + try { text.append(bytes, length); } catch (...) { return 0; } + return length; + }); + curl_easy_setopt(curl.get(), CURLOPT_WRITEDATA, &response); + curl_easy_setopt(curl.get(), CURLOPT_NOPROGRESS, 0L); + curl_easy_setopt(curl.get(), CURLOPT_XFERINFOFUNCTION, + +[](void* data, curl_off_t, curl_off_t, curl_off_t, curl_off_t) -> int { + return static_cast(data)->stopRequested() ? 1 : 0; + }); + curl_easy_setopt(curl.get(), CURLOPT_XFERINFODATA, &context); + auto transfer = curl_easy_perform(curl.get()); + // Transport errors cannot establish whether the remote mutation committed. + if (transfer != CURLE_OK) return ToolOutcome::unknown(curl_easy_strerror(transfer)); + long status = 0; + curl_easy_getinfo(curl.get(), CURLINFO_RESPONSE_CODE, &status); + try { + auto receipt = Json::parse(response); + if (receipt.at("key") != key) return ToolOutcome::unknown("Receipt identity mismatch"); + if (status == 200 && receipt.at("outcome") == "committed" && receipt.at("output").at("sku") == args.at("sku") && + receipt.at("output").at("quantity") == args.at("quantity")) + return ToolOutcome::committed(receipt.at("output")); + if (status == 409 && receipt.at("outcome") == "failed") + return ToolOutcome::failed(receipt.at("error").get()); + } catch (const Json::exception&) { /* A malformed receipt is not evidence of failure. */ } + return ToolOutcome::unknown("No authoritative service receipt"); +} +int integer(const char* text) { + const std::string value(text); + int result = 0; + auto parsed = std::from_chars(value.data(), value.data() + value.size(), result); + if (parsed.ec != std::errc{} || parsed.ptr != value.data() + value.size() || result <= 0 || result > 300000) + throw std::invalid_argument("Expected integer in 1..300000"); + return result; +} +} +int main(int argc, char** argv) { + if (argc != 7 && argc != 8) { + std::cerr << "Usage: inventory_agent BASE_URL STATE_DB SESSION REQUEST SKU QUANTITY [TIMEOUT_MS]\n"; + return 2; + } + try { + const std::string endpoint(argv[1]); + if (endpoint.rfind("http://", 0) != 0 && endpoint.rfind("https://", 0) != 0) return 2; + Json arguments{{"sku", argv[5]}, {"quantity", integer(argv[6])}}; + const auto input = arguments.dump(); + Runtime runtime; + auto opened = runtime.openStore(argv[2]); + if (!opened) { std::cout << Json{{"ok", false}, {"error", opened.error_code}} << '\n'; return 1; } + Json schema{{"type", "object"}, {"additionalProperties", false}, {"required", {"sku", "quantity"}}, + {"properties", {{"sku", {{"type", "string"}}}, {"quantity", {{"type", "integer"}, {"minimum", 1}, {"maximum", 300000}}}}}}; + if (!runtime.registerContextTool({"inventory.reserve", "Reserve real service inventory", schema, + [&](const ExecutionContext& context, const Json& args) { return reserve(endpoint, context, args); }})) return 2; + if (!runtime.registerSkill(input, "inventory.reserve", arguments)) return 2; + RunOptions options; + options.deadline = std::chrono::steady_clock::now() + std::chrono::milliseconds(argc == 8 ? integer(argv[7]) : 30000); + auto result = runtime.run({argv[3], argv[4], input}, options); + if (result) std::cout << Json{{"ok", true}, {"output", result.value->output}, {"replayed", result.value->replayed}} << '\n'; + else std::cout << Json{{"ok", false}, {"error", result.error->code}, {"message", result.error->message}} << '\n'; + return result ? 0 : 1; + } catch (const std::exception& e) { std::cerr << e.what() << '\n'; return 2; } +} diff --git a/examples/inventory_agent/service.py b/examples/inventory_agent/service.py new file mode 100644 index 0000000..507ae96 --- /dev/null +++ b/examples/inventory_agent/service.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +"""Loopback-only inventory example. SQLite atomically stores stock and receipts. +No authentication: use only as a local integration example, not a public service. +""" +import argparse +import json +import re +import sqlite3 +from http.server import BaseHTTPRequestHandler, HTTPServer +from pathlib import Path +from urllib.parse import urlsplit + + +def initialize(path): + with sqlite3.connect(path) as db: + db.execute('PRAGMA journal_mode=WAL') + db.execute('CREATE TABLE IF NOT EXISTS stock(sku TEXT PRIMARY KEY, quantity INTEGER NOT NULL)') + db.execute('CREATE TABLE IF NOT EXISTS receipts(key TEXT PRIMARY KEY, input TEXT NOT NULL, result TEXT NOT NULL, status INTEGER NOT NULL)') + db.execute("INSERT OR IGNORE INTO stock VALUES('widget',10)") + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument('--database', required=True) + parser.add_argument('--port', type=int, default=8765) + parser.add_argument('--ready-file') + parser.add_argument('--drop-first-response', action='store_true', help='Fault injection: commit then close socket without a response') + args = parser.parse_args() + initialize(args.database) + + class Handler(BaseHTTPRequestHandler): + dropped = False + + def log_message(self, *_): + pass + + def reply(self, status, result): + body = json.dumps(result).encode() + self.send_response(status) + self.send_header('Content-Type', 'application/json') + self.send_header('Content-Length', str(len(body))) + self.end_headers() + self.wfile.write(body) + + def do_GET(self): + path = urlsplit(self.path).path + with sqlite3.connect(args.database) as db: + if path == '/stock/widget': + return self.reply(200, {'quantity': db.execute("SELECT quantity FROM stock WHERE sku='widget'").fetchone()[0]}) + if path.startswith('/receipts/'): + receipt = db.execute('SELECT result FROM receipts WHERE key=?', (path.removeprefix('/receipts/'),)).fetchone() + if receipt: + return self.reply(200, json.loads(receipt[0])) + self.reply(404, {'error': 'not found'}) + + def do_POST(self): + if self.path != '/reservations': + return self.reply(404, {'error': 'not found'}) + try: + length = int(self.headers.get('Content-Length', '0')) + if not 0 < length <= 4096: + raise ValueError('Invalid body length') + key = self.headers.get('Idempotency-Key', '') + if not re.fullmatch(r'[0-9a-f]{2,512}\.[0-9a-f]{2,512}', key): + raise ValueError('Invalid idempotency key') + request = json.loads(self.rfile.read(length)) + if (not isinstance(request, dict) or set(request) != {'sku', 'quantity'} or + not isinstance(request['sku'], str) or type(request['quantity']) is not int or + not 1 <= request['quantity'] <= 300000): + raise ValueError('Invalid reservation') + except (ValueError, TypeError): + return self.reply(400, {'error': 'invalid request'}) + encoded = json.dumps(request, sort_keys=True) + with sqlite3.connect(args.database) as db: + db.execute('PRAGMA synchronous=FULL') + db.execute('BEGIN IMMEDIATE') + prior = db.execute('SELECT input,result,status FROM receipts WHERE key=?', (key,)).fetchone() + if prior: + if prior[0] != encoded: + return self.reply(409, {'key': key, 'outcome': 'conflict', 'error': 'key reused with different input'}) + return self.reply(prior[2], json.loads(prior[1])) + stock = db.execute('SELECT quantity FROM stock WHERE sku=?', (request['sku'],)).fetchone() + if not stock or stock[0] < request['quantity']: + result = {'key': key, 'outcome': 'failed', 'error': 'Insufficient stock or unknown SKU'} + status = 409 + else: + remaining = stock[0] - request['quantity'] + db.execute('UPDATE stock SET quantity=? WHERE sku=?', (remaining, request['sku'])) + result = {'key': key, 'outcome': 'committed', 'output': {**request, 'remaining': remaining}} + status = 200 + db.execute('INSERT INTO receipts VALUES(?,?,?,?)', (key, encoded, json.dumps(result), status)) + # The independent service transaction has committed before this fault. + if args.drop_first_response and not Handler.dropped: + Handler.dropped = True + self.close_connection = True + return + self.reply(status, result) + + server = HTTPServer(('127.0.0.1', args.port), Handler) + if args.ready_file: + Path(args.ready_file).write_text(str(server.server_port)) + print(f'Inventory service listening on http://127.0.0.1:{server.server_port}', flush=True) + server.serve_forever() + + +if __name__ == '__main__': + main() diff --git a/examples/reference_agent/main.cpp b/examples/reference_agent/main.cpp index 529e92f..9917170 100644 --- a/examples/reference_agent/main.cpp +++ b/examples/reference_agent/main.cpp @@ -26,8 +26,10 @@ int main(int argc, char** argv) { #endif int calls = 0; Json schema{{"type", "object"}, {"properties", Json::object()}, {"additionalProperties", false}}; - if (!runtime.registerTool({"ping", "Consumer integration", schema, - [&](const Json&) { ++calls; return Result::success("pong"); }})) return 1; + if (!runtime.registerContextTool({"ping", "Consumer integration", schema, + [&](const ExecutionContext& context, const Json&) { + if (context.idempotencyKey() != "636f6e73756d6572.31") return ToolOutcome::failed("Wrong identity"); + ++calls; return ToolOutcome::committed("pong"); }})) return 1; if (!runtime.registerSkill("ping", "ping", Json::object())) return 2; auto result = runtime.run({"consumer", "1", "ping"}); return result && result.value->output == "pong" && calls == 1 ? 0 : 3; diff --git a/include/master_agent/runtime/reference_runtime.h b/include/master_agent/runtime/reference_runtime.h index ec8f4b1..131a66b 100644 --- a/include/master_agent/runtime/reference_runtime.h +++ b/include/master_agent/runtime/reference_runtime.h @@ -3,6 +3,8 @@ #include "master_agent/common/types.h" #include #include +#include +#include #include #include #include @@ -25,6 +27,55 @@ struct Tool { ToolHandler execute; }; +// Copies share a thread-safe cancellation signal. Cancellation is cooperative. +class CancellationToken { +public: + void cancel() const { cancelled_->store(true); } + bool cancelled() const { return cancelled_->load(); } +private: + std::shared_ptr> cancelled_ = std::make_shared>(false); +}; + +struct RunOptions { + std::optional deadline; + CancellationToken cancellation; +}; + +struct ExecutionContext { + std::string session_id; + std::string request_id; + RunOptions options; + bool stopRequested() const; + // Collision-free encoding of session/request bytes; scope to one application/store. + std::string idempotencyKey() const; +}; + +enum class ToolState { Committed, Failed, Unknown }; + +// Failed means the external outcome is definitely unsuccessful. A lost response +// after sending a mutation is Unknown, even when the transport reports an error. +class ToolOutcome { +public: + static ToolOutcome committed(Json output); + static ToolOutcome failed(std::string message); + static ToolOutcome unknown(std::string message); + ToolState state() const { return state_; } +private: + friend class Runtime; + ToolOutcome(ToolState state, Json output, StructuredError error) + : state_(state), output_(std::move(output)), error_(std::move(error)) {} + ToolState state_; + Json output_; + StructuredError error_; +}; + +struct ContextTool { + std::string name; + std::string description; + Json parameters; + std::function execute; +}; + struct Turn { std::string session_id; std::string request_id; @@ -73,9 +124,10 @@ class Runtime { Status reconcile(const std::string& session_id, const std::string& request_id, const Resolution& resolution); Status registerTool(Tool tool); + Status registerContextTool(ContextTool tool); Status registerSkill(std::string phrase, std::string tool, Json arguments); Status setModel(ModelHandler model); - Result run(const Turn& turn); + Result run(const Turn& turn, const RunOptions& options = {}); // Discards history AND deduplication. Refuses sessions with UNKNOWN outcomes. Status clearSession(const std::string& session_id); @@ -91,11 +143,11 @@ class Runtime { std::map requests; std::vector history; }; - Result execute(const Turn& turn, Session& session); + Result execute(const Turn& turn, Session& session, const ExecutionContext& context); void remember(Session& session, const Turn& turn, const Reply& reply); Limits limits_; std::mutex mutex_; - std::map tools_; + std::map tools_; std::map skills_; std::map sessions_; ModelHandler model_; diff --git a/scripts/package.sh b/scripts/package.sh index 0a22fcf..13174af 100755 --- a/scripts/package.sh +++ b/scripts/package.sh @@ -12,6 +12,8 @@ cmake --install "${BUILD_DIR}" --prefix "${STAGING}/sdk" --config Release test -x "${STAGING}/sdk/bin/sparx" cp "${ROOT}/LICENSE" "${ROOT}/README.md" "${STAGING}/sdk/" cp -R "${ROOT}/examples/reference_agent" "${STAGING}/sdk/example" +cp -R "${ROOT}/examples/inventory_agent" "${STAGING}/sdk/example-inventory" +cp "${ROOT}/tests/test_inventory_contract.py" "${STAGING}/sdk/example-inventory/contract_test.py" python3 "${ROOT}/scripts/build_manifest.py" "${ROOT}" "${BUILD_DIR}" "${STAGING}/sdk/BUILD_INFO.json" tar -czf "${STAGING}/candidate.tar.gz" -C "${STAGING}/sdk" . mkdir "${STAGING}/unpacked" @@ -22,6 +24,19 @@ cmake -S "${STAGING}/unpacked/example" -B "${STAGING}/consumer" \ -DCMAKE_PREFIX_PATH="${STAGING}/unpacked" cmake --build "${STAGING}/consumer" --config Release ctest --test-dir "${STAGING}/consumer" -C Release --output-on-failure +if python3 - "${STAGING}/unpacked/BUILD_INFO.json" <<'PY_FEATURES' +import json, sys +features = json.load(open(sys.argv[1]))['features'] +sys.exit(0 if features['MASTER_AGENT_ENABLE_HTTP'] and features['MASTER_AGENT_ENABLE_STORAGE'] else 1) +PY_FEATURES +then + cmake -S "${STAGING}/unpacked/example-inventory" -B "${STAGING}/inventory-consumer" \ + -DCMAKE_PREFIX_PATH="${STAGING}/unpacked" + cmake --build "${STAGING}/inventory-consumer" --config Release + python3 "${STAGING}/unpacked/example-inventory/contract_test.py" \ + "${STAGING}/inventory-consumer/inventory_agent" "${STAGING}/unpacked/bin/sparx" \ + "${STAGING}/unpacked/example-inventory/service.py" +fi # Publish the file only after all verification has succeeded. cp "${STAGING}/candidate.tar.gz" "${ARCHIVE}" python3 - "${ARCHIVE}" <<'PY_CHECKSUM' diff --git a/tests/consumer/main.cpp b/tests/consumer/main.cpp index 529e92f..9917170 100644 --- a/tests/consumer/main.cpp +++ b/tests/consumer/main.cpp @@ -26,8 +26,10 @@ int main(int argc, char** argv) { #endif int calls = 0; Json schema{{"type", "object"}, {"properties", Json::object()}, {"additionalProperties", false}}; - if (!runtime.registerTool({"ping", "Consumer integration", schema, - [&](const Json&) { ++calls; return Result::success("pong"); }})) return 1; + if (!runtime.registerContextTool({"ping", "Consumer integration", schema, + [&](const ExecutionContext& context, const Json&) { + if (context.idempotencyKey() != "636f6e73756d6572.31") return ToolOutcome::failed("Wrong identity"); + ++calls; return ToolOutcome::committed("pong"); }})) return 1; if (!runtime.registerSkill("ping", "ping", Json::object())) return 2; auto result = runtime.run({"consumer", "1", "ping"}); return result && result.value->output == "pong" && calls == 1 ? 0 : 3; diff --git a/tests/test_execution_context.cpp b/tests/test_execution_context.cpp new file mode 100644 index 0000000..7bdcfcd --- /dev/null +++ b/tests/test_execution_context.cpp @@ -0,0 +1,79 @@ +#include "master_agent/runtime/reference_runtime.h" +#include "check.h" +#include +#include +using namespace master_agent; +using namespace master_agent::reference; +int main() { + Json schema{{"type", "object"}, {"properties", Json::object()}, {"additionalProperties", false}}; + Runtime runtime({1, 32, 8}); + int calls = 0; + CHECK(runtime.registerContextTool({"effect", "context-aware", schema, + [&](const ExecutionContext& context, const Json&) { + ++calls; + CHECK(context.session_id == "s" && context.request_id == "1"); + CHECK(context.idempotencyKey() == "73.31"); + context.options.cancellation.cancel(); // Cancellation after a confirmed effect. + return ToolOutcome::committed(42); + }})); + CHECK(runtime.registerSkill("effect", "effect", Json::object())); + RunOptions cancelled; + cancelled.cancellation.cancel(); + CHECK(runtime.run({"unused", "1", "effect"}, cancelled).error->code == "CANCELLED"); + RunOptions expired; + expired.deadline = std::chrono::steady_clock::now(); + CHECK(runtime.run({"unused", "1", "effect"}, expired).error->code == "DEADLINE_EXCEEDED"); + CHECK(calls == 0); + CHECK(runtime.run({"s", "1", "effect"}).value->output == 42); + CHECK(runtime.run({"s", "1", "effect"}, cancelled).value->replayed && calls == 1); + CHECK(runtime.registerContextTool({"rejected", "known failure", schema, + [](const ExecutionContext&, const Json&) { return ToolOutcome::failed("Insufficient stock"); }})); + CHECK(runtime.registerSkill("reject", "rejected", Json::object())); + CHECK(runtime.run({"s", "2", "reject"}).error->code == "TOOL_FAILED"); + CHECK(runtime.unresolved().value->empty()); + CHECK(runtime.registerContextTool({"lost", "lost receipt", schema, + [](const ExecutionContext&, const Json&) { return ToolOutcome::unknown("Service committed; response lost"); }})); + CHECK(runtime.registerSkill("lost", "lost", Json::object())); + CHECK(runtime.run({"s", "3", "lost"}).error->code == "UNKNOWN"); + CHECK(runtime.unresolved().value->size() == 1); + CHECK(runtime.reconcile("s", "3", {true, "verified", "External receipt found"})); + CHECK(runtime.run({"s", "3", "lost"}).value->output == "verified"); + // Delimiters and Unicode do not collide across the session/request boundary. + CHECK(ExecutionContext{"a:b", "c", {}}.idempotencyKey() != ExecutionContext{"a", "b:c", {}}.idempotencyKey()); + CHECK(ExecutionContext{"中", "文", {}}.idempotencyKey() == "e4b8ad.e69687"); + Runtime during_model; + RunOptions options; + int effects = 0; + CHECK(during_model.registerContextTool({"effect", "must not run", schema, + [&](const ExecutionContext&, const Json&) { ++effects; return ToolOutcome::committed(1); }})); + CHECK(during_model.setModel([&](const Json&) { + options.cancellation.cancel(); + return Result::success(R"({"tool":"effect","arguments":{}})"); + })); + CHECK(during_model.run({"s", "m", "model"}, options).error->code == "CANCELLED"); + CHECK(effects == 0); + RunOptions short_deadline; + short_deadline.deadline = std::chrono::steady_clock::now() + std::chrono::milliseconds(10); + CHECK(during_model.setModel([&](const Json&) { + while (std::chrono::steady_clock::now() < *short_deadline.deadline) std::this_thread::yield(); + return Result::success(R"({"tool":"effect","arguments":{}})"); + })); + CHECK(during_model.run({"s", "expired-model", "model"}, short_deadline).error->code == "DEADLINE_EXCEEDED"); + CHECK(effects == 0); + Runtime cooperative; + RunOptions signal; + std::promise entered; + auto entered_future = entered.get_future(); + CHECK(cooperative.registerContextTool({"wait", "cooperative cancellation", schema, + [&](const ExecutionContext& context, const Json&) { + entered.set_value(); + while (!context.stopRequested()) std::this_thread::yield(); + return ToolOutcome::failed("Cancelled before any side effect"); + }})); + CHECK(cooperative.registerSkill("wait", "wait", Json::object())); + auto running = std::async(std::launch::async, [&] { return cooperative.run({"s", "w", "wait"}, signal); }); + CHECK(entered_future.wait_for(std::chrono::seconds(2)) == std::future_status::ready); + signal.cancellation.cancel(); + CHECK(running.wait_for(std::chrono::seconds(2)) == std::future_status::ready); + CHECK(running.get().error->code == "TOOL_FAILED"); +} diff --git a/tests/test_inventory_contract.py b/tests/test_inventory_contract.py new file mode 100644 index 0000000..c2abf3b --- /dev/null +++ b/tests/test_inventory_contract.py @@ -0,0 +1,93 @@ +"""Independent service commits stock; runtime recovers a lost network receipt.""" +import json +import subprocess +import sys +import tempfile +import time +import urllib.error +import urllib.request +from pathlib import Path + +agent, cli, service = sys.argv[1:] +# Loopback integration must not inherit a developer's HTTP proxy. +http = urllib.request.build_opener(urllib.request.ProxyHandler({})) + + +def request(url, data=None, key=None): + headers = {'Content-Type': 'application/json'} + if key: + headers['Idempotency-Key'] = key + req = urllib.request.Request(url, data=None if data is None else json.dumps(data).encode(), headers=headers) + try: + with http.open(req, timeout=3) as response: + return response.status, json.load(response) + except urllib.error.HTTPError as error: + return error.code, json.load(error) + + +def run(command, code=0): + p = subprocess.run(command, capture_output=True, text=True, timeout=5) + assert p.returncode == code, (command, p.stdout, p.stderr) + return json.loads(p.stdout) + + +def start(root, drop=False): + ready = root / 'ready' + ready.unlink(missing_ok=True) + command = [sys.executable, service, '--database', str(root / 'inventory.db'), '--port', '0', '--ready-file', str(ready)] + if drop: + command += ['--drop-first-response'] + process = subprocess.Popen(command, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE) + for _ in range(150): + if ready.exists(): + return process, 'http://127.0.0.1:' + ready.read_text() + if process.poll() is not None: + raise AssertionError(process.stderr.read().decode()) + time.sleep(0.02) + process.terminate() + process.wait(timeout=5) + raise AssertionError('Service startup timed out') + + +with tempfile.TemporaryDirectory(prefix='sparx-inventory-') as directory: + root = Path(directory) + db = str(root / 'runtime.db') + process, endpoint = start(root, drop=True) + try: + args = [agent, endpoint, db, 'shop', 'order-1', 'widget', '2', '1000'] + assert run(args, 1)['error'] == 'UNKNOWN' + assert request(endpoint + '/stock/widget')[1]['quantity'] == 8 + # Fresh process replays UNKNOWN and never sends the mutation again. + assert run(args, 1)['error'] == 'UNKNOWN' + pending = run([cli, 'recover', '--state', db]) + assert len(pending['unresolved']) == 1, pending + key = 'shop'.encode().hex() + '.' + 'order-1'.encode().hex() + receipt = request(endpoint + '/receipts/' + key)[1] + assert receipt['outcome'] == 'committed' + run([cli, 'reconcile', '--state', db, '--session', 'shop', '--request-id', 'order-1', + '--outcome', 'committed', '--output', json.dumps(receipt['output']), + '--note', 'Verified independent service receipt ' + key]) + replay = run(args) + assert replay['replayed'] and replay['output']['remaining'] == 8 + # External idempotency is atomic with stock and detects changed arguments. + assert request(endpoint + '/reservations', {'sku': 'widget', 'quantity': 2}, key)[0] == 200 + assert request(endpoint + '/reservations', {'sku': 'widget', 'quantity': 3}, key)[1]['outcome'] == 'conflict' + assert request(endpoint + '/stock/widget')[1]['quantity'] == 8 + assert run([agent, endpoint, db, 'shop', 'order-2', 'widget', '20'], 1)['error'] == 'TOOL_FAILED' + assert not run([cli, 'recover', '--state', db])['unresolved'] + assert run([agent, endpoint, db, 'shop', 'order-3', 'widget', '3'])['output']['remaining'] == 5 + finally: + process.terminate() + process.wait(timeout=5) + process, endpoint = start(root) + try: + assert request(endpoint + '/stock/widget')[1]['quantity'] == 5 + # Even a fresh runtime store uses the stable key; the service does not + # repeat the reservation. The service receipt also survives its restart. + result = run([agent, endpoint, str(root / 'fresh.db'), 'shop', 'order-3', 'widget', '3']) + assert result['output']['remaining'] == 5 and not result['replayed'] + assert request(endpoint + '/stock/widget')[1]['quantity'] == 5 + finally: + process.terminate() + process.wait(timeout=5) +print('Inventory integration: lost receipt, reconciliation, atomic idempotency and restart passed') From 842121ead24d49c30ad4f1021fae40e0f42a0a36 Mon Sep 17 00:00:00 2001 From: Codex Date: Mon, 28 Sep 2026 15:02:46 +0800 Subject: [PATCH 4/4] fix(tests): avoid reverse DNS during loopback service startup --- examples/inventory_agent/service.py | 12 +++++++++++- tests/test_http_model.py | 11 ++++++++++- tests/test_inventory_contract.py | 7 ++++--- 3 files changed, 25 insertions(+), 5 deletions(-) diff --git a/examples/inventory_agent/service.py b/examples/inventory_agent/service.py index 507ae96..36d9544 100644 --- a/examples/inventory_agent/service.py +++ b/examples/inventory_agent/service.py @@ -6,6 +6,7 @@ import json import re import sqlite3 +from socketserver import TCPServer from http.server import BaseHTTPRequestHandler, HTTPServer from pathlib import Path from urllib.parse import urlsplit @@ -19,6 +20,15 @@ def initialize(path): db.execute("INSERT OR IGNORE INTO stock VALUES('widget',10)") +class LoopbackServer(HTTPServer): + def server_bind(self): + # HTTPServer normally does a reverse-DNS lookup for server_name. This + # loopback-only service does not need DNS and must start offline too. + TCPServer.server_bind(self) + self.server_name = 'localhost' + self.server_port = self.server_address[1] + + def main(): parser = argparse.ArgumentParser() parser.add_argument('--database', required=True) @@ -96,7 +106,7 @@ def do_POST(self): return self.reply(status, result) - server = HTTPServer(('127.0.0.1', args.port), Handler) + server = LoopbackServer(('127.0.0.1', args.port), Handler) if args.ready_file: Path(args.ready_file).write_text(str(server.server_port)) print(f'Inventory service listening on http://127.0.0.1:{server.server_port}', flush=True) diff --git a/tests/test_http_model.py b/tests/test_http_model.py index 37332cb..c92a08d 100644 --- a/tests/test_http_model.py +++ b/tests/test_http_model.py @@ -4,6 +4,7 @@ import subprocess import sys import threading +from socketserver import TCPServer import time @@ -40,7 +41,15 @@ def do_POST(self): pass -server = http.server.ThreadingHTTPServer(('127.0.0.1', 0), Server) +class LoopbackServer(http.server.ThreadingHTTPServer): + def server_bind(self): + # Keep the local protocol fixture independent of DNS configuration. + TCPServer.server_bind(self) + self.server_name = 'localhost' + self.server_port = self.server_address[1] + + +server = LoopbackServer(('127.0.0.1', 0), Server) thread = threading.Thread(target=server.serve_forever, daemon=True) thread.start() try: diff --git a/tests/test_inventory_contract.py b/tests/test_inventory_contract.py index c2abf3b..57e3096 100644 --- a/tests/test_inventory_contract.py +++ b/tests/test_inventory_contract.py @@ -38,15 +38,16 @@ def start(root, drop=False): if drop: command += ['--drop-first-response'] process = subprocess.Popen(command, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE) - for _ in range(150): + deadline = time.monotonic() + 10 + while time.monotonic() < deadline: if ready.exists(): return process, 'http://127.0.0.1:' + ready.read_text() if process.poll() is not None: raise AssertionError(process.stderr.read().decode()) time.sleep(0.02) process.terminate() - process.wait(timeout=5) - raise AssertionError('Service startup timed out') + _, errors = process.communicate(timeout=5) + raise AssertionError('Service startup timed out: ' + errors.decode()) with tempfile.TemporaryDirectory(prefix='sparx-inventory-') as directory: