From 106f71866f1e6bf28d8f19a4f89a000bef7e778e Mon Sep 17 00:00:00 2001 From: Vitor Castro Date: Mon, 28 Sep 2026 13:26:51 +0100 Subject: [PATCH 1/3] feat: bundle llama.cpp and vLLM compose overrides with an end-to-end memory check Adds docker-compose.llamacpp.yml (CPU) and docker-compose.vllm.yml (GPU), scripts/aha-chat.sh/.ps1 (two-turn session recall where the second request carries only the new question), and an e2e-llamacpp workflow that runs it against a real llama-server. Release-As: 0.2.0-beta --- .env.example | 34 ++++++++-- .github/workflows/e2e-llamacpp.yml | 105 +++++++++++++++++++++++++++++ docker-compose.llamacpp.yml | 50 ++++++++++++++ docker-compose.vllm.yml | 56 +++++++++++++++ docker-compose.yml | 3 +- scripts/aha-chat.ps1 | 50 ++++++++++++++ scripts/aha-chat.sh | 46 +++++++++++++ 7 files changed, 336 insertions(+), 8 deletions(-) create mode 100644 .github/workflows/e2e-llamacpp.yml create mode 100644 docker-compose.llamacpp.yml create mode 100644 docker-compose.vllm.yml create mode 100644 scripts/aha-chat.ps1 create mode 100755 scripts/aha-chat.sh diff --git a/.env.example b/.env.example index 2b5e60a..d6b89bf 100644 --- a/.env.example +++ b/.env.example @@ -5,18 +5,38 @@ ADMIN_PORT=5200 PUBLIC_API_URL=http://localhost:5100 ADMIN_PUBLIC_URL=http://localhost:5200 -# Preferred host-level LLM base URL (any OpenAI-compatible or Ollama-compatible engine). -# When empty, Compose falls back to OLLAMA_ENDPOINT below. -# LLM_ENDPOINT=http://host.docker.internal:8000 +# --- LLM engine ------------------------------------------------------------- +# Bundled engines (no host LLM needed): +# llama.cpp: docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build +# vLLM: docker compose -f docker-compose.yml -f docker-compose.vllm.yml up --build +# The overrides set LLM_ENDPOINT for you; the values below only matter for the plain compose file. -# DX default: Ollama on the host (used when LLM_ENDPOINT is unset) +# Model name sent to the engine. llama-server ignores it; vLLM uses it as --served-model-name; +# Ollama needs a pulled tag (e.g. qwen3.5:9b). Overrides default to local-model, plain compose to qwen3.5:9b. +# DEFAULT_LLM_MODEL=local-model + +# llama.cpp override +# LLAMACPP_HF_MODEL=bartowski/Qwen2.5-3B-Instruct-GGUF:Q4_K_M +# LLAMACPP_CTX_SIZE=8192 +# LLAMACPP_IMAGE=ghcr.io/ggml-org/llama.cpp:server + +# vLLM override +# VLLM_MODEL=Qwen/Qwen2.5-7B-Instruct +# VLLM_MAX_MODEL_LEN=16384 +# VLLM_TOOL_CALL_PARSER=hermes +# HF_TOKEN= + +# Engine already running on the host (any OpenAI-compatible /v1 server): +# LLM_ENDPOINT=http://host.docker.internal:8080 + +# Ollama on the host (used only when LLM_ENDPOINT is empty): OLLAMA_ENDPOINT=http://host.docker.internal:11434 -DEFAULT_LLM_MODEL=qwen3.5:9b -# Optional: OpenAI / vLLM / ExLlamaSharp / any OpenAI-compatible host defaults -# OPENAI_ENDPOINT=http://host.docker.internal:8000 +# Host defaults for apps with llmBackend openai / openai-compatible / vllm and no per-app endpoint +# OPENAI_ENDPOINT=https://api.openai.com # OPENAI_API_KEY= +# --- Keys (change before any real use) ------------------------------------- MASTER_KEY=cm_master_dev_key_change_me DEMO_APP_API_KEY=cm_live_dev_key_change_me diff --git a/.github/workflows/e2e-llamacpp.yml b/.github/workflows/e2e-llamacpp.yml new file mode 100644 index 0000000..21d4002 --- /dev/null +++ b/.github/workflows/e2e-llamacpp.yml @@ -0,0 +1,105 @@ +name: e2e-llamacpp + +# End-to-end memory check against a real llama.cpp server (CPU, small GGUF): +# gateway image built from this commit + llama-server + scripts/aha-chat.sh. + +on: + push: + branches: [main] + pull_request: + paths: + - "src/ContextMemory.Api/**" + - "src/ContextMemory.Adapters/**" + - "src/ContextMemory.Core/**" + - "src/ContextMemory.Infrastructure/**" + - "Dockerfile" + - "scripts/aha-chat.sh" + - ".github/workflows/e2e-llamacpp.yml" + schedule: + - cron: "0 4 * * *" + workflow_dispatch: + inputs: + hf_model: + description: "GGUF to test (:)" + required: false + default: "bartowski/Qwen2.5-1.5B-Instruct-GGUF:Q4_K_M" + +permissions: + contents: read + +concurrency: + group: e2e-llamacpp-${{ github.ref }} + cancel-in-progress: true + +env: + HF_MODEL: ${{ inputs.hf_model || 'bartowski/Qwen2.5-1.5B-Instruct-GGUF:Q4_K_M' }} + MODELS_DIR: ${{ github.workspace }}/.llama-models + +jobs: + aha: + runs-on: ubuntu-latest + timeout-minutes: 40 + steps: + - uses: actions/checkout@v4 + + - name: Cache GGUF + uses: actions/cache@v4 + with: + path: .llama-models + key: llama-gguf-${{ env.HF_MODEL }} + + - name: Start llama-server + run: | + mkdir -p "$MODELS_DIR" && chmod 777 "$MODELS_DIR" + docker network create cm-e2e + docker run -d --name llama-server --network cm-e2e -p 8081:8080 \ + -v "$MODELS_DIR:/models" -e LLAMA_CACHE=/models \ + ghcr.io/ggml-org/llama.cpp:server \ + --host 0.0.0.0 --port 8080 --jinja --ctx-size 8192 -hf "$HF_MODEL" + + - name: Build gateway image + run: docker build -t contextmemory-e2e -f Dockerfile . + + - name: Wait for llama-server + run: | + for i in $(seq 1 120); do + if curl -fsS http://localhost:8081/health >/dev/null 2>&1; then echo "llama-server ready"; exit 0; fi + sleep 5 + done + docker logs llama-server | tail -n 50 + exit 1 + + - name: Start gateway + run: | + docker run -d --name cm-api --network cm-e2e -p 5100:8080 \ + -e ContextMemory__PersistenceProvider=File \ + -e ContextMemory__DataPath=/app/data \ + -e ContextMemory__MasterKey=cm_master_e2e \ + -e ContextMemory__LlmEndpoint=http://llama-server:8080 \ + -e ContextMemory__DefaultLlmModel=local-model \ + -e ContextMemory__Apps__demo-dev__ApiKey=cm_live_e2e \ + -e ContextMemory__Apps__demo-dev__LlmBackend=openai-compatible \ + -e ContextMemory__Apps__demo-dev__LlmEndpoint=http://llama-server:8080 \ + -e ContextMemory__Apps__demo-dev__LlmModel=local-model \ + contextmemory-e2e + for i in $(seq 1 60); do + if curl -fsS http://localhost:5100/health >/dev/null 2>&1; then echo "gateway ready"; exit 0; fi + sleep 3 + done + docker logs cm-api | tail -n 80 + exit 1 + + - name: Run chat aha + env: + CONTEXTMEMORY_BASE_URL: http://localhost:5100 + CONTEXTMEMORY_API_KEY: cm_live_e2e + CONTEXTMEMORY_APP_ID: demo-dev + CONTEXTMEMORY_MODEL: local-model + CONTEXTMEMORY_TIMEOUT: "600" + run: bash scripts/aha-chat.sh + + - name: Container logs + if: failure() + run: | + echo "::group::cm-api"; docker logs cm-api 2>&1 | tail -n 200; echo "::endgroup::" + echo "::group::llama-server"; docker logs llama-server 2>&1 | tail -n 100; echo "::endgroup::" diff --git a/docker-compose.llamacpp.yml b/docker-compose.llamacpp.yml new file mode 100644 index 0000000..0ef7af4 --- /dev/null +++ b/docker-compose.llamacpp.yml @@ -0,0 +1,50 @@ +# Override: run llama.cpp (llama-server) next to the gateway — CPU, no host LLM required. +# Usage: docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build +# +# The model is pulled from Hugging Face on first start and cached in the llama_models volume. +# Pick another GGUF with LLAMACPP_HF_MODEL=: (see docs/self-host.md#engines). +# GPU build: set LLAMACPP_IMAGE=ghcr.io/ggml-org/llama.cpp:server-cuda and add a device reservation. + +services: + llama-server: + image: ${LLAMACPP_IMAGE:-ghcr.io/ggml-org/llama.cpp:server} + container_name: contextmemory-llama-server + restart: unless-stopped + # --jinja is required for OpenAI-style tool calling (agentic mode). + command: + - --host + - 0.0.0.0 + - --port + - "8080" + - --jinja + - --ctx-size + - "${LLAMACPP_CTX_SIZE:-8192}" + - -hf + - ${LLAMACPP_HF_MODEL:-bartowski/Qwen2.5-3B-Instruct-GGUF:Q4_K_M} + environment: + LLAMA_CACHE: /models + HF_TOKEN: ${HF_TOKEN:-} + volumes: + - llama_models:/models + networks: + - cm-net + healthcheck: + test: ["CMD", "curl", "-fsS", "http://127.0.0.1:8080/health"] + interval: 15s + timeout: 5s + retries: 40 + start_period: 60s + + api: + environment: + ContextMemory__LlmEndpoint: http://llama-server:8080 + ContextMemory__DefaultLlmModel: ${DEFAULT_LLM_MODEL:-local-model} + ContextMemory__Apps__demo-dev__LlmBackend: openai-compatible + ContextMemory__Apps__demo-dev__LlmEndpoint: http://llama-server:8080 + ContextMemory__Apps__demo-dev__LlmModel: ${DEFAULT_LLM_MODEL:-local-model} + depends_on: + llama-server: + condition: service_healthy + +volumes: + llama_models: diff --git a/docker-compose.vllm.yml b/docker-compose.vllm.yml new file mode 100644 index 0000000..2cafa39 --- /dev/null +++ b/docker-compose.vllm.yml @@ -0,0 +1,56 @@ +# Override: run vLLM next to the gateway — requires an NVIDIA GPU + nvidia-container-toolkit. +# Usage: docker compose -f docker-compose.yml -f docker-compose.vllm.yml up --build +# +# Weights are downloaded from Hugging Face on first start and cached in the vllm_cache volume. +# Gated models need HF_TOKEN. Tool calling needs --enable-auto-tool-choice + a parser that +# matches the model family (hermes for Qwen2.5/Qwen3, llama3_json for Llama 3.x, mistral for Mistral). + +services: + vllm: + image: ${VLLM_IMAGE:-vllm/vllm-openai:latest} + container_name: contextmemory-vllm + restart: unless-stopped + ipc: host + command: + - --model + - ${VLLM_MODEL:-Qwen/Qwen2.5-7B-Instruct} + - --served-model-name + - ${DEFAULT_LLM_MODEL:-local-model} + - --max-model-len + - "${VLLM_MAX_MODEL_LEN:-16384}" + - --enable-auto-tool-choice + - --tool-call-parser + - ${VLLM_TOOL_CALL_PARSER:-hermes} + environment: + HF_TOKEN: ${HF_TOKEN:-} + volumes: + - vllm_cache:/root/.cache/huggingface + deploy: + resources: + reservations: + devices: + - driver: nvidia + count: all + capabilities: [gpu] + networks: + - cm-net + healthcheck: + test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health')"] + interval: 15s + timeout: 5s + retries: 60 + start_period: 120s + + api: + environment: + ContextMemory__LlmEndpoint: http://vllm:8000 + ContextMemory__DefaultLlmModel: ${DEFAULT_LLM_MODEL:-local-model} + ContextMemory__Apps__demo-dev__LlmBackend: vllm + ContextMemory__Apps__demo-dev__LlmEndpoint: http://vllm:8000 + ContextMemory__Apps__demo-dev__LlmModel: ${DEFAULT_LLM_MODEL:-local-model} + depends_on: + vllm: + condition: service_healthy + +volumes: + vllm_cache: diff --git a/docker-compose.yml b/docker-compose.yml index 471502a..6317089 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -1,6 +1,7 @@ # One-command local stack: API (:5100) + Admin (:5200) + MCP runtime + code sandbox # Usage: docker compose up --build -# Requires Ollama on the host (default http://host.docker.internal:11434) +# LLM engine: add -f docker-compose.llamacpp.yml (CPU) or -f docker-compose.vllm.yml (GPU), +# or point LLM_ENDPOINT at any OpenAI-compatible /v1 server. Falls back to Ollama on the host. # # mcp-runtime = hosts MCP stdio servers (e.g. zuora-mcp) # sandbox-runtime = runs shell/python/node via POST /execute (self-hosted-sandbox) diff --git a/scripts/aha-chat.ps1 b/scripts/aha-chat.ps1 new file mode 100644 index 0000000..d53e65b --- /dev/null +++ b/scripts/aha-chat.ps1 @@ -0,0 +1,50 @@ +# Chat aha: tell the gateway a fact, then ask for it in a second request that carries +# ONLY the new question. The answer can only come from ContextMemory's session memory. +# Works with any engine behind the gateway (llama.cpp, vLLM, Ollama, OpenAI, ...). +# Usage: .\scripts\aha-chat.ps1 +$ErrorActionPreference = "Stop" + +$Base = if ($env:CONTEXTMEMORY_BASE_URL) { $env:CONTEXTMEMORY_BASE_URL } else { "http://localhost:5100" } +$Key = if ($env:CONTEXTMEMORY_API_KEY) { $env:CONTEXTMEMORY_API_KEY } else { "cm_live_dev_key_change_me" } +$App = if ($env:CONTEXTMEMORY_APP_ID) { $env:CONTEXTMEMORY_APP_ID } else { "demo-dev" } +$Model = if ($env:CONTEXTMEMORY_MODEL) { $env:CONTEXTMEMORY_MODEL } else { "local-model" } +$UserId = if ($env:CONTEXTMEMORY_USER_ID) { $env:CONTEXTMEMORY_USER_ID } else { "aha-user" } +$Timeout = if ($env:CONTEXTMEMORY_TIMEOUT) { [int]$env:CONTEXTMEMORY_TIMEOUT } else { 300 } +$Session = "aha-$([DateTimeOffset]::UtcNow.ToUnixTimeSeconds())" +$Fact = "postgres-staging-01" + +$headers = @{ + Authorization = "Bearer $Key" + "X-App-Id" = $App + "X-User-Id" = $UserId + "X-Session-Id" = $Session +} + +function Send-Chat([string]$Content) { + $body = @{ model = $Model; messages = @(@{ role = "user"; content = $Content }) } | ConvertTo-Json -Depth 5 + Invoke-RestMethod -Method Post -Uri "$Base/v1/chat/completions" -Headers $headers ` + -ContentType "application/json" -Body $body -TimeoutSec $Timeout +} + +Write-Host "==> Health ($Base)" +Invoke-RestMethod -Uri "$Base/health" | ConvertTo-Json -Compress +Write-Host "" + +Write-Host "==> Turn 1 (session $Session): state the fact" +$turn1 = Send-Chat "Remember this for later: our staging database host is $Fact. Reply with OK." +$turn1.choices[0].message.content +Write-Host "" + +Write-Host "==> Turn 2 (same session, body contains only the new question)" +$turn2 = Send-Chat "What is our staging database host? Reply with the host name only." +$answer = $turn2.choices[0].message.content +$answer +Write-Host "" + +if ($answer -match [regex]::Escape($Fact)) { + Write-Host "AHA OK — the client sent no history; the gateway remembered '$Fact'." + exit 0 +} + +Write-Host "AHA FAILED — '$Fact' not found in the turn 2 answer." +exit 1 diff --git a/scripts/aha-chat.sh b/scripts/aha-chat.sh new file mode 100755 index 0000000..805e4d4 --- /dev/null +++ b/scripts/aha-chat.sh @@ -0,0 +1,46 @@ +#!/usr/bin/env bash +# Chat aha: tell the gateway a fact, then ask for it in a second request that carries +# ONLY the new question. The answer can only come from ContextMemory's session memory. +# Works with any engine behind the gateway (llama.cpp, vLLM, Ollama, OpenAI, ...). +# Usage: ./scripts/aha-chat.sh +set -euo pipefail + +BASE="${CONTEXTMEMORY_BASE_URL:-http://localhost:5100}" +KEY="${CONTEXTMEMORY_API_KEY:-cm_live_dev_key_change_me}" +APP="${CONTEXTMEMORY_APP_ID:-demo-dev}" +MODEL="${CONTEXTMEMORY_MODEL:-local-model}" +USER_ID="${CONTEXTMEMORY_USER_ID:-aha-user}" +SESSION="aha-$(date +%s)" +FACT="postgres-staging-01" +OUT="$(mktemp -d)" + +chat() { + curl -sS --fail-with-body --max-time "${CONTEXTMEMORY_TIMEOUT:-300}" \ + -X POST "$BASE/v1/chat/completions" \ + -H "Authorization: Bearer $KEY" \ + -H "X-App-Id: $APP" \ + -H "X-User-Id: $USER_ID" \ + -H "X-Session-Id: $SESSION" \ + -H "Content-Type: application/json" \ + -d "{\"model\":\"$MODEL\",\"messages\":[{\"role\":\"user\",\"content\":\"$1\"}]}" +} + +echo "==> Health ($BASE)" +curl -sSf "$BASE/health" | head -c 300 +echo; echo + +echo "==> Turn 1 (session $SESSION): state the fact" +chat "Remember this for later: our staging database host is $FACT. Reply with OK." | tee "$OUT/turn1.json" +echo; echo + +echo "==> Turn 2 (same session, body contains only the new question)" +chat "What is our staging database host? Reply with the host name only." | tee "$OUT/turn2.json" +echo; echo + +if grep -q "$FACT" "$OUT/turn2.json"; then + echo "AHA OK — the client sent no history; the gateway remembered '$FACT'." + exit 0 +fi + +echo "AHA FAILED — '$FACT' not found in the turn 2 answer." +exit 1 From 7ebc4185328aa0b7097e71a0551ff0ea4f80f743 Mon Sep 17 00:00:00 2001 From: Vitor Castro Date: Mon, 28 Sep 2026 13:26:51 +0100 Subject: [PATCH 2/3] ci: keep the public tracker for users and make release-please able to open PRs content-cadence writes to the run summary instead of opening issues; sync-mirror is one-way to the backup repo and fails with an actionable message when the token is invalid; release-please accepts RELEASE_PLEASE_TOKEN; PR titles must be Conventional Commits; issue and PR templates. --- .github/ISSUE_TEMPLATE/bug_report.yml | 59 ++++++++++++++++++++++++ .github/ISSUE_TEMPLATE/config.yml | 5 ++ .github/ISSUE_TEMPLATE/engine_report.yml | 47 +++++++++++++++++++ .github/pull_request_template.md | 12 +++++ .github/workflows/content-cadence.yml | 24 +++------- .github/workflows/pr-title.yml | 32 +++++++++++++ .github/workflows/release-please.yml | 3 ++ .github/workflows/sync-mirror.yml | 39 ++++++---------- CONTRIBUTING.md | 7 ++- 9 files changed, 185 insertions(+), 43 deletions(-) create mode 100644 .github/ISSUE_TEMPLATE/bug_report.yml create mode 100644 .github/ISSUE_TEMPLATE/config.yml create mode 100644 .github/ISSUE_TEMPLATE/engine_report.yml create mode 100644 .github/pull_request_template.md create mode 100644 .github/workflows/pr-title.yml diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml new file mode 100644 index 0000000..025de28 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/bug_report.yml @@ -0,0 +1,59 @@ +name: Bug report +description: Something in the gateway, Admin, or MCP server does not work as documented. +labels: [bug] +body: + - type: textarea + id: what + attributes: + label: What happened + description: What you did, what you expected, and what you got instead. Do not paste API keys or private conversation content. + validations: + required: true + - type: textarea + id: repro + attributes: + label: Steps to reproduce + placeholder: | + 1. docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build + 2. ./scripts/aha-chat.sh + 3. ... + validations: + required: true + - type: dropdown + id: engine + attributes: + label: LLM engine + options: + - llama.cpp + - vLLM + - Ollama + - LM Studio + - OpenAI / Azure-compatible + - Other OpenAI-compatible + validations: + required: true + - type: input + id: model + attributes: + label: Model + placeholder: e.g. Qwen2.5-7B-Instruct Q4_K_M + - type: dropdown + id: deploy + attributes: + label: Deployment + options: + - Docker Compose (this repo) + - GHCR image + - dotnet run + - Kortexio Cloud + - type: input + id: version + attributes: + label: Version / commit + placeholder: v0.2.0-beta or commit SHA + - type: textarea + id: logs + attributes: + label: Relevant logs + description: Gateway logs around the failure (redact keys and message content). + render: text diff --git a/.github/ISSUE_TEMPLATE/config.yml b/.github/ISSUE_TEMPLATE/config.yml new file mode 100644 index 0000000..fb890c6 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/config.yml @@ -0,0 +1,5 @@ +blank_issues_enabled: true +contact_links: + - name: Kortexio Cloud / commercial license + url: https://kortexio.io + about: Hosted gateway, commercial terms, or enterprise deployment questions. diff --git a/.github/ISSUE_TEMPLATE/engine_report.yml b/.github/ISSUE_TEMPLATE/engine_report.yml new file mode 100644 index 0000000..e10c2bb --- /dev/null +++ b/.github/ISSUE_TEMPLATE/engine_report.yml @@ -0,0 +1,47 @@ +name: Engine / model compatibility report +description: Tell us how a specific engine + model behaves behind ContextMemory (works, partially works, or fails). +labels: [engine-compat] +body: + - type: dropdown + id: engine + attributes: + label: Engine + options: + - llama.cpp + - vLLM + - Ollama + - LM Studio + - SGLang + - TGI + - Other OpenAI-compatible + validations: + required: true + - type: input + id: engine_flags + attributes: + label: Engine version and flags + placeholder: "llama-server b6xxx --jinja --ctx-size 8192" + validations: + required: true + - type: input + id: model + attributes: + label: Model and quantization + placeholder: Qwen2.5-7B-Instruct Q4_K_M + validations: + required: true + - type: checkboxes + id: results + attributes: + label: What works + options: + - label: scripts/aha-chat.sh passes (session memory) + - label: Streaming responses + - label: Agentic mode with wiki_search + - label: Agentic mode with MCP tools + - label: Sandbox tools (shell / python / node) + - type: textarea + id: notes + attributes: + label: Notes + description: Failures, loops, malformed tool calls, context overflows — anything we should add to the engine docs. diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md new file mode 100644 index 0000000..172d3c6 --- /dev/null +++ b/.github/pull_request_template.md @@ -0,0 +1,12 @@ +## What and why + + + +## How it was tested + +- [ ] `dotnet test tests/ContextMemory.Api.Tests/ContextMemory.Api.Tests.csproj` +- [ ] `./scripts/aha-chat.sh` against a local engine (if the chat path changed) + +## Notes for reviewers + + diff --git a/.github/workflows/content-cadence.yml b/.github/workflows/content-cadence.yml index 542229a..f943979 100644 --- a/.github/workflows/content-cadence.yml +++ b/.github/workflows/content-cadence.yml @@ -1,7 +1,7 @@ name: content-cadence -# Suggests a short social hook (manual approve via issue). -# Does not auto-publish to social — safer for brand voice. +# Suggests a short social hook in the run summary (Actions tab). +# Does not auto-publish to social and does not open issues — the public tracker is for users. on: schedule: @@ -9,16 +9,14 @@ on: workflow_dispatch: permissions: - issues: write contents: read jobs: suggest: + if: github.repository == 'Kortexio/ContextMemory' runs-on: ubuntu-latest steps: - - name: Open hook suggestion issue - env: - GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + - name: Write hook suggestion to the run summary run: | HOOKS=( "Your coding agent forgets staging DB names. Fix that in five minutes." @@ -31,16 +29,8 @@ jobs: "Admin Playground: watch tool steps and HITL without a client." ) HOOK="${HOOKS[$RANDOM % ${#HOOKS[@]}]}" - BODY=$(printf '%s\n\n%s\n\n%s\n' \ + printf '%s\n\n%s\n\n%s\n' \ "## Suggested post" \ "$HOOK" \ - "Copy to LinkedIn/dev.to after editing. Do **not** call this RAG. CTA: https://github.com/Kortexio/ContextMemory") - gh issue create \ - --repo "$GITHUB_REPOSITORY" \ - --title "Content cadence: social hook suggestion" \ - --label "content-hook" \ - --body "$BODY" || \ - gh issue create \ - --repo "$GITHUB_REPOSITORY" \ - --title "Content cadence: social hook suggestion" \ - --body "$BODY" + "Copy to LinkedIn/dev.to after editing. Do **not** call this RAG. CTA: https://github.com/Kortexio/ContextMemory" \ + >> "$GITHUB_STEP_SUMMARY" diff --git a/.github/workflows/pr-title.yml b/.github/workflows/pr-title.yml new file mode 100644 index 0000000..175bfb0 --- /dev/null +++ b/.github/workflows/pr-title.yml @@ -0,0 +1,32 @@ +name: pr-title + +# release-please only counts Conventional Commits (feat:, fix:, docs:, chore:, ...). +# PRs are squash-merged with the PR title, so the title is what lands on main. + +on: + pull_request_target: + types: [opened, edited, synchronize, reopened] + +permissions: + pull-requests: read + +jobs: + conventional: + runs-on: ubuntu-latest + steps: + - uses: amannn/action-semantic-pull-request@v5 + env: + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + with: + types: | + feat + fix + perf + refactor + docs + test + build + ci + chore + revert + requireScope: false diff --git a/.github/workflows/release-please.yml b/.github/workflows/release-please.yml index 828cc63..e06fd13 100644 --- a/.github/workflows/release-please.yml +++ b/.github/workflows/release-please.yml @@ -10,6 +10,7 @@ permissions: jobs: release-please: + if: github.repository == 'Kortexio/ContextMemory' runs-on: ubuntu-latest outputs: release_created: ${{ steps.release.outputs.release_created }} @@ -21,3 +22,5 @@ jobs: id: release with: release-type: simple + # GITHUB_TOKEN cannot open PRs unless the org allows it (Settings > Actions > General). + token: ${{ secrets.RELEASE_PLEASE_TOKEN || github.token }} diff --git a/.github/workflows/sync-mirror.yml b/.github/workflows/sync-mirror.yml index e9dd843..39dedcf 100644 --- a/.github/workflows/sync-mirror.yml +++ b/.github/workflows/sync-mirror.yml @@ -1,5 +1,8 @@ name: Sync mirror +# One-way backup: Kortexio/ContextMemory (source of truth) -> vitorcastro78/ContextMemory. +# PEER_SYNC_TOKEN must be a PAT with Contents: write on the backup repository. + on: push: workflow_dispatch: @@ -9,26 +12,12 @@ permissions: jobs: sync: + if: github.repository == 'Kortexio/ContextMemory' runs-on: ubuntu-latest + env: + PEER_REPO: vitorcastro78/ContextMemory steps: - - name: Resolve peer repository - id: peer - run: | - case "${{ github.repository }}" in - Kortexio/ContextMemory) - echo "peer=vitorcastro78/ContextMemory" >> "$GITHUB_OUTPUT" - ;; - vitorcastro78/ContextMemory) - echo "peer=Kortexio/ContextMemory" >> "$GITHUB_OUTPUT" - ;; - *) - echo "Unknown repository ${{ github.repository }}; skipping." - echo "skip=true" >> "$GITHUB_OUTPUT" - ;; - esac - - name: Checkout - if: steps.peer.outputs.skip != 'true' uses: actions/checkout@v4 with: fetch-depth: 0 @@ -36,11 +25,9 @@ jobs: # any PAT embedded in a remote URL and causes 403 as the source org/bot. persist-credentials: false - - name: Mirror to peer - if: steps.peer.outputs.skip != 'true' + - name: Mirror to backup env: PEER_TOKEN: ${{ secrets.PEER_SYNC_TOKEN }} - PEER_REPO: ${{ steps.peer.outputs.peer }} run: | set -euo pipefail if [ -z "${PEER_TOKEN:-}" ]; then @@ -48,17 +35,19 @@ jobs: exit 0 fi - # Show which account the secret authenticates as (no token leakage). - TOKEN_USER="$(curl -sS -H "Authorization: Bearer ${PEER_TOKEN}" \ + STATUS="$(curl -sS -o /dev/null -w '%{http_code}' \ + -H "Authorization: Bearer ${PEER_TOKEN}" \ -H "Accept: application/vnd.github+json" \ - https://api.github.com/user | jq -r '.login // empty')" - echo "PEER_SYNC_TOKEN identity: ${TOKEN_USER:-unknown-or-not-a-user-token}" + "https://api.github.com/repos/${PEER_REPO}")" + if [ "$STATUS" != "200" ]; then + echo "::error::PEER_SYNC_TOKEN cannot access ${PEER_REPO} (HTTP ${STATUS}). Rotate the secret: fine-grained PAT, Contents: write on ${PEER_REPO}." + exit 1 + fi # Strip any leftover Authorization headers from checkout / runner config. git config --local --unset-all "http.https://github.com/.extraheader" || true git config --local --unset-all http.extraheader || true - # Prefer Authorization header with PEER_TOKEN over URL-embedded creds. BASIC="$(printf 'x-access-token:%s' "${PEER_TOKEN}" | base64 -w 0)" git config --local "http.https://github.com/.extraheader" "AUTHORIZATION: basic ${BASIC}" diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index b10d60a..912bd73 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -39,5 +39,10 @@ Public contracts are in `src/ContextMemory.Core/Contracts/` with XML summaries. ## Pull requests 1. Keep changes focused; match existing naming and DI patterns. -2. Run the full test suite before opening a PR. +2. Run the full test suite before opening a PR. If you touched the chat path, also run `./scripts/aha-chat.sh` against a local engine (e.g. `docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build`). 3. Do not commit secrets, `data/`, or local `appsettings.Development.json`. +4. Use [Conventional Commits](https://www.conventionalcommits.org/) for the PR title and for commits pushed to `main` (`feat: …`, `fix: …`, `docs: …`, `chore: …`). Releases and the changelog are generated by release-please from these prefixes; other messages are ignored. + +## Engine compatibility reports + +Tried ContextMemory with an engine/model we do not document yet? Open an issue with the **Engine / model compatibility report** template — those reports feed the table in [docs/self-host.md](docs/self-host.md#engines). From 6f06e5da44fdc07fdcc0a438146de4942cfa8f89 Mon Sep 17 00:00:00 2001 From: Vitor Castro Date: Mon, 28 Sep 2026 13:26:51 +0100 Subject: [PATCH 3/3] docs: lead with self-hosted llama.cpp / vLLM and a three-command quickstart --- README.md | 82 ++++++++++++++++++++--------------------- docs/self-host.md | 72 ++++++++++++++++++++++++++++++++---- docs/show-hn.md | 94 ++++++++++++++++++++++++++++++++++------------- 3 files changed, 173 insertions(+), 75 deletions(-) diff --git a/README.md b/README.md index 38af2bd..b3eed60 100644 --- a/README.md +++ b/README.md @@ -5,15 +5,13 @@

- Get Cloud key + Try it (3 commands) · - Self-host + Engines · Docs · - Aha storyboard (GIF) - · - Show HN + vs Mem0 / Zep / Letta

@@ -22,21 +20,40 @@ Docker GHCR Docker CI Tests - GitHub commit activity + e2e: llama.cpp

- Your agent forgets. Fix that with memory you can open like a wiki. + Self-hosted memory gateway for your llama.cpp / vLLM server.

- One OpenAI-compatible /v1 URL: wiki memory, agentic tool loop, skills/guardrails, MCP, sandbox, and HITL — - self-hosted or Cloud. Bring your own LLM (any OpenAI-compatible engine). - Not a vector black box. Not classic RAG inject. + Put one OpenAI-compatible /v1 URL in front of your engine. Your client sends only the new message; + the gateway keeps session memory as markdown you can open, edit, and diff. No vector DB, no client rewrite.

--- +## Try it in three commands + +```bash +git clone https://github.com/Kortexio/ContextMemory.git && cd ContextMemory +docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build -d # CPU; GPU: docker-compose.vllm.yml +./scripts/aha-chat.sh # Windows: .\scripts\aha-chat.ps1 +``` + +`aha-chat` sends two requests in the same session. The second one carries **only** the new question: + +```text +==> Turn 1: "Remember this for later: our staging database host is postgres-staging-01." +==> Turn 2: "What is our staging database host?" (no history in the request body) +AHA OK — the client sent no history; the gateway remembered 'postgres-staging-01'. +``` + +The same check runs in CI against a real `llama-server` on every push ([`e2e-llamacpp`](.github/workflows/e2e-llamacpp.yml)). First start downloads a ~2 GB GGUF; pick another model with `LLAMACPP_HF_MODEL` ([Engines](docs/self-host.md#engines)). Admin UI: `http://localhost:5200`. + +--- + ## What is ContextMemory? [ContextMemory](https://github.com/Kortexio/ContextMemory) is the open-source **agentic memory gateway** behind [Kortexio](https://kortexio.io). @@ -66,7 +83,7 @@ Your client (OpenAI SDK / Cursor MCP / curl) **Honest boundaries:** this is a **gateway + server-side harness**, not a client agent framework (LangGraph/CrewAI) and not an agent OS (Letta). You keep your OpenAI client; the loop runs on the server. -**LLM engines:** ContextMemory does **not** ship or lock to one inference stack. Per tenant you pick any OpenAI-compatible `/v1` host — Ollama, vLLM, LM Studio, [ExLlamaSharp](https://github.com/Kortexio/ExLlamaSharp), OpenAI, Azure-compatible, LiteLLM, custom. Compose defaults to Ollama only for zero-friction local DX; swap in **Admin → Config → LLM**. +**LLM engines:** ContextMemory does **not** ship or lock to one inference stack. Per tenant you pick any OpenAI-compatible `/v1` host — Ollama, vLLM, LM Studio, [ExLlamaSharp](https://github.com/Kortexio/ExLlamaSharp), OpenAI, Azure-compatible, LiteLLM, custom. Compose ships llama.cpp and vLLM overrides; swap engines per app in **Admin → Config → LLM**. How we compare (Mem0 / Zep / Letta / **why we are not RAG**): [`docs/compare.md`](docs/compare.md). @@ -112,11 +129,11 @@ Full detail: [`docs/architecture-and-features.md`](docs/architecture-and-feature --- -## Quickstart (5 minutes) +## More ways to run -**Bring your own LLM.** The gateway talks OpenAI-compatible `/v1` (and Ollama native `/api/chat` when you need `num_ctx`). Compose/GHCR defaults point at Ollama on the host for DX — change anytime in **Admin → Config → LLM** or `PATCH /admin/apps/{id}/config`. +The gateway talks OpenAI-compatible `/v1` to the engine (and Ollama native `/api/chat` when you need `num_ctx`). Change the engine anytime in **Admin → Config → LLM** or `PATCH /admin/apps/{id}/config`. -### 1a. Start with any `/v1` engine (vLLM, LM Studio, ExLlamaSharp, OpenAI, …) +### Engine already running (llama-server, vLLM, LM Studio, …) — API image only ```bash docker run --rm -p 5100:8080 \ @@ -124,33 +141,15 @@ docker run --rm -p 5100:8080 \ -e ContextMemory__MasterKey=cm_master_dev_key_change_me \ -e ContextMemory__Apps__demo-dev__ApiKey=cm_live_dev_key_change_me \ -e ContextMemory__Apps__demo-dev__LlmBackend=openai-compatible \ - -e ContextMemory__Apps__demo-dev__LlmModel=my-model \ - -e ContextMemory__Apps__demo-dev__LlmEndpoint=http://host.docker.internal:8000 \ - -e ContextMemory__OpenAiEndpoint=http://host.docker.internal:8000 \ + -e ContextMemory__Apps__demo-dev__LlmModel=local-model \ + -e ContextMemory__Apps__demo-dev__LlmEndpoint=http://host.docker.internal:8080 \ --add-host=host.docker.internal:host-gateway \ ghcr.io/kortexio/contextmemory:latest ``` -Prefer a host-level default for all apps: set `ContextMemory__LlmEndpoint` (alias; falls back to `OllamaEndpoint` for older Compose files). - -### 1b. Or start with Ollama on the host (zero-friction DX) - -```bash -docker run --rm -p 5100:8080 \ - -v contextmemory-data:/app/data \ - -e ContextMemory__MasterKey=cm_master_dev_key_change_me \ - -e ContextMemory__Apps__demo-dev__ApiKey=cm_live_dev_key_change_me \ - -e ContextMemory__Apps__demo-dev__LlmModel=qwen3.5:9b \ - -e ContextMemory__LlmEndpoint=http://host.docker.internal:11434 \ - --add-host=host.docker.internal:host-gateway \ - ghcr.io/kortexio/contextmemory:latest -``` - -Full stack (API + Admin + MCP + sandbox): [`docs/self-host.md`](docs/self-host.md) / `docker-compose.yml`. Admin: `http://localhost:5200`. - -No Docker? Use **[Kortexio Cloud](https://kortexio.io)** (`cmk_live_…`) and set `CONTEXTMEMORY_BASE_URL` to the cloud API. +Host-level default for all apps: `ContextMemory__LlmEndpoint`. Ollama on the host works too (`ContextMemory__LlmEndpoint=http://host.docker.internal:11434`). Engine flags that matter (llama.cpp `--jinja`, vLLM tool parser): [Engines](docs/self-host.md#engines). Full stack (API + Admin + MCP + sandbox): [`docs/self-host.md`](docs/self-host.md). -### 2. Wire MCP into Cursor +### Cursor / Claude memory (MCP) ```bash git clone https://github.com/Kortexio/ContextMemory.git @@ -159,16 +158,16 @@ cd ContextMemory/mcp-server && npm install && node print-mcp-config.mjs Paste into **Cursor → Settings → MCP** (or `~/.cursor/mcp.json`). Same snippet works for Claude Desktop. Details: [`mcp-server/README.md`](mcp-server/README.md). -### 3. Aha (memory wedge) +Then, in two separate chats: | Chat | You say | Agent should | |---|---|---| | **A** | `Remember: staging DB is postgres-staging-01` | `memory_save` | | **B** (new) | `What is our staging DB?` | `memory_search` + answer | -CLI: `./scripts/aha-demo.sh` or `.\scripts\aha-demo.ps1` · storyboard (for GIF recording): [`docs/aha-demo.html`](docs/aha-demo.html) +Same flow without Cursor (wiki API, no LLM): `./scripts/aha-demo.sh` or `.\scripts\aha-demo.ps1`. -### Cloud vs self-host +### No GPU, no ops: Kortexio Cloud | | **[Kortexio Cloud](https://kortexio.io)** | **Self-host (this repo)** | |---|---|---| @@ -186,7 +185,7 @@ curl -X POST http://localhost:5100/v1/chat/completions \ -H "Content-Type: application/json" \ -H "X-App-Id: demo-dev" -H "X-User-Id: user-42" -H "X-Session-Id: sess-abc" \ -H "Authorization: Bearer cm_live_dev_key_change_me" \ - -d '{"model":"qwen3.5:9b","messages":[{"role":"user","content":"Hello"}]}' + -d '{"model":"local-model","messages":[{"role":"user","content":"Hello"}]}' ``` Thin header helpers (**not** full SDKs): [`@kortexio/contextmemory`](https://www.npmjs.com/package/@kortexio/contextmemory) · [`kortexio-contextmemory`](https://pypi.org/project/kortexio-contextmemory/) @@ -198,7 +197,6 @@ Thin header helpers (**not** full SDKs): [`@kortexio/contextmemory`](https://www | Doc | Topic | |---|---| | [docs/compare.md](docs/compare.md) | Why it exists · vs Mem0 / Zep / Letta · **why we are not RAG** | -| [docs/show-hn.md](docs/show-hn.md) | Suggested Show HN title + blurb | | [docs/architecture-and-features.md](docs/architecture-and-features.md) | Wiki, temporal facts, agentic, skills, **LLM backends** | | [docs/admin-ui.md](docs/admin-ui.md) | Admin UI map | | [docs/hitl.md](docs/hitl.md) | Human-in-the-loop | @@ -213,4 +211,4 @@ Website: [kortexio.io](https://kortexio.io) · Email: [hello@kortexio.io](mailto ## License -**AGPL-3.0** for this open-source core. Commercial / hosted offerings: [kortexio.io](https://kortexio.io). See [docs/license-and-support.md](docs/license-and-support.md). +**AGPL-3.0** for this open-source core — self-host it freely, including commercially. Need to embed it in a closed-source product without AGPL obligations? Use [Kortexio Cloud](https://kortexio.io) or a commercial license. See [docs/license-and-support.md](docs/license-and-support.md). diff --git a/docs/self-host.md b/docs/self-host.md index 534cfa7..b188c81 100644 --- a/docs/self-host.md +++ b/docs/self-host.md @@ -4,7 +4,27 @@ Run the open-source gateway yourself — the path for **on-prem or fully local** deployments. Unlike Cloud (where the dashboard wires up your LLM for you), here **you point the gateway at your own LLM backend** — endpoint and model — in config. Same OpenAI-compatible chat body and `choices[]` response as Cloud; you supply the `X-App-Id` and use a `cm_live_` key. -### Fastest: one-liner from GHCR (API) +### Fastest: gateway + bundled engine (Compose) + +```bash +git clone https://github.com/Kortexio/ContextMemory.git && cd ContextMemory + +# CPU, no GPU needed — llama.cpp pulls a small GGUF on first start +docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build + +# or NVIDIA GPU — vLLM +docker compose -f docker-compose.yml -f docker-compose.vllm.yml up --build +``` + +Then prove memory works (the second request carries only the new question): + +```bash +./scripts/aha-chat.sh # Windows: .\scripts\aha-chat.ps1 +``` + +Engine flags, model choice, and gotchas: [Engines](#engines). + +### One-liner from GHCR (API only) Public images (published on every push to `main`): @@ -67,7 +87,7 @@ Or the helper scripts: ### Build from source: Docker Compose -Builds and starts the **API** (`:5100`), **Admin** (`:5200`), **mcp-runtime** (stdio MCP host), and **sandbox-runtime** (shell/python/node) locally. Requires [Docker](https://docs.docker.com/get-docker/) and an LLM reachable from the containers (Compose DX default = Ollama on the host; point `LLM_ENDPOINT` / `OPENAI_ENDPOINT` at any `/v1` engine instead). +Builds and starts the **API** (`:5100`), **Admin** (`:5200`), **mcp-runtime** (stdio MCP host), and **sandbox-runtime** (shell/python/node) locally. Requires [Docker](https://docs.docker.com/get-docker/) and an LLM: add a bundled engine override (see [Fastest](#fastest-gateway--bundled-engine-compose)), point `LLM_ENDPOINT` at any `/v1` engine, or fall back to Ollama on the host. ```bash git clone https://github.com/Kortexio/ContextMemory.git @@ -96,13 +116,12 @@ Ops triage (Azure Monitor / GitHub): see [inbound-mcp-guide.md](inbound-mcp-guid For a Postgres-backed network overlay (shared Docker network, extra tenants), see `docker-compose.network.yml`. -**LLM on the host** +**LLM engine** ```bash -# Example: Ollama DX default -ollama pull qwen3.5:9b -# Compose: OLLAMA_ENDPOINT=http://host.docker.internal:11434 -# Or any /v1 engine: LLM_ENDPOINT=http://host.docker.internal:8000 + Admin backend openai-compatible +# Bundled: add -f docker-compose.llamacpp.yml (CPU) or -f docker-compose.vllm.yml (GPU) +# Engine on the host: LLM_ENDPOINT=http://host.docker.internal:8080 + Admin backend openai-compatible +# Ollama on the host (fallback when LLM_ENDPOINT is empty): ollama pull qwen3.5:9b ``` **Useful Compose env vars** (see [`.env.example`](../.env.example)): @@ -131,6 +150,45 @@ curl -X POST http://localhost:5100/v1/chat/completions \ -d '{"model":"qwen3.5:9b","messages":[{"role":"user","content":"Hello"}]}' ``` +## Engines + +The gateway speaks OpenAI-compatible `/v1/chat/completions` to the engine. Any server that implements it works; these are the ones we test and document. Set the app backend to `openai-compatible` (or `vllm`) and the endpoint to the engine base URL — `/v1` is appended automatically. + +| Engine | Compose override | Required flags | Notes | +|---|---|---|---| +| **llama.cpp** (`llama-server`) | `docker-compose.llamacpp.yml` | `--jinja` | Without `--jinja`, tool calls come back as plain text and agentic mode degrades. Set `--ctx-size` ≥ 8192 for agentic use. Model name in the request is ignored. | +| **vLLM** | `docker-compose.vllm.yml` | `--enable-auto-tool-choice --tool-call-parser ` | Parser must match the model: `hermes` (Qwen2.5 / Qwen3), `llama3_json` (Llama 3.x), `mistral` (Mistral). `--served-model-name` must equal the app `llmModel`. | +| **Ollama** | _(host, default fallback)_ | — | Ollama `/v1` ignores `num_ctx`; when you set `llmOptions.numCtx` the gateway switches to native `/api/chat`. | +| LM Studio, SGLang, TGI, LiteLLM, OpenAI, Azure-compatible | — | tool calling enabled on the server | Point `LLM_ENDPOINT` (host default) or the app `llmEndpoint` at the server. | + +**Choosing a model** + +| Use | Minimum that works | Recommended | +|---|---|---| +| Session memory only (agentic off — the default) | 1.5B instruct (CI runs `Qwen2.5-1.5B-Instruct Q4_K_M`) | 3B–8B instruct | +| Agentic mode (wiki_search, MCP, sandbox) | 7B–9B instruct with native tool calling | 14B+ or a hosted model | + +Small models need the hardening preset in [small-model-guide.md](small-model-guide.md). Found an engine/model combination that works (or does not)? Open an **Engine / model compatibility report** issue. + +**Changing the llama.cpp model** + +```bash +LLAMACPP_HF_MODEL=bartowski/Qwen2.5-7B-Instruct-GGUF:Q4_K_M \ + docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up +``` + +**Engine already running elsewhere** (host, another box, a GPU server): + +```bash +LLM_ENDPOINT=http://gpu-box.lan:8000 docker compose up --build +``` + +Then in **Admin → Config → LLM** set backend `openai-compatible` and the model name the engine expects. + +The end-to-end check in CI ([`e2e-llamacpp`](../.github/workflows/e2e-llamacpp.yml)) builds the gateway, starts `llama-server`, and runs `scripts/aha-chat.sh` on every push to `main` and nightly. + +## Run from source (dotnet run) + ### Prerequisites (dotnet run) - .NET 9 SDK diff --git a/docs/show-hn.md b/docs/show-hn.md index c01ebee..1fd9739 100644 --- a/docs/show-hn.md +++ b/docs/show-hn.md @@ -1,44 +1,86 @@ > Part of the ContextMemory docs. [Back to README](../README.md). -# Show HN — suggested blurb +# Show HN — launch kit -One product only. Link the GitHub repo. Keep Cloud / other Kortexio apps as a single closing line. +Audience: people who self-host LLMs (llama.cpp, vLLM) and want memory without adopting an agent framework or a vector DB. One product only; Cloud is one closing line. -## Title +## Title (pick one) +```text +Show HN: ContextMemory – editable markdown memory for your llama.cpp/vLLM server +Show HN: A self-hosted /v1 gateway that gives local LLMs persistent memory ``` -Show HN: ContextMemory – OpenAI-compatible gateway with wiki memory (not RAG) -``` + +Keep "not RAG" out of the title — say it in the body and the first comment, where it can be explained. ## Body -``` -Agents forget. Most “memory” products hide facts in vector stores you can’t open. +```text +I run models on llama.cpp and vLLM and kept rebuilding the same thing in every app: +history trimming, summaries, "remember this" facts, per-user isolation. + +ContextMemory is an open-source gateway that sits in front of any OpenAI-compatible +engine. Your client keeps calling /v1/chat/completions and sends only the new message. +The gateway attaches session memory, calls your engine, and returns a normal +chat.completions response (streaming included). -ContextMemory is an open-source agentic memory gateway: your app keeps calling a -normal OpenAI-compatible /v1/chat/completions URL. The gateway attaches a session -markdown wiki + history, can run a server-side tool loop (sandbox, MCP, wiki search), -and returns a standard chat.completions response. +Memory is markdown on disk (or Postgres): you can open it, edit it, diff it. +No embeddings, no vector DB. When you want more, the same URL can run a server-side +tool loop (wiki search, MCP servers, a code sandbox) with human confirmation before +destructive actions. -Bring your own LLM — Ollama, vLLM, LM Studio, ExLlamaSharp, OpenAI, Azure-compatible, -or any /v1 host. Compose defaults to Ollama only for local DX; swap the backend per -tenant in Admin. +Try it in three commands (CPU is enough, llama.cpp pulls a small GGUF): -Not classic RAG inject. Not a client agent framework. Memory you can open like a wiki. + git clone https://github.com/Kortexio/ContextMemory && cd ContextMemory + docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build -d + ./scripts/aha-chat.sh -Aha with the Cursor MCP wedge (two chats): - A) “Remember: staging DB is postgres-staging-01” → memory_save - B) new chat “What is our staging DB?” → memory_search +The script sends two requests in one session; the second carries only the question, +so the answer can only come from the gateway's memory. CI runs the same check +against a real llama-server on every push. -Repo: https://github.com/Kortexio/ContextMemory -Self-host / Cloud: see README. GIF storyboard: docs/aha-demo.html +.NET 9, AGPL-3.0. There is a hosted version (Kortexio Cloud) if you do not want to run it. + +https://github.com/Kortexio/ContextMemory ``` -## Do / don’t +## First comment (post immediately after submitting) + +```text +Author here. A few things people usually ask: + +- Why not RAG / embeddings? Session memory is budgeted markdown the model sees every + turn. Shared docs go into a "Global Wiki" searched on demand with Postgres FTS + inside the tool loop. We chose lexical search over markdown you can audit; the + trade-off is weaker fuzzy recall on paraphrases. +- Why .NET? It is what I ship fastest in and it gives us a single small container. + You never touch it from your app — it is just an HTTP URL. +- Small models: session memory works with 1.5B instruct models (that is what CI uses). + The agentic tool loop needs 7B+ with native tool calling; llama.cpp needs --jinja. +- Honest limits: beta, small team, APIs may still move. Engine compatibility + reports are the most useful issue you can open. +``` + +## Checklist before posting + +- [ ] Latest release published and newer than the last big change (release-please green) +- [ ] `e2e-llamacpp` badge green on `main` +- [ ] No bot issues in the tracker; 3–5 `good first issue` items open +- [ ] GIF or asciinema of `aha-chat.sh` at the top of the README +- [ ] Fresh clone on a clean machine: three commands work as written (Linux + Windows) +- [ ] 4–6 hours free after posting to answer comments + +## Timing and follow-up + +- Post Tuesday–Thursday, 14:00–16:00 UTC. +- Cross-post to r/LocalLLaMA 24–48 h later with the llama.cpp angle (engine flags, model table). +- Measure for 14 days: unique visitors and referrers (GitHub Insights → Traffic), stars, issues from outside contributors. Clones are inflated by CI and are not a signal. + +## Do / don't -| Do | Don’t | +| Do | Don't | |---|---| -| Lead with wiki memory + `/v1` drop-in | List CompanyBrain / TriageHub / Fincheck in the title | -| Say BYO LLM explicitly | Imply the product is “an Ollama app” | -| Point to Compose + Admin LLM picker | Promise a public live chat sandbox unless it exists | -| One soft line on Kortexio Cloud at the end | Dump six repos into the first paragraph | +| Lead with "your engine + memory you can open" | List other Kortexio products | +| Show the three commands in the body | Promise features that are not on `main` | +| Answer "why not RAG" with the trade-off | Argue that RAG is wrong | +| One line on Cloud at the end | Link pricing |