diff --git a/.env.example b/.env.example
index 2b5e60a..d6b89bf 100644
--- a/.env.example
+++ b/.env.example
@@ -5,18 +5,38 @@ ADMIN_PORT=5200
PUBLIC_API_URL=http://localhost:5100
ADMIN_PUBLIC_URL=http://localhost:5200
-# Preferred host-level LLM base URL (any OpenAI-compatible or Ollama-compatible engine).
-# When empty, Compose falls back to OLLAMA_ENDPOINT below.
-# LLM_ENDPOINT=http://host.docker.internal:8000
+# --- LLM engine -------------------------------------------------------------
+# Bundled engines (no host LLM needed):
+# llama.cpp: docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build
+# vLLM: docker compose -f docker-compose.yml -f docker-compose.vllm.yml up --build
+# The overrides set LLM_ENDPOINT for you; the values below only matter for the plain compose file.
-# DX default: Ollama on the host (used when LLM_ENDPOINT is unset)
+# Model name sent to the engine. llama-server ignores it; vLLM uses it as --served-model-name;
+# Ollama needs a pulled tag (e.g. qwen3.5:9b). Overrides default to local-model, plain compose to qwen3.5:9b.
+# DEFAULT_LLM_MODEL=local-model
+
+# llama.cpp override
+# LLAMACPP_HF_MODEL=bartowski/Qwen2.5-3B-Instruct-GGUF:Q4_K_M
+# LLAMACPP_CTX_SIZE=8192
+# LLAMACPP_IMAGE=ghcr.io/ggml-org/llama.cpp:server
+
+# vLLM override
+# VLLM_MODEL=Qwen/Qwen2.5-7B-Instruct
+# VLLM_MAX_MODEL_LEN=16384
+# VLLM_TOOL_CALL_PARSER=hermes
+# HF_TOKEN=
+
+# Engine already running on the host (any OpenAI-compatible /v1 server):
+# LLM_ENDPOINT=http://host.docker.internal:8080
+
+# Ollama on the host (used only when LLM_ENDPOINT is empty):
OLLAMA_ENDPOINT=http://host.docker.internal:11434
-DEFAULT_LLM_MODEL=qwen3.5:9b
-# Optional: OpenAI / vLLM / ExLlamaSharp / any OpenAI-compatible host defaults
-# OPENAI_ENDPOINT=http://host.docker.internal:8000
+# Host defaults for apps with llmBackend openai / openai-compatible / vllm and no per-app endpoint
+# OPENAI_ENDPOINT=https://api.openai.com
# OPENAI_API_KEY=
+# --- Keys (change before any real use) -------------------------------------
MASTER_KEY=cm_master_dev_key_change_me
DEMO_APP_API_KEY=cm_live_dev_key_change_me
diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml
new file mode 100644
index 0000000..025de28
--- /dev/null
+++ b/.github/ISSUE_TEMPLATE/bug_report.yml
@@ -0,0 +1,59 @@
+name: Bug report
+description: Something in the gateway, Admin, or MCP server does not work as documented.
+labels: [bug]
+body:
+ - type: textarea
+ id: what
+ attributes:
+ label: What happened
+ description: What you did, what you expected, and what you got instead. Do not paste API keys or private conversation content.
+ validations:
+ required: true
+ - type: textarea
+ id: repro
+ attributes:
+ label: Steps to reproduce
+ placeholder: |
+ 1. docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build
+ 2. ./scripts/aha-chat.sh
+ 3. ...
+ validations:
+ required: true
+ - type: dropdown
+ id: engine
+ attributes:
+ label: LLM engine
+ options:
+ - llama.cpp
+ - vLLM
+ - Ollama
+ - LM Studio
+ - OpenAI / Azure-compatible
+ - Other OpenAI-compatible
+ validations:
+ required: true
+ - type: input
+ id: model
+ attributes:
+ label: Model
+ placeholder: e.g. Qwen2.5-7B-Instruct Q4_K_M
+ - type: dropdown
+ id: deploy
+ attributes:
+ label: Deployment
+ options:
+ - Docker Compose (this repo)
+ - GHCR image
+ - dotnet run
+ - Kortexio Cloud
+ - type: input
+ id: version
+ attributes:
+ label: Version / commit
+ placeholder: v0.2.0-beta or commit SHA
+ - type: textarea
+ id: logs
+ attributes:
+ label: Relevant logs
+ description: Gateway logs around the failure (redact keys and message content).
+ render: text
diff --git a/.github/ISSUE_TEMPLATE/config.yml b/.github/ISSUE_TEMPLATE/config.yml
new file mode 100644
index 0000000..fb890c6
--- /dev/null
+++ b/.github/ISSUE_TEMPLATE/config.yml
@@ -0,0 +1,5 @@
+blank_issues_enabled: true
+contact_links:
+ - name: Kortexio Cloud / commercial license
+ url: https://kortexio.io
+ about: Hosted gateway, commercial terms, or enterprise deployment questions.
diff --git a/.github/ISSUE_TEMPLATE/engine_report.yml b/.github/ISSUE_TEMPLATE/engine_report.yml
new file mode 100644
index 0000000..e10c2bb
--- /dev/null
+++ b/.github/ISSUE_TEMPLATE/engine_report.yml
@@ -0,0 +1,47 @@
+name: Engine / model compatibility report
+description: Tell us how a specific engine + model behaves behind ContextMemory (works, partially works, or fails).
+labels: [engine-compat]
+body:
+ - type: dropdown
+ id: engine
+ attributes:
+ label: Engine
+ options:
+ - llama.cpp
+ - vLLM
+ - Ollama
+ - LM Studio
+ - SGLang
+ - TGI
+ - Other OpenAI-compatible
+ validations:
+ required: true
+ - type: input
+ id: engine_flags
+ attributes:
+ label: Engine version and flags
+ placeholder: "llama-server b6xxx --jinja --ctx-size 8192"
+ validations:
+ required: true
+ - type: input
+ id: model
+ attributes:
+ label: Model and quantization
+ placeholder: Qwen2.5-7B-Instruct Q4_K_M
+ validations:
+ required: true
+ - type: checkboxes
+ id: results
+ attributes:
+ label: What works
+ options:
+ - label: scripts/aha-chat.sh passes (session memory)
+ - label: Streaming responses
+ - label: Agentic mode with wiki_search
+ - label: Agentic mode with MCP tools
+ - label: Sandbox tools (shell / python / node)
+ - type: textarea
+ id: notes
+ attributes:
+ label: Notes
+ description: Failures, loops, malformed tool calls, context overflows — anything we should add to the engine docs.
diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md
new file mode 100644
index 0000000..172d3c6
--- /dev/null
+++ b/.github/pull_request_template.md
@@ -0,0 +1,12 @@
+## What and why
+
+
+
+## How it was tested
+
+- [ ] `dotnet test tests/ContextMemory.Api.Tests/ContextMemory.Api.Tests.csproj`
+- [ ] `./scripts/aha-chat.sh` against a local engine (if the chat path changed)
+
+## Notes for reviewers
+
+
diff --git a/.github/workflows/content-cadence.yml b/.github/workflows/content-cadence.yml
index 542229a..f943979 100644
--- a/.github/workflows/content-cadence.yml
+++ b/.github/workflows/content-cadence.yml
@@ -1,7 +1,7 @@
name: content-cadence
-# Suggests a short social hook (manual approve via issue).
-# Does not auto-publish to social — safer for brand voice.
+# Suggests a short social hook in the run summary (Actions tab).
+# Does not auto-publish to social and does not open issues — the public tracker is for users.
on:
schedule:
@@ -9,16 +9,14 @@ on:
workflow_dispatch:
permissions:
- issues: write
contents: read
jobs:
suggest:
+ if: github.repository == 'Kortexio/ContextMemory'
runs-on: ubuntu-latest
steps:
- - name: Open hook suggestion issue
- env:
- GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ - name: Write hook suggestion to the run summary
run: |
HOOKS=(
"Your coding agent forgets staging DB names. Fix that in five minutes."
@@ -31,16 +29,8 @@ jobs:
"Admin Playground: watch tool steps and HITL without a client."
)
HOOK="${HOOKS[$RANDOM % ${#HOOKS[@]}]}"
- BODY=$(printf '%s\n\n%s\n\n%s\n' \
+ printf '%s\n\n%s\n\n%s\n' \
"## Suggested post" \
"$HOOK" \
- "Copy to LinkedIn/dev.to after editing. Do **not** call this RAG. CTA: https://github.com/Kortexio/ContextMemory")
- gh issue create \
- --repo "$GITHUB_REPOSITORY" \
- --title "Content cadence: social hook suggestion" \
- --label "content-hook" \
- --body "$BODY" || \
- gh issue create \
- --repo "$GITHUB_REPOSITORY" \
- --title "Content cadence: social hook suggestion" \
- --body "$BODY"
+ "Copy to LinkedIn/dev.to after editing. Do **not** call this RAG. CTA: https://github.com/Kortexio/ContextMemory" \
+ >> "$GITHUB_STEP_SUMMARY"
diff --git a/.github/workflows/e2e-llamacpp.yml b/.github/workflows/e2e-llamacpp.yml
new file mode 100644
index 0000000..21d4002
--- /dev/null
+++ b/.github/workflows/e2e-llamacpp.yml
@@ -0,0 +1,105 @@
+name: e2e-llamacpp
+
+# End-to-end memory check against a real llama.cpp server (CPU, small GGUF):
+# gateway image built from this commit + llama-server + scripts/aha-chat.sh.
+
+on:
+ push:
+ branches: [main]
+ pull_request:
+ paths:
+ - "src/ContextMemory.Api/**"
+ - "src/ContextMemory.Adapters/**"
+ - "src/ContextMemory.Core/**"
+ - "src/ContextMemory.Infrastructure/**"
+ - "Dockerfile"
+ - "scripts/aha-chat.sh"
+ - ".github/workflows/e2e-llamacpp.yml"
+ schedule:
+ - cron: "0 4 * * *"
+ workflow_dispatch:
+ inputs:
+ hf_model:
+ description: "GGUF to test (:)"
+ required: false
+ default: "bartowski/Qwen2.5-1.5B-Instruct-GGUF:Q4_K_M"
+
+permissions:
+ contents: read
+
+concurrency:
+ group: e2e-llamacpp-${{ github.ref }}
+ cancel-in-progress: true
+
+env:
+ HF_MODEL: ${{ inputs.hf_model || 'bartowski/Qwen2.5-1.5B-Instruct-GGUF:Q4_K_M' }}
+ MODELS_DIR: ${{ github.workspace }}/.llama-models
+
+jobs:
+ aha:
+ runs-on: ubuntu-latest
+ timeout-minutes: 40
+ steps:
+ - uses: actions/checkout@v4
+
+ - name: Cache GGUF
+ uses: actions/cache@v4
+ with:
+ path: .llama-models
+ key: llama-gguf-${{ env.HF_MODEL }}
+
+ - name: Start llama-server
+ run: |
+ mkdir -p "$MODELS_DIR" && chmod 777 "$MODELS_DIR"
+ docker network create cm-e2e
+ docker run -d --name llama-server --network cm-e2e -p 8081:8080 \
+ -v "$MODELS_DIR:/models" -e LLAMA_CACHE=/models \
+ ghcr.io/ggml-org/llama.cpp:server \
+ --host 0.0.0.0 --port 8080 --jinja --ctx-size 8192 -hf "$HF_MODEL"
+
+ - name: Build gateway image
+ run: docker build -t contextmemory-e2e -f Dockerfile .
+
+ - name: Wait for llama-server
+ run: |
+ for i in $(seq 1 120); do
+ if curl -fsS http://localhost:8081/health >/dev/null 2>&1; then echo "llama-server ready"; exit 0; fi
+ sleep 5
+ done
+ docker logs llama-server | tail -n 50
+ exit 1
+
+ - name: Start gateway
+ run: |
+ docker run -d --name cm-api --network cm-e2e -p 5100:8080 \
+ -e ContextMemory__PersistenceProvider=File \
+ -e ContextMemory__DataPath=/app/data \
+ -e ContextMemory__MasterKey=cm_master_e2e \
+ -e ContextMemory__LlmEndpoint=http://llama-server:8080 \
+ -e ContextMemory__DefaultLlmModel=local-model \
+ -e ContextMemory__Apps__demo-dev__ApiKey=cm_live_e2e \
+ -e ContextMemory__Apps__demo-dev__LlmBackend=openai-compatible \
+ -e ContextMemory__Apps__demo-dev__LlmEndpoint=http://llama-server:8080 \
+ -e ContextMemory__Apps__demo-dev__LlmModel=local-model \
+ contextmemory-e2e
+ for i in $(seq 1 60); do
+ if curl -fsS http://localhost:5100/health >/dev/null 2>&1; then echo "gateway ready"; exit 0; fi
+ sleep 3
+ done
+ docker logs cm-api | tail -n 80
+ exit 1
+
+ - name: Run chat aha
+ env:
+ CONTEXTMEMORY_BASE_URL: http://localhost:5100
+ CONTEXTMEMORY_API_KEY: cm_live_e2e
+ CONTEXTMEMORY_APP_ID: demo-dev
+ CONTEXTMEMORY_MODEL: local-model
+ CONTEXTMEMORY_TIMEOUT: "600"
+ run: bash scripts/aha-chat.sh
+
+ - name: Container logs
+ if: failure()
+ run: |
+ echo "::group::cm-api"; docker logs cm-api 2>&1 | tail -n 200; echo "::endgroup::"
+ echo "::group::llama-server"; docker logs llama-server 2>&1 | tail -n 100; echo "::endgroup::"
diff --git a/.github/workflows/pr-title.yml b/.github/workflows/pr-title.yml
new file mode 100644
index 0000000..175bfb0
--- /dev/null
+++ b/.github/workflows/pr-title.yml
@@ -0,0 +1,32 @@
+name: pr-title
+
+# release-please only counts Conventional Commits (feat:, fix:, docs:, chore:, ...).
+# PRs are squash-merged with the PR title, so the title is what lands on main.
+
+on:
+ pull_request_target:
+ types: [opened, edited, synchronize, reopened]
+
+permissions:
+ pull-requests: read
+
+jobs:
+ conventional:
+ runs-on: ubuntu-latest
+ steps:
+ - uses: amannn/action-semantic-pull-request@v5
+ env:
+ GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ with:
+ types: |
+ feat
+ fix
+ perf
+ refactor
+ docs
+ test
+ build
+ ci
+ chore
+ revert
+ requireScope: false
diff --git a/.github/workflows/release-please.yml b/.github/workflows/release-please.yml
index 828cc63..e06fd13 100644
--- a/.github/workflows/release-please.yml
+++ b/.github/workflows/release-please.yml
@@ -10,6 +10,7 @@ permissions:
jobs:
release-please:
+ if: github.repository == 'Kortexio/ContextMemory'
runs-on: ubuntu-latest
outputs:
release_created: ${{ steps.release.outputs.release_created }}
@@ -21,3 +22,5 @@ jobs:
id: release
with:
release-type: simple
+ # GITHUB_TOKEN cannot open PRs unless the org allows it (Settings > Actions > General).
+ token: ${{ secrets.RELEASE_PLEASE_TOKEN || github.token }}
diff --git a/.github/workflows/sync-mirror.yml b/.github/workflows/sync-mirror.yml
index e9dd843..39dedcf 100644
--- a/.github/workflows/sync-mirror.yml
+++ b/.github/workflows/sync-mirror.yml
@@ -1,5 +1,8 @@
name: Sync mirror
+# One-way backup: Kortexio/ContextMemory (source of truth) -> vitorcastro78/ContextMemory.
+# PEER_SYNC_TOKEN must be a PAT with Contents: write on the backup repository.
+
on:
push:
workflow_dispatch:
@@ -9,26 +12,12 @@ permissions:
jobs:
sync:
+ if: github.repository == 'Kortexio/ContextMemory'
runs-on: ubuntu-latest
+ env:
+ PEER_REPO: vitorcastro78/ContextMemory
steps:
- - name: Resolve peer repository
- id: peer
- run: |
- case "${{ github.repository }}" in
- Kortexio/ContextMemory)
- echo "peer=vitorcastro78/ContextMemory" >> "$GITHUB_OUTPUT"
- ;;
- vitorcastro78/ContextMemory)
- echo "peer=Kortexio/ContextMemory" >> "$GITHUB_OUTPUT"
- ;;
- *)
- echo "Unknown repository ${{ github.repository }}; skipping."
- echo "skip=true" >> "$GITHUB_OUTPUT"
- ;;
- esac
-
- name: Checkout
- if: steps.peer.outputs.skip != 'true'
uses: actions/checkout@v4
with:
fetch-depth: 0
@@ -36,11 +25,9 @@ jobs:
# any PAT embedded in a remote URL and causes 403 as the source org/bot.
persist-credentials: false
- - name: Mirror to peer
- if: steps.peer.outputs.skip != 'true'
+ - name: Mirror to backup
env:
PEER_TOKEN: ${{ secrets.PEER_SYNC_TOKEN }}
- PEER_REPO: ${{ steps.peer.outputs.peer }}
run: |
set -euo pipefail
if [ -z "${PEER_TOKEN:-}" ]; then
@@ -48,17 +35,19 @@ jobs:
exit 0
fi
- # Show which account the secret authenticates as (no token leakage).
- TOKEN_USER="$(curl -sS -H "Authorization: Bearer ${PEER_TOKEN}" \
+ STATUS="$(curl -sS -o /dev/null -w '%{http_code}' \
+ -H "Authorization: Bearer ${PEER_TOKEN}" \
-H "Accept: application/vnd.github+json" \
- https://api.github.com/user | jq -r '.login // empty')"
- echo "PEER_SYNC_TOKEN identity: ${TOKEN_USER:-unknown-or-not-a-user-token}"
+ "https://api.github.com/repos/${PEER_REPO}")"
+ if [ "$STATUS" != "200" ]; then
+ echo "::error::PEER_SYNC_TOKEN cannot access ${PEER_REPO} (HTTP ${STATUS}). Rotate the secret: fine-grained PAT, Contents: write on ${PEER_REPO}."
+ exit 1
+ fi
# Strip any leftover Authorization headers from checkout / runner config.
git config --local --unset-all "http.https://github.com/.extraheader" || true
git config --local --unset-all http.extraheader || true
- # Prefer Authorization header with PEER_TOKEN over URL-embedded creds.
BASIC="$(printf 'x-access-token:%s' "${PEER_TOKEN}" | base64 -w 0)"
git config --local "http.https://github.com/.extraheader" "AUTHORIZATION: basic ${BASIC}"
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index b10d60a..912bd73 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -39,5 +39,10 @@ Public contracts are in `src/ContextMemory.Core/Contracts/` with XML summaries.
## Pull requests
1. Keep changes focused; match existing naming and DI patterns.
-2. Run the full test suite before opening a PR.
+2. Run the full test suite before opening a PR. If you touched the chat path, also run `./scripts/aha-chat.sh` against a local engine (e.g. `docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build`).
3. Do not commit secrets, `data/`, or local `appsettings.Development.json`.
+4. Use [Conventional Commits](https://www.conventionalcommits.org/) for the PR title and for commits pushed to `main` (`feat: …`, `fix: …`, `docs: …`, `chore: …`). Releases and the changelog are generated by release-please from these prefixes; other messages are ignored.
+
+## Engine compatibility reports
+
+Tried ContextMemory with an engine/model we do not document yet? Open an issue with the **Engine / model compatibility report** template — those reports feed the table in [docs/self-host.md](docs/self-host.md#engines).
diff --git a/README.md b/README.md
index 38af2bd..b3eed60 100644
--- a/README.md
+++ b/README.md
@@ -5,15 +5,13 @@
- Get Cloud key
+ Try it (3 commands)
·
- Self-host
+ Engines
·
Docs
·
- Aha storyboard (GIF)
- ·
- Show HN
+ vs Mem0 / Zep / Letta
@@ -22,21 +20,40 @@
-
+
- Your agent forgets. Fix that with memory you can open like a wiki.
+ Self-hosted memory gateway for your llama.cpp / vLLM server.
- One OpenAI-compatible /v1 URL: wiki memory, agentic tool loop, skills/guardrails, MCP, sandbox, and HITL —
- self-hosted or Cloud. Bring your own LLM (any OpenAI-compatible engine).
- Not a vector black box. Not classic RAG inject.
+ Put one OpenAI-compatible /v1 URL in front of your engine. Your client sends only the new message;
+ the gateway keeps session memory as markdown you can open, edit, and diff. No vector DB, no client rewrite.
---
+## Try it in three commands
+
+```bash
+git clone https://github.com/Kortexio/ContextMemory.git && cd ContextMemory
+docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build -d # CPU; GPU: docker-compose.vllm.yml
+./scripts/aha-chat.sh # Windows: .\scripts\aha-chat.ps1
+```
+
+`aha-chat` sends two requests in the same session. The second one carries **only** the new question:
+
+```text
+==> Turn 1: "Remember this for later: our staging database host is postgres-staging-01."
+==> Turn 2: "What is our staging database host?" (no history in the request body)
+AHA OK — the client sent no history; the gateway remembered 'postgres-staging-01'.
+```
+
+The same check runs in CI against a real `llama-server` on every push ([`e2e-llamacpp`](.github/workflows/e2e-llamacpp.yml)). First start downloads a ~2 GB GGUF; pick another model with `LLAMACPP_HF_MODEL` ([Engines](docs/self-host.md#engines)). Admin UI: `http://localhost:5200`.
+
+---
+
## What is ContextMemory?
[ContextMemory](https://github.com/Kortexio/ContextMemory) is the open-source **agentic memory gateway** behind [Kortexio](https://kortexio.io).
@@ -66,7 +83,7 @@ Your client (OpenAI SDK / Cursor MCP / curl)
**Honest boundaries:** this is a **gateway + server-side harness**, not a client agent framework (LangGraph/CrewAI) and not an agent OS (Letta). You keep your OpenAI client; the loop runs on the server.
-**LLM engines:** ContextMemory does **not** ship or lock to one inference stack. Per tenant you pick any OpenAI-compatible `/v1` host — Ollama, vLLM, LM Studio, [ExLlamaSharp](https://github.com/Kortexio/ExLlamaSharp), OpenAI, Azure-compatible, LiteLLM, custom. Compose defaults to Ollama only for zero-friction local DX; swap in **Admin → Config → LLM**.
+**LLM engines:** ContextMemory does **not** ship or lock to one inference stack. Per tenant you pick any OpenAI-compatible `/v1` host — Ollama, vLLM, LM Studio, [ExLlamaSharp](https://github.com/Kortexio/ExLlamaSharp), OpenAI, Azure-compatible, LiteLLM, custom. Compose ships llama.cpp and vLLM overrides; swap engines per app in **Admin → Config → LLM**.
How we compare (Mem0 / Zep / Letta / **why we are not RAG**): [`docs/compare.md`](docs/compare.md).
@@ -112,11 +129,11 @@ Full detail: [`docs/architecture-and-features.md`](docs/architecture-and-feature
---
-## Quickstart (5 minutes)
+## More ways to run
-**Bring your own LLM.** The gateway talks OpenAI-compatible `/v1` (and Ollama native `/api/chat` when you need `num_ctx`). Compose/GHCR defaults point at Ollama on the host for DX — change anytime in **Admin → Config → LLM** or `PATCH /admin/apps/{id}/config`.
+The gateway talks OpenAI-compatible `/v1` to the engine (and Ollama native `/api/chat` when you need `num_ctx`). Change the engine anytime in **Admin → Config → LLM** or `PATCH /admin/apps/{id}/config`.
-### 1a. Start with any `/v1` engine (vLLM, LM Studio, ExLlamaSharp, OpenAI, …)
+### Engine already running (llama-server, vLLM, LM Studio, …) — API image only
```bash
docker run --rm -p 5100:8080 \
@@ -124,33 +141,15 @@ docker run --rm -p 5100:8080 \
-e ContextMemory__MasterKey=cm_master_dev_key_change_me \
-e ContextMemory__Apps__demo-dev__ApiKey=cm_live_dev_key_change_me \
-e ContextMemory__Apps__demo-dev__LlmBackend=openai-compatible \
- -e ContextMemory__Apps__demo-dev__LlmModel=my-model \
- -e ContextMemory__Apps__demo-dev__LlmEndpoint=http://host.docker.internal:8000 \
- -e ContextMemory__OpenAiEndpoint=http://host.docker.internal:8000 \
+ -e ContextMemory__Apps__demo-dev__LlmModel=local-model \
+ -e ContextMemory__Apps__demo-dev__LlmEndpoint=http://host.docker.internal:8080 \
--add-host=host.docker.internal:host-gateway \
ghcr.io/kortexio/contextmemory:latest
```
-Prefer a host-level default for all apps: set `ContextMemory__LlmEndpoint` (alias; falls back to `OllamaEndpoint` for older Compose files).
-
-### 1b. Or start with Ollama on the host (zero-friction DX)
-
-```bash
-docker run --rm -p 5100:8080 \
- -v contextmemory-data:/app/data \
- -e ContextMemory__MasterKey=cm_master_dev_key_change_me \
- -e ContextMemory__Apps__demo-dev__ApiKey=cm_live_dev_key_change_me \
- -e ContextMemory__Apps__demo-dev__LlmModel=qwen3.5:9b \
- -e ContextMemory__LlmEndpoint=http://host.docker.internal:11434 \
- --add-host=host.docker.internal:host-gateway \
- ghcr.io/kortexio/contextmemory:latest
-```
-
-Full stack (API + Admin + MCP + sandbox): [`docs/self-host.md`](docs/self-host.md) / `docker-compose.yml`. Admin: `http://localhost:5200`.
-
-No Docker? Use **[Kortexio Cloud](https://kortexio.io)** (`cmk_live_…`) and set `CONTEXTMEMORY_BASE_URL` to the cloud API.
+Host-level default for all apps: `ContextMemory__LlmEndpoint`. Ollama on the host works too (`ContextMemory__LlmEndpoint=http://host.docker.internal:11434`). Engine flags that matter (llama.cpp `--jinja`, vLLM tool parser): [Engines](docs/self-host.md#engines). Full stack (API + Admin + MCP + sandbox): [`docs/self-host.md`](docs/self-host.md).
-### 2. Wire MCP into Cursor
+### Cursor / Claude memory (MCP)
```bash
git clone https://github.com/Kortexio/ContextMemory.git
@@ -159,16 +158,16 @@ cd ContextMemory/mcp-server && npm install && node print-mcp-config.mjs
Paste into **Cursor → Settings → MCP** (or `~/.cursor/mcp.json`). Same snippet works for Claude Desktop. Details: [`mcp-server/README.md`](mcp-server/README.md).
-### 3. Aha (memory wedge)
+Then, in two separate chats:
| Chat | You say | Agent should |
|---|---|---|
| **A** | `Remember: staging DB is postgres-staging-01` | `memory_save` |
| **B** (new) | `What is our staging DB?` | `memory_search` + answer |
-CLI: `./scripts/aha-demo.sh` or `.\scripts\aha-demo.ps1` · storyboard (for GIF recording): [`docs/aha-demo.html`](docs/aha-demo.html)
+Same flow without Cursor (wiki API, no LLM): `./scripts/aha-demo.sh` or `.\scripts\aha-demo.ps1`.
-### Cloud vs self-host
+### No GPU, no ops: Kortexio Cloud
| | **[Kortexio Cloud](https://kortexio.io)** | **Self-host (this repo)** |
|---|---|---|
@@ -186,7 +185,7 @@ curl -X POST http://localhost:5100/v1/chat/completions \
-H "Content-Type: application/json" \
-H "X-App-Id: demo-dev" -H "X-User-Id: user-42" -H "X-Session-Id: sess-abc" \
-H "Authorization: Bearer cm_live_dev_key_change_me" \
- -d '{"model":"qwen3.5:9b","messages":[{"role":"user","content":"Hello"}]}'
+ -d '{"model":"local-model","messages":[{"role":"user","content":"Hello"}]}'
```
Thin header helpers (**not** full SDKs): [`@kortexio/contextmemory`](https://www.npmjs.com/package/@kortexio/contextmemory) · [`kortexio-contextmemory`](https://pypi.org/project/kortexio-contextmemory/)
@@ -198,7 +197,6 @@ Thin header helpers (**not** full SDKs): [`@kortexio/contextmemory`](https://www
| Doc | Topic |
|---|---|
| [docs/compare.md](docs/compare.md) | Why it exists · vs Mem0 / Zep / Letta · **why we are not RAG** |
-| [docs/show-hn.md](docs/show-hn.md) | Suggested Show HN title + blurb |
| [docs/architecture-and-features.md](docs/architecture-and-features.md) | Wiki, temporal facts, agentic, skills, **LLM backends** |
| [docs/admin-ui.md](docs/admin-ui.md) | Admin UI map |
| [docs/hitl.md](docs/hitl.md) | Human-in-the-loop |
@@ -213,4 +211,4 @@ Website: [kortexio.io](https://kortexio.io) · Email: [hello@kortexio.io](mailto
## License
-**AGPL-3.0** for this open-source core. Commercial / hosted offerings: [kortexio.io](https://kortexio.io). See [docs/license-and-support.md](docs/license-and-support.md).
+**AGPL-3.0** for this open-source core — self-host it freely, including commercially. Need to embed it in a closed-source product without AGPL obligations? Use [Kortexio Cloud](https://kortexio.io) or a commercial license. See [docs/license-and-support.md](docs/license-and-support.md).
diff --git a/docker-compose.llamacpp.yml b/docker-compose.llamacpp.yml
new file mode 100644
index 0000000..0ef7af4
--- /dev/null
+++ b/docker-compose.llamacpp.yml
@@ -0,0 +1,50 @@
+# Override: run llama.cpp (llama-server) next to the gateway — CPU, no host LLM required.
+# Usage: docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build
+#
+# The model is pulled from Hugging Face on first start and cached in the llama_models volume.
+# Pick another GGUF with LLAMACPP_HF_MODEL=: (see docs/self-host.md#engines).
+# GPU build: set LLAMACPP_IMAGE=ghcr.io/ggml-org/llama.cpp:server-cuda and add a device reservation.
+
+services:
+ llama-server:
+ image: ${LLAMACPP_IMAGE:-ghcr.io/ggml-org/llama.cpp:server}
+ container_name: contextmemory-llama-server
+ restart: unless-stopped
+ # --jinja is required for OpenAI-style tool calling (agentic mode).
+ command:
+ - --host
+ - 0.0.0.0
+ - --port
+ - "8080"
+ - --jinja
+ - --ctx-size
+ - "${LLAMACPP_CTX_SIZE:-8192}"
+ - -hf
+ - ${LLAMACPP_HF_MODEL:-bartowski/Qwen2.5-3B-Instruct-GGUF:Q4_K_M}
+ environment:
+ LLAMA_CACHE: /models
+ HF_TOKEN: ${HF_TOKEN:-}
+ volumes:
+ - llama_models:/models
+ networks:
+ - cm-net
+ healthcheck:
+ test: ["CMD", "curl", "-fsS", "http://127.0.0.1:8080/health"]
+ interval: 15s
+ timeout: 5s
+ retries: 40
+ start_period: 60s
+
+ api:
+ environment:
+ ContextMemory__LlmEndpoint: http://llama-server:8080
+ ContextMemory__DefaultLlmModel: ${DEFAULT_LLM_MODEL:-local-model}
+ ContextMemory__Apps__demo-dev__LlmBackend: openai-compatible
+ ContextMemory__Apps__demo-dev__LlmEndpoint: http://llama-server:8080
+ ContextMemory__Apps__demo-dev__LlmModel: ${DEFAULT_LLM_MODEL:-local-model}
+ depends_on:
+ llama-server:
+ condition: service_healthy
+
+volumes:
+ llama_models:
diff --git a/docker-compose.vllm.yml b/docker-compose.vllm.yml
new file mode 100644
index 0000000..2cafa39
--- /dev/null
+++ b/docker-compose.vllm.yml
@@ -0,0 +1,56 @@
+# Override: run vLLM next to the gateway — requires an NVIDIA GPU + nvidia-container-toolkit.
+# Usage: docker compose -f docker-compose.yml -f docker-compose.vllm.yml up --build
+#
+# Weights are downloaded from Hugging Face on first start and cached in the vllm_cache volume.
+# Gated models need HF_TOKEN. Tool calling needs --enable-auto-tool-choice + a parser that
+# matches the model family (hermes for Qwen2.5/Qwen3, llama3_json for Llama 3.x, mistral for Mistral).
+
+services:
+ vllm:
+ image: ${VLLM_IMAGE:-vllm/vllm-openai:latest}
+ container_name: contextmemory-vllm
+ restart: unless-stopped
+ ipc: host
+ command:
+ - --model
+ - ${VLLM_MODEL:-Qwen/Qwen2.5-7B-Instruct}
+ - --served-model-name
+ - ${DEFAULT_LLM_MODEL:-local-model}
+ - --max-model-len
+ - "${VLLM_MAX_MODEL_LEN:-16384}"
+ - --enable-auto-tool-choice
+ - --tool-call-parser
+ - ${VLLM_TOOL_CALL_PARSER:-hermes}
+ environment:
+ HF_TOKEN: ${HF_TOKEN:-}
+ volumes:
+ - vllm_cache:/root/.cache/huggingface
+ deploy:
+ resources:
+ reservations:
+ devices:
+ - driver: nvidia
+ count: all
+ capabilities: [gpu]
+ networks:
+ - cm-net
+ healthcheck:
+ test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health')"]
+ interval: 15s
+ timeout: 5s
+ retries: 60
+ start_period: 120s
+
+ api:
+ environment:
+ ContextMemory__LlmEndpoint: http://vllm:8000
+ ContextMemory__DefaultLlmModel: ${DEFAULT_LLM_MODEL:-local-model}
+ ContextMemory__Apps__demo-dev__LlmBackend: vllm
+ ContextMemory__Apps__demo-dev__LlmEndpoint: http://vllm:8000
+ ContextMemory__Apps__demo-dev__LlmModel: ${DEFAULT_LLM_MODEL:-local-model}
+ depends_on:
+ vllm:
+ condition: service_healthy
+
+volumes:
+ vllm_cache:
diff --git a/docker-compose.yml b/docker-compose.yml
index 471502a..6317089 100644
--- a/docker-compose.yml
+++ b/docker-compose.yml
@@ -1,6 +1,7 @@
# One-command local stack: API (:5100) + Admin (:5200) + MCP runtime + code sandbox
# Usage: docker compose up --build
-# Requires Ollama on the host (default http://host.docker.internal:11434)
+# LLM engine: add -f docker-compose.llamacpp.yml (CPU) or -f docker-compose.vllm.yml (GPU),
+# or point LLM_ENDPOINT at any OpenAI-compatible /v1 server. Falls back to Ollama on the host.
#
# mcp-runtime = hosts MCP stdio servers (e.g. zuora-mcp)
# sandbox-runtime = runs shell/python/node via POST /execute (self-hosted-sandbox)
diff --git a/docs/self-host.md b/docs/self-host.md
index 534cfa7..b188c81 100644
--- a/docs/self-host.md
+++ b/docs/self-host.md
@@ -4,7 +4,27 @@
Run the open-source gateway yourself — the path for **on-prem or fully local** deployments. Unlike Cloud (where the dashboard wires up your LLM for you), here **you point the gateway at your own LLM backend** — endpoint and model — in config. Same OpenAI-compatible chat body and `choices[]` response as Cloud; you supply the `X-App-Id` and use a `cm_live_` key.
-### Fastest: one-liner from GHCR (API)
+### Fastest: gateway + bundled engine (Compose)
+
+```bash
+git clone https://github.com/Kortexio/ContextMemory.git && cd ContextMemory
+
+# CPU, no GPU needed — llama.cpp pulls a small GGUF on first start
+docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build
+
+# or NVIDIA GPU — vLLM
+docker compose -f docker-compose.yml -f docker-compose.vllm.yml up --build
+```
+
+Then prove memory works (the second request carries only the new question):
+
+```bash
+./scripts/aha-chat.sh # Windows: .\scripts\aha-chat.ps1
+```
+
+Engine flags, model choice, and gotchas: [Engines](#engines).
+
+### One-liner from GHCR (API only)
Public images (published on every push to `main`):
@@ -67,7 +87,7 @@ Or the helper scripts:
### Build from source: Docker Compose
-Builds and starts the **API** (`:5100`), **Admin** (`:5200`), **mcp-runtime** (stdio MCP host), and **sandbox-runtime** (shell/python/node) locally. Requires [Docker](https://docs.docker.com/get-docker/) and an LLM reachable from the containers (Compose DX default = Ollama on the host; point `LLM_ENDPOINT` / `OPENAI_ENDPOINT` at any `/v1` engine instead).
+Builds and starts the **API** (`:5100`), **Admin** (`:5200`), **mcp-runtime** (stdio MCP host), and **sandbox-runtime** (shell/python/node) locally. Requires [Docker](https://docs.docker.com/get-docker/) and an LLM: add a bundled engine override (see [Fastest](#fastest-gateway--bundled-engine-compose)), point `LLM_ENDPOINT` at any `/v1` engine, or fall back to Ollama on the host.
```bash
git clone https://github.com/Kortexio/ContextMemory.git
@@ -96,13 +116,12 @@ Ops triage (Azure Monitor / GitHub): see [inbound-mcp-guide.md](inbound-mcp-guid
For a Postgres-backed network overlay (shared Docker network, extra tenants), see `docker-compose.network.yml`.
-**LLM on the host**
+**LLM engine**
```bash
-# Example: Ollama DX default
-ollama pull qwen3.5:9b
-# Compose: OLLAMA_ENDPOINT=http://host.docker.internal:11434
-# Or any /v1 engine: LLM_ENDPOINT=http://host.docker.internal:8000 + Admin backend openai-compatible
+# Bundled: add -f docker-compose.llamacpp.yml (CPU) or -f docker-compose.vllm.yml (GPU)
+# Engine on the host: LLM_ENDPOINT=http://host.docker.internal:8080 + Admin backend openai-compatible
+# Ollama on the host (fallback when LLM_ENDPOINT is empty): ollama pull qwen3.5:9b
```
**Useful Compose env vars** (see [`.env.example`](../.env.example)):
@@ -131,6 +150,45 @@ curl -X POST http://localhost:5100/v1/chat/completions \
-d '{"model":"qwen3.5:9b","messages":[{"role":"user","content":"Hello"}]}'
```
+## Engines
+
+The gateway speaks OpenAI-compatible `/v1/chat/completions` to the engine. Any server that implements it works; these are the ones we test and document. Set the app backend to `openai-compatible` (or `vllm`) and the endpoint to the engine base URL — `/v1` is appended automatically.
+
+| Engine | Compose override | Required flags | Notes |
+|---|---|---|---|
+| **llama.cpp** (`llama-server`) | `docker-compose.llamacpp.yml` | `--jinja` | Without `--jinja`, tool calls come back as plain text and agentic mode degrades. Set `--ctx-size` ≥ 8192 for agentic use. Model name in the request is ignored. |
+| **vLLM** | `docker-compose.vllm.yml` | `--enable-auto-tool-choice --tool-call-parser ` | Parser must match the model: `hermes` (Qwen2.5 / Qwen3), `llama3_json` (Llama 3.x), `mistral` (Mistral). `--served-model-name` must equal the app `llmModel`. |
+| **Ollama** | _(host, default fallback)_ | — | Ollama `/v1` ignores `num_ctx`; when you set `llmOptions.numCtx` the gateway switches to native `/api/chat`. |
+| LM Studio, SGLang, TGI, LiteLLM, OpenAI, Azure-compatible | — | tool calling enabled on the server | Point `LLM_ENDPOINT` (host default) or the app `llmEndpoint` at the server. |
+
+**Choosing a model**
+
+| Use | Minimum that works | Recommended |
+|---|---|---|
+| Session memory only (agentic off — the default) | 1.5B instruct (CI runs `Qwen2.5-1.5B-Instruct Q4_K_M`) | 3B–8B instruct |
+| Agentic mode (wiki_search, MCP, sandbox) | 7B–9B instruct with native tool calling | 14B+ or a hosted model |
+
+Small models need the hardening preset in [small-model-guide.md](small-model-guide.md). Found an engine/model combination that works (or does not)? Open an **Engine / model compatibility report** issue.
+
+**Changing the llama.cpp model**
+
+```bash
+LLAMACPP_HF_MODEL=bartowski/Qwen2.5-7B-Instruct-GGUF:Q4_K_M \
+ docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up
+```
+
+**Engine already running elsewhere** (host, another box, a GPU server):
+
+```bash
+LLM_ENDPOINT=http://gpu-box.lan:8000 docker compose up --build
+```
+
+Then in **Admin → Config → LLM** set backend `openai-compatible` and the model name the engine expects.
+
+The end-to-end check in CI ([`e2e-llamacpp`](../.github/workflows/e2e-llamacpp.yml)) builds the gateway, starts `llama-server`, and runs `scripts/aha-chat.sh` on every push to `main` and nightly.
+
+## Run from source (dotnet run)
+
### Prerequisites (dotnet run)
- .NET 9 SDK
diff --git a/docs/show-hn.md b/docs/show-hn.md
index c01ebee..1fd9739 100644
--- a/docs/show-hn.md
+++ b/docs/show-hn.md
@@ -1,44 +1,86 @@
> Part of the ContextMemory docs. [Back to README](../README.md).
-# Show HN — suggested blurb
+# Show HN — launch kit
-One product only. Link the GitHub repo. Keep Cloud / other Kortexio apps as a single closing line.
+Audience: people who self-host LLMs (llama.cpp, vLLM) and want memory without adopting an agent framework or a vector DB. One product only; Cloud is one closing line.
-## Title
+## Title (pick one)
+```text
+Show HN: ContextMemory – editable markdown memory for your llama.cpp/vLLM server
+Show HN: A self-hosted /v1 gateway that gives local LLMs persistent memory
```
-Show HN: ContextMemory – OpenAI-compatible gateway with wiki memory (not RAG)
-```
+
+Keep "not RAG" out of the title — say it in the body and the first comment, where it can be explained.
## Body
-```
-Agents forget. Most “memory” products hide facts in vector stores you can’t open.
+```text
+I run models on llama.cpp and vLLM and kept rebuilding the same thing in every app:
+history trimming, summaries, "remember this" facts, per-user isolation.
+
+ContextMemory is an open-source gateway that sits in front of any OpenAI-compatible
+engine. Your client keeps calling /v1/chat/completions and sends only the new message.
+The gateway attaches session memory, calls your engine, and returns a normal
+chat.completions response (streaming included).
-ContextMemory is an open-source agentic memory gateway: your app keeps calling a
-normal OpenAI-compatible /v1/chat/completions URL. The gateway attaches a session
-markdown wiki + history, can run a server-side tool loop (sandbox, MCP, wiki search),
-and returns a standard chat.completions response.
+Memory is markdown on disk (or Postgres): you can open it, edit it, diff it.
+No embeddings, no vector DB. When you want more, the same URL can run a server-side
+tool loop (wiki search, MCP servers, a code sandbox) with human confirmation before
+destructive actions.
-Bring your own LLM — Ollama, vLLM, LM Studio, ExLlamaSharp, OpenAI, Azure-compatible,
-or any /v1 host. Compose defaults to Ollama only for local DX; swap the backend per
-tenant in Admin.
+Try it in three commands (CPU is enough, llama.cpp pulls a small GGUF):
-Not classic RAG inject. Not a client agent framework. Memory you can open like a wiki.
+ git clone https://github.com/Kortexio/ContextMemory && cd ContextMemory
+ docker compose -f docker-compose.yml -f docker-compose.llamacpp.yml up --build -d
+ ./scripts/aha-chat.sh
-Aha with the Cursor MCP wedge (two chats):
- A) “Remember: staging DB is postgres-staging-01” → memory_save
- B) new chat “What is our staging DB?” → memory_search
+The script sends two requests in one session; the second carries only the question,
+so the answer can only come from the gateway's memory. CI runs the same check
+against a real llama-server on every push.
-Repo: https://github.com/Kortexio/ContextMemory
-Self-host / Cloud: see README. GIF storyboard: docs/aha-demo.html
+.NET 9, AGPL-3.0. There is a hosted version (Kortexio Cloud) if you do not want to run it.
+
+https://github.com/Kortexio/ContextMemory
```
-## Do / don’t
+## First comment (post immediately after submitting)
+
+```text
+Author here. A few things people usually ask:
+
+- Why not RAG / embeddings? Session memory is budgeted markdown the model sees every
+ turn. Shared docs go into a "Global Wiki" searched on demand with Postgres FTS
+ inside the tool loop. We chose lexical search over markdown you can audit; the
+ trade-off is weaker fuzzy recall on paraphrases.
+- Why .NET? It is what I ship fastest in and it gives us a single small container.
+ You never touch it from your app — it is just an HTTP URL.
+- Small models: session memory works with 1.5B instruct models (that is what CI uses).
+ The agentic tool loop needs 7B+ with native tool calling; llama.cpp needs --jinja.
+- Honest limits: beta, small team, APIs may still move. Engine compatibility
+ reports are the most useful issue you can open.
+```
+
+## Checklist before posting
+
+- [ ] Latest release published and newer than the last big change (release-please green)
+- [ ] `e2e-llamacpp` badge green on `main`
+- [ ] No bot issues in the tracker; 3–5 `good first issue` items open
+- [ ] GIF or asciinema of `aha-chat.sh` at the top of the README
+- [ ] Fresh clone on a clean machine: three commands work as written (Linux + Windows)
+- [ ] 4–6 hours free after posting to answer comments
+
+## Timing and follow-up
+
+- Post Tuesday–Thursday, 14:00–16:00 UTC.
+- Cross-post to r/LocalLLaMA 24–48 h later with the llama.cpp angle (engine flags, model table).
+- Measure for 14 days: unique visitors and referrers (GitHub Insights → Traffic), stars, issues from outside contributors. Clones are inflated by CI and are not a signal.
+
+## Do / don't
-| Do | Don’t |
+| Do | Don't |
|---|---|
-| Lead with wiki memory + `/v1` drop-in | List CompanyBrain / TriageHub / Fincheck in the title |
-| Say BYO LLM explicitly | Imply the product is “an Ollama app” |
-| Point to Compose + Admin LLM picker | Promise a public live chat sandbox unless it exists |
-| One soft line on Kortexio Cloud at the end | Dump six repos into the first paragraph |
+| Lead with "your engine + memory you can open" | List other Kortexio products |
+| Show the three commands in the body | Promise features that are not on `main` |
+| Answer "why not RAG" with the trade-off | Argue that RAG is wrong |
+| One line on Cloud at the end | Link pricing |
diff --git a/scripts/aha-chat.ps1 b/scripts/aha-chat.ps1
new file mode 100644
index 0000000..d53e65b
--- /dev/null
+++ b/scripts/aha-chat.ps1
@@ -0,0 +1,50 @@
+# Chat aha: tell the gateway a fact, then ask for it in a second request that carries
+# ONLY the new question. The answer can only come from ContextMemory's session memory.
+# Works with any engine behind the gateway (llama.cpp, vLLM, Ollama, OpenAI, ...).
+# Usage: .\scripts\aha-chat.ps1
+$ErrorActionPreference = "Stop"
+
+$Base = if ($env:CONTEXTMEMORY_BASE_URL) { $env:CONTEXTMEMORY_BASE_URL } else { "http://localhost:5100" }
+$Key = if ($env:CONTEXTMEMORY_API_KEY) { $env:CONTEXTMEMORY_API_KEY } else { "cm_live_dev_key_change_me" }
+$App = if ($env:CONTEXTMEMORY_APP_ID) { $env:CONTEXTMEMORY_APP_ID } else { "demo-dev" }
+$Model = if ($env:CONTEXTMEMORY_MODEL) { $env:CONTEXTMEMORY_MODEL } else { "local-model" }
+$UserId = if ($env:CONTEXTMEMORY_USER_ID) { $env:CONTEXTMEMORY_USER_ID } else { "aha-user" }
+$Timeout = if ($env:CONTEXTMEMORY_TIMEOUT) { [int]$env:CONTEXTMEMORY_TIMEOUT } else { 300 }
+$Session = "aha-$([DateTimeOffset]::UtcNow.ToUnixTimeSeconds())"
+$Fact = "postgres-staging-01"
+
+$headers = @{
+ Authorization = "Bearer $Key"
+ "X-App-Id" = $App
+ "X-User-Id" = $UserId
+ "X-Session-Id" = $Session
+}
+
+function Send-Chat([string]$Content) {
+ $body = @{ model = $Model; messages = @(@{ role = "user"; content = $Content }) } | ConvertTo-Json -Depth 5
+ Invoke-RestMethod -Method Post -Uri "$Base/v1/chat/completions" -Headers $headers `
+ -ContentType "application/json" -Body $body -TimeoutSec $Timeout
+}
+
+Write-Host "==> Health ($Base)"
+Invoke-RestMethod -Uri "$Base/health" | ConvertTo-Json -Compress
+Write-Host ""
+
+Write-Host "==> Turn 1 (session $Session): state the fact"
+$turn1 = Send-Chat "Remember this for later: our staging database host is $Fact. Reply with OK."
+$turn1.choices[0].message.content
+Write-Host ""
+
+Write-Host "==> Turn 2 (same session, body contains only the new question)"
+$turn2 = Send-Chat "What is our staging database host? Reply with the host name only."
+$answer = $turn2.choices[0].message.content
+$answer
+Write-Host ""
+
+if ($answer -match [regex]::Escape($Fact)) {
+ Write-Host "AHA OK — the client sent no history; the gateway remembered '$Fact'."
+ exit 0
+}
+
+Write-Host "AHA FAILED — '$Fact' not found in the turn 2 answer."
+exit 1
diff --git a/scripts/aha-chat.sh b/scripts/aha-chat.sh
new file mode 100755
index 0000000..805e4d4
--- /dev/null
+++ b/scripts/aha-chat.sh
@@ -0,0 +1,46 @@
+#!/usr/bin/env bash
+# Chat aha: tell the gateway a fact, then ask for it in a second request that carries
+# ONLY the new question. The answer can only come from ContextMemory's session memory.
+# Works with any engine behind the gateway (llama.cpp, vLLM, Ollama, OpenAI, ...).
+# Usage: ./scripts/aha-chat.sh
+set -euo pipefail
+
+BASE="${CONTEXTMEMORY_BASE_URL:-http://localhost:5100}"
+KEY="${CONTEXTMEMORY_API_KEY:-cm_live_dev_key_change_me}"
+APP="${CONTEXTMEMORY_APP_ID:-demo-dev}"
+MODEL="${CONTEXTMEMORY_MODEL:-local-model}"
+USER_ID="${CONTEXTMEMORY_USER_ID:-aha-user}"
+SESSION="aha-$(date +%s)"
+FACT="postgres-staging-01"
+OUT="$(mktemp -d)"
+
+chat() {
+ curl -sS --fail-with-body --max-time "${CONTEXTMEMORY_TIMEOUT:-300}" \
+ -X POST "$BASE/v1/chat/completions" \
+ -H "Authorization: Bearer $KEY" \
+ -H "X-App-Id: $APP" \
+ -H "X-User-Id: $USER_ID" \
+ -H "X-Session-Id: $SESSION" \
+ -H "Content-Type: application/json" \
+ -d "{\"model\":\"$MODEL\",\"messages\":[{\"role\":\"user\",\"content\":\"$1\"}]}"
+}
+
+echo "==> Health ($BASE)"
+curl -sSf "$BASE/health" | head -c 300
+echo; echo
+
+echo "==> Turn 1 (session $SESSION): state the fact"
+chat "Remember this for later: our staging database host is $FACT. Reply with OK." | tee "$OUT/turn1.json"
+echo; echo
+
+echo "==> Turn 2 (same session, body contains only the new question)"
+chat "What is our staging database host? Reply with the host name only." | tee "$OUT/turn2.json"
+echo; echo
+
+if grep -q "$FACT" "$OUT/turn2.json"; then
+ echo "AHA OK — the client sent no history; the gateway remembered '$FACT'."
+ exit 0
+fi
+
+echo "AHA FAILED — '$FACT' not found in the turn 2 answer."
+exit 1