From 0c256cc201147a7c57ea240e7e00575dee337079 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Mon, 7 Sep 2026 21:07:20 +0000 Subject: [PATCH 01/16] Bump undici from 7.29.0 to 8.10.2 Bumps [undici](https://github.com/nodejs/undici) from 7.29.0 to 8.10.2. - [Release notes](https://github.com/nodejs/undici/releases) - [Commits](https://github.com/nodejs/undici/compare/v7.29.0...v8.10.2) --- updated-dependencies: - dependency-name: undici dependency-version: 8.10.2 dependency-type: direct:production update-type: version-update:semver-major ... Signed-off-by: dependabot[bot] --- package-lock.json | 10 +++++----- package.json | 2 +- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/package-lock.json b/package-lock.json index 2f5335e..76fd066 100644 --- a/package-lock.json +++ b/package-lock.json @@ -9,7 +9,7 @@ "version": "0.2.3", "license": "MIT", "dependencies": { - "undici": "^7.28.0" + "undici": "^8.10.2" }, "bin": { "lloom": "bin/lloom.mjs", @@ -879,12 +879,12 @@ } }, "node_modules/undici": { - "version": "7.29.0", - "resolved": "https://registry.npmjs.org/undici/-/undici-7.29.0.tgz", - "integrity": "sha512-IDxfleLmmbSskfWSUATiN1nfn2rDuvnMOqb5CWR92iIfojA0Ud+ulOAAEQ57LPr9rWmsreUyf5lwyao+7GNNVw==", + "version": "8.10.2", + "resolved": "https://registry.npmjs.org/undici/-/undici-8.10.2.tgz", + "integrity": "sha512-/y4/bH9YNU5hi9NIrpOuvGXFcxrj3CMrV+/AYpowAYTpHn8gX/XPFjNy766FPoYY0miQhdW977JFWKGNhBdwyQ==", "license": "MIT", "engines": { - "node": ">=20.18.1" + "node": ">=22.19.0" } }, "node_modules/uri-js": { diff --git a/package.json b/package.json index 508045f..dbcca33 100644 --- a/package.json +++ b/package.json @@ -92,6 +92,6 @@ "prettier": "^3.9.5" }, "dependencies": { - "undici": "^7.28.0" + "undici": "^8.10.2" } } From d1e76d93f660084eaae79f92d74ddf86735897eb Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 21:06:00 +0000 Subject: [PATCH 02/16] Bump eslint from 10.10.0 to 10.11.0 Bumps [eslint](https://github.com/eslint/eslint) from 10.10.0 to 10.11.0. - [Release notes](https://github.com/eslint/eslint/releases) - [Commits](https://github.com/eslint/eslint/compare/v10.10.0...v10.11.0) --- updated-dependencies: - dependency-name: eslint dependency-version: 10.11.0 dependency-type: direct:development update-type: version-update:semver-minor ... Signed-off-by: dependabot[bot] --- package-lock.json | 8 ++++---- package.json | 2 +- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/package-lock.json b/package-lock.json index 31f9a55..50c1ecd 100644 --- a/package-lock.json +++ b/package-lock.json @@ -17,7 +17,7 @@ }, "devDependencies": { "@eslint/js": "^10.0.1", - "eslint": "^10.10.0", + "eslint": "^10.11.0", "prettier": "^3.9.6" }, "engines": { @@ -418,9 +418,9 @@ } }, "node_modules/eslint": { - "version": "10.10.0", - "resolved": "https://registry.npmjs.org/eslint/-/eslint-10.10.0.tgz", - "integrity": "sha512-NPXn6r5zl4uET1DAVPaOwzX3rut4c0wcmw3dWJAfOsTM5+TogXo0DDjz8pwm/hL8cyVNpHqeK4JpN0NjnyFFNw==", + "version": "10.11.0", + "resolved": "https://registry.npmjs.org/eslint/-/eslint-10.11.0.tgz", + "integrity": "sha512-P7a6UEEqb9G95MYAtqkmsTbVXIYyzIfl6NGOIJk162PaahFxFyeGcrlXYFSiagECg4sEm8IseJdZBKR3rx6MsQ==", "dev": true, "license": "MIT", "workspaces": [ diff --git a/package.json b/package.json index 8fcb693..10ae6a5 100644 --- a/package.json +++ b/package.json @@ -91,7 +91,7 @@ "license": "MIT", "devDependencies": { "@eslint/js": "^10.0.1", - "eslint": "^10.10.0", + "eslint": "^10.11.0", "prettier": "^3.9.6" }, "dependencies": { From ed393a68b464de8866115a19f20f5707fea58f38 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:49:57 +0000 Subject: [PATCH 03/16] Bump prettier from 3.9.6 to 3.9.8 Bumps [prettier](https://github.com/prettier/prettier) from 3.9.6 to 3.9.8. - [Release notes](https://github.com/prettier/prettier/releases) - [Changelog](https://github.com/prettier/prettier/blob/main/CHANGELOG.md) - [Commits](https://github.com/prettier/prettier/compare/3.9.6...3.9.8) --- updated-dependencies: - dependency-name: prettier dependency-version: 3.9.8 dependency-type: direct:development update-type: version-update:semver-patch ... Signed-off-by: dependabot[bot] --- package-lock.json | 8 ++++---- package.json | 2 +- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/package-lock.json b/package-lock.json index 31f9a55..34ff357 100644 --- a/package-lock.json +++ b/package-lock.json @@ -18,7 +18,7 @@ "devDependencies": { "@eslint/js": "^10.0.1", "eslint": "^10.10.0", - "prettier": "^3.9.6" + "prettier": "^3.9.8" }, "engines": { "node": ">=20.0.0" @@ -887,9 +887,9 @@ } }, "node_modules/prettier": { - "version": "3.9.6", - "resolved": "https://registry.npmjs.org/prettier/-/prettier-3.9.6.tgz", - "integrity": "sha512-OpN0zzVdiaiAhxpuuj5efpIS4sY9j7bY6uR5mnj5yPzGkdkjNKSJeUThPb60Jw29QuAZgA4o+/iB49kFiaBX6g==", + "version": "3.9.8", + "resolved": "https://registry.npmjs.org/prettier/-/prettier-3.9.8.tgz", + "integrity": "sha512-WRFq3Wn3WId7LLROfMLdH7xaFr2jR62wU8nLO6rQUOLOxNZUviyJQs1M0iIhLexSFy+L+w0ch66wtoO2jRjG0A==", "dev": true, "license": "MIT", "bin": { diff --git a/package.json b/package.json index 2df4fe7..fd75f3e 100644 --- a/package.json +++ b/package.json @@ -93,7 +93,7 @@ "devDependencies": { "@eslint/js": "^10.0.1", "eslint": "^10.10.0", - "prettier": "^3.9.6" + "prettier": "^3.9.8" }, "dependencies": { "undici": "^7.28.0" From a0387ee1e8bda68b7a4c5943304bac26936a1f82 Mon Sep 17 00:00:00 2001 From: Jason McCartney Date: Mon, 21 Sep 2026 21:00:38 -0700 Subject: [PATCH 04/16] Build guided setup and the LLooM browser experience --- .gitignore | 2 +- README.md | 29 +- SECURITY.md | 3 + bin/lloom.mjs | 47 +- docs/browser-experience.md | 38 ++ package.json | 7 +- src/bootstrap.mjs | 59 +- src/browser-setup.mjs | 219 ++++++++ src/dashboard-installation.mjs | 136 +++++ src/dashboard-presence-client.mjs | 313 +++++++++++ src/dashboard-presence.mjs | 130 +++++ src/dashboard-scene.mjs | 280 ++++++++++ src/dashboard.mjs | 64 ++- src/first-run-page.mjs | 775 +++++++++++++++++++++++++++ src/first-run.mjs | 254 +++++++++ src/init.mjs | 25 +- src/installation-jobs.mjs | 63 +++ src/installer.mjs | 64 ++- src/model-installation.mjs | 31 ++ src/runtime-manager.mjs | 56 +- src/runtime-policy.mjs | 21 +- src/runtime-preferences.mjs | 109 ++++ src/security.mjs | 66 ++- src/server.mjs | 73 ++- src/setup.mjs | 7 +- test/admin-browser-security.test.mjs | 42 ++ test/first-run.test.mjs | 346 ++++++++++++ test/installation-jobs.test.mjs | 140 +++++ test/runtime-preferences.test.mjs | 223 ++++++++ test/security.test.mjs | 23 +- 30 files changed, 3559 insertions(+), 86 deletions(-) create mode 100644 docs/browser-experience.md create mode 100644 src/browser-setup.mjs create mode 100644 src/dashboard-installation.mjs create mode 100644 src/dashboard-presence-client.mjs create mode 100644 src/dashboard-presence.mjs create mode 100644 src/dashboard-scene.mjs create mode 100644 src/first-run-page.mjs create mode 100644 src/first-run.mjs create mode 100644 src/installation-jobs.mjs create mode 100644 src/model-installation.mjs create mode 100644 src/runtime-preferences.mjs create mode 100644 test/admin-browser-security.test.mjs create mode 100644 test/first-run.test.mjs create mode 100644 test/installation-jobs.test.mjs create mode 100644 test/runtime-preferences.test.mjs diff --git a/.gitignore b/.gitignore index 4456512..8dbcd91 100644 --- a/.gitignore +++ b/.gitignore @@ -1,4 +1,4 @@ -node_modules/ +node_modules .lloom/ .DS_Store .env diff --git a/README.md b/README.md index b61fd4c..24cf0ec 100644 --- a/README.md +++ b/README.md @@ -3,7 +3,7 @@ [![CI](https://github.com/enntity/lloom/actions/workflows/ci.yml/badge.svg)](https://github.com/enntity/lloom/actions/workflows/ci.yml) [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE) -LLooM is a local-first LLM gateway for people who run serious open models on their own hardware. It treats NVIDIA systems—including DGX Spark / GB10—and Apple Silicon Macs as first-class platforms. LLooM sits in front of vLLM, SGLang, MLX, MTPLX, llama.cpp, Ollama, image generators, and other local runtimes, then exposes stable OpenAI-compatible and Anthropic-compatible APIs to agent tools. +LLooM installs, manages, and serves AI models on your hardware. It treats NVIDIA systems—including DGX Spark / GB10—and Apple Silicon Macs as first-class platforms. LLooM sits in front of vLLM, SGLang, MLX, MTPLX, llama.cpp, Ollama, image generators, and other local runtimes, then exposes stable OpenAI-compatible and Anthropic-compatible APIs to agent tools. The goal is simple: install one bridge, let it inspect the machine, choose the best agentic model recipe from the LLooM community library, install the backend needed for that recipe, download and configure the model, keep it warm, and point Codex, Claude Code, OMP, OpenCode, Hermes, Zero, or any OpenAI-compatible client at one base URL. @@ -15,11 +15,11 @@ The planned public community host is `https://lloom.enntity.com`; source checkou ## First-Class Platforms -| NVIDIA / DGX Spark | Apple Silicon | -| -------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------- | -| CUDA, Blackwell, DGX Spark / GB10, and Linux NVIDIA hosts | M-series Macs with unified memory | -| vLLM and SGLang are the primary high-throughput backends | MLX, MTPLX, OptiQ, and llama.cpp are the primary native backends | -| Managed Docker runtimes, GPU-memory admission, warm/on-demand lanes, and Spark recipes | Native processes, unified-memory-aware recipes, model-root reuse, and Mac recipes | +| NVIDIA / DGX Spark | Apple Silicon | +| --------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------- | +| CUDA, Blackwell, DGX Spark / GB10, and Linux NVIDIA hosts | M-series Macs with unified memory | +| vLLM and SGLang are the primary high-throughput backends | MLX, MTPLX, OptiQ, and llama.cpp are the primary native backends | +| Managed Docker runtimes, GPU-memory admission, warm/on-demand lanes, and Spark recipes | Native processes, unified-memory-aware recipes, model-root reuse, and Mac recipes | | See [`docs/dgx-spark.md`](docs/dgx-spark.md) and [`docs/clusters.md`](docs/clusters.md) | See the bundled `apple-silicon-*` recipes | Both platforms get the same gateway APIs, runtime policy, per-connection telemetry, live dashboard, client integrations, external-provider passthrough, and community recipe/benchmark workflow. Independent LLooM gateways can also form a heterogeneous lab behind one central endpoint; node profiles and optional GPU telemetry degrade cleanly across CUDA, Metal, ROCm, and CPU-only hosts. @@ -34,7 +34,6 @@ cd lloom npm ci npm link lloom -lloom up --go ``` `npm link` installs the same `lloom` and `lloom-host` commands from your checkout. After the first npm release, `npm install -g lloom` will be the supported package install path. @@ -48,7 +47,9 @@ curl -sS http://127.0.0.1:8100/v1/models Dashboard: [http://127.0.0.1:8100/](http://127.0.0.1:8100/) -That is the 1.0 path. A bare `lloom` is a dry run first: it inspects the machine, asks the LLooM community host for the best known recipe pack and backend catalog, shows what will be installed, and refuses writes until you rerun it with `--go`. `up` is the named alias for the same first-run flow. `--go` applies the plan, confirms noninteractive writes, and starts the selected keep-warm runtime after setup. Use `--offline` when you want to ignore the host and select from only the local recipe library. +On a new installation in an interactive terminal, `lloom` opens a local browser setup. It detects your hardware, asks what you want to use AI for, and recommends a compatible vendor recipe from the bundled library. Review the plan and choose **Set up my AI** to install it. The terminal stays open during setup. Chat setup checks a real response through the gateway before showing **Ready**; media installations show when output still needs verification. + +Use `lloom up --browser` to request browser setup explicitly, or `lloom ui` to open an installed gateway. The CLI remains available: `lloom --no-browser` previews the community-based plan; `lloom up --go` installs, integrates, and starts it. Scripted, JSON, offline, and explicit recipe commands keep their CLI behavior. See [the browser experience](docs/browser-experience.md) for the flow and current boundaries. The default gateway endpoint is `127.0.0.1:8100`; managed backend runtimes default to `8201-8299`. This source checkout defaults community lookup to the local development host at `127.0.0.1:8110`, starts it automatically if it is not already running, serves signed seed host data from `community/`, and requires signed recipe packs by default. Local imports still land in `recipes/` and `benchmarks/community/`. A production package should point at the signed public LLooM host. Most users should not need to care. @@ -60,12 +61,12 @@ Use this path when validating the repository before a package release: ```bash npm install -g . -lloom up +lloom up --no-browser lloom up --go lloom doctor --no-runtimes ``` -`lloom up` should show the detected machine profile, the trusted community recommendation, the selected recipe, benchmark evidence, and the exact apply command. +`lloom up --no-browser` should show the detected machine profile, the trusted community recommendation, the selected recipe, benchmark evidence, and the exact apply command. - On NVIDIA Linux, LLooM detects CUDA devices, compute capability, Blackwell, and DGX Spark / GB10 markers. Spark recipes use vLLM or SGLang, managed Docker containers, and explicit GPU-memory/runtime policy. The checked-in Spark deployment demonstrates a warm primary chat model, warm embedding model, and an on-demand alternate chat lane. - On a 96 GB Apple Silicon machine, the bundled development host should recommend `apple-silicon-qwen36-35b-a3b-mtplx` and select `Youssofal/Qwen3.6-35B-A3B-MTPLX-Optimized-Speed-FP16`. Lower-memory Macs should fall back to the 27B MTPLX recipe. @@ -99,14 +100,16 @@ Then open OMP normally. The generated OMP config points at `http://127.0.0.1:810 - Backend recipes for vLLM, SGLang, MTPLX, MLX LM, llama.cpp, Ollama, OptiQ, and stable-diffusion.cpp, with dedicated DGX Spark / GB10 and Apple Silicon recipes. - Community recipe packs and hardware-matched benchmark evidence so machines can select the best known model/backend recipe automatically instead of blindly chasing global tok/s. - Generated client profiles for OMP, OpenCode, Codex-compatible, Claude-compatible, Hermes, Zero, and any OpenAI-compatible client. -- A small dashboard at `/` for local status and guarded setup actions, with a live topology that can switch between the default columnar racks and an action view whose camera and cards follow live models. +- A browser dashboard with Live, Models, Machines, Clients, and Settings. Inspect real topology, install models, choose their readiness, load or unload managed runtimes, and try chat through the gateway. The action camera follows serving models while preserving manual zoom. ## Daily Commands Primary ladder (see `lloom help`; full catalog under `lloom help advanced`): ```bash -lloom # preview plan +lloom # browser setup on first interactive run +lloom ui # open the installed gateway +lloom --no-browser # preview the CLI plan lloom up --go # install + integrate + start lloom down # stop the gateway and all managed model backends lloom doctor --no-runtimes @@ -123,7 +126,7 @@ lloom add-model 'openai:http://127.0.0.1:8000/v1#my-model' --default --apply --y lloom serve --config ~/.lloom/config.json ``` -Bare `lloom`, `up`, and `onboard` all route to the same first-run flow. By default, the community request asks for the best known `agentic-coding` recipe with `tools`, `reasoning`, and `long-context`; use repeated `--workload`, `--capability`, or `--tag` flags to target a different kind of local model. `doctor` is the readiness view for humans and automation. `integrate` repairs or writes client configs from the registry. `add-model` imports an ad hoc Hugging Face, local, or Ollama model outside the community recipe library. +The CLI `onboard` flow remains available independently of browser setup. By default, the community request asks for the best known `agentic-coding` recipe with `tools`, `reasoning`, and `long-context`; use repeated `--workload`, `--capability`, or `--tag` flags to target a different kind of local model. `doctor` is the readiness view for humans and automation. `integrate` repairs or writes client configs from the registry. `add-model` imports an ad hoc Hugging Face, local, or Ollama model outside the community recipe library. After `~/.lloom/config.json` exists, operational commands such as `doctor`, `models`, `serve`, `integrate`, `add-model`, and runtime controls automatically read that installed config when `--config` is not supplied. Before that file exists, those commands return a `not-installed` report with the exact `lloom up` command to run, instead of silently operating on bundled model defaults. Read-only planning commands such as `lloom`, `lloom up`, `onboard`, and `integrations` can still preview from the packaged gateway shell plus community data. Use `--config` whenever you want to inspect or operate a different config file. diff --git a/SECURITY.md b/SECURITY.md index 92dbe80..e35b57c 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -24,8 +24,11 @@ The maintainers will acknowledge reports as soon as practical, validate the issu - A signature proves which key signed a recipe pack. Trusting keys served by the same community host is equivalent to trusting that host and its TLS connection; use explicit local trusted keys for stronger publisher pinning. - Development keys and the checked-in public seed key are not production trust roots. - Model weights and external runtimes have their own licenses and security posture. LLooM does not make untrusted model code safe. +- Browser management requests must come from the gateway's own origin. First-run setup additionally requires a local session token and applies only reviewed plans. Keep the setup terminal and tokens private. See [the browser security boundary](docs/browser-experience.md#local-security). - The public community MVP is read-only by design. Proposals arrive through reviewed pull requests; anonymous recipe and benchmark uploads are disabled in production. - Remote community feeds must use HTTPS and a locally pinned public signing key. Do not treat a signing key downloaded from the same remote host as an independent trust root. - The production community deployment is isolated from inference, databases, and Docker control-plane access. See [`deploy/community/README.md`](deploy/community/README.md). See [docs/architecture.md](docs/architecture.md) for route authorization and network-binding defaults. + +Remote admin writes require `security.allowRemoteAdmin=true` and a key in `security.adminApiKeys`. Inference credentials alone do not grant remote process or installation control. diff --git a/bin/lloom.mjs b/bin/lloom.mjs index 40566c2..463f10d 100755 --- a/bin/lloom.mjs +++ b/bin/lloom.mjs @@ -38,6 +38,8 @@ import { import { loadConfig } from '../src/config.mjs'; import { runtimeControlTimeoutMs } from '../src/control-timeout.mjs'; import { createDoctorReport } from '../src/doctor.mjs'; +import { createBrowserSetup } from '../src/browser-setup.mjs'; +import { shouldOpenSetup, openLocalBrowser } from '../src/first-run.mjs'; import { ClusterCoordinator, currentNodeId, @@ -78,6 +80,7 @@ const __filename = fileURLToPath(import.meta.url); /** Command tiers used for help + installed-config policy. Aliases resolve before dispatch. */ const COMMAND_REGISTRY = [ { name: 'up', aliases: [], tier: 'primary', needsInstalledConfig: false }, + { name: 'ui', aliases: [], tier: 'primary', needsInstalledConfig: true }, { name: 'onboard', aliases: [], tier: 'advanced', needsInstalledConfig: false }, { name: 'down', aliases: [], tier: 'primary', needsInstalledConfig: true }, { name: 'doctor', aliases: [], tier: 'primary', needsInstalledConfig: true }, @@ -147,8 +150,11 @@ function usage() { return `LLooM — local-first LLM gateway Primary commands: - lloom / lloom up Start installed LLooM; preview setup on first run + lloom / lloom up Start installed LLooM; guided setup on first run lloom up --go First run: install, integrate, and start the model + lloom up --browser Open guided local setup on first run + lloom up --no-browser Keep setup in the terminal + lloom ui Open the installed gateway dashboard lloom down Stop the gateway and all managed model backends lloom doctor Readiness report (blockers, warnings, next actions) lloom serve Run the gateway (reads ~/.lloom/config.json) @@ -1118,15 +1124,22 @@ function gatewayProcessPaths(configPath) { }; } -async function startGatewayBackground(configPath) { +async function startGatewayBackground(configPath, { requireOwned = false } = {}) { const gatewayConfig = await loadConfig(configPath); const url = gatewayUrlFor(gatewayConfig); const alreadyHealthy = await gatewayHealth(gatewayConfig); const paths = gatewayProcessPaths(configPath); if (alreadyHealthy?.ok) { + const ownedPid = + requireOwned && existsSync(paths.pidPath) ? Number(readFileSync(paths.pidPath, 'utf8').trim()) : null; + if (requireOwned && (!ownedPid || alreadyHealthy.pid !== ownedPid)) + throw new Error( + 'The gateway port is already used by another server. Choose another port or stop that server before retrying setup.' + ); return { status: 'already-running', url, + pid: alreadyHealthy.pid, health: alreadyHealthy, ...paths }; @@ -1158,6 +1171,8 @@ async function startGatewayBackground(configPath) { ...paths }; } + if (requireOwned && health.pid !== child.pid) + throw new Error('Another server answered on the gateway port. The installed gateway has not been verified.'); return { status: 'started', url, @@ -1448,6 +1463,24 @@ async function main() { }, up: async (context) => { const { args, configPath } = context; + if ( + shouldOpenSetup(args, { + installed: Boolean(configPath), + interactive: Boolean(process.stdin.isTTY && process.stdout.isTTY) + }) && + (!wantsOnboardingPlan(args) || hasFlag(args, '--browser')) + ) { + const options = await firstRunCliOptions([...args, '--offline']); + const setup = createBrowserSetup(context.config, options, { + gatewayStarter: (configPath) => startGatewayBackground(configPath, { requireOwned: true }) + }); + await setup.listen(); + const url = setup.bootstrapUrl(); + console.log('LLooM setup: ' + url + '\nKeep this terminal open while setup runs. Press Ctrl+C to stop.'); + const opened = await openLocalBrowser(url); + if (!opened) console.log('Open the local setup URL above in your browser.'); + return; + } if (!configPath || wantsOnboardingPlan(args)) return handlers.onboard(context); const gateway = await startGatewayBackground(configPath); const report = { @@ -1473,6 +1506,16 @@ async function main() { } if (!report.ok) process.exitCode = 1; }, + ui: async ({ args, config }) => { + const url = gatewayUrlFor(config); + if (wantsJson(args)) { + console.log(JSON.stringify({ url })); + return; + } + console.log('LLooM: ' + url); + if (!hasFlag(args, '--no-browser') && !(await openLocalBrowser(url))) + console.log('Open the gateway URL above in your browser.'); + }, onboard: async ({ args, config, command: _command }) => { const go = wantsGo(args); const apply = hasFlag(args, '--apply') || go; diff --git a/docs/browser-experience.md b/docs/browser-experience.md new file mode 100644 index 0000000..025654e --- /dev/null +++ b/docs/browser-experience.md @@ -0,0 +1,38 @@ +# The LLooM browser experience + +LLooM chooses and operates vendor recipes. Recipe vendors own model execution, backend tuning, and distributed execution contracts. The dashboard presents the hardware, installation, readiness, and serving decisions that belong to LLooM. + +## First run + +After installing the CLI from the repository, run `lloom` in an interactive terminal. With no installed configuration, a loopback setup server opens a browser. Choose Chat & write, Code, Images, or Voice; review the compatible recipe and installation details; then choose **Set up my AI**. + +The setup page uses the bundled recipe library. Unsupported hardware or workloads produce an explicit explanation. Unknown download sizes, credentials, or license information stay unknown. A recipe metadata license does not cover the model weights. + +Installation runs as a background job in the setup process. Refreshing the page resumes its progress. Keep the terminal open. A failed bootstrap can retry only the exact, unchanged configuration created by that session, within the original 30-minute review window. After a process restart, use `lloom bootstrap --apply --yes` or the installed dashboard to continue. Setup never silently replaces another configuration. + +After installation, LLooM starts the gateway and checks a chat request through its normal inference API. Health alone does not mean the model is ready. Media recipes require checking their actual output; the page identifies that remaining step. + +`lloom --no-browser`, `lloom onboard`, and `lloom up --go` preserve the CLI paths. JSON, offline, and explicit recipe flags do not unexpectedly launch a browser. `lloom ui` opens an installed gateway. + +## Daily use + +- **Live** shows clients, gateway nodes, models, and observed traffic. A luminous gateway connects client cards to model rows grouped by physical machine. Active requests drive the light trails. Follow activity focuses the scene on serving models. Detailed topology retains the diagnostic canvas and its manual camera controls. Both views honor reduced motion. +- **Models** searches and filters the actual gateway catalog. The inspector offers Load, Warm up, Unload, readiness policy, and a small chat trial. Runtime details remain available in a disclosure. +- **Add model** reviews a vendor recipe or custom model reference before starting a background installation. Plan IDs bind the reviewed input, recipe/backend catalog, and configuration version. Concurrent or stale changes fail visibly. Downloads remain reusable. The new installation API does not accept arbitrary config paths or shell commands. +- **Machines** shows configured physical nodes and their individual memory readings. Memory is never presented as one interchangeable pool. Distributed model execution requires a compatible vendor recipe. +- **Clients** supplies the gateway base URL, exact model IDs, an example request, and CLI integration commands. Admin credentials do not belong in client applications. +- **Settings** retains the advanced recipe, backend, runtime, and setup tools. + +Readiness policies express intent: **Auto** loads on demand, **Prefer ready** uses the existing idle residency reconciler, and **Always ready** prevents automatic eviction. All loading still passes through memory admission. Changing readiness does not restart a model or interrupt active work. The API returns a pending job while the current admission completes; the page reports completion or failure. Queued residency starts recheck the saved policy, and pending hard pins protect eviction victims. Use Load when you want to start a cold model immediately. + +## Local security + +First-run setup binds only to loopback and uses a random session token passed through the URL fragment. It removes the fragment immediately and keeps the token in that tab's session storage. Setup rejects foreign origins, alternate authorities, oversized bodies, and unreviewed apply inputs. Configuration publication cannot overwrite a file created concurrently. + +The installed dashboard rejects cross-origin management requests and DNS-rebound loopback authorities. Its page cannot be framed. Remote management writes require both explicit remote-admin enablement and a configured admin key; inference keys alone cannot authorize them. Inference compatibility is unchanged. Local same-user processes remain inside the local trust boundary. + +Installation jobs and reviewed plan IDs live in the gateway process. Browser refresh is supported; a gateway restart requires a fresh plan. Existing installer stage state and downloaded files support subsequent retries. The config mutation store detects concurrent writes but is not a cross-process transaction lock. + +## Remaining product work + +The repository is still the distribution source; a signed single-command installer and published npm package are not part of this change. Nearby discovery and approval-based pairing between independently installed gateways need an authenticated node identity protocol. The Machines page does not invent peers or automatically grant access. Existing configured federation remains available through the CLI and gateway. diff --git a/package.json b/package.json index 2df4fe7..500d64d 100644 --- a/package.json +++ b/package.json @@ -64,7 +64,7 @@ "check": "npm run check:cluster && npm run check:js && npm run check:entity-stagger && npm run check:python && npm run check:shell && npm run check:spark-deploy && npm run check:workers", "check:cluster": "node --check src/cluster.mjs && node --check test/cluster.test.mjs", "check:entity-stagger": "node --check scripts/entity-stagger-benchmark.mjs && node --check test/entity-stagger-benchmark.test.mjs", - "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs", + "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs && node --check src/dashboard-scene.mjs && node --check src/dashboard-presence.mjs && node --check src/dashboard-presence-client.mjs && node --check src/first-run.mjs && node --check src/first-run-page.mjs && node --check src/browser-setup.mjs && node --check src/runtime-preferences.mjs && node --check src/installation-jobs.mjs && node --check src/model-installation.mjs && node --check src/dashboard-installation.mjs", "check:python": "python3 -m py_compile backends/dspark-vllm/apply-patch-pack.py backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/vision_exp/*.py backends/qwen38-vllm/apply-ple-fp8-patch.py backends/qwen38-sglang/*.py backends/glm53-exl3/*.py backends/mlx-audio/lloom_audio_server.py backends/hear/lloom_hear_server.py backends/chatterbox/lloom_chatterbox_server.py patches/apply_mtplx_longctx_fix.py backends/comfyui-media/bridge/*.py backends/comfyui-media/graphs/*.py backends/comfyui-media/build/*.py backends/comfyui-media/install.py backends/qwen-image-diffusers/*.py backends/hear/test/test_hear_security.py", "check:shell": "bash -n backends/dspark-vllm/entrypoint.sh backends/dspark-vllm/runtime/build-hardened-image.sh backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/apply-runtime.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.sh backends/qwen38-vllm/entrypoint.sh backends/qwen38-sglang/entrypoint.sh backends/glm53-exl3/entrypoint.sh backends/mlx-audio/install.sh backends/hear/install.sh backends/chatterbox/install.sh clients/examples/lloom-claude clients/examples/lloom-codex clients/examples/lloom-hermes clients/examples/lloom-opencode clients/examples/lloom-zero scripts/remote-install-spark.sh scripts/remote-stage-spark-worker.sh", "check:spark-deploy": "node --check test/spark-deploy.test.mjs && node test/spark-deploy.test.mjs", @@ -73,7 +73,7 @@ "community:check": "node scripts/check-community-deploy.mjs", "package:check": "node scripts/check-package.mjs", "test:adoption": "node test/resource-fit.test.mjs && node test/model-acquisition.test.mjs && node test/dashboard-status.test.mjs", - "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs", + "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs && npm run test:ux", "test:cluster": "node test/cluster.test.mjs", "test": "npm run test:cluster && npm run test:unit && npm run test:entity-stagger && npm run smoke && npm run community:check", "test:entity-stagger": "node test/entity-stagger-benchmark.test.mjs", @@ -84,7 +84,8 @@ "test:workers": "node --test test/research-workers.test.mjs test/research-worker-process.test.mjs", "check:workers": "node --check clients/examples/research-workers/runner.mjs && node --check clients/examples/research-workers/process.mjs && node --check clients/examples/research-workers/process-guard.mjs", "test:comfyui": "python3 scripts/test-comfyui-media.py", - "test:hear": "python3 -m unittest discover -s backends/hear/test -v" + "test:hear": "python3 -m unittest discover -s backends/hear/test -v", + "test:ux": "node --test test/first-run.test.mjs test/runtime-preferences.test.mjs test/admin-browser-security.test.mjs test/installation-jobs.test.mjs" }, "engines": { "node": ">=20.0.0" diff --git a/src/bootstrap.mjs b/src/bootstrap.mjs index 05d0375..144d940 100644 --- a/src/bootstrap.mjs +++ b/src/bootstrap.mjs @@ -4,7 +4,7 @@ import { selectIntegrationArtifacts, writeGeneratedIntegrationArtifacts } from './client-integrations.mjs'; -import { applyBackend, applyRecipe, defaultInstallStatePathFor } from './installer.mjs'; +import { applyBackend, applyRecipe, pinDownloadCommands, defaultInstallStatePathFor } from './installer.mjs'; import { backendIds, defaultBackendVariables, @@ -175,17 +175,25 @@ export async function createBootstrapPlan( name: recipe.name, backendId: recipe.backend?.id }, + // Frozen executable evidence: the exact backend/recipe step plans plus the + // selected recipe and modelRoot. Retries and reviewed-plan execution must + // replay these untouched rather than re-profiling or re-selecting. + reviewedRecipe: recipe, + reviewedBackend: backend, + modelRoot: selectedModelRoot, backend: await planBackend(backend, { variables: backendVariables, checkCommands: true }), - recipe: planRecipe(recipe, config, { - modelRoot: selectedModelRoot, - backendIds: backendIds(catalog), - benchmarkEvidence, - benchmarksRoot: selectedBenchmarksRoot, - benchmarkValidationErrors: benchmarkErrors - }), + recipe: await pinDownloadCommands( + planRecipe(recipe, config, { + modelRoot: selectedModelRoot, + backendIds: backendIds(catalog), + benchmarkEvidence, + benchmarksRoot: selectedBenchmarksRoot, + benchmarkValidationErrors: benchmarkErrors + }) + ), benchmarks: { root: selectedBenchmarksRoot, validationErrors: benchmarkErrors, @@ -212,6 +220,7 @@ export async function applyBootstrap( generatedRoot, backendVariables = defaultBackendVariables(process.env), _benchmarkDocuments = [], + reviewedPlan, recipesRoot, recipeDocuments = [], backendCatalogPath, @@ -223,11 +232,27 @@ export async function applyBootstrap( throw new Error('Refusing to bootstrap without yes=true. Re-run with --yes after reviewing the dry-run plan.'); } - const profile = await profileMachine(); - const recipes = [...recipeDocuments, ...(await loadRecipes(recipesRoot))]; - const recipe = await selectRecipe({ recipeId, recipes, profile, recipesRoot }); - const catalog = await loadBackendCatalog(backendCatalogPath); - const backend = getBackend(catalog, recipe.backend?.id); + // A reviewed plan carries the exact backend/recipe plans that were shown to + // and approved by the user. When supplied, reuse the frozen evidence so a + // later hardware/catalog change cannot silently swap in different commands. + const reviewed = reviewedPlan ?? null; + let recipe; + let backend; + let backendPlan = null; + if (reviewed) { + if (!reviewed.reviewedRecipe || !reviewed.reviewedBackend || !reviewed.backend?.steps || !reviewed.recipe?.steps) + throw new Error('Reviewed bootstrap plan is incomplete. Review a fresh plan.'); + recipe = reviewed.reviewedRecipe; + backend = reviewed.reviewedBackend; + backendPlan = reviewed.backend; + } else { + const profile = await profileMachine(); + const recipes = [...recipeDocuments, ...(await loadRecipes(recipesRoot))]; + recipe = await selectRecipe({ recipeId, recipes, profile, recipesRoot }); + const catalog = await loadBackendCatalog(backendCatalogPath); + backend = getBackend(catalog, recipe.backend?.id); + } + if (!recipe) throw new Error('Bootstrap requires a reviewed recipe or recipeId.'); if (!backend) throw new Error(`Recipe ${recipe.id} references unknown backend ${recipe.backend?.id}`); const registry = createRegistry(config); @@ -237,7 +262,8 @@ export async function applyBootstrap( yes, statePath, variables: backendVariables, - env: commandEnv + env: commandEnv, + ...(backendPlan ? { reviewedPlan: backendPlan } : {}) }); backendResult.summary = phaseSummary('backend', backendResult); @@ -249,7 +275,10 @@ export async function applyBootstrap( env: commandEnv, onProgress, stdio, - ...(modelRoot ? { modelRoot } : {}) + ...((reviewed?.modelRoot ?? modelRoot) ? { modelRoot: reviewed?.modelRoot ?? modelRoot } : {}), + ...((reviewed?.recipePlan ?? (reviewed?.recipe?.steps ? reviewed.recipe : null)) + ? { reviewedPlan: reviewed.recipePlan ?? reviewed.recipe } + : {}) }) : blockedPhase('recipe', 'backend phase failed', { dryRun }); recipeResult.summary ??= phaseSummary('recipe', recipeResult); diff --git a/src/browser-setup.mjs b/src/browser-setup.mjs new file mode 100644 index 0000000..d27e787 --- /dev/null +++ b/src/browser-setup.mjs @@ -0,0 +1,219 @@ +import fs from 'node:fs/promises'; +import { createHash } from 'node:crypto'; +import { createFirstRunServer } from './first-run.mjs'; +import { createOnboardingPlan } from './onboarding.mjs'; +import { applySetup } from './setup.mjs'; +import { applyBootstrap } from './bootstrap.mjs'; +import { profileMachine, rankRecipes } from './machine-profile.mjs'; +import { loadRecipes, loadRecipeById } from './recipes.mjs'; +import { loadBackendCatalog } from './backend-catalog.mjs'; +import { loadConfig } from './config.mjs'; + +const fingerprint = (value) => createHash('sha256').update(JSON.stringify(value)).digest('hex'); +export function recipeSupportsWorkload(recipe, workload) { + const values = new Set([ + ...(recipe.capabilities || []), + ...(recipe.models || []).flatMap((model) => [model.kind, ...(model.capabilities || [])]) + ]); + if (workload === 'images') return [...values].some((value) => /^image/.test(value)); + if (workload === 'voice') return [...values].some((value) => /^audio|speech|transcription|tts|stt/.test(value)); + const chat = ['chat', 'responses', 'anthropic-messages'].some((value) => values.has(value)); + return workload === 'code' + ? chat && (values.has('tools') || (recipe.keywords || []).some((value) => /cod/.test(value))) + : chat; +} + +export function createBrowserSetup(config, options, { gatewayStarter, serverOptions = {} } = {}) { + const evidence = async (recipeId) => ({ + recipe: await loadRecipeById(recipeId, options.recipesRoot), + backendCatalog: await loadBackendCatalog(options.backendCatalogPath) + }); + return createFirstRunServer({ + ...serverOptions, + gatewayStarter, + async planBuilder({ workloadId, recipeId }) { + const [recipes, profile] = await Promise.all([loadRecipes(options.recipesRoot), profileMachine()]); + const ranked = await rankRecipes( + recipes.filter((recipe) => recipeSupportsWorkload(recipe, workloadId)), + profile, + { checkCommands: true } + ); + const candidates = ranked.filter((candidate) => candidate.selectable); + const selected = recipeId ? candidates.find((candidate) => candidate.recipeId === recipeId) : candidates[0]; + if (!selected) + throw new Error( + 'No compatible ' + + workloadId + + ' recipe is available for this hardware. Choose another use or add a vendor recipe through the CLI.' + ); + const selectedOptions = { + ...options, + recipeId: selected.recipeId, + offline: true, + start: false, + includeRuntimes: false + }; + const plan = await createOnboardingPlan(config, selectedOptions); + const source = await evidence(selected.recipeId); + plan.browserSetup = { options: selectedOptions, fingerprint: fingerprint(source) }; + const recipe = source.recipe; + const option = (candidate) => ({ + id: candidate.recipeId, + name: candidate.name, + reason: candidate.reasons.length + ? candidate.reasons.join('; ') + : 'Compatible with ' + (profile.cpuBrand || profile.platformId) + '.', + memoryRequiredGb: candidate.memoryRequiredGb, + downloadSizeBytes: null, + license: + recipes.find((item) => item.id === candidate.recipeId)?.license?.name || + recipes.find((item) => item.id === candidate.recipeId)?.license?.id || + null, + credentials: null + }); + return { + plan, + view: { + workloadId, + selected: option(selected), + alternatives: candidates + .filter((c) => c.recipeId !== selected.recipeId) + .slice(0, 8) + .map(option), + machine: { + name: profile.cpuBrand || profile.platformId, + platformId: profile.platformId, + totalMemoryGb: profile.totalMemoryGb, + accelerators: profile.accelerators || [] + }, + paths: { configPath: plan.configPath, modelRoot: plan.modelRoot }, + ports: plan.ports, + review: { + stages: plan.stages, + doctorCommands: plan.next?.doctor, + packages: (plan.setup?.phases?.bootstrap?.backend?.steps || []).map((step) => step.label || step.id), + models: (recipe.models || []).map((model) => model.gatewayModel || model.model) + } + } + }; + }, + applyRunner: createReviewedSetupRunner(config, { evidence }), + async gatewayProbe(report, plan, gateway) { + const installed = await loadConfig(report.configPath); + const url = gateway?.url || report.dashboardUrl; + const key = installed.security?.apiKeys?.[0] || installed.security?.adminApiKeys?.[0]; + const headers = { 'content-type': 'application/json', ...(key ? { authorization: 'Bearer ' + key } : {}) }; + const healthReply = await fetch(new URL('/health', url), { signal: AbortSignal.timeout(15000) }); + const healthData = healthReply.ok ? await healthReply.json() : null; + const health = healthData?.ok === true && (!gateway?.pid || healthData.pid === gateway.pid); + if (!health) + return { + healthy: false, + inferenceVerified: false, + endpoint: url, + detail: 'Installed, but the gateway did not pass its health check.' + }; + const model = installed.models.find( + (item) => + (item.kind || 'chat') === 'chat' && + (plan.setup?.phases?.init?.config?.models || []).some((candidate) => candidate.id === item.id) + ); + if (!model) + return { + healthy: true, + inferenceVerified: false, + endpoint: url, + detail: 'Gateway is running. Try the installed media model to verify its output.' + }; + try { + const response = await fetch(new URL('/v1/chat/completions', url), { + method: 'POST', + headers, + body: JSON.stringify({ + model: model.id, + messages: [{ role: 'user', content: 'Reply with Ready.' }], + max_tokens: 16, + stream: false + }), + signal: AbortSignal.timeout(180000) + }); + const body = await response.json(); + const verified = + response.ok && + Array.isArray(body.choices) && + body.choices.some((choice) => typeof choice.message?.content === 'string' && choice.message.content.trim()); + return { + healthy: true, + inferenceVerified: verified, + endpoint: url, + detail: verified + ? 'A model answered through your gateway.' + : body.error?.message || 'Gateway is running, but the model did not return a text response.' + }; + } catch (error) { + return { + healthy: true, + inferenceVerified: false, + endpoint: url, + detail: 'Gateway is running. Model verification needs attention: ' + error.message + }; + } + } + }); +} + +export function createReviewedSetupRunner(config, { evidence, apply = applySetup, bootstrap = applyBootstrap }) { + let ownedConfig = null; + return async (plan, { onProgress }) => { + const details = plan.browserSetup; + if (!details || fingerprint(await evidence(plan.selectedRecipe.id)) !== details.fingerprint) + throw new Error('The recipe or backend catalog changed. Review a fresh plan.'); + // A first-run wizard must never silently replace a newly created or + // concurrently installed configuration. + let existing; + try { + existing = JSON.parse(await fs.readFile(plan.configPath, 'utf8')); + } catch (error) { + if (error.code !== 'ENOENT') throw error; + } + if (existing) { + if ( + !ownedConfig || + ownedConfig.path !== plan.configPath || + ownedConfig.recipeId !== plan.selectedRecipe.id || + fingerprint(existing) !== ownedConfig.fingerprint + ) + throw new Error('A configuration now exists or changed. Open LLooM to manage it; setup will not overwrite it.'); + if (fingerprint(plan.setup.phases.init.config) !== ownedConfig.fingerprint) + throw new Error( + 'The installation plan changed after configuration was created. Continue with lloom bootstrap --apply --yes to resume the installed recipe.' + ); + // Retry only the bootstrap for the exact configuration this session + // successfully created. Never replace it after a partial download. + const installed = await loadConfig(plan.configPath); + const result = await bootstrap(installed, { + ...details.options, + modelRoot: plan.modelRoot, + // Retry replays the exact reviewed bootstrap evidence; never re-plan. + reviewedPlan: plan.setup?.phases?.bootstrap, + dryRun: false, + yes: true, + onProgress + }); + return { ...result, configPath: plan.configPath, dashboardUrl: plan.dashboardUrl }; + } + const result = await apply(config, { + ...details.options, + reviewedPlan: plan.setup, + dryRun: false, + yes: true, + start: false, + onProgress, + exclusiveConfig: true, + onConfigWritten(file, value) { + ownedConfig = { path: file, recipeId: plan.selectedRecipe.id, fingerprint: fingerprint(value) }; + } + }); + return { ...result, dashboardUrl: plan.dashboardUrl }; + }; +} diff --git a/src/dashboard-installation.mjs b/src/dashboard-installation.mjs new file mode 100644 index 0000000..e9dcae9 --- /dev/null +++ b/src/dashboard-installation.mjs @@ -0,0 +1,136 @@ +import { createHash } from 'node:crypto'; +import fs from 'node:fs/promises'; +import { createInstallationJobs } from './installation-jobs.mjs'; +import { createModelImportPlan } from './model-intake.mjs'; +import { createSetupPlan, applySetup } from './setup.mjs'; +import { loadRecipeById } from './recipes.mjs'; +import { loadBackendCatalog, getBackend, planBackend } from './backend-catalog.mjs'; +import { loadConfig } from './config.mjs'; +import { mutateConfigSource } from './config-mutation.mjs'; +import { pinDownloadCommands } from './installer.mjs'; +import { installImportedModelAssets } from './model-installation.mjs'; + +const digest = (value) => createHash('sha256').update(JSON.stringify(value)).digest('hex'); +const invalid = (message) => Object.assign(new Error(message), { statusCode: 409 }); +export function createDashboardInstallation({ getConfig, reload }) { + const source = async () => { + if (!getConfig().sourcePath) throw invalid('Installations need a writable LLooM configuration.'); + return loadConfig(getConfig().sourcePath); + }; + const evidence = async (recipeId) => ({ + catalog: await loadBackendCatalog(), + recipe: recipeId ? await loadRecipeById(recipeId) : null + }); + return createInstallationJobs({ + async plan(input) { + if (!input || Object.keys(input).some((key) => !['recipeId', 'modelRef', 'backend', 'name'].includes(key))) + throw invalid('Choose a vendor recipe or model reference.'); + if (Boolean(input.recipeId) === Boolean(input.modelRef)) + throw invalid('Choose exactly one recipe or model reference.'); + for (const value of Object.values(input)) + if (typeof value !== 'string' || value.length > 2000) throw invalid('Invalid model reference.'); + const config = await source(); + const baseline = digest(JSON.parse(await fs.readFile(config.sourcePath, 'utf8'))); + const sources = await evidence(input.recipeId); + const options = { + ...input, + configPath: config.sourcePath, + additive: true, + offline: true, + start: false, + includeRuntimes: false + }; + const plan = input.recipeId ? await createSetupPlan(config, options) : createModelImportPlan(config, options); + if (!input.recipeId && plan.additions.runtimeId) { + const backend = getBackend(sources.catalog, plan.inference.backend); + if (!backend) throw invalid('Unknown vendor backend.'); + plan.backendPlan = await planBackend(backend); + } + if (!input.recipeId && plan.download?.command?.[0] === 'hf') { + const command = plan.download.command; + const pinned = await pinDownloadCommands({ + steps: [ + { + action: 'download-model', + provider: 'huggingface', + model: command[2], + destination: command[command.indexOf('--local-dir') + 1], + command, + commands: [command] + } + ] + }); + plan.download.command = pinned.steps[0].command; + delete plan.download.shellCommand; + } + return { + input, + options, + plan, + baseline, + sources, + digest: digest(sources), + view: { + kind: input.recipeId ? 'recipe' : 'model', + summary: + 'Installs the vendor backend and model files, then adds the model to this gateway. Loading uses normal memory admission.', + details: input.recipeId + ? { + recipe: plan.selectedRecipe, + configPath: plan.configPath, + modelRoot: plan.modelRoot, + ports: plan.ports, + backend: plan.phases.bootstrap.backend, + models: plan.phases.bootstrap.recipe, + integrations: plan.phases.bootstrap.integrations + } + : { + reference: plan.reference, + backend: plan.inference, + backendInstallation: plan.backendPlan, + additions: plan.additions, + download: plan.download, + configPath: plan.configPath + } + } + }; + }, + async apply(prepared, onProgress) { + if (digest(await evidence(prepared.input.recipeId)) !== prepared.digest) + throw invalid('The vendor recipe changed. Review a fresh plan.'); + const config = await source(); + if (digest(JSON.parse(await fs.readFile(config.sourcePath, 'utf8'))) !== prepared.baseline) + throw invalid('Your configuration changed. Review a fresh installation plan.'); + const writeConfig = async (file, value) => { + if (file !== config.sourcePath) throw invalid('The installation destination changed.'); + await mutateConfigSource(config, (raw) => { + if (digest(raw) !== prepared.baseline) + throw invalid( + 'Your configuration changed during installation. Files are retained; review a fresh plan to continue.' + ); + for (const key of Object.keys(raw)) delete raw[key]; + Object.assign(raw, value); + }); + }; + let result; + if (prepared.input.recipeId) { + result = await applySetup(config, { + ...prepared.options, + reviewedPlan: prepared.plan, + dryRun: false, + yes: true, + onProgress, + writeConfig + }); + } else { + await installImportedModelAssets(prepared.plan, { onProgress, backendCatalog: prepared.sources.catalog }); + await writeConfig(config.sourcePath, prepared.plan.config); + result = { ok: true }; + } + // Even failed bootstrap can have written configuration. Surface reload + // failures, and make the installed/failed state visible in the dashboard. + await reload(); + return result; + } + }); +} diff --git a/src/dashboard-presence-client.mjs b/src/dashboard-presence-client.mjs new file mode 100644 index 0000000..09ced01 --- /dev/null +++ b/src/dashboard-presence-client.mjs @@ -0,0 +1,313 @@ +// Runs inside the dashboard's existing script, sharing its authenticated API +// helpers and observed state. It never invents topology or hardware readings. +export const presenceScript = String.raw` + let presenceView = "live"; + let presenceImport = null; + let presencePlanVersion = 0; + let presenceLastFocus = null; + let presenceBusy = false; + let presenceRenderKey = ""; + let presenceMachineKey = ""; + let presenceInstallTimer = null; + let presenceInstallSeen = null; + const presencePolicyNames = { auto:"Auto", preferred:"Prefer ready", always:"Always ready" }; + function presenceNotice(message, error = false) { + const toast = $("#presence-toast"); + toast.querySelector("span").textContent = message; + toast.dataset.error = String(error); + toast.hidden = false; + } + function presenceSetView(name) { + if (!["live","models","machines","clients","settings"].includes(name)) name = "live"; + presenceView = name; + document.querySelectorAll(".presence-nav [data-view]").forEach(button => { + if (button.dataset.view === name) button.setAttribute("aria-current","page"); + else button.removeAttribute("aria-current"); + }); + document.querySelectorAll("[data-presence-panel]").forEach(panel => panel.hidden = panel.dataset.presencePanel !== name); + for (const id of ["models","machines","clients"]) $("#view-" + id).hidden = id !== name; + $(".operations-dock").open = name === "settings"; + closeModelInspector(); closeNodeInspector(); + if (name === "clients") presenceLoadIntegrations(); + if (name === "machines") presenceDiscover(); + try { history.replaceState(null,"","#" + name); } catch {} + renderPresence(); + window.scrollTo({top:0,behavior:"instant"}); + } + function presenceRuntime(model) { + return model.runtime ? state.status?.runtimeManager?.runtimes?.[model.runtime] : null; + } + function presencePolicy(runtime) { + return runtime?.keepWarm ? "always" : runtime?.preferredWarm ? "preferred" : "auto"; + } + function presenceModelLabel(model) { + const runtime = presenceRuntime(model); + if (model.alias) return "Route"; + if (!model.runtime) return model.federated ? "Shared model" : "External provider"; + if (runtime?.maintenance) return "Paused"; + const transition={starting:"Loading",warming:"Warming",queued:"Queued",stopping:"Unloading",draining:"Finishing work",failed:"Needs attention",unreachable:"Unavailable",disabled:"Disabled"}[runtime?.status]; + if(transition)return transition; + if (runtime?.activeRequests > 0 && runtime?.healthy) return "Serving"; + if (runtime?.healthy) return "Ready"; + return "On demand"; + } + function renderPresenceModels() { + const search = $("#presence-search").value.trim().toLowerCase(); + const kind = $("#presence-kind").value; + const models = (state.physicalModels || []).filter(model => + (!search || (model.name + " " + model.id).toLowerCase().includes(search)) && + (!kind || (model.kind || "chat").startsWith(kind)) + ).sort((a,b) => Number(Boolean(b.runtime)) - Number(Boolean(a.runtime)) || Number(Boolean(presenceRuntime(b)?.healthy)) - Number(Boolean(presenceRuntime(a)?.healthy))); + $("#presence-model-count").textContent = models.length + (models.length === 1 ? " model" : " models"); + const key = JSON.stringify(models.map(model => {const rt=presenceRuntime(model);return [model.id,model.name,model.kind,model.contextWindow,model.runtime,presenceModelLabel(model),rt?.memoryGb,rt?.node,rt?.keepWarm,rt?.preferredWarm];})); + if (key === presenceRenderKey) return; + presenceRenderKey = key; + $("#presence-models").innerHTML = models.map(model => { + const rt = presenceRuntime(model), policy = presencePolicy(rt); + const location = rt?.node || (model.targets || []).map(t => t.node).filter(Boolean).join(", ") || (model.runtime ? "This machine" : "Upstream"); + return '
' + escapeHtml(({chat:"Chat & code",audio_speech:"Speech",audio_transcription:"Transcription",audio_generation:"Music & audio",embedding:"Embeddings",image:"Images",video:"Video"})[model.kind] || model.kind || "Chat") + '' + escapeHtml(presenceModelLabel(model)) + '

' + escapeHtml(model.name || model.id) + '

' + escapeHtml(model.id) + '

Runs on
' + escapeHtml(location) + '
Memory estimate
' + (rt?.memoryGb != null ? escapeHtml(rt.memoryGb) + ' GB' : 'Not reported') + '
Availability
' + escapeHtml(rt ? presencePolicyNames[policy] : 'Managed upstream') + '
'; + }).join("") || '
' + (state.models.length ? 'No models match these filters.' : 'No configured models yet. Add a model to get started.') + '
'; + } + function renderPresenceMachines() { + const nodes = Object.values(state.status?.cluster?.nodes || {}); + const rows = nodes.length ? nodes : (state.topologySummary?.host ? [{id:"local",name:"This machine",local:true,telemetry:state.topologySummary.host}] : []); + const key=JSON.stringify(rows.map(node=>[node.id,node.name,node.local,node.reachable,node.profile?.cpuBrand,node.telemetry?.memory?.usedBytes,node.telemetry?.memory?.availableBytes,node.telemetry?.memory?.totalBytes])); + if(key===presenceMachineKey)return; + presenceMachineKey=key; + $("#presence-machines").innerHTML = rows.map(node => { + const memory = node.telemetry?.memory || {}; + const total = Number(memory.totalBytes); + const available = Number(memory.availableBytes); + const used = Number.isFinite(available) ? total - available : Number(memory.usedBytes); + const hasMemory = total > 0 && Number.isFinite(used); + return '
' + (node.local ? 'This machine' : 'Configured peer') + '' + (node.reachable === false ? 'Unavailable' : 'Connected') + '

' + escapeHtml(node.name || node.id) + '

' + escapeHtml(node.profile?.cpuBrand || node.telemetry?.cpu?.model || node.profile?.platformId || 'Hardware details unavailable') + '

' + (hasMemory ? '

' + escapeHtml(formatBytes(Number.isFinite(available) ? available : used)) + (Number.isFinite(available) ? ' available / ' : ' used / ') + escapeHtml(formatBytes(total)) + '

' : '

Memory reading unavailable

') + '
'; + }).join("") || '
Waiting for hardware telemetry.
'; + } + function presenceClientExample() { + const model = $("#presence-client-model").value; + const body = JSON.stringify({model,messages:[{role:"user",content:"Hello"}]}, null, 2); + $("#presence-client-example").textContent = "POST " + endpoint + "/v1/chat/completions\nContent-Type: application/json\nAuthorization: Bearer YOUR_LLOOM_KEY\n\n" + body; + } + function renderPresenceClients() { + $("#presence-client-url").value = endpoint + "/v1"; + const select = $("#presence-client-model"), selected = select.value; + const models = (state.models || []).filter(model => (model.kind || "chat") === "chat"); + const key = models.map(m=>m.id).join("\n"); + if (select.dataset.models !== key) { + select.innerHTML = models.map(m=>'').join(""); + if (models.some(m=>m.id === selected)) select.value = selected; + select.dataset.models = key; + } + $("#presence-client-auth").textContent = state.security?.authRequired ? "Use a configured inference key. Keep your admin key out of client applications." : "This loopback gateway allows local clients without a key. Remote access requires authenticated configuration."; + presenceClientExample(); + } + function renderPresence() { + if (!$("#presence-models")) return; + renderPresenceModels(); renderPresenceMachines(); renderPresenceClients(); renderPresencePolicy(); + } + function renderPresencePolicy() { + const model = state.physicalModels.find(m=>m.id === state.selectedModelId); + const runtime = model && presenceRuntime(model); + if(model) $("#model-inspector-state span:last-child").textContent=presenceModelLabel(model); + $("#presence-policy").hidden = !model; + $("#presence-trial").hidden = !model || (model.kind || "chat") !== "chat"; + for (const button of document.querySelectorAll("[data-residency]")) { + button.disabled = presenceBusy || !runtime || runtime.enabled === false || runtime.management === "external" || runtime.remote === true || Boolean(runtime.maintenance); + button.setAttribute("aria-pressed",String(Boolean(runtime) && presencePolicy(runtime) === button.dataset.residency)); + } + $("#presence-policy-hint").textContent = runtime ? "Readiness remains subject to memory admission. Use Load to start a cold model. Always ready prevents automatic eviction." : "Availability is managed by the upstream provider."; + } + async function presenceDiscover() { + const button = $("#presence-discover"); + if (button.disabled) return; + button.disabled = true; + try { + const result = await getJson("/gateway/discovery"); + $("#nearby-machines").innerHTML = (result.peers || []).map(peer=>'

' + escapeHtml(peer.name || peer.id) + '

Discovered · not connected

' + escapeHtml(peer.host || '') + '

').join("") || escapeHtml(result.warning || (result.enabled ? "No other LLooM installations found yet." : "Network discovery is not enabled on this gateway.")); + } catch { $("#nearby-machines").textContent = "This gateway does not provide nearby discovery yet. Your configured peers are shown above."; } + finally { button.hidden = true; button.disabled = false; } + } + async function presenceLoadIntegrations() { + try { + const manifest = await getJson("/gateway/integrations"); + $("#presence-integrations").innerHTML = (manifest.clients || []).map(client=>'
' + escapeHtml(client.name || client.id) + '
' + escapeHtml("lloom integrate " + client.id) + '

Preview first. Add --apply --yes only after reviewing the generated configuration.

').join("") || "No client profiles available."; + } catch(error) { $("#presence-integrations").textContent = error.message; } + } + function presenceInvalidatePlan() { + presenceImport = null; presencePlanVersion++; + $("#presence-add-review").hidden = true; $("#presence-add-apply").hidden = true; + $("#presence-add-plan").hidden = false; + } + function presenceOpenAdd() { + presenceLastFocus = document.activeElement; + presenceInvalidatePlan(); + $("#presence-add-error").textContent = ""; + const selected = state.library?.selected; + $("#presence-recommendation").innerHTML = selected + ? '
Recommended for this hardware

' + escapeHtml(selected.name || selected.recipeId) + '

' + escapeHtml((selected.reasons || []).join(" · ") || "Matched by the local recipe library.") + '

' + : '

No compatible recipe recommendation is available yet.

'; + $("#presence-add").hidden = false; $("#presence-model-ref").focus(); + } + function presenceCloseAdd() { + if (presenceBusy) return; + $("#presence-add").hidden = true; presenceLastFocus?.focus(); + } + async function presenceReviewRecipe() { + const selected = state.library?.selected; + if (!selected) return; + const recipeId = selected.recipeId; + const version = presencePlanVersion; + const plan = await postJson("/gateway/installations/plan",{recipeId}); + if(version!==presencePlanVersion)return; + presenceImport = {planId:plan.planId}; + presenceShowPlan(plan); + } + function presenceShowPlan(plan) { + $("#presence-plan-json").textContent = JSON.stringify(plan.details || plan,null,2); + $("#presence-plan-summary").textContent = plan.summary || "Review the packages, model files, and configuration this installation will use."; + $("#presence-add-review").hidden = false; + $("#presence-add-apply").hidden = plan.ok === false; + $("#presence-add-plan").hidden = plan.ok !== false; + } + async function presenceRun(button, action) { + if (presenceBusy) return; + presenceBusy = true; + if (button) button.disabled = true; + renderPresencePolicy(); + try { return await action(); } + catch(error) { presenceNotice(error.message,true); $("#presence-add-error").textContent = error.message; } + finally { presenceBusy = false; if(button) button.disabled=false; renderPresencePolicy(); } + } + async function presencePollInstallation() { + clearTimeout(presenceInstallTimer); + try { + const {job}=await getJson("/gateway/installations"); + if(!job)return; + if(job.status==="running") { + presenceNotice(job.detail); + presenceInstallTimer=setTimeout(presencePollInstallation,1500); + } else if(presenceInstallSeen!==job.id) { + presenceInstallSeen=job.id; + presenceNotice(job.error || job.detail,job.status==="failed"); + await refresh(); + } + } catch(error) { presenceNotice("Could not read installation progress: "+error.message,true); } + } + // Keep the model inspector available from every view, rather than clipping + // it when the canvas is hidden. + document.body.append($("#model-inspector"),$("#node-inspector")); + const policy = document.createElement("section"); + policy.id = "presence-policy"; + policy.innerHTML = '

Keep this model available

' + Object.entries(presencePolicyNames).map(([id,name])=>'').join("") + '

'; + $("#model-inspector .model-inspector-body").append(policy); + const trial = document.createElement("section"); + trial.id = "presence-trial"; trial.className = "presence-trial"; + trial.innerHTML = '
';
+    $("#model-inspector .model-inspector-body").append(trial);
+    const inspectorBody=$("#model-inspector .model-inspector-body");
+    const technical=document.createElement("details");
+    technical.innerHTML="Recipe & runtime details";
+    technical.append($("#model-inspector-details"),$("#model-inspector-tags"));
+    inspectorBody.append(technical);
+    inspectorBody.prepend(policy,$("#model-inspector .model-inspector-actions"),trial);
+    $("#model-start").textContent="Load"; $("#model-warm").textContent="Warm up"; $("#model-stop").textContent="Unload";
+    const oldRenderModels = renderModels;
+    renderModels = function() { oldRenderModels(); renderPresence(); };
+    const oldRenderInspector = renderModelInspector;
+    renderModelInspector = function() { oldRenderInspector(); renderPresencePolicy(); };
+    const oldShowOutput = showOutput;
+    showOutput = function(value) { oldShowOutput(value); presenceNotice(value?.error?.message || value?.error || "Operation complete. Details are in Settings.",Boolean(value?.error)); };
+    const keyField = $("#api-key").closest("label");
+    keyField.style.cssText = "";
+    const settingsHeader = document.createElement("div"); settingsHeader.className="presence-heading";
+    settingsHeader.innerHTML='

Settings & details.

Authentication, recipes, runtimes, and installation plans.

'; + $(".operations-content").prepend(settingsHeader,keyField); + $("#api-key").addEventListener("change",()=>{ sessionStorage.setItem("lloom_api_key",$("#api-key").value.trim()); localStorage.removeItem("lloom_api_key"); refresh(); }); + $(".presence-nav").addEventListener("click",event=>{const button=event.target.closest("[data-view]"); if(button) presenceSetView(button.dataset.view);}); + $("#presence-toast button").addEventListener("click",()=>$("#presence-toast").hidden=true); + $("#presence-search").addEventListener("input",()=>renderPresenceModels()); + $("#presence-kind").addEventListener("change",()=>renderPresenceModels()); + $("#presence-discover").addEventListener("click",presenceDiscover); + $("#presence-client-model").addEventListener("change",presenceClientExample); + $("#presence-copy-url").addEventListener("click",event=>presenceRun(event.currentTarget,async()=>{await navigator.clipboard.writeText(endpoint+"/v1");presenceNotice("Gateway URL copied.");})); + $("#presence-copy-example").addEventListener("click",event=>presenceRun(event.currentTarget,async()=>{await navigator.clipboard.writeText($("#presence-client-example").textContent);presenceNotice("Example request copied. Replace the key placeholder in your client.");})); + $("#presence-add-close").addEventListener("click",presenceCloseAdd); + $("#presence-add-form").addEventListener("input",presenceInvalidatePlan); + $("#presence-add-form").addEventListener("submit",event=>{ + event.preventDefault(); const button=$("#presence-add-plan"); + presenceRun(button,async()=>{ + const version=presencePlanVersion; + const input={modelRef:$("#presence-model-ref").value.trim(),backend:$("#presence-model-backend").value.trim()||undefined,name:$("#presence-model-name").value.trim()||undefined}; + const plan=await postJson("/gateway/installations/plan",input); + if(version!==presencePlanVersion)return; + presenceImport={planId:plan.planId}; presenceShowPlan(plan); + }); + }); + $("#presence-add-apply").addEventListener("click",event=>presenceRun(event.currentTarget,async()=>{ + if(!presenceImport || !ensureAdminKeyIfNeeded()) return; + await postJson("/gateway/installations",{...presenceImport,yes:true}); + presenceBusy=false; presenceCloseAdd(); presencePollInstallation(); + })); + document.addEventListener("click",event=>{ + const add=event.target.closest("[data-add-model]"); + if(add) presenceOpenAdd(); + const model=event.target.closest("[data-presence-model]"); + if(model) {openModelInspector(model.dataset.presenceModel);$("#model-inspector-close").focus();} + const machine=event.target.closest("[data-presence-node]"); + if(machine) openNodeInspector(machine.dataset.presenceNode); + if(event.target.closest("#presence-use-recipe")) presenceRun(event.target.closest("button"),presenceReviewRecipe); + const residency=event.target.closest("[data-residency]"); + if(residency) presenceRun(residency,async()=>{ + const model=state.physicalModels.find(m=>m.id===state.selectedModelId); + if(!model?.runtime || !ensureAdminKeyIfNeeded()) return; + const path="/gateway/runtimes/"+encodeURIComponent(model.runtime)+"/residency"; + const accepted=await postJson(path,{policy:residency.dataset.residency,yes:true}); + presenceNotice("Readiness saved. Waiting for the current load to finish before applying it."); + const poll=async()=>{ + try { + const {job}=await getJson(path); + if(!job || job.id!==accepted.id)return; + if(job.status==='pending'){setTimeout(poll,1500);return;} + await refresh(); + presenceNotice(job.status==='failed'?"Readiness saved, but could not be applied: "+job.error:"Readiness applied: "+presencePolicyNames[job.policy]+".",job.status==='failed'); + }catch(error){presenceNotice("Could not check readiness: "+error.message,true);} + }; + setTimeout(poll,500); + }); + }); + // Capture legacy runtime actions once, add bounded busy/error handling, and + // use normal admission for Load instead of a forced process start. + document.addEventListener("click",event=>{ + const button=event.target.closest("button[data-runtime]"); + if(!button) return; + event.stopImmediatePropagation(); + if(button.disabled || !button.dataset.runtime || !ensureAdminKeyIfNeeded()) return; + presenceRun(button,async()=>{ + const load=button.dataset.action==="start"; + const action=load?"admit":button.dataset.action; + const result=await postJson("/gateway/runtimes/"+encodeURIComponent(button.dataset.runtime)+"/"+action,load?{apply:true,yes:true,force:false,warmup:true}:{}); + oldShowOutput(result);await refresh();presenceNotice("Model operation complete."); + }); + },true); + $("#presence-send").addEventListener("click",event=>presenceRun(event.currentTarget,async()=>{ + const prompt=$("#presence-prompt").value.trim(), model=state.selectedModelId; + if(!prompt || !model) return; + $("#presence-answer").textContent="Waiting for "+model+"…"; + try { + const result=await postJson("/v1/chat/completions",{model,messages:[{role:"user",content:prompt}],max_tokens:256,stream:false}); + $("#presence-answer").textContent=result.choices?.[0]?.message?.content || "No text response returned."; + } catch(error) { $("#presence-answer").textContent=error.message; throw error; } + })); + document.addEventListener("keydown",event=>{ + if(event.key==="Escape") {presenceCloseAdd();closeModelInspector();closeNodeInspector();} + const dialog=!$("#presence-add").hidden?$("#presence-add .presence-dialog"):null; + if(event.key==="Tab" && dialog) { + const items=[...dialog.querySelectorAll('button:not([disabled]),input,select,textarea,summary')].filter(el=>el.getClientRects().length); + const first=items[0],last=items.at(-1); + if(event.shiftKey && document.activeElement===first){event.preventDefault();last?.focus();} + else if(!event.shiftKey && document.activeElement===last){event.preventDefault();first?.focus();} + } + }); + presenceSetView(location.hash.slice(1)); + presencePollInstallation(); +`; diff --git a/src/dashboard-presence.mjs b/src/dashboard-presence.mjs new file mode 100644 index 0000000..a421e37 --- /dev/null +++ b/src/dashboard-presence.mjs @@ -0,0 +1,130 @@ +export const presenceStyles = ` + :root { color-scheme:dark; --bg:#080d12; --band:#0d151d; --panel:#111c25; --panel-2:#15232e; --line:#243641; --text:#edf5f8; --muted:#9eb3c0; --accent:#38dff5; --accent-2:#85c8ff; --danger:#fb927e; } + body { font-family:Inter,-apple-system,BlinkMacSystemFont,"Segoe UI",sans-serif; font-size:14px; letter-spacing:0; } + body > header { margin-left:190px; min-height:72px; padding:16px 28px; background:var(--bg); border-bottom:1px solid var(--line); } + body > header .brand { display:none; } + .topline { width:100%; justify-content:flex-end; } + main { margin-left:190px; width:auto; padding:24px 28px; max-width:none; min-width:0; } + main > * { min-width:0; } + button.primary { background:#1e4953; border-color:#36717d; } + button,input,select,textarea { font-family:inherit; border-radius:9px; } + button { text-transform:none; font-size:13px; letter-spacing:0; font-weight:500; } + button:focus-visible, a:focus-visible, input:focus-visible, select:focus-visible { outline:2px solid var(--accent); outline-offset:3px; } + h1,h2,h3,strong { font-weight:500; } + .pill { text-transform:none; letter-spacing:0; font-size:12px; border-radius:20px; } + .band,.empty { border-radius:14px; } + .band-head h2,label { text-transform:none; letter-spacing:0; } + .band-body { padding:22px; } + .presence-nav { position:fixed; top:0; bottom:0; left:0; width:190px; padding:26px 18px; background:#0b1219; border-right:1px solid var(--line); display:flex; flex-direction:column; gap:7px; z-index:20; } + .presence-brand { padding:0 12px 30px; font-size:24px; letter-spacing:-1px; font-weight:500; } + .presence-brand small { display:block; color:var(--muted); font-size:10px; letter-spacing:2px; margin-top:3px; } + .presence-nav button { display:flex; align-items:center; gap:12px; text-align:left; background:transparent; border:1px solid transparent; min-height:43px; color:var(--muted); padding:10px 12px; } + .presence-nav button[aria-current="page"] { color:var(--text); background:#18303a; border-color:#25515e; } + .presence-nav button:last-child { margin-top:auto; } + .presence-nav svg { width:18px; height:18px; fill:none; stroke:currentColor; stroke-width:1.6; } + .presence-heading { display:flex; justify-content:space-between; align-items:center; flex-wrap:wrap; gap:16px; margin-bottom:24px; } + .presence-heading h2 { margin:0 0 7px; font-size:clamp(26px,3vw,38px); letter-spacing:-1.2px; font-weight:500; } + .presence-heading p { margin:0; color:var(--muted); line-height:1.6; } + .presence-view[hidden], [hidden] { display:none!important; } + .presence-toolbar { display:flex; flex-wrap:wrap; gap:12px; align-items:center; margin-bottom:22px; } + .presence-toolbar input { min-width:180px; max-width:380px; flex:1; margin:0; } + .presence-toolbar select { width:auto; min-width:140px; margin:0; } + .presence-grid { display:grid; grid-template-columns:repeat(auto-fill,minmax(265px,1fr)); gap:16px; } + .presence-card { background:linear-gradient(140deg,#14222c,#101a23); border:1px solid var(--line); border-radius:16px; padding:22px; display:flex; flex-direction:column; gap:14px; min-width:0; } + .presence-card h3 { font-size:17px; margin:0; line-height:1.4; overflow-wrap:anywhere; } + .presence-card p { color:var(--muted); margin:0; line-height:1.65; overflow-wrap:anywhere; } + .presence-card .actions { margin-top:auto; } + .presence-eyebrow { font-size:11px; color:var(--muted); letter-spacing:1.1px; text-transform:uppercase; } + .presence-card-top { display:flex; justify-content:space-between; align-items:center; gap:8px; } + .presence-status { font-size:12px; color:var(--muted); display:flex; align-items:center; gap:7px; } + .presence-status::before { content:""; width:6px; height:6px; background:currentColor; border-radius:50%; flex-shrink:0; } + .presence-status[data-ready="true"] { color:var(--accent); } + .presence-card dl { display:grid; grid-template-columns:1fr 1fr; gap:7px; margin:0; font-size:12px; } + .presence-card dt { color:var(--muted); } + .presence-card dd { margin:0; text-align:right; overflow-wrap:anywhere; } + .presence-memory { height:5px; background:#263640; border-radius:4px; overflow:hidden; } + .presence-memory > span { display:block; height:100%; background:var(--accent); } + .presence-section-title { font-size:18px; margin:30px 0 16px; } + .presence-message { color:var(--muted); padding:20px 0; line-height:1.7; max-width:800px; } + .presence-notice { position:fixed; z-index:60; bottom:20px; left:218px; right:28px; max-width:660px; background:#162b35; color:var(--text); border:1px solid #345767; border-radius:12px; padding:16px 48px 16px 18px; box-shadow:0 10px 35px #0008; overflow-wrap:anywhere; } + .presence-notice[data-error="true"] { border-color:#a36050; } + .presence-notice button { position:absolute; right:7px; top:7px; background:none; border:0; } + .presence-connection { display:grid; grid-template-columns:minmax(0,1.3fr) minmax(260px,1fr); gap:20px; } + .presence-connection pre { white-space:pre-wrap; overflow-wrap:anywhere; font-size:12px; background:#091118; border-radius:10px; padding:18px; max-height:360px; overflow:auto; } + .presence-connection label { display:block; font-size:12px; margin-top:12px; } + .presence-connection select { margin-top:8px; } + .presence-readiness { display:flex; flex-wrap:wrap; gap:6px; } + .presence-readiness button { flex:1; font-size:12px; padding:9px 7px; } + .presence-readiness button[aria-pressed="true"] { background:#20414b; border-color:var(--accent); } + .presence-drawer { position:fixed; z-index:45; inset:0; display:grid; place-items:center; padding:20px; background:#02080cc9; } + .presence-dialog { width:min(620px,100%); max-height:90vh; overflow:auto; border:1px solid var(--line); background:var(--panel); border-radius:20px; padding:28px; } + .presence-dialog h2 { margin-top:0; letter-spacing:-.5px; } + .presence-dialog label { display:block; margin:14px 0; } + .presence-dialog pre { max-height:230px; overflow:auto; white-space:pre-wrap; overflow-wrap:anywhere; font-size:12px; } + .presence-dialog details { margin:16px 0; color:var(--muted); } + .presence-dialog .actions { justify-content:flex-end; flex-wrap:wrap; } + .presence-trial { margin-top:18px; } + .presence-trial textarea { width:100%; padding:12px; background:var(--bg); color:var(--text); border:1px solid var(--line); resize:vertical; min-height:90px; } + .presence-trial pre { white-space:pre-wrap; font-size:13px; line-height:1.7; } + .topology { border:1px solid var(--line); border-radius:18px; min-height:600px; background:#080f15; box-shadow:none; } + .topology::before,.topology::after { display:none; } + .topology-canvas { min-height:600px; height:calc(100vh - 230px); background:transparent; } + .topology-hud { top:16px; left:16px; right:16px; flex-wrap:wrap; gap:8px; } + .topology-hud-panel { flex:1 1 220px; } + .topology-hud-right { flex-wrap:wrap; max-width:100%; } + .topology-hud-panel .mono { font-family:inherit; font-size:12px; } + .topology-hud-panel { background:#101b24e8; border-color:var(--line); border-radius:10px; box-shadow:none; } + .fabric-title { font-family:inherit; font-size:13px; letter-spacing:0; font-weight:500; } + .topology-model-filter,.topology-metrics,.topology-zoom { border-radius:9px; } + .fabric-totals { gap:14px; } + .fabric-total strong { font-family:inherit; font-weight:500; } + .model-inspector { position:fixed; z-index:40; top:88px; right:24px; bottom:24px; max-height:calc(100vh - 112px); width:min(390px,calc(100vw - 32px)); border-radius:18px; background:#101b24; border-color:#345461; box-shadow:0 24px 80px #0008; } + .model-inspector:not(.open) { visibility:hidden; } + .model-inspector-title { font-weight:500; font-size:22px; } + .model-detail-grid { grid-template-columns:1fr 1fr; } + .model-detail strong { font-weight:400; } + .model-inspector-actions { grid-template-columns:repeat(3,minmax(0,1fr)); } + .operations-dock { margin:0; border-radius:16px; } + .operations-dock > summary { display:none; } + .operations-content { padding:0; } + .operations-content .grid.two { grid-template-columns:1fr; } + @media(max-width:1000px) { .presence-nav { width:155px; padding:24px 10px; } body > header, main { margin-left:155px; } main { padding:22px 18px; } .presence-notice { left:175px; } .topology-hud { flex-wrap:wrap; gap:8px; } .topology-hud-right { flex-wrap:wrap; } .topology-hud-panel > .muted { display:none; } .presence-connection { grid-template-columns:1fr; } } + @media(max-width:640px) { .presence-nav { position:sticky; width:100%; top:0; bottom:auto; flex-direction:row; padding:8px; gap:4px; border-right:0; border-bottom:1px solid var(--line); } .presence-brand { display:none; } .presence-nav button { flex:1; flex-direction:column; justify-content:center; padding:6px 2px; gap:4px; font-size:11px; } .presence-nav button:last-child { margin-top:0; } body > header, main { margin-left:0; } body > header { padding:12px 16px; min-height:0; } .topline { justify-content:space-between; gap:8px; } #endpoint { max-width:65%; overflow:hidden; text-overflow:ellipsis; } main { padding:20px 12px; } .topology-hud { top:10px; left:10px; right:10px; } .topology-canvas { min-height:550px; height:65vh; } .topology { min-height:550px; } .topology-zoom { left:8px; right:auto; } .presence-heading { margin-bottom:20px; } .presence-grid { grid-template-columns:1fr; } .presence-notice { left:12px; right:12px; bottom:12px; } .model-inspector { top:76px; bottom:12px; right:12px; width:calc(100vw - 24px); max-height:calc(100vh - 88px); } .presence-dialog { padding:20px; } .presence-drawer { padding:12px; } } + @media(prefers-reduced-motion:reduce) { *,*::before,*::after { animation:none!important; transition:none!important; scroll-behavior:auto!important; } } +`; + +const icons = { + live: '', + models: '', + machines: '', + clients: '', + settings: '' +}; +export const presenceNav = ``; + +export const presenceViews = ` + + + + + +`; diff --git a/src/dashboard-scene.mjs b/src/dashboard-scene.mjs new file mode 100644 index 0000000..cf030bd --- /dev/null +++ b/src/dashboard-scene.mjs @@ -0,0 +1,280 @@ +// The product composition: physical machines around a single gateway, with +// motion driven by observed requests. The old diagnostic canvas stays available. +export const sceneStyles = ` + body { background:radial-gradient(ellipse at 48% 8%,#0a1c24 0,transparent 48%),#050b0f; } + body > header { min-height:38px;padding:8px 26px;background:transparent;border:0; } + body > header .topline {font-size:11px;opacity:.65} + body > header #refresh {padding:4px 10px;font-size:11px} + main {padding-top:8px} + .presence-nav {background:linear-gradient(180deg,#081218,#070d12);padding-top:24px} + .presence-nav button[aria-current] {color:#2be1f6;background:linear-gradient(90deg,#11313c,#102029);position:relative;border-color:#193540} + .presence-nav button[aria-current]::before {content:"";position:absolute;left:-9px;top:8px;bottom:8px;width:3px;border-radius:4px;background:#2be1f6;box-shadow:0 0 14px #2be1f666} + .presence-brand {display:flex;gap:12px;align-items:center;padding:0 8px 30px;font-weight:600;letter-spacing:-.6px} + .presence-brand small {font-weight:400;text-transform:none;letter-spacing:0;font-size:12px} + .presence-brand .scene-logo {width:30px;height:40px;filter:drop-shadow(0 0 8px #29dff344)} + .presence-heading h2 {font-size:36px;font-weight:600;letter-spacing:-1.1px} + .presence-heading {margin-bottom:22px} + button.primary,.scene-primary {background:linear-gradient(115deg,#29def3,#26cfee);border:1px solid #64e5f3;color:#03202a;box-shadow:0 5px 23px #0dcced16;font-weight:600} + button.primary:hover,.scene-primary:hover {background:#72eafb;box-shadow:0 0 22px #2ddaf32b} + .scene-icon {display:inline-flex;align-items:center;justify-content:center;flex-shrink:0} + .scene-icon svg {width:22px;height:22px;fill:none;stroke:currentColor;stroke-width:1.5;stroke-linecap:round;stroke-linejoin:round} + .scene-layout {display:grid;grid-template-columns:minmax(0,1fr) 310px;gap:14px;align-items:stretch} + .scene-diagram,.scene-detail,.scene-analytics,.scene-memory-panel,.scene-network {border:1px solid #24343f;border-radius:16px;background:linear-gradient(135deg,#0b171d88,#060d12b0);box-shadow:inset 0 1px #8ad7f304} + .scene-diagram {position:relative;height:clamp(440px,calc(100vh - 320px),620px);overflow:hidden} + .scene-columns {position:relative;z-index:1;display:grid;grid-template-columns:minmax(130px,.78fr) minmax(100px,.66fr) minmax(240px,1.42fr);gap:18px;height:100%;padding:22px 20px} + .scene-column {min-width:0;min-height:0;display:flex;flex-direction:column} + .scene-column > h3 {font-size:14px;margin:0 0 6px;font-weight:500} + .scene-column > p {font-size:12px;color:#9fb5c7;margin:0} + .scene-clients {display:flex;flex:1;flex-direction:column;justify-content:space-evenly;gap:18px;padding:38px 0} + .scene-client {display:flex;gap:12px;align-items:center;padding:17px 13px;border:1px solid #284553;border-radius:13px;background:linear-gradient(135deg,#162933aa,#08151bec);box-shadow:0 12px 24px #0002;min-height:72px} + .scene-client .scene-icon {width:34px;height:38px;border-radius:9px;background:linear-gradient(135deg,#263a47,#11212c);color:#c9ecf8} + .scene-client strong {display:block;font-size:13px;font-weight:500;overflow-wrap:anywhere} + .scene-client small {display:block;margin-top:6px;color:#9cb5c7;font-size:11px;line-height:1.4} + .scene-client[data-active="true"] {border-color:#2d7283} + .scene-gateway-column {text-align:center} + .scene-gateway-wrap {flex:1;display:flex;flex-direction:column;align-items:center;justify-content:center;padding-bottom:28px} + .scene-gateway {position:relative;width:98px;height:98px;border-radius:50%;border:2px solid #32def7;background:radial-gradient(circle at 32% 25%,#143642,#03131c 70%);display:grid;place-items:center;box-shadow:0 0 0 7px #0a2c3629,0 0 26px #14d3f22b,inset 0 0 24px #10cbea13;isolation:isolate} + .scene-gateway::before,.scene-gateway::after {content:"";position:absolute;inset:-12px;border-radius:50%;border:1px solid #28d4ef18;pointer-events:none} + .scene-gateway::after {inset:-28px;border-color:#27d9f208} + .scene-gateway[data-active="true"] {animation:scene-breathe 3s ease-in-out infinite;box-shadow:0 0 0 7px #0a2c3629,0 0 38px #14d3f258,inset 0 0 28px #10cbea25} + .scene-gateway svg {width:37px;height:48px;filter:drop-shadow(0 0 8px #21dff36b)} + .scene-gateway-wrap > strong {font-size:17px;margin-top:18px;font-weight:500} + .scene-gateway-wrap > small {font-size:12px;color:#a4c0d2;margin-top:8px;text-align:center} + .scene-machines {display:flex;flex-direction:column;justify-content:center;gap:12px;flex:1;min-height:0;overflow:auto;padding-top:18px;justify-content:flex-start} + .scene-machine {border:1px solid #29404d;border-radius:13px;background:linear-gradient(125deg,#10222d80,#0a141cdb);padding:13px;min-width:0} + .scene-machine-header {display:flex;gap:10px;align-items:center;margin-bottom:12px} + .scene-machine-header .scene-icon {color:#b9d8e9} + .scene-machine-header strong {display:block;font-size:13px;font-weight:500} + .scene-machine-header small {display:block;color:#9bb6c8;font-size:11px;margin-top:4px;line-height:1.4} + .scene-machine-header > div:nth-child(2) {flex:1;min-width:0;overflow-wrap:anywhere} + .scene-mini-memory {width:76px;text-align:right;flex-shrink:0} + .scene-mini-memory > i {height:5px;border-radius:5px;background:#1a303e;display:block;overflow:hidden;margin-bottom:5px} + .scene-mini-memory b {height:100%;display:block;border-radius:5px;background:linear-gradient(90deg,#0dc7e7,#4ce5f5);box-shadow:0 0 8px #37d7f85c} + .scene-mini-memory small {font-size:10px!important} + .scene-model {width:100%;display:flex;align-items:center;gap:10px;margin-top:7px;border:1px solid #203440;border-radius:10px;background:linear-gradient(100deg,#11212a8a,#09151ad9);padding:10px 9px;text-align:left;min-height:43px;box-shadow:inset 0 1px #b5eaff03} + .scene-model[data-serving="true"],.scene-model[data-selected="true"] {border-color:#27cbe6;background:linear-gradient(100deg,#07303e,#0a1720);box-shadow:inset 0 0 20px #22dffa07,0 0 15px #19cfea0a} + .scene-model > .scene-icon svg {width:18px;height:18px} + .scene-model-name {flex:1;min-width:0;white-space:nowrap;overflow:hidden;text-overflow:ellipsis;font-size:12px} + .scene-model-state {font-size:10px;color:#a9c2d4;background:#1a2c37;border:1px solid #223d4b;border-radius:15px;padding:4px 8px;white-space:nowrap} + [data-serving="true"] > .scene-model-state {color:#49e5ee;border-color:#087d91;background:#06313b} + .scene-more {border:0;background:none;font-size:11px;color:#9ec2d4;padding:10px 4px 0;display:block} + .scene-links {position:absolute;inset:0;width:100%;height:100%;pointer-events:none;z-index:0;overflow:visible} + .scene-detail {min-width:0;padding:22px;height:clamp(440px,calc(100vh - 320px),620px);overflow:auto} + .scene-placeholder {height:100%;display:flex;flex-direction:column;justify-content:center;text-align:center;align-items:center;gap:18px} + .scene-placeholder .scene-icon {width:72px;height:72px;border:1px solid #28414e;border-radius:50%;color:#7fdfea;background:radial-gradient(circle at 40% 30%,#203d4b,#0b1922)} + .scene-placeholder .scene-icon svg {width:32px;height:32px} + .scene-placeholder h3 {font-size:22px;margin:0} + .scene-placeholder p {font-size:13px;color:#99b2c5;line-height:1.7;margin:0} + .scene-detail .model-inspector {position:static;visibility:visible;width:100%;max-height:none;height:auto;border:0;border-radius:0;background:transparent;box-shadow:none;overflow:visible} + .scene-detail .model-inspector:not(.open) {display:none} + .scene-detail .model-inspector-header {padding:0 0 22px;border-bottom:1px solid #233642} + .scene-detail .model-inspector-body {padding:0;overflow:visible} + .scene-detail .model-inspector-title {font-size:22px;line-height:1.35} + .scene-detail .model-inspector-state {margin-bottom:12px} + .scene-detail #presence-policy {padding:20px 0;border-bottom:1px solid #233642} + .scene-detail #presence-policy h3 {font-size:14px;margin:0 0 14px} + .presence-readiness {gap:0;border:1px solid #2c4757;border-radius:10px;overflow:hidden} + .presence-readiness button {border-radius:0;border:0;border-right:1px solid #243c49;padding:12px 4px;background:transparent;font-size:11px;white-space:nowrap} + .presence-readiness button:last-child {border:0} + .presence-readiness button[aria-pressed="true"] {background:#25d3ee;color:#012333} + #presence-policy-hint {font-size:12px;line-height:1.6;margin:16px 0 0} + .scene-detail .model-inspector-actions {padding:20px 0;gap:7px} + .scene-detail details {border-top:1px solid #233642;padding-top:18px;margin-top:18px} + .scene-detail summary {font-size:13px;cursor:pointer} + .scene-inspector-memory {padding:18px 0;border-bottom:1px solid #233642;display:flex;justify-content:space-between;gap:12px;font-size:12px;color:#a9c1cf} + .scene-inspector-memory strong {color:#e1f2f6;font-weight:400} + .scene-analytics {margin-top:14px;display:grid;grid-template-columns:1fr 1fr 1fr;padding:20px 24px;gap:24px} + .scene-chart {min-width:0} + .scene-chart + .scene-chart {border-left:1px solid #20323e;padding-left:24px} + .scene-chart h3 {display:flex;justify-content:space-between;gap:10px;margin:0 0 16px;font-size:13px;font-weight:500} + .scene-chart h3 span {font-size:11px;color:#9cbed2;font-weight:400} + .scene-chart svg {width:100%;height:80px;overflow:visible} + .scene-chart-caption {font-size:10px;color:#6e899d;margin-top:6px} + .scene-memory-row {display:grid;grid-template-columns:minmax(70px,1fr) 1.4fr auto;align-items:center;gap:12px;font-size:11px;margin:12px 0;color:#b9cedb} + .scene-memory-row > span:first-child {overflow:hidden;text-overflow:ellipsis;white-space:nowrap} + .scene-memory-row i {height:7px;border-radius:4px;background:#182f3c;overflow:hidden} + .scene-memory-row b {display:block;height:100%;border-radius:4px;background:linear-gradient(90deg,#11ccea,#53e7f6)} + .scene-tools {display:flex;align-items:center;justify-content:space-between;gap:12px;margin:17px 0 4px;color:#7998ad;font-size:11px;flex-wrap:wrap} + .scene-tools .actions {align-items:center;gap:8px;margin:0} + .scene-tools button {padding:8px 12px;font-size:11px;background:#0b171f;border-color:#223a48} + .scene-follow {display:flex;gap:8px;align-items:center;font-size:11px;color:#acccdc} + .scene-follow input {accent-color:#22d9f3;width:auto;margin:0} + .scene-memory-panel {padding:22px;margin-bottom:24px} + .scene-memory-panel h3 {display:flex;justify-content:space-between;font-size:16px;margin:0 0 18px;align-items:center} + .scene-memory-panel select {width:auto;max-width:180px;font-size:11px;padding:5px 8px;background:#0a161e} + .scene-memory-bar {display:flex;gap:2px;min-height:72px;border-radius:10px;overflow:hidden} + .scene-memory-segment {min-width:100px;flex:1;padding:14px 18px;background:linear-gradient(110deg,#35444f,#253945);color:#dde9f0;font-size:12px} + .scene-memory-segment strong {display:block;font-size:23px;margin-top:6px;font-weight:600} + .scene-memory-segment.available {background:linear-gradient(100deg,#349f9c,#55b4a7);color:#042322} + .scene-model-layout {display:grid;grid-template-columns:minmax(0,1fr) 310px;gap:22px} + .scene-model-list {display:flex;flex-direction:column;gap:9px} + .scene-model-row {display:grid;grid-template-columns:44px minmax(130px,1.4fr) minmax(75px,.7fr) auto;gap:16px;align-items:center;border:1px solid #203541;border-radius:13px;padding:19px;background:linear-gradient(120deg,#111d2490,#09141b80);cursor:pointer;text-align:left;min-height:92px} + .scene-model-row[data-selected="true"] {border-color:#29d3eb;background:linear-gradient(100deg,#0a2a37c0,#0c1c2690)} + .scene-model-row > .scene-icon {width:44px;height:48px;background:linear-gradient(135deg,#24374688,#101d2a);border:1px solid #2b414d;border-radius:11px;color:#d6e9f4} + .scene-model-row > .scene-icon svg {width:25px;height:25px} + .scene-model-row h3 {margin:0;font-size:16px;overflow-wrap:anywhere;font-weight:500;line-height:1.35} + .scene-model-row p {font-size:12px;color:#96aebe;margin:6px 0 0} + .scene-row-status {font-size:12px;color:#a6c2d1;display:flex;align-items:center;gap:7px} + .scene-row-status::before {content:"";width:7px;height:7px;border-radius:50%;background:#718b9c;flex-shrink:0} + .scene-model-row[data-ready="true"] .scene-row-status::before {background:#44dccf;box-shadow:0 0 8px #44dccf22} + .scene-row-policy {font-size:10px;color:#a1bbc9;border:1px solid #28414e;border-radius:8px;padding:7px 9px;white-space:nowrap} + .scene-model-tabs {display:flex;align-items:center;gap:25px;border-bottom:1px solid #273b45;margin:0 0 18px} + .scene-model-tabs strong {padding:0 0 13px;font-size:16px;border-bottom:3px solid #1ed5f0} + .scene-model-tabs button {background:none;border:0;color:#8aa5b7;padding:0 0 15px} + .scene-chips {display:flex;gap:8px;flex-wrap:wrap;margin:0 0 18px} + .scene-chips button {border-radius:22px;padding:8px 16px;background:#0b141a;font-size:12px;border-color:#34454e} + .scene-chips button[aria-pressed="true"] {border-color:#29dcee;background:#0d2c36;color:#55e9f5} + .scene-add-capability {display:flex;gap:15px;align-items:center;border:1px dashed #35515d;border-radius:13px;padding:22px;margin-top:18px;background:#09131977} + .scene-add-capability > .scene-icon {width:40px;height:40px;border:1px solid #8aa9b8;border-radius:50%} + .scene-add-capability strong {font-size:16px;font-weight:500;display:block} + .scene-add-capability p {font-size:12px;color:#98b3c4;margin:7px 0 0} + .scene-add-capability button {margin-left:auto;padding:11px 16px} + .scene-network {display:flex;align-items:center;justify-content:center;gap:0;padding:40px 30px;margin-bottom:30px;min-height:190px;overflow:auto} + .scene-device {min-width:110px;text-align:center} + .scene-device .scene-icon {display:flex;filter:drop-shadow(0 0 12px #20d3f72b);color:#a7edff;margin-bottom:12px} + .scene-device svg {width:88px;height:70px;stroke-width:.65} + .scene-device strong {font-size:13px;font-weight:500} + .scene-wire {height:1px;min-width:60px;flex:1;max-width:160px;background:linear-gradient(90deg,#22d3f4,#1c8fa5);box-shadow:0 0 9px #20d5ed60;position:relative;margin:0 16px 22px} + .scene-wire::before,.scene-wire::after {content:"";position:absolute;top:-3px;width:7px;height:7px;border-radius:50%;background:#28def4;box-shadow:0 0 12px #25d5ee} + .scene-wire::after {right:0} + .scene-wire span {position:absolute;top:-22px;left:0;right:0;text-align:center;font-size:10px;color:#72dbe9;white-space:nowrap} + .scene-machine-list {display:flex;flex-direction:column;gap:12px} + .scene-machine-list .presence-card {display:grid;grid-template-columns:54px minmax(120px,1fr) minmax(150px,1fr) auto;gap:24px;align-items:center;padding:23px} + .scene-machine-list .presence-card > .scene-icon {color:#aae9f8;filter:drop-shadow(0 0 9px #1cc5df40)} + .scene-machine-list .presence-card > .scene-icon svg {width:52px;height:45px;stroke-width:.8} + .scene-machine-list .presence-card h3 {font-size:17px} + .scene-machine-list .presence-card p {font-size:12px;margin-top:7px} + @keyframes scene-breathe {50%{box-shadow:0 0 0 10px #0a2c3630,0 0 46px #14d3f262,inset 0 0 28px #10cbea25}} + @media(min-width:1650px) {.scene-layout,.scene-model-layout {grid-template-columns:minmax(0,1fr) 360px}.scene-columns {gap:30px;padding:30px}.scene-diagram,.scene-columns,.scene-detail {min-height:610px}.scene-model-name{font-size:13px}} + @media(max-width:1220px) {.scene-layout,.scene-model-layout {grid-template-columns:minmax(0,1fr) 280px;gap:12px}.scene-columns {grid-template-columns:120px 90px minmax(210px,1fr);gap:8px;padding:20px 15px}.scene-gateway {width:78px;height:78px}.scene-detail {padding:18px}.scene-mini-memory{width:58px}.scene-model-row {grid-template-columns:35px minmax(0,1fr) auto;gap:12px;padding:15px}.scene-row-policy{display:none}.scene-model-row > .scene-icon{width:35px;height:40px}.scene-model-row h3{font-size:14px}.scene-model-state{font-size:9px;padding:4px 6px}.scene-model-name{font-size:11px}} + @media(max-width:1050px) {.scene-layout,.scene-model-layout {grid-template-columns:1fr}.scene-detail {min-height:0}.scene-detail:has(.scene-placeholder:not([hidden])){display:none}.scene-columns{grid-template-columns:minmax(125px,.8fr) minmax(100px,.8fr) minmax(230px,1.4fr);gap:20px}.scene-analytics{padding:20px;gap:18px}.scene-chart + .scene-chart{padding-left:18px}.scene-machine-list .presence-card{grid-template-columns:44px 1fr auto;gap:18px}.scene-machine-list .presence-card .machine-memory{grid-column:2 / -1;grid-row:2}.scene-model-row {grid-template-columns:40px minmax(0,1fr) auto auto}.scene-row-policy{display:block}} + @media(max-width:680px) {.presence-brand{display:none}.presence-nav{padding:8px}.presence-nav button[aria-current]::before{left:12px;right:12px;top:auto;bottom:0;width:auto;height:2px}body > header{display:none}main{padding-top:22px}.presence-heading h2{font-size:29px}.scene-columns{grid-template-columns:1fr 1fr;gap:18px;padding:20px;min-height:0}.scene-gateway-column{grid-column:2;grid-row:1}.scene-client-column{grid-column:1;grid-row:1}.scene-machine-column{grid-column:1/-1}.scene-gateway-wrap{min-height:210px;padding:30px 0 0}.scene-clients{padding:25px 0;gap:12px}.scene-diagram{min-height:0}.scene-machines{padding-top:16px}.scene-analytics{grid-template-columns:1fr;gap:22px}.scene-chart + .scene-chart{border-left:0;border-top:1px solid #20323e;padding:20px 0 0}.scene-chart svg{height:74px}.scene-model-row{grid-template-columns:34px minmax(0,1fr) auto;padding:15px 12px;gap:10px}.scene-model-row .scene-row-policy{display:none}.scene-row-status{font-size:10px}.scene-memory-panel{padding:18px}.scene-memory-segment{padding:12px;font-size:11px;min-width:80px}.scene-memory-segment strong{font-size:21px}.scene-add-capability{padding:18px;flex-wrap:wrap}.scene-add-capability button{width:100%;margin:0}.scene-network{padding:25px 18px;justify-content:flex-start}.scene-machine-list .presence-card{grid-template-columns:40px 1fr;gap:14px;padding:18px}.scene-machine-list .presence-card .actions{grid-column:1/-1}.scene-machine-list .presence-card .machine-memory{grid-column:1/-1;grid-row:auto}.scene-links{opacity:.85}.scene-model-name{font-size:12px}.scene-model-state{font-size:10px}} + @media(max-width:680px){.scene-detail:has(.model-inspector.open){position:fixed;z-index:80;left:12px;right:12px;bottom:12px;height:auto;max-height:calc(100dvh - 96px);background:linear-gradient(145deg,#122530,#09141c);box-shadow:0 -20px 70px #0009,0 0 0 1px #42616b66;padding:22px}.scene-detail:has(.model-inspector.open) .model-inspector-header{position:sticky;top:-22px;margin-top:-22px;padding-top:22px;background:#10212b;z-index:1}} + @media(prefers-reduced-motion:reduce){.scene-gateway{animation:none!important}.scene-links .scene-particle{display:none}.scene-model{transition:none}} +`; + +export const sceneScript = String.raw` + const scenePaths={ + model:'', + chat:'', + code:'', + image:'', + audio:'', + laptop:'', + server:'', + cloud:'', + terminal:'', + plus:'' + }; + function sceneIcon(kind){return '';} + const sceneLogo=''; + function sceneKind(model){return ({chat:'Chat & code',embedding:'Search & retrieval',audio_speech:'Speech',audio_transcription:'Transcription',audio_generation:'Music & audio',image:'Generate & edit',video:'Video'})[model.kind] || model.kind || 'Chat & code';} + function sceneModelIcon(model){return (model.kind||'').startsWith('audio')?'audio':(model.kind||'').startsWith('image')?'image':model.kind==='chat'?'model':'code';} + function sceneMemory(node){ + const memory=node?.telemetry?.memory || {}; + const total=Number(memory.totalBytes), available=Number(memory.availableBytes), rawUsed=Number(memory.usedBytes); + const used=memory.availableBytes!=null&&Number.isFinite(available)?total-available:memory.usedBytes!=null?rawUsed:NaN; + return {total,used,available,known:total>0&&Number.isFinite(used),percent:Math.max(0,Math.min(100,used/total*100))}; + } + function sceneNodes(){const nodes=Object.values(state.status?.cluster?.nodes || {});return nodes.length?nodes:state.topologySummary?.host?[{id:'local',local:true,name:'This machine',telemetry:state.topologySummary.host}]:[];} + function sceneNodeName(node){return node.local?'This machine':node.name || node.id;} + function sceneMiniMemory(node){const m=sceneMemory(node);return m.known?'
'+escapeHtml(formatBytes(m.used))+' / '+escapeHtml(formatBytes(m.total))+'
':'';} + function sceneDeviceIcon(node){return node.local || /apple|mac/i.test(node.profile?.platformId || node.profile?.cpuBrand || '')?'laptop':'server';} + document.querySelector('.presence-brand').innerHTML=sceneLogo+'LLooMby Enntity'; + const scene=document.createElement('section');scene.id='presence-scene';scene.dataset.presencePanel='live'; + scene.innerHTML='

Clients

Waiting for telemetry

LLooM Gateway

Routes and balances

'+sceneLogo+'
GatewayConnecting…

Models on your machines

Requests

Observed during this session

Response time

Recent completed requests · includes generation time

Machine memory Used / total

Waiting for gateway telemetry
'; + $('.topology').before(scene); + $('.topology').dataset.presencePanel='diagnostic';$('.topology').hidden=true; + const modelLeft=document.createElement('div');modelLeft.className='scene-model-main'; + const modelLayout=document.createElement('div');modelLayout.className='scene-model-layout';$('#view-models').append(modelLayout);modelLayout.append(modelLeft); + const memoryPanel=document.createElement('section');memoryPanel.className='scene-memory-panel';memoryPanel.innerHTML='

Your memory

Downloads stay on disk when models leave memory.

'; + modelLeft.append(memoryPanel); + const modelTabs=document.createElement('div');modelTabs.className='scene-model-tabs';modelTabs.innerHTML='Installed'; + modelLeft.append(modelTabs,$('.presence-toolbar')); + const chips=document.createElement('div');chips.className='scene-chips';chips.innerHTML=[['','All'],['chat','Chat & code'],['image','Images'],['audio','Voice'],['embedding','Search'],['video','Video']].map(([id,name])=>'').join(''); + modelLeft.append(chips,$('#presence-models'));$('#presence-models').className='scene-model-list';$('#presence-kind').hidden=true; + const capability=document.createElement('div');capability.className='scene-add-capability';capability.innerHTML=sceneIcon('plus')+'
Add a capability

Get more done with another model.

';modelLeft.append(capability); + const modelDetail=document.createElement('aside');modelDetail.id='scene-model-detail';modelDetail.className='scene-detail';modelDetail.innerHTML='
'+sceneIcon('model')+'

Make room for more.

Select a model to choose how it uses memory. Your downloaded files stay on disk.

'; + modelLayout.append(modelLeft,modelDetail);$('#view-models').append(modelLayout); + const network=document.createElement('div');network.id='scene-network';network.className='scene-network';$('#presence-machines').before(network);$('#presence-machines').className='scene-machine-list'; + const inspectorMemory=document.createElement('div');inspectorMemory.className='scene-inspector-memory';inspectorMemory.id='scene-inspector-memory';$('#presence-policy').before(inspectorMemory); + let sceneClientKey='',sceneMachineKey='',sceneModelKey='',sceneNodeKey='',sceneLinkKey='',sceneSamples=[],sceneSampleAt=0; + let sceneMemoryNode=null,sceneFollowing=false; + function sceneMoveInspector(){ + const slot=presenceView==='models'?modelDetail:$('#scene-live-detail'); + if($('#model-inspector').parentElement!==slot)slot.append($('#model-inspector')); + document.querySelectorAll('.scene-placeholder').forEach(item=>item.hidden=Boolean(state.selectedModelId)&&item.parentElement===slot); + if(state.selectedModelId){const model=state.physicalModels.find(m=>m.id===state.selectedModelId),rt=model&&presenceRuntime(model);inspectorMemory.innerHTML='Memory estimate'+(rt?.memoryGb!=null?escapeHtml(rt.memoryGb)+' GB':'Not reported')+'';} + } + const scenePreviousView=presenceSetView; + presenceSetView=function(name){scenePreviousView(name);sceneMoveInspector();renderScene();}; + const scenePreviousInspector=renderModelInspector; + renderModelInspector=function(){scenePreviousInspector();sceneMoveInspector();}; + renderPresenceModels=function(){ + const search=$('#presence-search').value.trim().toLowerCase(),kind=$('#presence-kind').value; + const models=(state.physicalModels||[]).filter(m=>(!search||(m.name+' '+m.id).toLowerCase().includes(search))&&(!kind||(m.kind||'chat').startsWith(kind))).sort((a,b)=>Number(Boolean(b.runtime))-Number(Boolean(a.runtime))||Number(Boolean(presenceRuntime(b)?.healthy))-Number(Boolean(presenceRuntime(a)?.healthy))); + $('#presence-model-count').textContent=models.length+(models.length===1?' model':' models'); + const key=JSON.stringify(models.map(m=>[m.id,m.name,sceneKind(m),presenceModelLabel(m),presencePolicy(presenceRuntime(m)),state.selectedModelId===m.id]));if(key===sceneModelKey)return;sceneModelKey=key; + $('#presence-models').innerHTML=models.map(model=>{const rt=presenceRuntime(model);return '';}).join('')||'
No models match. Choose another filter or add a model.
'; + }; + renderPresenceMachines=function(){ + const nodes=sceneNodes(),key=JSON.stringify(nodes.map(n=>[n.id,n.name,n.local,n.reachable,n.profile?.cpuBrand,sceneMemory(n)]));if(key===sceneNodeKey)return;sceneNodeKey=key; + network.innerHTML=nodes.map((node,index)=>(index?'
'+(node.reachable===false?'Unavailable':'Configured connection')+'
':'')+'
'+sceneIcon(sceneDeviceIcon(node))+''+escapeHtml(sceneNodeName(node))+'
').join('')||'

Waiting for machine telemetry

'; + $('#presence-machines').innerHTML=nodes.map(node=>{const m=sceneMemory(node);return '
'+sceneIcon(sceneDeviceIcon(node))+'

'+escapeHtml(sceneNodeName(node))+'

'+escapeHtml(node.profile?.cpuBrand||node.id)+' · '+(node.reachable===false?'Unavailable':node.local?'This computer':'Configured peer')+'

'+(m.known?escapeHtml(formatBytes(m.total))+' memory':'Memory unavailable')+'

'+(m.known?escapeHtml(formatBytes(m.used))+' used / '+escapeHtml(formatBytes(m.total)):'No reading')+'

';}).join(''); + }; + function sceneChart(selector,values,color){ + const svg=$(selector),width=300,height=78,max=Math.max(1,...values),points=values.map((value,index)=>[values.length>1?index/(values.length-1)*width:0,height-4-(value/max)*(height-10)]); + const path=points.map((p,i)=>(i?'L':'M')+p[0].toFixed(1)+','+p[1].toFixed(1)).join(' '); + svg.setAttribute('viewBox','0 0 300 80');svg.setAttribute('preserveAspectRatio','none'); + svg.innerHTML=''+[0,26,52,78].map(y=>'').join('')+(values.length>1?'':''); + } + function renderScene(){ + if(!scene.isConnected)return; + const nodes=sceneNodes(),models=state.physicalModels||[],connections=(state.topologyConnections||[]).filter(c=>c.live),summary=state.topologySummary||{}; + const clients=new Map();for(const connection of connections){const id=connection.caller||connection.requester||'API client';if(!clients.has(id))clients.set(id,{name:id,count:0});clients.get(id).count++;} + const clientRows=[...clients.values()].slice(0,4),clientKey=JSON.stringify(clientRows); + if(clientKey!==sceneClientKey||!$('#scene-clients').children.length){sceneClientKey=clientKey;$('#scene-clients').innerHTML=clientRows.map(client=>'
'+sceneIcon(/code|terminal|cli|agent/i.test(client.name)?'terminal':'chat')+'
'+escapeHtml(client.name)+''+client.count+' active request'+(client.count===1?'':'s')+'
').join('')||'
'+sceneIcon('terminal')+'
Ready for a clientConnect an app to see its requests here.
';sceneLinkKey='';} + $('#scene-client-count').textContent=connections.length?clients.size+' active client'+(clients.size===1?'':'s'):'No active requests'; + $('#scene-active').textContent=(summary.active||0)?summary.active+' active request'+(summary.active===1?'':'s'):'Ready when you are';$('#scene-gateway').dataset.active=String(Boolean(summary.active)); + $('#scene-machine-count').textContent=nodes.length+' machine'+(nodes.length===1?'':'s')+' · '+models.length+' models'; + const groups=nodes.map(node=>({node,models:models.filter(model=>{const topology=(state.topologyCatalogModels||[]).find(m=>m.id===model.id);const rt=presenceRuntime(model),ids=[...new Set([...(topology?.nodes||[]),...(model.targets||[]).map(t=>t.node),...(rt?.members||[]).map(m=>m.node),rt?.node].filter(Boolean))];return ids.includes(node.id)||(!ids.length&&Boolean(model.runtime)&&node.local);})})); + const assigned=new Set(groups.flatMap(g=>g.models.map(m=>m.id)));const external=models.filter(m=>!assigned.has(m.id));if(external.length)groups.push({node:{id:'external',name:'External providers'},models:external,external:true}); + const visibleGroups=groups.map(group=>({...group,models:group.models.slice().sort((a,b)=>Number(presenceRuntime(b)?.activeRequests>0)-Number(presenceRuntime(a)?.activeRequests>0)||Number(Boolean(presenceRuntime(b)?.healthy))-Number(Boolean(presenceRuntime(a)?.healthy)))})); + const machineKey=JSON.stringify(visibleGroups.map(g=>[g.node.id,g.node.reachable,g.models.map(m=>[m.id,m.name,presenceModelLabel(m),connections.some(c=>c.model===m.id),m.id===state.selectedModelId]),sceneFollowing])); + if(machineKey!==sceneMachineKey){sceneMachineKey=machineKey;$('#scene-machines').innerHTML=visibleGroups.filter(g=>!sceneFollowing||!summary.active||g.models.some(m=>presenceRuntime(m)?.activeRequests>0||connections.some(c=>c.model===m.id))).map(group=>{ + const ranked=sceneFollowing&&summary.active?group.models.filter(m=>presenceRuntime(m)?.activeRequests>0||connections.some(c=>c.model===m.id)):group.models; + const shown=ranked.slice(0,group.external?2:3); + return '
'+sceneIcon(group.external?'cloud':sceneDeviceIcon(group.node))+'
'+escapeHtml(sceneNodeName(group.node))+''+escapeHtml(group.external?'Through your gateway':group.node.profile?.cpuBrand||group.node.id)+'
'+sceneMiniMemory(group.node)+'
'+shown.map(model=>{const label=presenceModelLabel(model),active=connections.some(c=>c.model===model.id),serving=label==='Serving'||(active&&!model.runtime&&!model.federated);return '';}).join('')+(group.models.length>shown.length?'':'')+'
'; + }).join('')||'
Add a model to bring your hardware to life.
';sceneLinkKey='';} + for(const group of visibleGroups){const box=[...document.querySelectorAll('[data-scene-node]')].find(el=>el.dataset.sceneNode===group.node.id),mini=box?.querySelector('.scene-mini-memory');if(mini){const m=sceneMemory(group.node);mini.querySelector('b').style.width=m.percent+'%';mini.querySelector('small').textContent=formatBytes(m.used)+' / '+formatBytes(m.total);}} + const minute=state.metrics?.rolling?.minute||{};$('#scene-requests-label').textContent=(summary.active||0)+' active'+(minute.requests!=null?' · '+minute.requests+'/min':''); + const now=Date.now();if(now-sceneSampleAt>1900){sceneSamples.push(Number(summary.active||0));if(sceneSamples.length>60)sceneSamples.shift();sceneSampleAt=now;} + sceneChart('#scene-request-chart',sceneSamples,'#29d9f5'); + const durations=(state.metrics?.recent||[]).slice(0,40).reverse().map(r=>Number(r.durationMs)/1000).filter(n=>Number.isFinite(n)&&n>=0);sceneChart('#scene-latency-chart',durations,'#47e7cc');$('#scene-latency-label').textContent=durations.length?(durations.reduce((a,b)=>a+b,0)/durations.length).toFixed(1)+'s avg':'No completed requests'; + $('#scene-memory-list').innerHTML=nodes.slice(0,4).map(node=>{const m=sceneMemory(node);return '
'+escapeHtml(sceneNodeName(node))+''+(m.known?escapeHtml(formatBytes(m.used))+' / '+escapeHtml(formatBytes(m.total)):'Unknown')+'
';}).join('')||'

Memory telemetry unavailable

'; + $('#scene-health').textContent=state.status?.error?'Gateway telemetry needs attention':nodes.length?'Live gateway telemetry · '+(summary.recentErrors||0)+' errors in the last minute':'Connecting to your gateway…'; + const machineSelect=$('#scene-memory-machine');const optionsKey=nodes.map(n=>n.id).join('|');if(machineSelect.dataset.nodes!==optionsKey){machineSelect.dataset.nodes=optionsKey;machineSelect.innerHTML=nodes.map(n=>'').join('');if(nodes.some(n=>n.id===sceneMemoryNode))machineSelect.value=sceneMemoryNode;} + sceneMemoryNode=machineSelect.value;const memory=sceneMemory(nodes.find(n=>n.id===sceneMemoryNode));$('#scene-memory-bar').innerHTML=memory.known?'
In use'+escapeHtml(formatBytes(memory.used))+'
Available'+escapeHtml(formatBytes(memory.total-memory.used))+'
':'
Waiting for a physical machine memory reading.
'; + sceneMoveInspector();requestAnimationFrame(sceneDrawLinks); + } + function sceneDrawLinks(){ + if(scene.hidden||document.hidden)return; + const root=$('.scene-diagram'),bounds=root.getBoundingClientRect(),gateway=$('#scene-gateway').getBoundingClientRect(),gx=gateway.left-bounds.left+gateway.width/2,gy=gateway.top-bounds.top+gateway.height/2; + const curves=[];for(const client of document.querySelectorAll('.scene-client')){const box=client.getBoundingClientRect();curves.push({x:box.right-bounds.left,y:box.top-bounds.top+box.height/2,toX:gx-gateway.width/2,toY:gy,active:client.dataset.active==='true'});} + for(const model of document.querySelectorAll('#scene-machines .scene-model')){const box=model.getBoundingClientRect(),viewport=$('#scene-machines').getBoundingClientRect();if(box.topviewport.bottom)continue;curves.push({x:gx+gateway.width/2,y:gy,toX:box.left-bounds.left,toY:box.top-bounds.top+box.height/2,active:model.dataset.active==='true'});} + const key=JSON.stringify(curves.map(c=>[Math.round(c.x),Math.round(c.y),Math.round(c.toX),Math.round(c.toY),c.active]));if(key===sceneLinkKey)return;sceneLinkKey=key; + const reduced=window.matchMedia('(prefers-reduced-motion: reduce)').matches; + $('#scene-links').setAttribute('viewBox','0 0 '+bounds.width+' '+bounds.height); + $('#scene-links').innerHTML=''+curves.map((c,index)=>{const bend=(c.toX-c.x)*.5;const d='M'+c.x+','+c.y+' C'+(c.x+bend)+','+c.y+' '+(c.toX-bend)+','+c.toY+' '+c.toX+','+c.toY;return (c.active?'':'')+''+(c.active&&!reduced?[0,1,2].map(n=>'').join(''):'')+'';}).join(''); + } + const sceneOriginalPresence=renderPresence;renderPresence=function(){sceneOriginalPresence();renderScene();}; + const sceneOriginalActivity=renderActivity;renderActivity=function(){sceneOriginalActivity();renderScene();}; + $('#scene-memory-machine').addEventListener('change',()=>{sceneMemoryNode=$('#scene-memory-machine').value;renderScene();}); + $('#scene-follow').addEventListener('change',()=>{sceneFollowing=$('#scene-follow').checked;sceneMachineKey='';renderScene();}); + $('#scene-diagnostic').addEventListener('click',()=>{const detail=$('.topology');detail.hidden=!detail.hidden;if(!detail.hidden)detail.scrollIntoView({behavior:window.matchMedia('(prefers-reduced-motion: reduce)').matches?'auto':'smooth',block:'start'});}); + document.addEventListener('click',event=>{const kind=event.target.closest('[data-scene-kind]');if(kind){$('#presence-kind').value=kind.dataset.sceneKind;document.querySelectorAll('[data-scene-kind]').forEach(b=>b.setAttribute('aria-pressed',String(b===kind)));renderPresenceModels();}if(event.target.closest('[data-scene-all]'))presenceSetView('models');const selected=event.target.closest('[data-presence-model]');if(selected){sceneModelKey='';sceneMachineKey='';renderPresenceModels();renderScene();}}); + $('#scene-machines').addEventListener('scroll',()=>{sceneLinkKey='';requestAnimationFrame(sceneDrawLinks);}); + window.addEventListener('resize',()=>{sceneLinkKey='';requestAnimationFrame(sceneDrawLinks);}); + document.addEventListener('visibilitychange',()=>{if(!document.hidden){sceneLinkKey='';renderScene();}}); + presenceSetView(location.hash.slice(1)); +`; diff --git a/src/dashboard.mjs b/src/dashboard.mjs index 7f9a34a..70618ee 100644 --- a/src/dashboard.mjs +++ b/src/dashboard.mjs @@ -1,3 +1,7 @@ +import { presenceStyles, presenceNav, presenceViews } from './dashboard-presence.mjs'; +import { sceneStyles, sceneScript } from './dashboard-scene.mjs'; +import { presenceScript } from './dashboard-presence-client.mjs'; + const DASHBOARD_HTML = String.raw` @@ -996,7 +1000,7 @@ const DASHBOARD_HTML = String.raw` state.topologyCamera.frameKey = ""; } else fitTopologyCameraToModels(); } - if (state.selectedModelId && !state.topologyModels.some(model => model.id === state.selectedModelId)) closeModelInspector(); + if (state.selectedModelId && !state.physicalModels.some(model => model.id === state.selectedModelId)) closeModelInspector(); renderTopologyModelFilter(); renderModelInspector(); } @@ -1700,8 +1704,10 @@ const DASHBOARD_HTML = String.raw` const canvas = $("#topology-canvas"); if (!canvas || !canvas.isConnected) return; const viewportWidth = Math.max(1, canvas.clientWidth), viewportHeight = Math.max(1, canvas.clientHeight); - if (canvas.width !== Math.round(viewportWidth) || canvas.height !== Math.round(viewportHeight)) { canvas.width = Math.round(viewportWidth); canvas.height = Math.round(viewportHeight); } + const pixelRatio = Math.max(1, Math.min(3, Number(window.devicePixelRatio) || 1)); + if (canvas.width !== Math.round(viewportWidth * pixelRatio) || canvas.height !== Math.round(viewportHeight * pixelRatio)) { canvas.width = Math.round(viewportWidth * pixelRatio); canvas.height = Math.round(viewportHeight * pixelRatio); } const ctx = canvas.getContext("2d"); + ctx.setTransform(pixelRatio,0,0,pixelRatio,0,0); ctx.clearRect(0, 0, viewportWidth, viewportHeight); const models = state.topologyModels || []; const clusterNodes = state.status?.cluster?.enabled ? Object.values(state.status?.cluster?.nodes || {}) : []; @@ -1761,7 +1767,7 @@ const DASHBOARD_HTML = String.raw` return hashUnit(aSeed * 29) - hashUnit(bSeed * 29); }); const threadField = { left: 24, right: gate.left - 24, top: 112, bottom: height - 45 }; - ctx.font = '11px "SFMono-Regular",monospace'; + ctx.font = '12px system-ui,sans-serif'; ctx.textAlign = "left"; const connectionLabels = new Map(orderedConnections.map(connection => { const outputRate = smoothRate("connection:" + connection.id + ":out", connection.outputRate, now); @@ -1921,7 +1927,7 @@ const DASHBOARD_HTML = String.raw` if (active) { ctx.shadowColor = "rgba(47,230,200,.55)"; ctx.shadowBlur = 12; } ctx.beginPath(); ctx.roundRect(cardLeft, cardTop, nodeCardWidth, point.cardHeight, 6); ctx.fill(); ctx.stroke(); ctx.shadowBlur = 0; - ctx.textAlign = "left"; ctx.font = '700 10px "SFMono-Regular",monospace'; + ctx.textAlign = "left"; ctx.font = '600 11px system-ui,sans-serif'; ctx.fillStyle = node.reachable === false ? "#ff6f7d" : "#e9fffb"; ctx.fillText(fitCanvasText(ctx, node.name || node.id, clusterEnabled ? 116 : 94), cardLeft + 10, cardTop + 19); ctx.textAlign = "right"; ctx.fillStyle = node.local ? "#8fb4ff" : "rgba(153,163,176,.9)"; @@ -1930,10 +1936,10 @@ const DASHBOARD_HTML = String.raw` if (clusterEnabled) { const platform = node.profile?.platformId || [node.system?.platform, node.system?.arch].filter(Boolean).join("-") || "unknown architecture"; const accelerator = node.profile?.accelerators?.[0] || node.labels?.hardware || "cpu"; - ctx.textAlign = "left"; ctx.font = '8px "SFMono-Regular",monospace'; ctx.fillStyle = "rgba(143,180,255,.72)"; + ctx.textAlign = "left"; ctx.font = '8px system-ui,sans-serif'; ctx.fillStyle = "rgba(143,180,255,.72)"; ctx.fillText(fitCanvasText(ctx, platform + " · " + accelerator, nodeCardWidth - 20), cardLeft + 10, cardTop + 35); } - ctx.font = '9px "SFMono-Regular",monospace'; + ctx.font = '9px system-ui,sans-serif'; point.resources.forEach((row, rowIndex) => { const y = cardTop + 58 + rowIndex * 18; const barLeft = cardLeft + 39; @@ -1992,16 +1998,16 @@ const DASHBOARD_HTML = String.raw` ctx.beginPath(); ctx.roundRect(cardLeft - 5, cardTop - 5, cardWidth + 10, 78, 8); ctx.stroke(); ctx.restore(); } - ctx.beginPath(); ctx.roundRect(cardLeft, cardTop, cardWidth, 68, 5); ctx.fill(); ctx.stroke(); + ctx.beginPath(); ctx.roundRect(cardLeft, cardTop, cardWidth, 68, 12); ctx.fill(); ctx.stroke(); if (processing) { const scanX = cardLeft + ((now * .08) % (cardWidth + 36)) - 18; const scan = ctx.createLinearGradient(scanX - 16, 0, scanX + 16, 0); const scanRgb = externalProcessing ? "192,153,255" : "243,189,79"; scan.addColorStop(0, "rgba(" + scanRgb + ",0)"); scan.addColorStop(.5, "rgba(" + scanRgb + ",.13)"); scan.addColorStop(1, "rgba(" + scanRgb + ",0)"); - ctx.save(); ctx.beginPath(); ctx.roundRect(cardLeft, cardTop, cardWidth, 68, 5); ctx.clip(); ctx.fillStyle = scan; ctx.fillRect(scanX - 16, cardTop, 32, 68); ctx.restore(); + ctx.save(); ctx.beginPath(); ctx.roundRect(cardLeft, cardTop, cardWidth, 68, 12); ctx.clip(); ctx.fillStyle = scan; ctx.fillRect(scanX - 16, cardTop, 32, 68); ctx.restore(); } ctx.fillStyle = serving ? "#42d77d" : externalProcessing ? "#f3bd4f" : external ? "#c099ff" : hot ? "#2fe6c8" : warming ? "#f3bd4f" : evicting ? "#ff7e66" : unavailable ? "#ff6f7d" : "#8fb4ff"; ctx.fillRect(cardLeft, cardTop, 4, 68); - ctx.textAlign = "left"; ctx.font = '11px "SFMono-Regular",monospace'; + ctx.textAlign = "left"; ctx.font = '12px system-ui,sans-serif'; const vendor = modelFamily(point.model.id).toUpperCase(); const vendorText = fitCanvasText(ctx, vendor, 54); const titleWidth = Math.max(24, cardWidth - 12 - 10 - ctx.measureText(vendorText).width - 12); @@ -2059,8 +2065,16 @@ const DASHBOARD_HTML = String.raw` const instantaneousOutputRate = Math.max(0, Number(summary.outputRate || 0)); ctx.textAlign = "center"; if (clusterEnabled) { - ctx.fillStyle = "#e9fffb"; ctx.font = '700 17px "SFMono-Regular",monospace'; ctx.fillText("LLooM", center.x, gate.top + 20); - ctx.fillStyle = "rgba(153,163,176,.9)"; ctx.font = '9px "SFMono-Regular",monospace'; + ctx.save(); + ctx.shadowColor = "rgba(56,223,245,.45)"; + ctx.shadowBlur = summary.active ? 26 : 12; + ctx.fillStyle = "rgba(18,43,54,.92)"; + ctx.strokeStyle = "rgba(56,223,245,.45)"; + ctx.lineWidth = 1; + ctx.beginPath(); ctx.roundRect(center.x - nodeCardWidth / 2, gate.top - 6, nodeCardWidth, 80, 20); ctx.fill(); ctx.stroke(); + ctx.restore(); + ctx.fillStyle = "#e9fffb"; ctx.font = '700 17px system-ui,sans-serif'; ctx.fillText("LLooM", center.x, gate.top + 20); + ctx.fillStyle = "rgba(153,163,176,.9)"; ctx.font = '9px system-ui,sans-serif'; ctx.fillText(fitCanvasText(ctx, (state.status?.cluster?.id || "CLUSTER") + " · ROUTING", nodeCardWidth), center.x, gate.top + 35); const clusterTraffic = [ (summary.active || 0) + " ACTIVE", @@ -2077,13 +2091,13 @@ const DASHBOARD_HTML = String.raw` ctx.beginPath(); ctx.roundRect(gate.left, gate.top, gate.right - gate.left, gate.bottom - gate.top, 9); ctx.fill(); ctx.stroke(); ctx.strokeStyle = "rgba(143,180,255,.35)"; ctx.lineWidth = 1; for (let i = 0; i < 9; i++) { const y = gate.top + 30 + i * 25; ctx.beginPath(); ctx.moveTo(gate.left + 13, y); ctx.lineTo(gate.right - 13, y); ctx.stroke(); } - ctx.fillStyle = "#e9fffb"; ctx.font = '700 18px "SFMono-Regular",monospace'; ctx.fillText("LLooM", center.x, gate.top + 39); - ctx.fillStyle = "rgba(153,163,176,.9)"; ctx.font = '10px "SFMono-Regular",monospace'; ctx.fillText("ROUTING LOOM", center.x, gate.top + 57); + ctx.fillStyle = "#e9fffb"; ctx.font = '700 18px system-ui,sans-serif'; ctx.fillText("LLooM", center.x, gate.top + 39); + ctx.fillStyle = "rgba(153,163,176,.9)"; ctx.font = '10px system-ui,sans-serif'; ctx.fillText("ROUTING LOOM", center.x, gate.top + 57); const statRows = [["ACTIVE", summary.active || 0], ["PROMPT", promptTokens > 0 ? (summary.promptEstimated ? "~" : "") + formatCompact(Math.round(promptTokens)) + " tok" : "—"], ["OUTPUT", formatRate(instantaneousOutputRate) + " ~t/s"], ["ERRORS/1M", summary.recentErrors || 0]]; - ctx.font = '11px "SFMono-Regular",monospace'; + ctx.font = '12px system-ui,sans-serif'; statRows.forEach((row, index) => { const y = gate.top + 94 + index * 27; ctx.textAlign = "left"; ctx.fillStyle = "rgba(153,163,176,.9)"; ctx.fillText(row[0], gate.left + 18, y); ctx.textAlign = "right"; ctx.fillStyle = "rgba(242,245,247,.95)"; ctx.fillText(String(row[1]), gate.right - 18, y); }); const resourceRows = hostResourceRows(summary.host); - ctx.font = '9px "SFMono-Regular",monospace'; + ctx.font = '9px system-ui,sans-serif'; resourceRows.forEach((row, index) => { const y = gate.top + 218 + index * 24, value = row[1]; ctx.textAlign = "left"; ctx.fillStyle = "rgba(153,163,176,.9)"; ctx.fillText(row[0], gate.left + 17, y); @@ -2097,7 +2111,7 @@ const DASHBOARD_HTML = String.raw` } function animateTopology() { - if (!document.hidden) { + if (!document.hidden && !$(".topology").hidden) { const reducedMotion = window.matchMedia("(prefers-reduced-motion: reduce)").matches; // Reduced motion still resolves the layout, it just settles it and // snaps the camera instead of easing both across frames. @@ -2287,6 +2301,7 @@ const DASHBOARD_HTML = String.raw` } async function refreshActivity() { + if (document.hidden) return; try { state.metrics = await getJson("/gateway/metrics?period=" + encodeURIComponent(state.metricsPeriod)); renderActivity(); @@ -2306,6 +2321,7 @@ const DASHBOARD_HTML = String.raw` } async function refresh() { + if (document.hidden) return; const healthPill = $("#health"); healthPill?.classList.add("refreshing"); try { @@ -2715,5 +2731,19 @@ const DASHBOARD_HTML = String.raw` `; export function renderDashboardPage() { - return DASHBOARD_HTML; + return DASHBOARD_HTML.replace('', () => presenceStyles + sceneStyles + '\n ') + .replace('', () => '' + presenceNav) + .replace( + '
', + '

Your AI, in motion.

Your hardware. Your models. One gateway.

' + ) + .replace('
', + () => presenceViews + '
- + `; diff --git a/src/installer.mjs b/src/installer.mjs index 2abfa74..b35a0b5 100644 --- a/src/installer.mjs +++ b/src/installer.mjs @@ -113,16 +113,29 @@ async function resolveHuggingFaceDownloadCommand(step, { env = process.env } = { // Bind browser-reviewed downloads to the executable selected at review. If a // recipe installs its own CLI later, its deterministic path remains reviewed. -export async function pinDownloadCommands(plan, { env = process.env } = {}) { +export async function pinDownloadCommands(plan, { env = process.env, requireAvailable = false } = {}) { for (const step of plan.steps ?? []) { - if (step.action !== 'download-model') continue; - const resolved = await resolveHuggingFaceDownloadCommand(step, { env }); - let executable = resolved?.[0]; + const legacyDownload = + step.action === 'command' && + Array.isArray(step.command) && + ['hf', 'huggingface-cli'].includes(path.basename(step.command[0])) && + step.command[1] === 'download'; + if (step.action !== 'download-model' && !legacyDownload) continue; + const destinationIndex = step.command?.indexOf('--local-dir') ?? -1; + const destination = step.destination ?? (destinationIndex >= 0 ? step.command[destinationIndex + 1] : null); + const resolved = await resolveHuggingFaceDownloadCommand({ ...step, destination }, { env }); + let executable = legacyDownload && path.isAbsolute(step.command[0]) ? step.command[0] : resolved?.[0]; if (executable && !path.isAbsolute(executable)) { const which = await runCommand('/usr/bin/which', [executable], { env, allowFailure: true }); executable = which.stdout.trim(); } - step.downloadExecutable = executable || path.join(path.dirname(step.destination), '.hf-cli', 'bin', 'hf'); + if (!executable && requireAvailable) + throw new Error( + 'The Hugging Face downloader is not installed. Choose a vendor recipe or install the Hugging Face CLI, then review again.' + ); + if (!executable && !destination) + throw new Error('A reviewed Hugging Face download needs an installed CLI or an explicit --local-dir.'); + step.downloadExecutable = executable || path.join(path.dirname(destination), '.hf-cli', 'bin', 'hf'); step.commands = (step.commands ?? [step.command]).map((command) => [step.downloadExecutable, ...command.slice(1)]); step.command = step.commands[0]; } diff --git a/src/model-installation.mjs b/src/model-installation.mjs index 87a4f18..39108fa 100644 --- a/src/model-installation.mjs +++ b/src/model-installation.mjs @@ -1,6 +1,7 @@ import { defaultBackendVariables, getBackend, loadBackendCatalog } from './backend-catalog.mjs'; import { applyBackend } from './installer.mjs'; import { runCommand } from './process-control.mjs'; +import { modelAcquisitionStatus, prepareModelAcquisition, finalizeModelAcquisition } from './model-acquisition.mjs'; export async function installImportedModelAssets(plan, { onProgress, backendCatalog, env = process.env } = {}) { const variables = defaultBackendVariables(env); @@ -24,8 +25,17 @@ export async function installImportedModelAssets(plan, { onProgress, backendCata } if (!plan.download?.command) return; onProgress?.({ message: 'Downloading model files. Existing files are reused.' }); - const [command, ...args] = plan.download.command; + const acquisition = plan.download.acquisition; + if (plan.reference?.type === 'huggingface' && !acquisition) + throw new Error('Hugging Face imports require a reviewed acquisition plan.'); + if (acquisition && (await modelAcquisitionStatus(acquisition)).complete) return; + const prepared = acquisition ? await prepareModelAcquisition(acquisition) : null; + const planned = plan.download.command; + const [command, ...args] = prepared + ? planned.map((arg, index) => (planned[index - 1] === '--local-dir' ? prepared.workPath : arg)) + : planned; const result = await runCommand(command, args, { allowFailure: true, env: installEnv, stdio: 'inherit' }); if (result.code !== 0) throw new Error(result.stderr || 'Model download failed. Check the vendor requirements and credentials.'); + if (prepared) await finalizeModelAcquisition(acquisition, prepared); } diff --git a/test/first-run.test.mjs b/test/first-run.test.mjs index 211a87a..476c398 100644 --- a/test/first-run.test.mjs +++ b/test/first-run.test.mjs @@ -344,3 +344,47 @@ test('reviewed downloads retain their executable when the environment changes', await fs.rm(dir, { recursive: true, force: true }); } }); + +test('legacy command downloads retain the reviewed executable and all vendor arguments', async () => { + const { pinDownloadCommands, applyRecipe } = await import('../src/installer.mjs'); + const dir = await fs.mkdtemp(path.join(os.tmpdir(), 'lloom-reviewed-legacy-')); + try { + const executable = path.join(dir, 'hf'); + await fs.writeFile( + executable, + '#!/bin/sh\nfor last do :; done\nmkdir -p "$last"\nprintf "reviewed" > "$last/model.gguf"\n', + { mode: 0o755 } + ); + const destination = path.join(dir, 'model'); + const args = [ + 'download', + 'owner/model', + 'model.gguf', + '--revision', + '1234567890abcdef1234567890abcdef12345678', + '--local-dir', + destination + ]; + const reviewed = await pinDownloadCommands( + { validationErrors: [], steps: [{ id: 'legacy', action: 'command', command: ['hf', ...args] }] }, + { env: { PATH: dir + ':/bin:/usr/bin' } } + ); + assert.deepEqual(reviewed.steps[0].command, [executable, ...args]); + const report = await applyRecipe( + { id: 'legacy-reviewed' }, + {}, + { + reviewedPlan: reviewed, + dryRun: false, + yes: true, + statePath: path.join(dir, 'state.json'), + env: { PATH: '/bin:/usr/bin', LLOOM_HF_BIN: '/must-not-run' } + } + ); + assert.equal(report.results[0].status, 'completed', JSON.stringify(report.results)); + assert.deepEqual(report.results[0].command, [executable, ...args]); + assert.equal(await fs.readFile(path.join(destination, 'model.gguf'), 'utf8'), 'reviewed'); + } finally { + await fs.rm(dir, { recursive: true, force: true }); + } +}); diff --git a/test/installation-jobs.test.mjs b/test/installation-jobs.test.mjs index 9a7feee..6254f59 100644 --- a/test/installation-jobs.test.mjs +++ b/test/installation-jobs.test.mjs @@ -6,6 +6,8 @@ import path from 'node:path'; import { createInstallationJobs } from '../src/installation-jobs.mjs'; import { createDashboardInstallation } from '../src/dashboard-installation.mjs'; import { loadConfig } from '../src/config.mjs'; +import { installImportedModelAssets } from '../src/model-installation.mjs'; +import { MODEL_ACQUISITION_MANIFEST } from '../src/model-acquisition.mjs'; test('review is read-only; apply is bound, serialized and idempotent across refresh', async () => { let release, @@ -138,3 +140,88 @@ test('dashboard installation routes enforce browser origin and publish the revie await fs.rm(dir, { recursive: true, force: true }); } }); + +test('browser imports require an immutable Hugging Face revision before creating a plan', async () => { + const dir = await fs.mkdtemp(path.join(os.tmpdir(), 'lloom-import-review-')); + try { + const file = path.join(dir, 'config.json'); + await fs.writeFile(file, JSON.stringify({ models: [], runtimes: {}, paths: { modelRoot: dir } })); + const config = await loadConfig(file); + const executable = path.join(dir, 'hf'); + const jobs = createDashboardInstallation({ + getConfig: () => config, + reload: async () => {}, + env: { PATH: dir, LLOOM_HF_BIN: executable } + }); + for (const modelRef of ['owner/model-gguf', 'https://huggingface.co/owner/model-gguf/tree/main']) + await assert.rejects(jobs.review({ modelRef }), /pinned to a commit/); + const revision = '1234567890abcdef1234567890abcdef12345678'; + await assert.rejects( + jobs.review({ modelRef: 'https://huggingface.co/owner/model-gguf/resolve/' + revision + '/model.gguf' }), + /downloader is not installed/ + ); + await fs.writeFile(executable, '#!/bin/sh\nexit 0\n', { mode: 0o755 }); + const reviewed = await jobs.review({ + modelRef: 'https://huggingface.co/owner/model-gguf/resolve/' + revision + '/model.gguf' + }); + assert.equal(reviewed.details.download.acquisition.revision, revision); + assert.deepEqual(reviewed.details.download.acquisition.include, ['model.gguf']); + assert(path.isAbsolute(reviewed.details.download.command[0])); + assert.deepEqual( + (await fs.readdir(dir)).sort(), + ['config.json', 'hf'], + 'review must not create download directories' + ); + } finally { + await fs.rm(dir, { recursive: true, force: true }); + } +}); + +test('import acquisition keeps failed payloads staged and verifies before publishing', async () => { + const dir = await fs.mkdtemp(path.join(os.tmpdir(), 'lloom-import-acquisition-')); + try { + const destination = path.join(dir, 'model'); + const executable = path.join(dir, 'hf'); + const revision = '1234567890abcdef1234567890abcdef12345678'; + const acquisition = { + provider: 'huggingface', + model: 'owner/model', + revision, + destination, + include: ['model.gguf'] + }; + const command = [ + executable, + 'download', + acquisition.model, + '--revision', + revision, + '--include', + 'model.gguf', + '--local-dir', + destination + ]; + const plan = { additions: {}, reference: { type: 'huggingface' }, download: { command, acquisition } }; + // A successful subprocess that writes the wrong payload must not publish it. + await fs.writeFile(executable, '#!/bin/sh\nfor last do :; done\nprintf "partial" > "$last/wrong.gguf"\n', { + mode: 0o755 + }); + await assert.rejects(installImportedModelAssets(plan), /verification failed/); + await assert.rejects(fs.access(destination), { code: 'ENOENT' }); + await fs.access(destination + '.incomplete/wrong.gguf'); + await fs.writeFile(executable, '#!/bin/sh\nfor last do :; done\nprintf "weights" > "$last/model.gguf"\n', { + mode: 0o755 + }); + await installImportedModelAssets(plan); + await assert.rejects(fs.access(destination + '.incomplete'), { code: 'ENOENT' }); + const manifest = JSON.parse(await fs.readFile(path.join(destination, MODEL_ACQUISITION_MANIFEST), 'utf8')); + assert.equal(manifest.revision, revision); + assert.deepEqual(manifest.include, ['model.gguf']); + assert.equal(await fs.readFile(path.join(destination, 'model.gguf'), 'utf8'), 'weights'); + // Verified data is reusable even after the original downloader disappears. + await fs.unlink(executable); + await installImportedModelAssets(plan); + } finally { + await fs.rm(dir, { recursive: true, force: true }); + } +}); diff --git a/test/smoke.mjs b/test/smoke.mjs index 103e2d6..93db712 100644 --- a/test/smoke.mjs +++ b/test/smoke.mjs @@ -1083,7 +1083,10 @@ assert.equal(libraryJson.recipes[0].id, 'apple-silicon-flux2-klein-4b'); if (process.platform === 'darwin' && process.arch === 'arm64') { assert.equal(libraryJson.selected.recipeId, 'apple-silicon-qwen36-35b-a3b-optiq'); } else { - assert.equal(libraryJson.selected, null); + // CPU-only hosts can now select the lightweight Hear recipe. A missing GPU + // does not imply that the entire vendor library is incompatible. + const compatible = libraryJson.candidates.filter((candidate) => candidate.selectable); + assert.equal(libraryJson.selected?.recipeId ?? null, compatible[0]?.recipeId ?? null); } const addModelCli = await runCommand(process.execPath, [ path.join(process.cwd(), 'bin', 'lloom.mjs'), @@ -1928,7 +1931,7 @@ assert(helpCli.includes('lloom integrate')); assert(!helpCli.includes('recipe-submit ')); const helpFlagCli = (await runCommand(process.execPath, [path.join(process.cwd(), 'bin', 'lloom.mjs'), '--help'])) .stdout; -assert(helpFlagCli.includes('Start installed LLooM; preview setup on first run')); +assert(helpFlagCli.includes('Start installed LLooM; guided setup on first run')); const advancedHelpCli = ( await runCommand(process.execPath, [path.join(process.cwd(), 'bin', 'lloom.mjs'), 'help', 'advanced']) ).stdout; From 4612fe21a11f7e4b2db47a922481e87cbb7b6a5f Mon Sep 17 00:00:00 2001 From: Jason McCartney Date: Mon, 21 Sep 2026 21:17:44 -0700 Subject: [PATCH 06/16] Accept compatible CPU recipes in dashboard library smoke check --- test/smoke.mjs | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/test/smoke.mjs b/test/smoke.mjs index 93db712..f9f2eab 100644 --- a/test/smoke.mjs +++ b/test/smoke.mjs @@ -5966,7 +5966,8 @@ if (listened) { if (process.platform === 'darwin' && process.arch === 'arm64') { assert.equal(libraryPlanJson.selected.recipeId, 'apple-silicon-qwen36-35b-a3b-optiq'); } else { - assert.equal(libraryPlanJson.selected, null); + const compatible = libraryPlanJson.candidates.filter((candidate) => candidate.selectable); + assert.equal(libraryPlanJson.selected?.recipeId ?? null, compatible[0]?.recipeId ?? null); } assert.equal( libraryPlanJson.recipes.find((recipe) => recipe.id === 'apple-silicon-qwen36-35b-a3b-optiq')?.commands From 5cc4f4c65a97f89054109059ffd60f8f185be370 Mon Sep 17 00:00:00 2001 From: Jason McCartney Date: Mon, 21 Sep 2026 22:36:38 -0700 Subject: [PATCH 07/16] Abort unsafe model loads with supervised memory thresholds --- config/default.json | 7 +- docs/browser-experience.md | 10 + package.json | 7 +- src/config.mjs | 24 ++ src/dashboard-presence-client.mjs | 3 + src/dashboard-presence.mjs | 1 + src/dashboard.mjs | 2 +- src/host-memory.mjs | 11 +- src/runtime-manager.mjs | 142 ++++++++++- src/runtime-memory-safety.mjs | 137 +++++++++++ src/runtime-policy.mjs | 14 +- src/runtime-supervisor.mjs | 67 ++++- test/runtime-memory-safety.test.mjs | 365 ++++++++++++++++++++++++++++ test/runtime-policy.test.mjs | 69 +++++- test/smoke.mjs | 20 +- 15 files changed, 846 insertions(+), 33 deletions(-) create mode 100644 src/runtime-memory-safety.mjs create mode 100644 test/runtime-memory-safety.test.mjs diff --git a/config/default.json b/config/default.json index e84e8de..feccb4d 100644 --- a/config/default.json +++ b/config/default.json @@ -31,7 +31,12 @@ "enabled": true, "autoEvict": false, "reserveMemoryGb": 12, - "protectActiveRequests": true + "protectActiveRequests": true, + "memorySafety": { + "mode": "enforce", + "maxMemoryUtilization": 0.9, + "pollIntervalMs": 250 + } }, "community": { "hostUrl": "http://127.0.0.1:8110", diff --git a/docs/browser-experience.md b/docs/browser-experience.md index eb21d26..c0ca77e 100644 --- a/docs/browser-experience.md +++ b/docs/browser-experience.md @@ -25,6 +25,16 @@ After installation, LLooM starts the gateway and checks a chat request through i Readiness policies express intent: **Auto** loads on demand, **Prefer ready** uses the existing idle residency reconciler, and **Always ready** prevents automatic eviction. All loading still passes through memory admission. Changing readiness does not restart a model or interrupt active work. The API returns a pending job while the current admission completes; the page reports completion or failure. Queued residency starts recheck the saved policy, and pending hard pins protect eviction victims. Use Load when you want to start a cold model immediately. +### Memory protection + +Admission counts live host memory use, including other applications, even when the policy specifies only a reserve or an absolute budget. Model estimates are planning inputs, not allocation limits. + +Memory protection is enabled by default. A newly started backend is checked before launch and monitored during loading and warmup. If available memory reaches the hard reserve or host utilization reaches the ceiling, LLooM aborts that load, cleans up its processes, and reports the failed threshold. Ordinary Load, forced starts, and disabling automatic eviction do not disable this protection. Automatic retries are blocked until a manual retry or gateway restart. Suspend a model to keep it blocked across restarts. + +The installed config accepts `runtimePolicy.memorySafety` with `mode`, `minAvailableMemoryGb`, `maxMemoryUtilization` (a fraction), and `pollIntervalMs`. The normal mode is `"enforce"`. The default ceiling is 90%; the host reserve can impose a stricter limit. Sampling is a userspace safeguard, not a kernel-enforced allocation quota: a backend can allocate between samples. + +For deliberate manual experiments, `"mode": "yolo"` disables memory admission and the hard load guard. It leaves authentication, runtime ownership, and maintenance gates in place. The dashboard displays a persistent YOLO warning. Restore `"enforce"` before normal operation; YOLO can exhaust the host's memory. + ## Local security First-run setup binds only to loopback and uses a random session token passed through the URL fragment. It removes the fragment immediately and keeps the token in that tab's session storage. Setup rejects foreign origins, alternate authorities, oversized bodies, and unreviewed apply inputs. Configuration publication cannot overwrite a file created concurrently. diff --git a/package.json b/package.json index 500d64d..e09b166 100644 --- a/package.json +++ b/package.json @@ -64,7 +64,7 @@ "check": "npm run check:cluster && npm run check:js && npm run check:entity-stagger && npm run check:python && npm run check:shell && npm run check:spark-deploy && npm run check:workers", "check:cluster": "node --check src/cluster.mjs && node --check test/cluster.test.mjs", "check:entity-stagger": "node --check scripts/entity-stagger-benchmark.mjs && node --check test/entity-stagger-benchmark.test.mjs", - "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs && node --check src/dashboard-scene.mjs && node --check src/dashboard-presence.mjs && node --check src/dashboard-presence-client.mjs && node --check src/first-run.mjs && node --check src/first-run-page.mjs && node --check src/browser-setup.mjs && node --check src/runtime-preferences.mjs && node --check src/installation-jobs.mjs && node --check src/model-installation.mjs && node --check src/dashboard-installation.mjs", + "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs && node --check src/dashboard-scene.mjs && node --check src/dashboard-presence.mjs && node --check src/dashboard-presence-client.mjs && node --check src/first-run.mjs && node --check src/first-run-page.mjs && node --check src/browser-setup.mjs && node --check src/runtime-preferences.mjs && node --check src/installation-jobs.mjs && node --check src/model-installation.mjs && node --check src/dashboard-installation.mjs && node --check src/runtime-memory-safety.mjs", "check:python": "python3 -m py_compile backends/dspark-vllm/apply-patch-pack.py backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/vision_exp/*.py backends/qwen38-vllm/apply-ple-fp8-patch.py backends/qwen38-sglang/*.py backends/glm53-exl3/*.py backends/mlx-audio/lloom_audio_server.py backends/hear/lloom_hear_server.py backends/chatterbox/lloom_chatterbox_server.py patches/apply_mtplx_longctx_fix.py backends/comfyui-media/bridge/*.py backends/comfyui-media/graphs/*.py backends/comfyui-media/build/*.py backends/comfyui-media/install.py backends/qwen-image-diffusers/*.py backends/hear/test/test_hear_security.py", "check:shell": "bash -n backends/dspark-vllm/entrypoint.sh backends/dspark-vllm/runtime/build-hardened-image.sh backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/apply-runtime.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.sh backends/qwen38-vllm/entrypoint.sh backends/qwen38-sglang/entrypoint.sh backends/glm53-exl3/entrypoint.sh backends/mlx-audio/install.sh backends/hear/install.sh backends/chatterbox/install.sh clients/examples/lloom-claude clients/examples/lloom-codex clients/examples/lloom-hermes clients/examples/lloom-opencode clients/examples/lloom-zero scripts/remote-install-spark.sh scripts/remote-stage-spark-worker.sh", "check:spark-deploy": "node --check test/spark-deploy.test.mjs && node test/spark-deploy.test.mjs", @@ -73,7 +73,7 @@ "community:check": "node scripts/check-community-deploy.mjs", "package:check": "node scripts/check-package.mjs", "test:adoption": "node test/resource-fit.test.mjs && node test/model-acquisition.test.mjs && node test/dashboard-status.test.mjs", - "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs && npm run test:ux", + "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs && npm run test:ux && npm run test:memory-safety", "test:cluster": "node test/cluster.test.mjs", "test": "npm run test:cluster && npm run test:unit && npm run test:entity-stagger && npm run smoke && npm run community:check", "test:entity-stagger": "node test/entity-stagger-benchmark.test.mjs", @@ -85,7 +85,8 @@ "check:workers": "node --check clients/examples/research-workers/runner.mjs && node --check clients/examples/research-workers/process.mjs && node --check clients/examples/research-workers/process-guard.mjs", "test:comfyui": "python3 scripts/test-comfyui-media.py", "test:hear": "python3 -m unittest discover -s backends/hear/test -v", - "test:ux": "node --test test/first-run.test.mjs test/runtime-preferences.test.mjs test/admin-browser-security.test.mjs test/installation-jobs.test.mjs" + "test:ux": "node --test test/first-run.test.mjs test/runtime-preferences.test.mjs test/admin-browser-security.test.mjs test/installation-jobs.test.mjs", + "test:memory-safety": "node --test test/runtime-memory-safety.test.mjs" }, "engines": { "node": ">=20.0.0" diff --git a/src/config.mjs b/src/config.mjs index f496a2f..9ed7402 100644 --- a/src/config.mjs +++ b/src/config.mjs @@ -370,6 +370,30 @@ function validateConfig(config, sourcePath, env) { } } + const memorySafety = config.runtimePolicy?.memorySafety; + if (memorySafety != null) { + if (typeof memorySafety !== 'object' || Array.isArray(memorySafety)) { + errors.push('runtimePolicy.memorySafety must be an object'); + } else { + if (memorySafety.mode != null && !['enforce', 'yolo'].includes(memorySafety.mode)) { + errors.push('runtimePolicy.memorySafety.mode must be enforce or yolo'); + } + for (const key of ['minAvailableMemoryGb', 'maxMemoryUtilization', 'pollIntervalMs']) { + const value = memorySafety[key]; + if (value == null) continue; + if ( + typeof value !== 'number' || + !Number.isFinite(value) || + value <= 0 || + (key === 'maxMemoryUtilization' && value >= 1) || + (key === 'pollIntervalMs' && (value < 50 || value > 1000)) + ) { + errors.push(`runtimePolicy.memorySafety.${key} is outside its safe range`); + } + } + } + } + const distributedMemberGroups = new Map(); for (const [runtimeId, runtime] of Object.entries(config.runtimes ?? {})) { if (runtime?.placement?.mode !== 'distributed') continue; diff --git a/src/dashboard-presence-client.mjs b/src/dashboard-presence-client.mjs index ce0b736..24cb78d 100644 --- a/src/dashboard-presence-client.mjs +++ b/src/dashboard-presence-client.mjs @@ -102,6 +102,9 @@ export const presenceScript = String.raw` } function renderPresence() { if (!$("#presence-models")) return; + const safety = state.status?.runtimeManager?.memorySafety; + const warning = $("#presence-memory-safety"); + if(warning) warning.hidden = safety?.mode !== "yolo"; renderPresenceModels(); renderPresenceMachines(); renderPresenceClients(); renderPresencePolicy(); } function renderPresencePolicy() { diff --git a/src/dashboard-presence.mjs b/src/dashboard-presence.mjs index 9d940dc..f739ce3 100644 --- a/src/dashboard-presence.mjs +++ b/src/dashboard-presence.mjs @@ -15,6 +15,7 @@ export const presenceStyles = ` .band,.empty { border-radius:14px; } .band-head h2,label { text-transform:none; letter-spacing:0; } .band-body { padding:22px; } + #presence-memory-safety { border:1px solid #b67c37; background:#392915; color:#ffd8a0; border-radius:12px; padding:14px 18px; margin-bottom:20px; } .presence-nav { position:fixed; top:0; bottom:0; left:0; width:190px; padding:26px 18px; background:#0b1219; border-right:1px solid var(--line); display:flex; flex-direction:column; gap:7px; z-index:20; } .presence-brand { padding:0 12px 30px; font-size:24px; letter-spacing:-1px; font-weight:500; } .presence-brand small { display:block; color:var(--muted); font-size:10px; letter-spacing:2px; margin-top:3px; } diff --git a/src/dashboard.mjs b/src/dashboard.mjs index 70618ee..258723e 100644 --- a/src/dashboard.mjs +++ b/src/dashboard.mjs @@ -2735,7 +2735,7 @@ export function renderDashboardPage() { .replace('', () => '' + presenceNav) .replace( '
', - '

Your AI, in motion.

Your hardware. Your models. One gateway.

' + '

Your AI, in motion.

Your hardware. Your models. One gateway.

' ) .replace('
+ (cleanupPromise ??= (async () => { + if (owned.child?.pid && owned.child.exitCode == null && owned.child.signalCode == null) { + if (owned.child.connected) owned.child.send({ type: 'memory-safety-abort' }, () => {}); + const result = await terminateProcessTree([owned.child.pid], { termTimeoutMs: 150, killTimeoutMs: 1000 }); + if (result.survivors.length || result.failed.length) + throw new Error('Memory safety cleanup could not stop the owned process tree'); + } + if (owned.containerId) { + await execFileAsync('docker', ['kill', owned.containerId], { timeout: 5000 }).catch(async (error) => { + const state = await dockerContainerState(runtime); + if (state.id === owned.containerId && state.running) throw error; + }); + } + })()); + const guard = createMemorySafetyGuard({ + policy, + runtimeId, + sample: this.memorySampler, + onAbort: (error) => { + this.memorySafetyFailures.set(runtimeId, error); + this.record({ runtimeId, event: 'memory-safety-abort', message: error.message, snapshot: error.snapshot }); + this.logger.error?.(error.message); + void cleanup().catch(() => {}); + } + }); + const combinedSignal = signal ? AbortSignal.any([signal, guard.signal]) : guard.signal; + try { + await guard.check(); + combinedSignal.throwIfAborted(); + guard.start(); + const result = await this.startLocalUnlocked(runtimeId, runtime, { + force, + warmup, + reason, + signal: combinedSignal, + owned, + guard, + policy + }); + await guard.check(); + combinedSignal.throwIfAborted(); + if (owned.child?.connected) owned.child.send({ type: 'memory-safety-complete' }, () => {}); + return result; + } catch (cause) { + const error = guard.signal.aborted ? guard.signal.reason : cause; + if (owned.child || owned.containerId || guard.signal.aborted || combinedSignal.aborted) { + // A container start may have completed after the first abort cleanup. + await cleanupPromise?.catch(() => {}); + cleanupPromise = null; + await cleanup(); + const state = this.stateFor(runtimeId); + state.lastError = error.message; + this.setStatus( + runtimeId, + 'failed', + error.code === 'runtime_memory_safety_abort' ? 'memory-safety-abort' : 'start-aborted' + ); + } + throw error; + } finally { + await guard.stop(); + } + } + + async startLocalUnlocked(runtimeId, runtime, { force, warmup, reason, signal, owned, guard, policy }) { + const state = this.stateFor(runtimeId); if (runtimeAdapter(runtime) === 'docker') { if (runtimeManagement(runtime) !== 'managed') { return { runtimeId, started: false, healthy: false, reason: 'externally-managed' }; @@ -1853,7 +1936,11 @@ export class RuntimeManager { this.record({ runtimeId, event: 'docker-create', bootstrapResult, reason }); } this.setStatus(runtimeId, 'starting', reason); + container = await dockerContainerState(runtime); + if (!container.running) owned.containerId = container.id; + signal.throwIfAborted(); const processResult = await dockerLifecycle('start', runtime); + signal.throwIfAborted(); state.starts += 1; state.startedAt = nowIso(); this.record({ runtimeId, event: 'docker-start', processResult, reason }); @@ -1889,16 +1976,28 @@ export class RuntimeManager { // CLI managers disable capture so their detached runtime does not keep the // short-lived command process open through an inherited pipe. const supervised = process.platform !== 'win32'; + signal.throwIfAborted(); const child = spawn( supervised ? process.execPath : runtime.command, supervised ? [path.join(packageRoot, 'src', 'runtime-supervisor.mjs'), runtime.command, ...args] : args, { cwd: runtime.cwd, - env: runtimeEnvironment(this.config, runtime), - stdio: ['ignore', this.captureOutput ? 'pipe' : 'ignore', this.captureOutput ? 'pipe' : 'ignore'], + env: { ...runtimeEnvironment(this.config, runtime), LLOOM_MEMORY_SAFETY_POLICY: JSON.stringify(policy) }, + stdio: [ + 'ignore', + this.captureOutput ? 'pipe' : 'ignore', + this.captureOutput ? 'pipe' : 'ignore', + ...(supervised ? ['ipc'] : []) + ], detached: true } ); + owned.child = child; + child.on('message', (message) => { + if (message?.type === 'memory-safety-abort') + guard.trip(new RuntimeMemorySafetyError(message.message, { runtimeId, snapshot: message.snapshot, policy })); + }); + child.channel?.unref(); child.unref(); this.processes.set(runtimeId, child); state.starts += 1; @@ -1924,6 +2023,13 @@ export class RuntimeManager { this.record({ runtimeId, event: 'error', message: state.lastError }); }); child.on('exit', (code, signal) => { + if (code === 78 && policy.mode === 'enforce') + guard.trip( + new RuntimeMemorySafetyError('Backend supervisor rejected the load to protect host memory.', { + runtimeId, + policy + }) + ); const expectedStop = state.status === 'stopping' || ['SIGTERM', 'SIGKILL'].includes(signal); state.status = code === 0 || expectedStop ? 'stopped' : 'failed'; state.stoppedAt = nowIso(); @@ -1936,6 +2042,34 @@ export class RuntimeManager { } }); + if (supervised) { + // Health cannot authorize a load until the independent supervisor has + // checked memory and actually spawned this backend. + await new Promise((resolve, reject) => { + const finish = (error) => { + clearTimeout(timer); + child.off('message', onMessage); + child.off('exit', onExit); + child.off('error', onError); + signal.removeEventListener('abort', onAbort); + if (error) reject(error); + else resolve(); + }; + const onMessage = (message) => { + if (message?.type === 'memory-safety-ready') finish(); + }; + const onExit = () => + finish(signal.reason ?? new Error(`runtime ${runtimeId} supervisor exited before startup`)); + const onError = (error) => finish(error); + const onAbort = () => finish(signal.reason); + const timer = setTimeout(() => finish(new Error(`runtime ${runtimeId} supervisor startup timed out`)), 5000); + child.on('message', onMessage); + child.once('exit', onExit); + child.once('error', onError); + signal.addEventListener('abort', onAbort, { once: true }); + if (signal.aborted) onAbort(); + }); + } const result = await this.waitForHealth(runtimeId, runtime, child, { signal }); let warmupResult = null; if (result.healthy && warmup && runtime.warmup) { @@ -1955,6 +2089,7 @@ export class RuntimeManager { while (Date.now() < deadline) { if (signal?.aborted) throw signal.reason ?? new Error(`runtime ${runtimeId} start aborted`); if (await runtimeHealthOk(runtime)) { + signal?.throwIfAborted?.(); this.setStatus(runtimeId, 'running'); this.record({ runtimeId, event: 'healthy' }); return { runtimeId, healthy: true }; @@ -2025,6 +2160,7 @@ export class RuntimeManager { signal }); const text = await response.text().catch(() => ''); + signal?.throwIfAborted?.(); const result = { runtimeId, warmed: response.ok, diff --git a/src/runtime-memory-safety.mjs b/src/runtime-memory-safety.mjs new file mode 100644 index 0000000..e4257f4 --- /dev/null +++ b/src/runtime-memory-safety.mjs @@ -0,0 +1,137 @@ +import os from 'node:os'; +import { readHostMemory } from './host-memory.mjs'; + +const GiB = 1024 ** 3; + +export class RuntimeMemorySafetyError extends Error { + constructor(message, { runtimeId, snapshot, policy } = {}) { + super(message); + this.name = 'RuntimeMemorySafetyError'; + this.code = 'runtime_memory_safety_abort'; + this.type = 'runtime_memory_safety_error'; + this.statusCode = 503; + this.temporary = false; + this.runtimeId = runtimeId; + this.snapshot = snapshot; + this.policy = policy; + } +} + +export function memorySafetyPolicy(config, totalMemoryGb = os.totalmem() / GiB) { + const input = config.runtimePolicy?.memorySafety ?? {}; + const configuredReserve = config.runtimePolicy?.reserveMemoryGb; + const minAvailableMemoryGb = input.minAvailableMemoryGb ?? Math.min(configuredReserve ?? 8, totalMemoryGb * 0.15); + const maxMemoryUtilization = Math.min( + input.maxMemoryUtilization ?? 0.9, + config.runtimePolicy?.maxMemoryUtilization ?? 0.9 + ); + const pollIntervalMs = input.pollIntervalMs ?? 250; + const mode = input.mode ?? 'enforce'; + if ( + !['enforce', 'yolo'].includes(mode) || + !Number.isFinite(minAvailableMemoryGb) || + minAvailableMemoryGb <= 0 || + !Number.isFinite(maxMemoryUtilization) || + maxMemoryUtilization <= 0 || + maxMemoryUtilization >= 1 || + !Number.isFinite(pollIntervalMs) || + pollIntervalMs < 50 || + pollIntervalMs > 1000 + ) { + throw new RuntimeMemorySafetyError('Invalid memory safety limits; refusing to load a model.'); + } + return { mode, minAvailableMemoryGb, maxMemoryUtilization, pollIntervalMs }; +} + +export function assertMemorySafety(policy, snapshot, runtimeId) { + if (policy.mode === 'yolo') return; + const total = snapshot?.totalBytes; + const available = snapshot?.availableBytes; + let detail; + if (!Number.isFinite(total) || total <= 0 || !Number.isFinite(available) || available < 0 || available > total) { + detail = 'host memory could not be measured'; + } else if (available <= policy.minAvailableMemoryGb * GiB) { + detail = `${(available / GiB).toFixed(1)} GB available reached the ${policy.minAvailableMemoryGb.toFixed(1)} GB hard reserve`; + } else if (1 - available / total >= policy.maxMemoryUtilization) { + detail = `host memory use reached the ${(policy.maxMemoryUtilization * 100).toFixed(1)}% hard ceiling`; + } + if (detail) + throw new RuntimeMemorySafetyError(`Load aborted to protect this machine: ${detail}.`, { + runtimeId, + snapshot, + policy + }); +} + +// One operation owns one guard. Samples never overlap, and stop() waits for a +// pending sample so a stale callback cannot kill a later operation. +export function createMemorySafetyGuard({ + policy, + runtimeId, + sample = () => readHostMemory({ strict: true }), + onAbort = () => {} +}) { + const controller = new AbortController(); + let stopped = false; + let timer; + let inFlight; + const trip = (cause) => { + if (stopped || controller.signal.aborted) return; + const error = + cause instanceof RuntimeMemorySafetyError + ? cause + : new RuntimeMemorySafetyError('Load aborted: host memory could not be measured.', { runtimeId, policy }); + controller.abort(error); + onAbort(error); + }; + const check = () => { + if (policy.mode === 'yolo' || stopped) return Promise.resolve(); + controller.signal.throwIfAborted(); + if (!inFlight) { + inFlight = (async () => { + // A stuck sampler is also unsafe; do not leave the load unguarded. + let deadline; + try { + const snapshot = await Promise.race([ + Promise.resolve().then(sample), + new Promise((_, reject) => { + deadline = setTimeout(() => reject(new Error('Memory sampler timed out')), 1000); + }) + ]); + if (!stopped) assertMemorySafety(policy, snapshot, runtimeId); + } catch (error) { + trip(error); + } finally { + clearTimeout(deadline); + } + })().finally(() => { + inFlight = null; + }); + } + return inFlight.then(() => controller.signal.throwIfAborted()); + }; + const tick = async () => { + try { + await check(); + } catch { + return; + } + if (!stopped) { + timer = setTimeout(tick, policy.pollIntervalMs); + timer.unref?.(); + } + }; + return { + signal: controller.signal, + check, + trip, + start() { + if (!stopped && policy.mode !== 'yolo') void tick(); + }, + async stop() { + stopped = true; + clearTimeout(timer); + await inFlight; + } + }; +} diff --git a/src/runtime-policy.mjs b/src/runtime-policy.mjs index 92a588b..67e6cb9 100644 --- a/src/runtime-policy.mjs +++ b/src/runtime-policy.mjs @@ -169,7 +169,7 @@ function policyConfig(config, profile = {}) { ? Math.max(0, totalMemoryGb * maxMemoryUtilization) : Math.max(0, totalMemoryGb - reserveMemoryGb)); return { - enabled: policy.enabled !== false, + enabled: policy.enabled !== false && policy.memorySafety?.mode !== 'yolo', autoEvict: policy.autoEvict === true, totalMemoryGb, reserveMemoryGb, @@ -338,8 +338,10 @@ function clusterRuntimePolicyPlan( 0, totalMemoryGb - (numberOrNull(nodeProfile.availableMemoryGb) ?? totalMemoryGb) ); - const predictive = maxMemoryUtilization != null && nodeProfile.availableMemoryGb != null; - const projectedMemoryGb = (predictive ? actualUsedMemoryGb : loadedMemoryGb) + requestedAddsMemoryGb; + // The host reserve protects every process, even when the operator only + // configured a reserve or an absolute budget (rather than a percentage). + const predictive = nodeProfile.availableMemoryGb != null; + const projectedMemoryGb = Math.max(actualUsedMemoryGb, loadedMemoryGb) + requestedAddsMemoryGb; const overBudgetGb = Math.max(0, projectedMemoryGb - memoryBudgetGb); nodes[nodeId] = { ...nodeProfile, @@ -370,7 +372,7 @@ function clusterRuntimePolicyPlan( } const policy = { - enabled: policyTemplate.enabled !== false, + enabled: policyTemplate.enabled !== false && policyTemplate.memorySafety?.mode !== 'yolo', autoEvict: policyTemplate.autoEvict === true, protectActiveRequests: policyTemplate.protectActiveRequests !== false, clustered: true @@ -505,8 +507,8 @@ export async function createRuntimePolicyPlan( 0, policy.totalMemoryGb - (numberOrNull(memoryProfile.availableMemoryGb) ?? policy.totalMemoryGb) ); - const predictive = policy.maxMemoryUtilization != null; - const projectedMemoryGb = (predictive ? actualUsedMemoryGb : loadedMemoryGb) + requestedAddsMemory; + const predictive = memoryProfile.availableMemoryGb != null; + const projectedMemoryGb = Math.max(actualUsedMemoryGb, loadedMemoryGb) + requestedAddsMemory; let overBudgetGb = requested?.loaded ? 0 : Math.max(0, projectedMemoryGb - policy.memoryBudgetGb); const actions = []; diff --git a/src/runtime-supervisor.mjs b/src/runtime-supervisor.mjs index bc3ac4f..495b836 100644 --- a/src/runtime-supervisor.mjs +++ b/src/runtime-supervisor.mjs @@ -2,6 +2,8 @@ // alive too, so an exited backend cannot leave model workers behind. import { spawn } from 'node:child_process'; import { setTimeout as delay } from 'node:timers/promises'; +import { memorySafetyPolicy, assertMemorySafety, createMemorySafetyGuard } from './runtime-memory-safety.mjs'; +import { readHostMemory } from './host-memory.mjs'; const [command, ...args] = process.argv.slice(2); if (!command || process.platform === 'win32') { @@ -9,12 +11,29 @@ if (!command || process.platform === 'win32') { process.exit(2); } +let memoryPolicy = null; +try { + if (process.env.LLOOM_MEMORY_SAFETY_POLICY) { + memoryPolicy = memorySafetyPolicy({ + runtimePolicy: { memorySafety: JSON.parse(process.env.LLOOM_MEMORY_SAFETY_POLICY) } + }); + if (memoryPolicy.mode !== 'yolo') assertMemorySafety(memoryPolicy, await readHostMemory({ strict: true })); + } +} catch (error) { + process.send?.({ type: 'memory-safety-abort', message: error.message, snapshot: error.snapshot }); + process.stderr.write(`${error.message}\n`); + process.exit(78); +} + // Separate the backend group from the supervisor, allowing escalation even // after the backend exits or while a worker ignores SIGTERM. const child = spawn(command, args, { detached: true, stdio: ['ignore', 'pipe', 'pipe'] }); let exitCode = 1; let stopping = false; let cleanupPromise; +let memoryGuard; +let memoryComplete = false; +let memoryAborted = false; function forward(source, destination) { source.pipe(destination, { end: false }); @@ -40,6 +59,14 @@ function signalGroup(signal) { function cleanup() { cleanupPromise ??= (async () => { + await memoryGuard?.stop(); + if (memoryAborted) { + // SIGKILL was already delivered to the owned group. Reap the child; + // probing a dying process group can return EPERM on macOS. + const deadline = Date.now() + 1000; + while (Date.now() < deadline && child.exitCode == null && child.signalCode == null) await delay(10); + process.exit(78); + } signalGroup('SIGTERM'); const deadline = Date.now() + 2000; while (Date.now() < deadline && signalGroup(0)) await delay(50); @@ -54,12 +81,50 @@ function cleanup() { return cleanupPromise; } +function abortForMemory(error) { + if (memoryAborted || memoryComplete) return; + memoryAborted = true; + exitCode = 78; + if (process.connected) + process.send({ type: 'memory-safety-abort', message: error.message, snapshot: error.snapshot }); + if (!process.stderr.destroyed) process.stderr.write(`${error.message}\n`); + // This group belongs to this supervisor. Do not wait for the gateway or a + // graceful backend shutdown while the host is running out of memory. + signalGroup('SIGKILL'); + void cleanup(); +} + +if (memoryPolicy?.mode === 'enforce') { + memoryGuard = createMemorySafetyGuard({ policy: memoryPolicy, onAbort: abortForMemory }); + memoryGuard.start(); +} +process.on('message', (message) => { + if (message?.type === 'memory-safety-complete') { + memoryComplete = true; + void memoryGuard?.stop(); + // Disconnecting synchronously while Node drains queued IPC messages can + // crash its message dispatcher, leaving the backend without a supervisor. + setImmediate(() => { + if (process.connected) process.disconnect(); + }); + } else if (message?.type === 'memory-safety-abort') { + abortForMemory(new Error('Gateway aborted this load to protect host memory.')); + } +}); +process.on('disconnect', () => { + if (memoryGuard && !memoryComplete && !memoryAborted) + abortForMemory(new Error('Gateway disconnected during a guarded load.')); +}); + child.on('error', (error) => { if (!process.stderr.destroyed) process.stderr.write(`runtime launch failed: ${error.message}\n`); void cleanup(); }); +child.on('spawn', () => { + if (process.connected) process.send({ type: 'memory-safety-ready' }, () => {}); +}); child.on('exit', (code) => { - exitCode = code ?? 1; + if (!memoryAborted) exitCode = code ?? 1; void cleanup(); }); for (const signal of ['SIGTERM', 'SIGINT', 'SIGHUP']) { diff --git a/test/runtime-memory-safety.test.mjs b/test/runtime-memory-safety.test.mjs new file mode 100644 index 0000000..3b6a8f0 --- /dev/null +++ b/test/runtime-memory-safety.test.mjs @@ -0,0 +1,365 @@ +import test from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; +import http from 'node:http'; +import { spawn } from 'node:child_process'; +import { setTimeout as delay } from 'node:timers/promises'; +import { RuntimeManager } from '../src/runtime-manager.mjs'; +import { memorySafetyPolicy, assertMemorySafety, createMemorySafetyGuard } from '../src/runtime-memory-safety.mjs'; +import { readHostMemory } from '../src/host-memory.mjs'; +import { terminateProcessTree } from '../src/process-control.mjs'; + +const GiB = 1024 ** 3; +const snapshot = (available) => ({ totalBytes: 96 * GiB, availableBytes: available * GiB }); +const healthy = snapshot(60); +const pressure = snapshot(3); +const alive = (pid) => { + try { + process.kill(pid, 0); + return true; + } catch { + return false; + } +}; +async function until(fn) { + for (let i = 0; i < 100; i++) { + if (await fn()) return; + await delay(25); + } + throw new Error('Timed out waiting for small test process'); +} + +async function fixture(t, { warmup = false, yolo = false } = {}) { + const dir = await fs.mkdtemp(path.join(os.tmpdir(), 'lloom-memory-guard-')); + const receipt = path.join(dir, 'receipt.json'); + const marker = path.join(dir, 'warmup'); + const code = path.join(dir, 'backend.mjs'); + const reserve = http.createServer(); + await new Promise((resolve) => reserve.listen(0, '127.0.0.1', resolve)); + const port = reserve.address().port; + await new Promise((resolve) => reserve.close(resolve)); + await fs.writeFile( + code, + ` + import http from 'node:http'; + import fs from 'node:fs'; + import {spawn} from 'node:child_process'; + const worker=spawn(process.execPath,['-e',"process.on('SIGTERM',()=>{});setInterval(()=>{},1000)"],{stdio:'ignore'}); + fs.writeFileSync(process.argv[2],JSON.stringify({backend:process.pid,worker:worker.pid})); + http.createServer((req,res)=>{ + if(req.url==='/warmup'){fs.writeFileSync(process.argv[3],'warming');return;} + res.writeHead(${warmup || yolo ? 200 : 503});res.end('{}'); + }).listen(${port},'127.0.0.1'); + ` + ); + const config = { + runtimePolicy: { + enabled: false, + reserveMemoryGb: 12, + memorySafety: { mode: yolo ? 'yolo' : 'enforce', pollIntervalMs: 50 } + }, + runtimes: { + demo: { + enabled: true, + command: process.execPath, + args: [code, receipt, marker], + port, + healthUrl: `http://127.0.0.1:${port}/health`, + startupTimeoutMs: 5000, + ...(warmup ? { warmup: { url: `http://127.0.0.1:${port}/warmup`, method: 'POST', body: {} } } : {}) + } + } + }; + const manager = new RuntimeManager(config, { logger: { error() {} }, memorySampler: async () => healthy }); + const ids = async () => JSON.parse(await fs.readFile(receipt, 'utf8')); + t.after(async () => { + const pids = await ids().catch(() => null); + if (pids) await terminateProcessTree([pids.backend, pids.worker], { termTimeoutMs: 100 }); + const supervisor = manager.processes.get('demo'); + if (supervisor?.pid) await terminateProcessTree([supervisor.pid], { termTimeoutMs: 100 }); + await fs.rm(dir, { recursive: true, force: true }); + }); + return { manager, receipt, marker, ids, port }; +} + +test('hard limits include the host reserve, reject malformed telemetry, and use inclusive boundaries', () => { + const policy = memorySafetyPolicy({ runtimePolicy: { reserveMemoryGb: 12 } }, 96); + assert.equal(policy.minAvailableMemoryGb, 12); + assert.throws(() => assertMemorySafety(policy, snapshot(12)), /hard reserve/); + assert.throws(() => assertMemorySafety(policy, { totalBytes: 1, availableBytes: 2 }), /could not be measured/); + assert.throws(() => assertMemorySafety(policy, null), /could not be measured/); + assert.doesNotThrow(() => assertMemorySafety(policy, healthy)); + assert.equal( + memorySafetyPolicy({ runtimePolicy: { memorySafety: { minAvailableMemoryGb: 20 } } }, 96).minAvailableMemoryGb, + 20 + ); + assert.throws(() => memorySafetyPolicy({ runtimePolicy: { memorySafety: { mode: 'YOLO-ish' } } }), /Invalid/); +}); + +test('strict Mac telemetry fails closed instead of substituting free memory', async () => { + await assert.rejects( + readHostMemory({ + platform: 'darwin', + strict: true, + execFileImpl: async () => { + throw new Error('sampler unavailable'); + } + }), + /sampler unavailable/ + ); + await assert.rejects( + readHostMemory({ platform: 'darwin', strict: true, execFileImpl: async () => ({ stdout: 'invalid' }) }), + /unavailable/ + ); +}); + +test('force and disabled predictive policy cannot bypass hard preflight', async (t) => { + const f = await fixture(t); + f.manager.memorySampler = async () => pressure; + await assert.rejects(f.manager.start('demo', { force: true }), (e) => e.code === 'runtime_memory_safety_abort'); + assert.equal(f.manager.processes.size, 0); + assert.equal( + await fs.access(f.receipt).then( + () => true, + () => false + ), + false + ); +}); + +for (const warmup of [false, true]) + test(`pressure during blocked ${warmup ? 'warmup' : 'health'} kills only the new process tree`, async (t) => { + const f = await fixture(t, { warmup }); + const sentinel = spawn(process.execPath, ['-e', 'setInterval(()=>{},1000)'], { stdio: 'ignore' }); + t.after(() => terminateProcessTree([sentinel.pid], { termTimeoutMs: 100 })); + let trip = false; + f.manager.memorySampler = async () => (trip ? pressure : healthy); + const starting = f.manager.start('demo', { reason: 'model-request' }); + const rejected = assert.rejects(starting, (e) => e.code === 'runtime_memory_safety_abort' && !e.temporary); + await until(() => + fs.access(warmup ? f.marker : f.receipt).then( + () => true, + () => false + ) + ); + const pids = await f.ids(); + assert(alive(pids.backend)); + trip = true; + await rejected; + await until(() => !alive(pids.backend) && !alive(pids.worker)); + assert(alive(sentinel.pid), 'unrelated process must survive'); + assert.equal(f.manager.stateFor('demo').status, 'failed'); + assert.match(f.manager.stateFor('demo').lastError, /protect this machine/); + f.manager.memorySampler = async () => healthy; + await assert.rejects( + f.manager.start('demo', { reason: 'model-request' }), + (e) => e.code === 'runtime_memory_safety_abort' + ); + assert.equal(f.manager.stateFor('demo').starts, 1, 'automatic retry must not spawn again'); + }); + +test('a reused healthy endpoint remains usable under pressure', async (t) => { + const server = http.createServer((req, res) => res.end('{}')); + await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve)); + t.after(() => new Promise((resolve) => server.close(resolve))); + const manager = new RuntimeManager( + { runtimes: { existing: { enabled: true, healthUrl: `http://127.0.0.1:${server.address().port}/health` } } }, + { memorySampler: async () => pressure } + ); + const result = await manager.start('existing', { warmup: false }); + assert.equal(result.reason, 'already-healthy'); + assert(server.listening); +}); + +test('explicit YOLO is the only guard bypass', async (t) => { + const f = await fixture(t, { yolo: true }); + f.manager.memorySampler = async () => { + throw new Error('must not be sampled in YOLO'); + }; + const result = await f.manager.start('demo'); + assert.equal(result.healthy, true); + assert.equal((await f.manager.status()).memorySafety.mode, 'yolo'); + assert(alive((await f.ids()).backend)); +}); + +test('a stale sample cannot abort a completed operation', async () => { + let release; + let aborts = 0; + const guard = createMemorySafetyGuard({ + policy: memorySafetyPolicy({}), + sample: () => + new Promise((resolve) => { + release = resolve; + }), + onAbort: () => aborts++ + }); + const checking = guard.check(); + await until(() => Boolean(release)); + const stopping = guard.stop(); + release(pressure); + await Promise.all([checking, stopping]); + assert.equal(aborts, 0); +}); + +test('the independent supervisor enforces the transmitted reserve without gateway sampling', async (t) => { + const dir = await fs.mkdtemp(path.join(os.tmpdir(), 'lloom-independent-guard-')); + const source = path.resolve('src'); + for (const name of ['runtime-supervisor.mjs', 'runtime-memory-safety.mjs']) + await fs.copyFile(path.join(source, name), path.join(dir, name)); + // Replace only the OS telemetry dependency in this isolated subprocess test. + await fs.writeFile( + path.join(dir, 'host-memory.mjs'), + `import fs from 'node:fs/promises'; export async function readHostMemory(){ return JSON.parse(await fs.readFile(process.env.SAMPLE_FILE,'utf8')); }` + ); + const sampleFile = path.join(dir, 'sample.json'); + const receipt = path.join(dir, 'receipt.json'); + await fs.writeFile(sampleFile, JSON.stringify(healthy)); + const backend = `const fs=require('fs'),{spawn}=require('child_process');const worker=spawn(process.execPath,['-e','setInterval(()=>{},1000)'],{stdio:'ignore'});fs.writeFileSync(process.argv[1],JSON.stringify({backend:process.pid,worker:worker.pid}));setInterval(()=>{},1000);`; + const child = spawn( + process.execPath, + [path.join(dir, 'runtime-supervisor.mjs'), process.execPath, '-e', backend, receipt], + { + env: { + ...process.env, + SAMPLE_FILE: sampleFile, + LLOOM_MEMORY_SAFETY_POLICY: JSON.stringify({ + mode: 'enforce', + minAvailableMemoryGb: 20, + maxMemoryUtilization: 0.9, + pollIntervalMs: 50 + }) + }, + stdio: ['ignore', 'ignore', 'pipe', 'ipc'] + } + ); + let stderr = ''; + child.stderr.on('data', (data) => { + stderr += data; + }); + let aborted; + child.on('message', (message) => { + if (message.type === 'memory-safety-abort') aborted = message; + }); + const exited = new Promise((resolve) => child.once('exit', (code, signal) => resolve({ code, signal }))); + t.after(async () => { + const pids = await fs.readFile(receipt, 'utf8').then(JSON.parse, () => ({})); + await terminateProcessTree([child.pid, pids.backend, pids.worker].filter(Boolean), { termTimeoutMs: 100 }); + await fs.rm(dir, { recursive: true, force: true }); + }); + await until(() => + fs.access(receipt).then( + () => true, + () => false + ) + ); + const pids = JSON.parse(await fs.readFile(receipt, 'utf8')); + await fs.writeFile(sampleFile, JSON.stringify(snapshot(18))); + const result = await Promise.race([ + exited, + delay(3000).then(() => { + throw new Error('Independent cutoff did not fire'); + }) + ]); + assert.equal(result.code, 78, stderr); + assert.match(aborted?.message ?? '', /20.0 GB hard reserve/); + await until(() => !alive(pids.backend) && !alive(pids.worker)); +}); + +test('owned Docker startup is killed by immutable container ID on a hard breach', async (t) => { + const dir = await fs.mkdtemp(path.join(os.tmpdir(), 'lloom-docker-memory-')); + const started = path.join(dir, 'started'); + const killed = path.join(dir, 'killed'); + const docker = path.join(dir, 'docker'); + await fs.writeFile( + docker, + `#!/usr/bin/env node +const fs=require('fs');const started=${JSON.stringify(started)},killed=${JSON.stringify(killed)}; +switch(process.argv[2]) { +case 'inspect': console.log(JSON.stringify({Id:'owned-id',State:{Running:fs.existsSync(started)&&!fs.existsSync(killed),Status:'created'}})); break; +case 'start': fs.writeFileSync(started,'started'); break; +case 'kill': if(process.argv[3]!=='owned-id')process.exit(3);fs.writeFileSync(killed,'killed');break; +default: process.exit(4); +} +`, + { mode: 0o755 } + ); + const oldPath = process.env.PATH; + process.env.PATH = dir + path.delimiter + oldPath; + t.after(async () => { + process.env.PATH = oldPath; + await fs.rm(dir, { recursive: true, force: true }); + }); + const manager = new RuntimeManager( + { + runtimePolicy: { memorySafety: { pollIntervalMs: 50 } }, + runtimes: { + demo: { + enabled: true, + adapter: 'docker', + management: 'managed', + containerName: 'test-name', + healthUrl: 'http://127.0.0.1:1/health', + healthTimeoutMs: 50 + } + } + }, + { + logger: { error() {} }, + memorySampler: async () => + await fs.access(started).then( + () => pressure, + () => healthy + ) + } + ); + await assert.rejects(manager.start('demo'), (e) => e.code === 'runtime_memory_safety_abort'); + assert.equal(await fs.readFile(killed, 'utf8'), 'killed'); +}); + +test('supervisor preflight rejection cannot race an immediate health success', async (t) => { + const f = await fixture(t); + const totalBytes = os.totalmem(); + f.manager.config.runtimePolicy.memorySafety.minAvailableMemoryGb = totalBytes / GiB; + f.manager.memorySampler = async () => ({ totalBytes: totalBytes * 4, availableBytes: totalBytes * 3 }); + let probed = false; + f.manager.waitForHealth = async () => { + probed = true; + return { healthy: true }; + }; + await assert.rejects(f.manager.start('demo'), (e) => e.code === 'runtime_memory_safety_abort'); + assert.equal(probed, false); + assert.equal( + await fs.access(f.receipt).then( + () => true, + () => false + ), + false + ); +}); + +test('late health success after a safety abort cannot become a successful load', async (t) => { + const f = await fixture(t); + let release; + f.manager.waitForHealth = () => + new Promise((resolve) => { + release = resolve; + }); + let trip = false; + f.manager.memorySampler = async () => (trip ? pressure : healthy); + const starting = f.manager.start('demo'); + const rejected = assert.rejects(starting, (e) => e.code === 'runtime_memory_safety_abort'); + await until(() => + fs.access(f.receipt).then( + () => Boolean(release), + () => false + ) + ); + const pids = await f.ids(); + trip = true; + await until(() => !alive(pids.backend)); + release({ healthy: true }); + await rejected; + assert.equal(f.manager.stateFor('demo').status, 'failed'); +}); diff --git a/test/runtime-policy.test.mjs b/test/runtime-policy.test.mjs index 6f38003..e8bc7ba 100644 --- a/test/runtime-policy.test.mjs +++ b/test/runtime-policy.test.mjs @@ -1,5 +1,14 @@ import assert from 'node:assert/strict'; -import { applyRuntimePolicyPlan, createRuntimePolicyPlan, runtimeAdmissionBlockers } from '../src/runtime-policy.mjs'; +import { + applyRuntimePolicyPlan as applyPolicy, + createRuntimePolicyPlan as createPolicy, + runtimeAdmissionBlockers +} from '../src/runtime-policy.mjs'; + +// Keep synthetic capacity tests independent of the developer machine's load. +const profile = { totalMemoryGb: 128, availableMemoryGb: 128 }; +const createRuntimePolicyPlan = (config, options) => createPolicy(config, { profile, ...options }); +const applyRuntimePolicyPlan = (config, manager, options) => applyPolicy(config, manager, { profile, ...options }); const runtimePolicyConfig = { runtimePolicy: { @@ -248,6 +257,64 @@ const predictiveConfig = { requested: { enabled: true, memoryGb: 64 } } }; + +const macReserveConfig = { + runtimePolicy: { enabled: true, autoEvict: false, reserveMemoryGb: 12 }, + runtimes: { bonsai: { enabled: true, memoryGb: 16 } } +}; +const macIdleStatus = { runtimes: { bonsai: { healthy: false, status: 'idle' } } }; +for (const runtimePolicy of [macReserveConfig.runtimePolicy, { enabled: true, memoryBudgetGb: 84 }]) { + const pressured = await createRuntimePolicyPlan( + { ...macReserveConfig, runtimePolicy }, + { + requestedRuntimeId: 'bonsai', + status: macIdleStatus, + profile: { totalMemoryGb: 96, availableMemoryGb: 16 } + } + ); + assert.equal(pressured.admission.predictive, true); + assert.equal(pressured.admission.projectedMemoryGb, 96); + assert.equal(pressured.admission.allowed, false, 'other applications count without a percentage policy'); +} +const macEstimateFits = await createRuntimePolicyPlan(macReserveConfig, { + requestedRuntimeId: 'bonsai', + status: macIdleStatus, + profile: { totalMemoryGb: 96, availableMemoryGb: 35 } +}); +assert.equal(macEstimateFits.admission.projectedMemoryGb, 77); +assert.equal(macEstimateFits.admission.allowed, true, 'live startup guard must catch an underestimated load later'); + +const clusteredReserve = await createRuntimePolicyPlan( + { + ...macReserveConfig, + cluster: { nodeId: 'mac', leaderNode: 'mac', nodes: { mac: { resources: { memoryGb: 96 } } } }, + runtimes: { bonsai: { enabled: true, node: 'mac', memoryGb: 16 } } + }, + { + requestedRuntimeId: 'bonsai', + requesterNode: 'mac', + status: { + ...macIdleStatus, + cluster: { + nodes: { + mac: { + local: true, + reachable: true, + telemetry: { memory: { totalBytes: 96 * 1024 ** 3, availableBytes: 16 * 1024 ** 3 } } + } + } + } + }, + profile: { totalMemoryGb: 96, availableMemoryGb: 16 } + } +); +assert.equal( + clusteredReserve.admission.allowed, + false, + 'local cluster node also counts host pressure without a percentage' +); +assert.equal(clusteredReserve.admission.nodes.mac.predictive, true); + const predictiveStatus = { runtimes: { loaded: { healthy: true, status: 'running', activeRequests: 0 }, diff --git a/test/smoke.mjs b/test/smoke.mjs index f9f2eab..2d5a61c 100644 --- a/test/smoke.mjs +++ b/test/smoke.mjs @@ -1080,14 +1080,10 @@ const libraryCli = await runCommand(process.execPath, [ const libraryJson = JSON.parse(libraryCli.stdout); assert.equal(libraryJson.index.id, 'lloom-community-recipes'); assert.equal(libraryJson.recipes[0].id, 'apple-silicon-flux2-klein-4b'); -if (process.platform === 'darwin' && process.arch === 'arm64') { - assert.equal(libraryJson.selected.recipeId, 'apple-silicon-qwen36-35b-a3b-optiq'); -} else { - // CPU-only hosts can now select the lightweight Hear recipe. A missing GPU - // does not imply that the entire vendor library is incompatible. - const compatible = libraryJson.candidates.filter((candidate) => candidate.selectable); - assert.equal(libraryJson.selected?.recipeId ?? null, compatible[0]?.recipeId ?? null); -} +// Selection depends on the host's live memory and already-running backends. +// Verify the ranked decision without requiring a particular local model lane. +const compatibleLibraryCandidates = libraryJson.candidates.filter((candidate) => candidate.selectable); +assert.equal(libraryJson.selected?.recipeId ?? null, compatibleLibraryCandidates[0]?.recipeId ?? null); const addModelCli = await runCommand(process.execPath, [ path.join(process.cwd(), 'bin', 'lloom.mjs'), 'add-model', @@ -5963,12 +5959,8 @@ if (listened) { assert.equal(libraryResponse.status, 200); const libraryPlanJson = await libraryResponse.json(); assert.equal(libraryPlanJson.index.id, 'lloom-community-recipes'); - if (process.platform === 'darwin' && process.arch === 'arm64') { - assert.equal(libraryPlanJson.selected.recipeId, 'apple-silicon-qwen36-35b-a3b-optiq'); - } else { - const compatible = libraryPlanJson.candidates.filter((candidate) => candidate.selectable); - assert.equal(libraryPlanJson.selected?.recipeId ?? null, compatible[0]?.recipeId ?? null); - } + const compatible = libraryPlanJson.candidates.filter((candidate) => candidate.selectable); + assert.equal(libraryPlanJson.selected?.recipeId ?? null, compatible[0]?.recipeId ?? null); assert.equal( libraryPlanJson.recipes.find((recipe) => recipe.id === 'apple-silicon-qwen36-35b-a3b-optiq')?.commands .installApply, From 534c80ac1cca819b434b886fd166e6b8ea369520 Mon Sep 17 00:00:00 2001 From: Jason McCartney Date: Mon, 21 Sep 2026 22:44:20 -0700 Subject: [PATCH 08/16] Wait for the measured stall deadline before reporting watchdog failure --- src/server.mjs | 45 ++++++++++++++++++++-------------- test/runtime-watchdog.test.mjs | 1 + 2 files changed, 27 insertions(+), 19 deletions(-) diff --git a/src/server.mjs b/src/server.mjs index 13eb482..940c065 100644 --- a/src/server.mjs +++ b/src/server.mjs @@ -2589,25 +2589,32 @@ export function createLloomServer(config, { logger = console, runtimeManager = n if (!watchdogConfig.enabled || stream !== true) return; watchdogArmed = true; clearWatchdogTimer(); - watchdogTimer = setTimeout( - () => { - watchdogTimer = null; - noteRuntimeRequestOutcome(resolved.model.runtime, { - id: connectionId, - route, - model: resolved.model.id, - status: 504, - ok: false, - durationMs: Date.now() - started, - runtimeDurationMs: runtimeStartedAt == null ? 0 : Date.now() - runtimeStartedAt, - stallDurationMs: Date.now() - lastProgressAt, - hadContent: watchdogHadContent, - responseBytes: 0, - stalled: true - }); - }, - watchdogHadContent ? watchdogConfig.idleContentTimeoutMs : watchdogConfig.firstContentTimeoutMs - ); + const timeoutMs = watchdogHadContent ? watchdogConfig.idleContentTimeoutMs : watchdogConfig.firstContentTimeoutMs; + const onTimeout = () => { + watchdogTimer = null; + const stallDurationMs = Date.now() - lastProgressAt; + // A timer may wake slightly before the wall-clock deadline. Keep + // waiting rather than report a stall that classification must ignore. + if (stallDurationMs < timeoutMs) { + watchdogTimer = setTimeout(onTimeout, timeoutMs - stallDurationMs); + watchdogTimer.unref?.(); + return; + } + noteRuntimeRequestOutcome(resolved.model.runtime, { + id: connectionId, + route, + model: resolved.model.id, + status: 504, + ok: false, + durationMs: Date.now() - started, + runtimeDurationMs: runtimeStartedAt == null ? 0 : Date.now() - runtimeStartedAt, + stallDurationMs, + hadContent: watchdogHadContent, + responseBytes: 0, + stalled: true + }); + }; + watchdogTimer = setTimeout(onTimeout, timeoutMs); watchdogTimer.unref?.(); }; const progress = (patch) => { diff --git a/test/runtime-watchdog.test.mjs b/test/runtime-watchdog.test.mjs index 8a938cb..d251fd9 100644 --- a/test/runtime-watchdog.test.mjs +++ b/test/runtime-watchdog.test.mjs @@ -398,6 +398,7 @@ for (const mode of ['observe', 'defer', 'force', 'disable-during-drain', 'recove const idleStall = observations.find((item) => item.stalled); assert.ok(idleStall, 'shorter idle budget detects silence after content'); assert.equal(idleStall.hadContent, true); + assert.ok(idleStall.stallDurationMs >= 20, 'timer must not report a stall before its elapsed-time threshold'); assert.ok(idleStall.stallDurationMs < idleStall.runtimeDurationMs); assert.equal( classifyRuntimeWatchdogOutcome(config.runtimes['watchdog-runtime'], idleStall).kind, From 9781b5de6f58f9dc0a0db9483fd9b96b0714ffba Mon Sep 17 00:00:00 2001 From: Jason McCartney Date: Tue, 22 Sep 2026 00:15:47 -0700 Subject: [PATCH 09/16] Show interactive memory blocks and make model use automatic --- README.md | 2 +- docs/architecture.md | 9 + package.json | 6 +- src/cluster.mjs | 6 +- src/darwin-memory-usage.py | 41 +++ src/dashboard-memory.mjs | 422 +++++++++++++++++++++++++++++ src/dashboard-presence-client.mjs | 67 +++-- src/dashboard-presence.mjs | 10 +- src/dashboard-scene.mjs | 204 ++++++++++++-- src/dashboard.mjs | 9 +- src/runtime-manager.mjs | 23 +- src/runtime-memory-usage.mjs | 395 +++++++++++++++++++++++++++ src/server.mjs | 4 +- test/dashboard-memory.test.mjs | 358 ++++++++++++++++++++++++ test/runtime-memory-usage.test.mjs | 365 +++++++++++++++++++++++++ 15 files changed, 1866 insertions(+), 55 deletions(-) create mode 100644 src/darwin-memory-usage.py create mode 100644 src/dashboard-memory.mjs create mode 100644 src/runtime-memory-usage.mjs create mode 100644 test/dashboard-memory.test.mjs create mode 100644 test/runtime-memory-usage.test.mjs diff --git a/README.md b/README.md index 24cf0ec..08484a1 100644 --- a/README.md +++ b/README.md @@ -100,7 +100,7 @@ Then open OMP normally. The generated OMP config points at `http://127.0.0.1:810 - Backend recipes for vLLM, SGLang, MTPLX, MLX LM, llama.cpp, Ollama, OptiQ, and stable-diffusion.cpp, with dedicated DGX Spark / GB10 and Apple Silicon recipes. - Community recipe packs and hardware-matched benchmark evidence so machines can select the best known model/backend recipe automatically instead of blindly chasing global tok/s. - Generated client profiles for OMP, OpenCode, Codex-compatible, Claude-compatible, Hermes, Zero, and any OpenAI-compatible client. -- A browser dashboard with Live, Models, Machines, Clients, and Settings. Inspect real topology, install models, choose their readiness, load or unload managed runtimes, and try chat through the gateway. The action camera follows serving models while preserving manual zoom. +- A browser dashboard with Live, Models, Machines, Clients, and Settings. Inspect real topology and memory blocks, preview a model’s expected footprint by pointing or focusing, and use models directly through chat or connected apps. LLooM prepares models automatically; optional readiness and memory controls live under Options & details. The action camera follows serving models while preserving manual zoom. ## Daily Commands diff --git a/docs/architecture.md b/docs/architecture.md index 52bfe22..668b505 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -318,3 +318,12 @@ The current repository includes a static `lloom-host` development server that se Setup composes initialization, backend setup, recipe setup, generated clients, and client integration writes into one audited plan. It does not bypass the lower-level safety gates: dry-run is the default, and real execution requires explicit `--apply --yes`. Bootstrap remains the lower-level backend/model/client phase for an existing config. The default generated gateway port is `8100`; selected backend runtimes occupy the default backend range beginning at `8201`. `setup --port` and `setup --backend-port-range` retarget the generated provider URL, backend base URLs, runtime ports, health URLs, and warmup URLs together so custom port layouts remain internally consistent. + + +### Dashboard memory map + +The Models view divides one machine’s physical memory into running backends, system/other applications, and available space. Protected headroom is an overlay within available memory, not additional capacity. Hover, keyboard focus, and selection preview the incoming model’s configured peak estimate against that machine’s live memory and safety reserve. These interactions are read-only. Normal inference prepares the model through admission; optional manual preparation uses the same guarded admission endpoint. + +Dashboard and node-status requests opt into passive memory attribution. A five-second shared cache reads process memory and loopback listeners with bounded subprocesses. Overlapping process trees are counted once; Docker PIDs are used only on Linux. Known local Ollama and LLooM audio servers also expose their loaded-model lists through bounded, read-only requests. A healthy server with an empty model cache is not a resident model. Routing and admission status do not request this dashboard sampling. + +On macOS, an optional bounded Python helper reads Darwin’s physical-footprint accounting, which includes charged Metal and compressed memory. If unavailable, attribution falls back to RSS. Process memory is approximate and does not describe all shared or cached allocations. Unmeasured running backends use visibly marked estimates. If attribution exceeds measured host use, blocks are reconciled to that measured total. The host’s available-memory reading remains authoritative. Remote nodes use their own observations and safety policy; distributed capacity is never pooled into a single fit promise. Preview estimates never override admission or the live memory safety guard. diff --git a/package.json b/package.json index e09b166..a91dba0 100644 --- a/package.json +++ b/package.json @@ -64,8 +64,8 @@ "check": "npm run check:cluster && npm run check:js && npm run check:entity-stagger && npm run check:python && npm run check:shell && npm run check:spark-deploy && npm run check:workers", "check:cluster": "node --check src/cluster.mjs && node --check test/cluster.test.mjs", "check:entity-stagger": "node --check scripts/entity-stagger-benchmark.mjs && node --check test/entity-stagger-benchmark.test.mjs", - "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs && node --check src/dashboard-scene.mjs && node --check src/dashboard-presence.mjs && node --check src/dashboard-presence-client.mjs && node --check src/first-run.mjs && node --check src/first-run-page.mjs && node --check src/browser-setup.mjs && node --check src/runtime-preferences.mjs && node --check src/installation-jobs.mjs && node --check src/model-installation.mjs && node --check src/dashboard-installation.mjs && node --check src/runtime-memory-safety.mjs", - "check:python": "python3 -m py_compile backends/dspark-vllm/apply-patch-pack.py backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/vision_exp/*.py backends/qwen38-vllm/apply-ple-fp8-patch.py backends/qwen38-sglang/*.py backends/glm53-exl3/*.py backends/mlx-audio/lloom_audio_server.py backends/hear/lloom_hear_server.py backends/chatterbox/lloom_chatterbox_server.py patches/apply_mtplx_longctx_fix.py backends/comfyui-media/bridge/*.py backends/comfyui-media/graphs/*.py backends/comfyui-media/build/*.py backends/comfyui-media/install.py backends/qwen-image-diffusers/*.py backends/hear/test/test_hear_security.py", + "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs && node --check src/dashboard-scene.mjs && node --check src/dashboard-presence.mjs && node --check src/dashboard-presence-client.mjs && node --check src/first-run.mjs && node --check src/first-run-page.mjs && node --check src/browser-setup.mjs && node --check src/runtime-preferences.mjs && node --check src/installation-jobs.mjs && node --check src/model-installation.mjs && node --check src/dashboard-installation.mjs && node --check src/runtime-memory-safety.mjs && node --check src/dashboard-memory.mjs && node --check src/runtime-memory-usage.mjs", + "check:python": "python3 -m py_compile backends/dspark-vllm/apply-patch-pack.py backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/vision_exp/*.py backends/qwen38-vllm/apply-ple-fp8-patch.py backends/qwen38-sglang/*.py backends/glm53-exl3/*.py backends/mlx-audio/lloom_audio_server.py backends/hear/lloom_hear_server.py backends/chatterbox/lloom_chatterbox_server.py patches/apply_mtplx_longctx_fix.py backends/comfyui-media/bridge/*.py backends/comfyui-media/graphs/*.py backends/comfyui-media/build/*.py backends/comfyui-media/install.py backends/qwen-image-diffusers/*.py backends/hear/test/test_hear_security.py src/darwin-memory-usage.py", "check:shell": "bash -n backends/dspark-vllm/entrypoint.sh backends/dspark-vllm/runtime/build-hardened-image.sh backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/apply-runtime.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.sh backends/qwen38-vllm/entrypoint.sh backends/qwen38-sglang/entrypoint.sh backends/glm53-exl3/entrypoint.sh backends/mlx-audio/install.sh backends/hear/install.sh backends/chatterbox/install.sh clients/examples/lloom-claude clients/examples/lloom-codex clients/examples/lloom-hermes clients/examples/lloom-opencode clients/examples/lloom-zero scripts/remote-install-spark.sh scripts/remote-stage-spark-worker.sh", "check:spark-deploy": "node --check test/spark-deploy.test.mjs && node test/spark-deploy.test.mjs", "generate:clients": "node scripts/generate-client-configs.mjs", @@ -73,7 +73,7 @@ "community:check": "node scripts/check-community-deploy.mjs", "package:check": "node scripts/check-package.mjs", "test:adoption": "node test/resource-fit.test.mjs && node test/model-acquisition.test.mjs && node test/dashboard-status.test.mjs", - "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs && npm run test:ux && npm run test:memory-safety", + "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs && npm run test:ux && npm run test:memory-safety && node --test test/dashboard-memory.test.mjs test/runtime-memory-usage.test.mjs", "test:cluster": "node test/cluster.test.mjs", "test": "npm run test:cluster && npm run test:unit && npm run test:entity-stagger && npm run smoke && npm run community:check", "test:entity-stagger": "node test/entity-stagger-benchmark.test.mjs", diff --git a/src/cluster.mjs b/src/cluster.mjs index 726f2ff..45c7743 100644 --- a/src/cluster.mjs +++ b/src/cluster.mjs @@ -682,7 +682,7 @@ export class ClusterCoordinator { } } - async localNodeStatus({ runtimeStatus = null } = {}) { + async localNodeStatus({ runtimeStatus = null, includeMemoryUsage = false } = {}) { const profile = typeof this.profile === 'function' ? await this.profile() : this.profile; const models = typeof this.models === 'function' ? await this.models() : this.models; return { @@ -704,7 +704,9 @@ export class ClusterCoordinator { telemetry: this.telemetry ? await this.telemetry.snapshot() : null, runtimeManager: runtimeStatus ?? - (this.runtimeManager ? await this.runtimeManager.status({ localOnly: true }) : { runtimes: {} }) + (this.runtimeManager + ? await this.runtimeManager.status({ localOnly: true, includeMemoryUsage }) + : { runtimes: {} }) }; } diff --git a/src/darwin-memory-usage.py b/src/darwin-memory-usage.py new file mode 100644 index 0000000..5944cc9 --- /dev/null +++ b/src/darwin-memory-usage.py @@ -0,0 +1,41 @@ +"""Read Darwin's per-process physical footprint; never inspect process contents. + +The rusage_info_v2 ABI is defined by Apple's sys/resource.h (macOS 10.9+). +Physical footprint includes charged memory that RSS omits, notably Metal and +compressed allocations. A failed per-PID observation stays absent. +""" +import ctypes +import json +import sys + + +class RusageInfoV2(ctypes.Structure): + _fields_ = [('uuid', ctypes.c_uint8 * 16)] + [ + (name, ctypes.c_uint64) for name in ( + 'user_time', 'system_time', 'pkg_idle_wkups', 'interrupt_wkups', + 'pageins', 'wired_size', 'resident_size', 'phys_footprint', + 'proc_start_abstime', 'proc_exit_abstime', 'child_user_time', + 'child_system_time', 'child_pkg_idle_wkups', 'child_interrupt_wkups', + 'child_pageins', 'child_elapsed_abstime', 'diskio_bytesread', + 'diskio_byteswritten' + ) + ] + + +def main(): + libproc = ctypes.CDLL('/usr/lib/libproc.dylib') + libproc.proc_pid_rusage.argtypes = [ctypes.c_int, ctypes.c_int, ctypes.c_void_p] + libproc.proc_pid_rusage.restype = ctypes.c_int + result = {} + for value in sys.argv[1:]: + pid = int(value) + if not 0 < pid < 2 ** 31: + continue + usage = RusageInfoV2() + if libproc.proc_pid_rusage(pid, 2, ctypes.byref(usage)) == 0: + result[str(pid)] = usage.phys_footprint + print(json.dumps(result)) + + +if __name__ == '__main__': + main() diff --git a/src/dashboard-memory.mjs b/src/dashboard-memory.mjs new file mode 100644 index 0000000..8195cad --- /dev/null +++ b/src/dashboard-memory.mjs @@ -0,0 +1,422 @@ +function buildMemoryMap({ node = null, runtimes = {}, models = [], previewModelId = null, memorySafety = null } = {}) { + if (node?.local !== true && node?.runtimeManager?.runtimes) { + runtimes = { ...runtimes }; + for (const [id, observed] of Object.entries(node.runtimeManager.runtimes)) { + runtimes[id] = { ...runtimes[id], ...observed, node: node.id, remote: false }; + } + } + const GiB = 1024 * 1024 * 1024; + + function numberOrNull(value) { + return typeof value === 'number' && Number.isFinite(value) ? value : null; + } + + function nonNegativeOrNull(value) { + const number = numberOrNull(value); + return number !== null && number >= 0 ? number : null; + } + + function bounded(value, minimum, maximum) { + const number = numberOrNull(value); + return number === null || number < minimum || number > maximum ? null : number; + } + + function positiveOrNull(value) { + const number = numberOrNull(value); + return number !== null && number > 0 ? number : null; + } + + function sameNode(runtime, currentNode) { + if ( + !runtime || + runtime.remote === true || + runtime.distributed === true || + runtime.placement?.mode === 'distributed' + ) + return false; + const placementNode = runtimeNode(runtime); + if (placementNode) return placementNode === currentNode; + return node?.local === true; + } + + function runtimeNode(runtime) { + return runtime?.node ?? runtime?.placement?.node ?? null; + } + + function safetyReserve(policy, total) { + if (!policy) return null; + if (policy.mode === 'yolo') return 0; + const minAvailable = nonNegativeOrNull(policy.minAvailableMemoryGb); + const maxUtilization = nonNegativeOrNull(policy.maxMemoryUtilization); + if (minAvailable === null || maxUtilization === null || maxUtilization > 1) return null; + return Math.min(total, Math.max(minAvailable * GiB, total * (1 - maxUtilization))); + } + + function colorIndex(value) { + const text = String(value ?? ''); + let hash = 2166136261; + for (let index = 0; index < text.length; index += 1) { + hash ^= text.charCodeAt(index); + hash = (hash * 16777619) >>> 0; + } + return hash % 8; + } + + function targetNodes(model) { + return (Array.isArray(model?.targets) ? model.targets : []) + .map((target) => { + if (typeof target === 'string') return target; + return target?.node ?? target?.id ?? null; + }) + .filter(Boolean); + } + + function uniqueSorted(values) { + return [...new Set(values.filter(Boolean))].sort((left, right) => String(left).localeCompare(String(right))); + } + + function roundPercent(value) { + return Number.isFinite(value) ? Math.round(value * 1e6) / 1e6 : null; + } + + const nodeId = node?.id ?? null; + const telemetryMemory = node?.telemetry?.memory; + const totalBytes = nonNegativeOrNull(telemetryMemory?.totalBytes); + const availableBytes = telemetryMemory ? bounded(telemetryMemory.availableBytes, 0, totalBytes) : null; + const reportedUsedBytes = telemetryMemory ? bounded(telemetryMemory.usedBytes, 0, totalBytes) : null; + const hasCapacity = typeof totalBytes === 'number' && totalBytes > 0; + const invalidAvailable = telemetryMemory?.availableBytes != null && availableBytes === null; + const invalidUsed = telemetryMemory?.usedBytes != null && reportedUsedBytes === null; + const hasMemoryFacts = + hasCapacity && !invalidAvailable && !invalidUsed && (availableBytes !== null || reportedUsedBytes !== null); + const known = + (node?.reachable === true || (node?.local === true && node?.reachable !== false)) && hasCapacity && hasMemoryFacts; + let usedBytes; + let availableMemoryBytes; + if (hasCapacity && availableBytes !== null) { + availableMemoryBytes = availableBytes; + usedBytes = totalBytes - availableBytes; + } else if (hasCapacity && reportedUsedBytes !== null) { + usedBytes = reportedUsedBytes; + availableMemoryBytes = totalBytes - usedBytes; + } else { + usedBytes = null; + availableMemoryBytes = null; + } + const activePolicy = node?.local === true ? memorySafety : node?.runtimeManager?.memorySafety; + const reserveBytes = hasMemoryFacts ? safetyReserve(activePolicy, totalBytes) : null; + const usableBytes = hasMemoryFacts && reserveBytes !== null ? Math.max(0, availableMemoryBytes - reserveBytes) : null; + + if (!known) { + return { + known: false, + totalBytes: hasCapacity ? totalBytes : null, + usedBytes: null, + availableBytes: hasCapacity ? availableBytes : null, + reserveBytes: null, + usableBytes: null, + nodeId, + segments: [], + attributionNote: 'Memory telemetry is unavailable; no allocation is inferred.', + preview: null + }; + } + + const runtimeIds = Object.keys(runtimes ?? {}).sort((left, right) => left.localeCompare(right)); + const includedRuntimeIds = []; + for (const runtimeId of runtimeIds) { + const runtime = runtimes[runtimeId]; + const status = String(runtime?.status ?? '').toLowerCase(); + const active = ['running', 'healthy', 'starting', 'warming', 'external', 'stopping', 'draining'].includes(status); + if (!active || runtime?.paused === true) continue; + if (!sameNode(runtime, nodeId)) continue; + includedRuntimeIds.push(runtimeId); + } + + const runtimeGroups = new Map(); + for (const runtimeId of includedRuntimeIds) { + const runtime = runtimes[runtimeId]; + const usage = runtime?.memoryUsage; + const groupId = typeof usage?.groupId === 'string' && usage.groupId ? usage.groupId : `runtime:${runtimeId}`; + if (!runtimeGroups.has(groupId)) { + runtimeGroups.set(groupId, { runtimeIds: [], usage: null }); + } + const group = runtimeGroups.get(groupId); + group.runtimeIds.push(runtimeId); + if (!group.usage) group.usage = usage ?? null; + } + + const attributed = []; + for (const [groupId, group] of runtimeGroups) { + const primaryRuntimeId = group.runtimeIds[0]; + const runtime = runtimes[primaryRuntimeId]; + const usages = group.runtimeIds.map((runtimeId) => runtimes[runtimeId]?.memoryUsage).filter(Boolean); + const rssValues = usages.map((usage) => nonNegativeOrNull(usage.residentBytes)).filter((value) => value !== null); + const rssBytes = rssValues.length ? Math.max(...rssValues) : null; + const estimateValues = group.runtimeIds + .map((runtimeId) => positiveOrNull(runtimes[runtimeId]?.memoryGb)) + .filter((value) => value !== null) + .map((value) => value * GiB); + const bytes = rssBytes !== null ? rssBytes : estimateValues.length ? Math.max(...estimateValues) : null; + if (bytes === null) continue; + const modelIds = new Set(); + for (const runtimeId of group.runtimeIds) { + for (const model of models ?? []) { + if (model?.runtime === runtimeId) { + if (model.id) modelIds.add(model.id); + } + } + } + const names = uniqueSorted([...modelIds].map((id) => models.find((model) => model.id === id)?.name || id)); + const label = names.length + ? names.slice(0, 2).join(' + ') + (names.length > 2 ? ' +' + (names.length - 2) : '') + : (runtime?.id ?? primaryRuntimeId); + const estimated = rssBytes === null; + attributed.push({ + id: groupId, + kind: 'runtime', + label, + bytes: Math.round(bytes), + percent: roundPercent((bytes / totalBytes) * 100), + runtimeId: primaryRuntimeId, + modelIds: uniqueSorted([...modelIds]), + estimated + }); + } + + let reconciliationScale = 1; + let attributedBytes = attributed.reduce((sum, segment) => sum + segment.bytes, 0); + if (attributedBytes > usedBytes) { + reconciliationScale = usedBytes / attributedBytes; + for (const segment of attributed) { + segment.bytes = Math.round(segment.bytes * reconciliationScale); + segment.estimated = true; + } + attributedBytes = attributed.reduce((sum, segment) => sum + segment.bytes, 0); + } + attributed.sort((left, right) => String(left.runtimeId).localeCompare(String(right.runtimeId))); + const colors = new Set(); + for (const segment of attributed) { + let index = colorIndex(segment.runtimeId); + if (colors.size < 8) while (colors.has(index)) index = (index + 1) % 8; + colors.add(index); + segment.colorIndex = index; + segment.percent = roundPercent((segment.bytes / totalBytes) * 100); + } + + if (attributedBytes > usedBytes && attributed.length) { + const excess = attributedBytes - usedBytes; + const largest = attributed.reduce((left, right) => (right.bytes > left.bytes ? right : left)); + largest.bytes = Math.max(0, largest.bytes - excess); + attributedBytes = attributed.reduce((sum, segment) => sum + segment.bytes, 0); + } + for (const segment of attributed) segment.percent = roundPercent((segment.bytes / totalBytes) * 100); + const systemBytes = Math.max(0, usedBytes - attributedBytes); + + const segments = [...attributed]; + if (systemBytes > 0 || usedBytes === 0) { + segments.push({ + id: 'system', + kind: 'system', + label: 'System & other apps', + bytes: systemBytes, + percent: roundPercent((systemBytes / totalBytes) * 100), + runtimeId: null, + modelIds: [], + estimated: false, + colorIndex: 8 + }); + } + const availableSegmentBytes = totalBytes - usedBytes; + segments.push({ + id: 'available', + kind: 'available', + label: 'Available', + bytes: availableSegmentBytes, + percent: roundPercent((availableSegmentBytes / totalBytes) * 100), + runtimeId: null, + modelIds: [], + estimated: false, + colorIndex: 9 + }); + + const attributionParts = ['Live process memory is approximate. Shared models use one block. Previews are estimates.']; + if (reconciliationScale !== 1) + attributionParts.push('App measurements overlap; blocks are adjusted to match total memory in use.'); + if ( + [...runtimeGroups.values()].some((group) => + group.runtimeIds.every((runtimeId) => { + const usage = runtimes[runtimeId]?.memoryUsage; + return !usage || nonNegativeOrNull(usage.residentBytes) === null; + }) + ) + ) + attributionParts.push('Striped blocks use an estimate until a live reading is available.'); + + const preview = previewModelId ? previewModel(previewModelId) : null; + function previewModel(modelId) { + const model = (models ?? []).find((item) => item?.id === modelId); + if (!model) return null; + const runtimeId = model.runtime ?? null; + const runtime = runtimeId ? runtimes[runtimeId] : null; + const label = model.name ?? model.id; + const base = { + modelId, + label, + runtimeId, + status: 'unknown', + additionalBytes: null, + remainingBytes: null, + projectedUsedBytes: null, + percent: null, + nodeId, + shared: false, + message: '' + }; + const targets = targetNodes(model); + const otherTargets = targets.filter((target) => target !== nodeId); + const allTargetsAreOther = targets.length > 0 && otherTargets.length === targets.length; + const runtimePlacementNode = runtime ? runtimeNode(runtime) : null; + const runtimeIsOtherNode = + (runtime?.remote === true && runtimePlacementNode === null) || + (runtimePlacementNode !== null && runtimePlacementNode !== nodeId); + if (runtimeIsOtherNode || allTargetsAreOther) { + return { + ...base, + status: 'other-node', + additionalBytes: 0, + projectedUsedBytes: usedBytes, + remainingBytes: availableMemoryBytes, + percent: roundPercent((usedBytes / totalBytes) * 100), + nodeId: targets[0] ?? runtimePlacementNode, + message: `Runs on node ${targets[0] ?? runtimePlacementNode ?? 'another node'}; no impact on this machine.` + }; + } + if (runtime?.distributed === true || runtime?.placement?.mode === 'distributed' || new Set(targets).size > 1) { + return { ...base, status: 'unknown', shared: true, message: 'Needs room on each machine.' }; + } + if ( + runtime?.maintenance || + runtime?.enabled === false || + runtime?.paused === true || + runtime?.status === 'paused' || + model.paused === true + ) { + return { ...base, status: 'paused', message: 'Paused; automatic starts are disabled.' }; + } + if (!runtimeId) { + return { + ...base, + status: 'external', + additionalBytes: 0, + projectedUsedBytes: usedBytes, + remainingBytes: availableMemoryBytes, + percent: roundPercent((usedBytes / totalBytes) * 100), + message: 'No local model load is expected.' + }; + } + if (!runtime) { + return { ...base, status: 'unknown', message: 'No runtime status is available.' }; + } + const usage = runtime.memoryUsage; + const loadedIds = Array.isArray(usage?.loadedModelIds) ? usage.loadedModelIds : null; + const wantedId = model.upstreamModel ?? model.id; + const residentConfirmed = + runtime.healthy === true && + usage?.residencyKnown === true && + loadedIds !== null && + (loadedIds.includes(wantedId) || loadedIds.includes(model.id)); + const commandParts = String(runtime.command ?? '') + .trim() + .split(/\s+/); + const commandBase = commandParts[0]?.split(/[\\/]/).pop() ?? ''; + const lazyBackend = [commandBase, ...(runtime.args ?? [])].some((part) => + /(?:^|[/\\])(ollama|lloom-audio-server|lloom_audio_server(?:\.py)?)$/.test(String(part)) + ); + const sharedBackend = usage?.sharedRuntimeIds?.length > 1 || lazyBackend; + const modelsOnRuntime = (models ?? []).filter((item) => item?.runtime === runtimeId); + const runtimeIsSharedGroup = runtimeGroups.get(usage?.groupId ?? `runtime:${runtimeId}`)?.runtimeIds.length > 1; + if (residentConfirmed) { + return { + ...base, + status: 'resident', + additionalBytes: 0, + projectedUsedBytes: usedBytes, + remainingBytes: availableMemoryBytes, + percent: roundPercent((usedBytes / totalBytes) * 100), + shared: sharedBackend || runtimeIsSharedGroup, + message: + sharedBackend || runtimeIsSharedGroup + ? 'Already available; shared backend footprint can change.' + : 'Already available' + }; + } + if (sharedBackend && runtime.healthy === true && !(usage?.residencyKnown && loadedIds !== null)) { + return { ...base, status: 'unknown', shared: true, message: 'Shared backend residency is unknown.' }; + } + const memoryGb = positiveOrNull(runtime.memoryGb); + if (memoryGb === null) { + return { ...base, status: 'unknown', message: 'Cold start estimate is unknown.' }; + } + // A healthy server may load weights lazily. Only confirmed model-cache + // observations promise residency; otherwise forecast growth toward its peak. + const measured = + !sharedBackend && !runtimeIsSharedGroup && modelsOnRuntime.length === 1 && runtime.healthy === true + ? nonNegativeOrNull(usage?.residentBytes) + : null; + const incoming = Math.max(0, memoryGb * GiB - (measured ?? 0)); + const remaining = availableMemoryBytes - incoming; + const reserve = reserveBytes ?? null; + if (reserve === null) { + return { + ...base, + status: 'unknown', + additionalBytes: incoming, + remainingBytes: remaining, + projectedUsedBytes: usedBytes + incoming, + percent: roundPercent((incoming / totalBytes) * 100), + shared: sharedBackend, + message: 'Memory reserve is unknown; no fit is guaranteed.' + }; + } + if (remaining <= reserve) { + return { + ...base, + status: 'blocked', + additionalBytes: incoming, + remainingBytes: remaining, + projectedUsedBytes: usedBytes + incoming, + percent: roundPercent((incoming / totalBytes) * 100), + shared: sharedBackend, + message: 'Does not fit within the reserve.' + }; + } + const tightLimit = reserve + Math.max(GiB, totalBytes * 0.02); + const tight = remaining <= tightLimit; + return { + ...base, + status: tight ? 'tight' : 'fits', + additionalBytes: incoming, + remainingBytes: remaining, + projectedUsedBytes: usedBytes + incoming, + percent: roundPercent((incoming / totalBytes) * 100), + shared: sharedBackend, + message: tight ? 'Expected to fit, with little reserve headroom.' : 'Expected to fit' + }; + } + + return { + known: true, + totalBytes, + usedBytes, + availableBytes: availableMemoryBytes, + reserveBytes, + usableBytes, + nodeId, + segments, + attributionNote: attributionParts.join(' '), + preview + }; +} + +export { buildMemoryMap }; diff --git a/src/dashboard-presence-client.mjs b/src/dashboard-presence-client.mjs index 24cb78d..f2bc681 100644 --- a/src/dashboard-presence-client.mjs +++ b/src/dashboard-presence-client.mjs @@ -10,7 +10,7 @@ export const presenceScript = String.raw` let presenceMachineKey = ""; let presenceInstallTimer = null; let presenceInstallSeen = null; - const presencePolicyNames = { auto:"Auto", preferred:"Prefer ready", always:"Always ready" }; + const presencePolicyNames = { auto:"Automatic", preferred:"Prefer instant replies", always:"Keep ready" }; function presenceNotice(message, error = false) { const toast = $("#presence-toast"); toast.querySelector("span").textContent = message; @@ -39,16 +39,22 @@ export const presenceScript = String.raw` function presencePolicy(runtime) { return runtime?.keepWarm ? "always" : runtime?.preferredWarm ? "preferred" : "auto"; } + function presenceModelResident(model) { + const rt=presenceRuntime(model),usage=rt?.memoryUsage; + if(!rt?.healthy)return false; + if(usage?.residencyKnown&&Array.isArray(usage.loadedModelIds))return usage.loadedModelIds.some(id=>id===model.id||id===model.upstreamModel); + return true; + } function presenceModelLabel(model) { const runtime = presenceRuntime(model); if (model.alias) return "Route"; if (!model.runtime) return model.federated ? "Shared model" : "External provider"; if (runtime?.maintenance) return "Paused"; - const transition={starting:"Loading",warming:"Warming",queued:"Queued",stopping:"Unloading",draining:"Finishing work",failed:"Needs attention",unreachable:"Unavailable",disabled:"Disabled"}[runtime?.status]; + const transition={starting:"Getting ready",warming:"Getting ready",queued:"Waiting for room",stopping:"Freeing memory",draining:"Finishing work",failed:"Needs attention",unreachable:"Unavailable",disabled:"Disabled"}[runtime?.status]; if(transition)return transition; if (runtime?.activeRequests > 0 && runtime?.healthy) return "Serving"; - if (runtime?.healthy) return "Ready"; - return "On demand"; + if (presenceModelResident(model)) return "Ready to use"; + return "Starts when needed"; } function renderPresenceModels() { const search = $("#presence-search").value.trim().toLowerCase(); @@ -84,13 +90,16 @@ export const presenceScript = String.raw` } function presenceClientExample() { const model = $("#presence-client-model").value; - const body = JSON.stringify({model,messages:[{role:"user",content:"Hello"}]}, null, 2); - $("#presence-client-example").textContent = "POST " + endpoint + "/v1/chat/completions\nContent-Type: application/json\nAuthorization: Bearer YOUR_LLOOM_KEY\n\n" + body; + const selected=(state.physicalModels||[]).find(m=>m.id===model),kind=selected?.kind||'chat'; + const templates={chat:['/v1/chat/completions',{model,messages:[{role:'user',content:'Hello'}]}],embedding:['/v1/embeddings',{model,input:'Text to search'}],image:['/v1/images/generations',{model,prompt:'A quiet mountain lake'}],audio_speech:['/v1/audio/speech',{model,input:'Hello there',voice:'default'}],audio_generation:['/v1/audio/generations',{model,prompt:'Gentle piano'}],video:['/v1/videos/generations',{model,prompt:'A quiet mountain lake'}]}; + if(kind==='audio_transcription'){$("#presence-client-example").textContent="POST "+endpoint+"/v1/audio/transcriptions\nAuthorization: Bearer YOUR_LLOOM_KEY\nMultipart form: model="+model+", file=YOUR_AUDIO_FILE";return;} + const [route,body]=templates[kind]||templates.chat; + $("#presence-client-example").textContent = "POST " + endpoint + route + "\nContent-Type: application/json\nAuthorization: Bearer YOUR_LLOOM_KEY\n\n" + JSON.stringify(body,null,2); } function renderPresenceClients() { $("#presence-client-url").value = endpoint + "/v1"; const select = $("#presence-client-model"), selected = select.value; - const models = (state.models || []).filter(model => (model.kind || "chat") === "chat"); + const models = state.physicalModels || []; const key = models.map(m=>m.id).join("\n"); if (select.dataset.models !== key) { select.innerHTML = models.map(m=>'').join(""); @@ -117,7 +126,15 @@ export const presenceScript = String.raw` button.disabled = presenceBusy || !runtime || runtime.enabled === false || runtime.management === "external" || runtime.remote === true || Boolean(runtime.maintenance); button.setAttribute("aria-pressed",String(Boolean(runtime) && presencePolicy(runtime) === button.dataset.residency)); } - $("#presence-policy-hint").textContent = runtime ? "Readiness remains subject to memory admission. Use Load to start a cold model. Always ready prevents automatic eviction." : "Availability is managed by the upstream provider."; + const managed=runtime&&runtime.enabled!==false&&runtime.management!=="external"&&!runtime.remote&&!runtime.distributed&&!runtime.maintenance; + const transitioning=runtime&&['starting','warming','queued','draining','stopping'].includes(runtime.status); + for(const id of ['model-start','model-stop'])$("#"+id).disabled=presenceBusy||!managed||transitioning; + $("#model-start").textContent=transitioning?"Getting ready…":"Make ready now"; + $("#model-stop").disabled||=Boolean(runtime?.activeRequests||runtime?.queuedRequests||runtime?.keepWarm); + $("#presence-send").disabled=presenceBusy||Boolean(runtime?.maintenance)||runtime?.enabled===false; + $("#presence-availability").textContent=!runtime?"Available through your connected provider.":runtime.maintenance?"Paused. This model is protected from starting.":transitioning?"LLooM is getting this ready for you.":runtime.status==='failed'?"This model needs attention. Open details to see what happened.":"Just use it. LLooM prepares this automatically when your app asks."; + $("#presence-policy-hint").textContent = !runtime?"Your provider manages availability.":presencePolicy(runtime)==="always"?"Kept ready for quick replies. Choose Automatic if you want LLooM to reclaim its memory.":presencePolicy(runtime)==="preferred"?"Stays ready when there is room. LLooM makes space when another model needs it.":"Recommended: LLooM gets this ready when needed. Your downloaded files stay on disk."; + $("#presence-model-error").textContent=runtime?.lastError||""; } async function presenceLoadIntegrations() { try { @@ -188,21 +205,19 @@ export const presenceScript = String.raw` // Keep the model inspector available from every view, rather than clipping // it when the canvas is hidden. document.body.append($("#model-inspector"),$("#node-inspector")); - const policy = document.createElement("section"); - policy.id = "presence-policy"; - policy.innerHTML = '

Keep this model available

' + Object.entries(presencePolicyNames).map(([id,name])=>'').join("") + '

'; - $("#model-inspector .model-inspector-body").append(policy); - const trial = document.createElement("section"); - trial.id = "presence-trial"; trial.className = "presence-trial"; - trial.innerHTML = '
';
-    $("#model-inspector .model-inspector-body").append(trial);
     const inspectorBody=$("#model-inspector .model-inspector-body");
-    const technical=document.createElement("details");
-    technical.innerHTML="Recipe & runtime details";
-    technical.append($("#model-inspector-details"),$("#model-inspector-tags"));
-    inspectorBody.append(technical);
-    inspectorBody.prepend(policy,$("#model-inspector .model-inspector-actions"),trial);
-    $("#model-start").textContent="Load"; $("#model-warm").textContent="Warm up"; $("#model-stop").textContent="Unload";
+    const availability=document.createElement("p");availability.id="presence-availability";availability.className="presence-availability";inspectorBody.prepend(availability);
+    const policy = document.createElement("section");policy.id = "presence-policy";
+    policy.innerHTML = '

When should this stay ready?

' + Object.entries(presencePolicyNames).map(([id,name])=>'').join("") + '

'; + const trial = document.createElement("section");trial.id = "presence-trial"; trial.className = "presence-trial"; + trial.innerHTML = '
';
+    inspectorBody.append(trial);
+    const connect=document.createElement("button");connect.type="button";connect.id="presence-connect";connect.textContent="Connect an app";inspectorBody.append(connect);
+    connect.addEventListener("click",()=>{const id=state.selectedModelId;presenceSetView("clients");if([...$("#presence-client-model").options].some(option=>option.value===id))$("#presence-client-model").value=id;presenceClientExample();});
+    const technical=document.createElement("details");technical.className="presence-advanced";
+    technical.innerHTML='Options & details

'; + technical.append(policy,$("#model-inspector .model-inspector-actions"),$("#model-inspector-details"),$("#model-inspector-tags"));inspectorBody.append(technical); + $("#model-start").textContent="Make ready now";$("#model-warm").remove();$("#model-stop").textContent="Free up memory"; const oldRenderModels = renderModels; renderModels = function() { oldRenderModels(); renderPresence(); }; const oldRenderInspector = renderModelInspector; @@ -253,21 +268,21 @@ export const presenceScript = String.raw` if(!model?.runtime || !ensureAdminKeyIfNeeded()) return; const path="/gateway/runtimes/"+encodeURIComponent(model.runtime)+"/residency"; const accepted=await postJson(path,{policy:residency.dataset.residency,yes:true}); - presenceNotice("Readiness saved. Waiting for the current load to finish before applying it."); + presenceNotice("Preference saved. LLooM will take care of it."); const poll=async()=>{ try { const {job}=await getJson(path); if(!job || job.id!==accepted.id)return; if(job.status==='pending'){setTimeout(poll,1500);return;} await refresh(); - presenceNotice(job.status==='failed'?"Readiness saved, but could not be applied: "+job.error:"Readiness applied: "+presencePolicyNames[job.policy]+".",job.status==='failed'); + presenceNotice(job.status==='failed'?"Could not update availability: "+job.error:job.policy==='auto'?"LLooM will prepare this when your app needs it.":"Availability updated: "+presencePolicyNames[job.policy]+".",job.status==='failed'); }catch(error){presenceNotice("Could not check readiness: "+error.message,true);} }; setTimeout(poll,500); }); }); // Capture legacy runtime actions once, add bounded busy/error handling, and - // use normal admission for Load instead of a forced process start. + // keep preparation behind the normal admission and safety checks. document.addEventListener("click",event=>{ const button=event.target.closest("button[data-runtime]"); if(!button) return; @@ -277,7 +292,7 @@ export const presenceScript = String.raw` const load=button.dataset.action==="start"; const action=load?"admit":button.dataset.action; const result=await postJson("/gateway/runtimes/"+encodeURIComponent(button.dataset.runtime)+"/"+action,load?{apply:true,yes:true,force:false,warmup:true}:{}); - oldShowOutput(result);await refresh();presenceNotice("Model operation complete."); + oldShowOutput(result);await refresh();presenceNotice(load?"Ready for your apps.":"Memory released. The model stays installed."); }); },true); $("#presence-send").addEventListener("click",event=>presenceRun(event.currentTarget,async()=>{ diff --git a/src/dashboard-presence.mjs b/src/dashboard-presence.mjs index f739ce3..63a863a 100644 --- a/src/dashboard-presence.mjs +++ b/src/dashboard-presence.mjs @@ -66,6 +66,7 @@ export const presenceStyles = ` .presence-dialog .actions { justify-content:flex-end; flex-wrap:wrap; } .presence-trial { margin-top:18px; } .presence-trial textarea { width:100%; padding:12px; background:var(--bg); color:var(--text); border:1px solid var(--line); resize:vertical; min-height:90px; } + .presence-trial pre:empty { display:none; } .presence-trial pre { white-space:pre-wrap; font-size:13px; line-height:1.7; } .topology { border:1px solid var(--line); border-radius:18px; min-height:600px; background:#080f15; box-shadow:none; } .topology::before,.topology::after { display:none; } @@ -84,7 +85,14 @@ export const presenceStyles = ` .model-inspector-title { font-weight:500; font-size:22px; } .model-detail-grid { grid-template-columns:1fr 1fr; } .model-detail strong { font-weight:400; } - .model-inspector-actions { grid-template-columns:repeat(3,minmax(0,1fr)); } + .model-inspector-actions { grid-template-columns:repeat(2,minmax(0,1fr)); } + .presence-availability {color:#acc8d6;font-size:13px;line-height:1.65;margin:0 0 14px} + .presence-advanced {margin-top:22px;border-top:1px solid var(--line);padding-top:14px} + .presence-advanced summary {color:#8faebd;font-size:12px;cursor:pointer} + .presence-advanced .model-inspector-actions {margin:16px 0} + #presence-connect {width:100%;margin-top:12px} + #presence-send {width:100%;margin-top:8px} + #presence-model-error:empty {display:none} .operations-dock { margin:0; border-radius:16px; } .operations-dock > summary { display:none; } .operations-content { padding:0; } diff --git a/src/dashboard-scene.mjs b/src/dashboard-scene.mjs index cf030bd..f58f99f 100644 --- a/src/dashboard-scene.mjs +++ b/src/dashboard-scene.mjs @@ -99,13 +99,97 @@ export const sceneStyles = ` .scene-tools button {padding:8px 12px;font-size:11px;background:#0b171f;border-color:#223a48} .scene-follow {display:flex;gap:8px;align-items:center;font-size:11px;color:#acccdc} .scene-follow input {accent-color:#22d9f3;width:auto;margin:0} - .scene-memory-panel {padding:22px;margin-bottom:24px} - .scene-memory-panel h3 {display:flex;justify-content:space-between;font-size:16px;margin:0 0 18px;align-items:center} - .scene-memory-panel select {width:auto;max-width:180px;font-size:11px;padding:5px 8px;background:#0a161e} - .scene-memory-bar {display:flex;gap:2px;min-height:72px;border-radius:10px;overflow:hidden} - .scene-memory-segment {min-width:100px;flex:1;padding:14px 18px;background:linear-gradient(110deg,#35444f,#253945);color:#dde9f0;font-size:12px} - .scene-memory-segment strong {display:block;font-size:23px;margin-top:6px;font-weight:600} - .scene-memory-segment.available {background:linear-gradient(100deg,#349f9c,#55b4a7);color:#042322} + .scene-memory-panel {position:relative;z-index:6;background:linear-gradient(150deg,#0a1a2299,#060d12d8);padding:22px 24px 20px} + .scene-memory-panel h3 {display:flex;justify-content:space-between;gap:14px;font-size:15px;margin:0 0 6px;align-items:baseline;flex-wrap:wrap} + .scene-memory-panel h3 small {font-size:11px;color:#7d99ac;font-weight:400} + .scene-memory-panel select {width:auto;max-width:190px;font-size:11px;padding:5px 8px;background:#0a161e} + .scene-mem-sub {font-size:11.5px;color:#8ea9bb;margin:0 0 16px;line-height:1.5;max-width:66ch} + .scene-mem-head {display:flex;align-items:flex-end;justify-content:space-between;gap:16px;flex-wrap:wrap;margin:0 0 14px} + .scene-mem-total {display:flex;align-items:baseline;gap:9px;font-size:26px;font-weight:600;letter-spacing:-.6px;line-height:1} + .scene-mem-total span {font-size:12px;font-weight:400;color:#7f9cb0;letter-spacing:0;display:block;line-height:1.15} + .scene-mem-total em {font-style:normal;font-size:13px;color:#9fc4d6;font-weight:400} + .scene-mem-readouts {display:flex;gap:18px;flex-wrap:wrap;justify-content:flex-end} + .scene-mem-readout {text-align:right;min-width:86px} + .scene-mem-readout b {display:block;font-size:15px;font-weight:600;letter-spacing:-.2px;line-height:1.15} + .scene-mem-readout span {font-size:10px;letter-spacing:.7px;text-transform:uppercase;color:#6f8b9e} + .scene-mem-readout.used b {color:#e6f4fa} + .scene-mem-readout.free b {color:#5fe6d0} + .scene-mem-readout span.scene-mem-dot {display:inline-flex;align-items:center;gap:6px} + .scene-mem-readout span.scene-mem-dot::before {content:"";width:6px;height:6px;border-radius:50%;background:currentColor;opacity:.85} + .scene-mem-readout.free span.scene-mem-dot {color:#4fd8c4} + .scene-mem-readout.used span.scene-mem-dot {color:#8fb2c5} + .scene-mem-instrument {position:relative;border:1px solid #1f3a47;border-radius:12px;background:linear-gradient(180deg,#08131a,#060e13);overflow:hidden;isolation:isolate} + .scene-mem-ruler {position:relative;height:20px;border-bottom:1px solid #16303d} + .scene-mem-ticks {position:absolute;inset:0;display:flex;pointer-events:none} + .scene-mem-tick {flex:1 0 0;min-width:0;border-left:1px solid #16303d;position:relative} + .scene-mem-tick span {position:absolute;left:6px;top:5px;font-size:9px;color:#5d7c8e;white-space:nowrap} + .scene-mem-grid {position:absolute;inset:20px 0 0;pointer-events:none;display:flex;opacity:.5} + .scene-mem-grid i {flex:1 0 0;min-width:0;border-left:1px solid #11eaf50a} + .scene-mem-bar {position:relative;display:flex;height:142px;cursor:default;margin-top:2px} + .scene-mem-bar[data-known="false"] {height:70px} + .scene-mem-block {position:relative;flex:0 0 auto;min-width:0;border:0;padding:0;margin:0;background:transparent;color:#dceaf2;font:inherit;text-align:left;overflow:hidden;cursor:pointer;transition:filter .22s ease,opacity .22s ease} + .scene-mem-block > .scene-mem-fill {position:absolute;inset:0;background:var(--seg-fill);opacity:.9;transition:opacity .22s ease,box-shadow .22s ease} + .scene-mem-block > .scene-mem-rim {position:absolute;inset:0;border-right:1px solid #04121a99;background:linear-gradient(180deg,#ffffff14,#ffffff00 42%,#00000038)} + .scene-mem-block > .scene-mem-face {position:relative;z-index:2;display:flex;flex-direction:column;justify-content:space-between;height:100%;padding:11px 12px;gap:6px} + .scene-mem-block em {font-style:normal;font-size:10px;letter-spacing:.5px;color:#eaf7fc;text-shadow:0 1px 3px #04121ad9;white-space:nowrap} + .scene-mem-block em i {font-style:normal;opacity:.72;margin-left:5px;font-size:9px} + .scene-mem-block b {font-size:15px;font-weight:600;letter-spacing:-.2px;text-shadow:0 1px 3px #04121ad9;white-space:nowrap} + .scene-mem-block.wide > .scene-mem-face {padding:11px 14px} + .scene-mem-block.narrow em,.scene-mem-block.narrow b {display:none} + .scene-mem-block[data-kind="available"] {color:#eafffb} + .scene-mem-block[data-kind="available"] > .scene-mem-rim {background:linear-gradient(180deg,#ffffff1f,#ffffff00 40%,#0000001f)} + .scene-mem-block.system > .scene-mem-fill {background-image:repeating-linear-gradient(135deg,#ffffff10 0 7px,#ffffff00 7px 14px)} + .scene-mem-block[data-estimated="true"] > .scene-mem-fill {opacity:.72;background-image:repeating-linear-gradient(115deg,#ffffff12 0 6px,#ffffff00 6px 13px)} + .scene-mem-block:hover > .scene-mem-fill,.scene-mem-block[data-preview="true"] > .scene-mem-fill {opacity:1;box-shadow:inset 0 0 30px #ffffff1f} + .scene-mem-block[data-selected="true"] > .scene-mem-rim {box-shadow:inset 0 0 0 1px #ffffff4d} + .scene-mem-block:focus-visible {outline:2px solid #6ff0ff;outline-offset:-2px} + .scene-mem-ghost {position:absolute;top:0;bottom:0;left:0;width:0;pointer-events:none;transition:width .34s cubic-bezier(.22,1,.36,1),left .34s cubic-bezier(.22,1,.36,1)} + .scene-mem-ghost > .scene-mem-ghost-fill {position:absolute;inset:0;border:1px dashed #7ce8ffd9;border-left:0;border-radius:0 8px 8px 0;background:repeating-linear-gradient(115deg,#7ce8ff2e 0 6px,#7ce8ff0d 6px 12px);box-shadow:0 0 22px #23dcf61f,inset 0 0 24px #23dcf614} + .scene-mem-ghost[data-overflow="true"] > .scene-mem-ghost-fill {border-color:#ffb487ee;background:repeating-linear-gradient(115deg,#ff9d6a3a 0 6px,#ff9d6a12 6px 12px);box-shadow:0 0 22px #ff9d6a26} + .scene-mem-ghost[data-mode="resident"] > .scene-mem-ghost-fill,.scene-mem-ghost[data-mode="external"] > .scene-mem-ghost-fill {border-style:solid;border-color:#7ce8ff77;background:#7ce8ff14} + .scene-mem-ghost-label {position:absolute;right:8px;bottom:8px;font-size:10px;color:#bdf1ff;background:#062028e0;border:1px solid #2a6273;border-radius:7px;padding:4px 8px;white-space:nowrap;pointer-events:none;box-shadow:0 6px 18px #0006} + .scene-mem-ghost[data-overflow="true"] .scene-mem-ghost-label {color:#ffd6bd;border-color:#9b5a39;background:#2a140ce8} + .scene-mem-overflow {position:absolute;inset:0;z-index:3;pointer-events:none;opacity:0;transition:opacity .25s ease;background:repeating-linear-gradient(135deg,#ff9d6a1c 0 8px,#ff9d6a00 8px 16px)} + .scene-mem-overflow[data-on="true"] {opacity:1} + .scene-mem-overflow b {position:absolute;right:8px;top:8px;font-size:10px;font-weight:500;color:#ffd0b4;background:#2a120ae0;border:1px solid #9b5a39;border-radius:7px;padding:4px 8px;white-space:nowrap} + .scene-mem-overlay {position:absolute;top:0;bottom:0;pointer-events:none;border-left:1px dashed #48e9c7aa;background:linear-gradient(90deg,#48e9c729,#48e9c700 75%);transition:opacity .22s ease} + .scene-mem-overlay b {position:absolute;left:7px;top:8px;font-size:9px;letter-spacing:.5px;color:#8bf0da;white-space:nowrap;text-shadow:0 1px 3px #04121ad9} + .scene-mem-reserve {position:absolute;top:0;bottom:0;left:auto;right:0;pointer-events:none} + .scene-mem-reserve::before {content:"";position:absolute;top:0;bottom:0;left:0;width:1px;background:linear-gradient(180deg,#ffd79a00,#ffd79acc 18%,#ffd79acc 82%,#ffd79a00)} + .scene-mem-reserve b {position:absolute;left:0;bottom:6px;font-size:9px;letter-spacing:.5px;color:#e7bd82;white-space:nowrap;transform:translateX(-100%) translateX(-6px);text-shadow:0 1px 3px #04121ad9} + .scene-mem-bar[data-mode="external"] .scene-mem-overlay,.scene-mem-bar[data-mode="unknown"] .scene-mem-overlay {background:linear-gradient(90deg,#8fa8b833,#8fa8b800 75%);border-left-color:#9fb6c4aa} + .scene-mem-forecast {margin:13px 0 0;font-size:12px;line-height:1.55;color:#a9c3d3;display:flex;gap:9px;align-items:flex-start} + .scene-mem-forecast .scene-mem-forecast-icon {flex-shrink:0;width:8px;height:8px;margin-top:5px;border-radius:50%;background:#7f9aab} + .scene-mem-forecast[data-status="fits"] .scene-mem-forecast-icon {background:#3fe3cf;box-shadow:0 0 10px #3fe3cf7a} + .scene-mem-forecast[data-status="tight"] .scene-mem-forecast-icon {background:#f5c86a;box-shadow:0 0 10px #f5c86a7a} + .scene-mem-forecast[data-status="blocked"] .scene-mem-forecast-icon {background:#ff9d6a;box-shadow:0 0 10px #ff9d6a7a} + .scene-mem-forecast[data-status="resident"] .scene-mem-forecast-icon {background:#4fe0f2;box-shadow:0 0 10px #4fe0f27a} + .scene-mem-forecast[data-status="external"] .scene-mem-forecast-icon,.scene-mem-forecast[data-status="other-node"] .scene-mem-forecast-icon {background:#9db7c7} + .scene-mem-forecast strong {color:#eaf6fb;font-weight:500} + .scene-mem-forecast p {margin:0;flex:1} + .scene-mem-forecast .scene-mem-hint {color:#7d99ac;font-size:11px;margin-top:4px;display:block} + .scene-mem-legend {display:grid;grid-template-columns:repeat(auto-fit,minmax(196px,1fr));gap:6px;margin:15px 0 0} + .scene-mem-key {display:flex;align-items:center;gap:10px;width:100%;text-align:left;padding:9px 11px;border:1px solid #1c333f;border-radius:10px;background:#0a151c;color:#d3e4ee;font:inherit;font-size:12px;min-height:38px;transition:border-color .2s ease,background .2s ease,box-shadow .2s ease} + .scene-mem-key:hover {border-color:#2c6072;background:#0d1d26} + .scene-mem-key[data-selected="true"] {border-color:#3ed4ea;box-shadow:0 0 0 1px #28b9d126,0 6px 20px #0ad0f014} + .scene-mem-key[data-preview="true"] {border-color:#2e7f92;background:#0e222c} + .scene-mem-key:focus-visible {outline:2px solid #6ff0ff;outline-offset:2px} + .scene-mem-swatch {width:12px;height:12px;border-radius:4px;flex-shrink:0;background:var(--seg-fill);box-shadow:0 0 0 1px #ffffff1a} + .scene-mem-key-name {flex:1;min-width:0;overflow:hidden;text-overflow:ellipsis;white-space:nowrap} + .scene-mem-key-size {font-size:11px;color:#9db9c9;white-space:nowrap} + .scene-mem-key[data-estimated="true"] .scene-mem-key-size::after {content:" est.";color:#7d99ac} + .scene-mem-key em {font-style:normal;font-size:10px;color:#7d99ac;letter-spacing:.4px} + .scene-mem-rel {font-size:10px;color:#7d99ac;padding:4px 7px;border:1px solid #24404d;border-radius:7px;white-space:nowrap} + .scene-mem-key[data-preview="true"] .scene-mem-rel {color:#8bf0da;border-color:#2b6f7d} + .scene-mem-note {font-size:11px;color:#7d99ac;line-height:1.6;margin:13px 0 0;display:flex;gap:9px;flex-wrap:wrap;align-items:center} + .scene-mem-note span.scene-mem-chip {display:inline-flex;align-items:center;gap:6px;border:1px solid #1f3a47;border-radius:20px;padding:4px 10px;background:#08131a} + .scene-mem-note span.scene-mem-chip::before {content:"";width:8px;height:8px;border-radius:3px;background:#3d5b6b} + .scene-mem-note span.scene-mem-chip.measured::before {background:#2ec9e0} + .scene-mem-note span.scene-mem-chip.estimated::before {background:repeating-linear-gradient(115deg,#f3c268 0 3px,#f3c26800 3px 6px)} + .scene-mem-empty {padding:22px;font-size:12px;color:#93aebe;line-height:1.6} + @keyframes sceneMemBreathe {0%,100%{opacity:.7}50%{opacity:1}} + .scene-mem-bar[data-preview="true"] .scene-mem-ghost > .scene-mem-ghost-fill {animation:sceneMemBreathe 2.6s ease-in-out infinite} + .scene-mem-sticky {position:sticky;top:12px;align-self:start;margin-bottom:16px} .scene-model-layout {display:grid;grid-template-columns:minmax(0,1fr) 310px;gap:22px} .scene-model-list {display:flex;flex-direction:column;gap:9px} .scene-model-row {display:grid;grid-template-columns:44px minmax(130px,1.4fr) minmax(75px,.7fr) auto;gap:16px;align-items:center;border:1px solid #203541;border-radius:13px;padding:19px;background:linear-gradient(120deg,#111d2490,#09141b80);cursor:pointer;text-align:left;min-height:92px} @@ -148,9 +232,27 @@ export const sceneStyles = ` @media(min-width:1650px) {.scene-layout,.scene-model-layout {grid-template-columns:minmax(0,1fr) 360px}.scene-columns {gap:30px;padding:30px}.scene-diagram,.scene-columns,.scene-detail {min-height:610px}.scene-model-name{font-size:13px}} @media(max-width:1220px) {.scene-layout,.scene-model-layout {grid-template-columns:minmax(0,1fr) 280px;gap:12px}.scene-columns {grid-template-columns:120px 90px minmax(210px,1fr);gap:8px;padding:20px 15px}.scene-gateway {width:78px;height:78px}.scene-detail {padding:18px}.scene-mini-memory{width:58px}.scene-model-row {grid-template-columns:35px minmax(0,1fr) auto;gap:12px;padding:15px}.scene-row-policy{display:none}.scene-model-row > .scene-icon{width:35px;height:40px}.scene-model-row h3{font-size:14px}.scene-model-state{font-size:9px;padding:4px 6px}.scene-model-name{font-size:11px}} @media(max-width:1050px) {.scene-layout,.scene-model-layout {grid-template-columns:1fr}.scene-detail {min-height:0}.scene-detail:has(.scene-placeholder:not([hidden])){display:none}.scene-columns{grid-template-columns:minmax(125px,.8fr) minmax(100px,.8fr) minmax(230px,1.4fr);gap:20px}.scene-analytics{padding:20px;gap:18px}.scene-chart + .scene-chart{padding-left:18px}.scene-machine-list .presence-card{grid-template-columns:44px 1fr auto;gap:18px}.scene-machine-list .presence-card .machine-memory{grid-column:2 / -1;grid-row:2}.scene-model-row {grid-template-columns:40px minmax(0,1fr) auto auto}.scene-row-policy{display:block}} - @media(max-width:680px) {.presence-brand{display:none}.presence-nav{padding:8px}.presence-nav button[aria-current]::before{left:12px;right:12px;top:auto;bottom:0;width:auto;height:2px}body > header{display:none}main{padding-top:22px}.presence-heading h2{font-size:29px}.scene-columns{grid-template-columns:1fr 1fr;gap:18px;padding:20px;min-height:0}.scene-gateway-column{grid-column:2;grid-row:1}.scene-client-column{grid-column:1;grid-row:1}.scene-machine-column{grid-column:1/-1}.scene-gateway-wrap{min-height:210px;padding:30px 0 0}.scene-clients{padding:25px 0;gap:12px}.scene-diagram{min-height:0}.scene-machines{padding-top:16px}.scene-analytics{grid-template-columns:1fr;gap:22px}.scene-chart + .scene-chart{border-left:0;border-top:1px solid #20323e;padding:20px 0 0}.scene-chart svg{height:74px}.scene-model-row{grid-template-columns:34px minmax(0,1fr) auto;padding:15px 12px;gap:10px}.scene-model-row .scene-row-policy{display:none}.scene-row-status{font-size:10px}.scene-memory-panel{padding:18px}.scene-memory-segment{padding:12px;font-size:11px;min-width:80px}.scene-memory-segment strong{font-size:21px}.scene-add-capability{padding:18px;flex-wrap:wrap}.scene-add-capability button{width:100%;margin:0}.scene-network{padding:25px 18px;justify-content:flex-start}.scene-machine-list .presence-card{grid-template-columns:40px 1fr;gap:14px;padding:18px}.scene-machine-list .presence-card .actions{grid-column:1/-1}.scene-machine-list .presence-card .machine-memory{grid-column:1/-1;grid-row:auto}.scene-links{opacity:.85}.scene-model-name{font-size:12px}.scene-model-state{font-size:10px}} - @media(max-width:680px){.scene-detail:has(.model-inspector.open){position:fixed;z-index:80;left:12px;right:12px;bottom:12px;height:auto;max-height:calc(100dvh - 96px);background:linear-gradient(145deg,#122530,#09141c);box-shadow:0 -20px 70px #0009,0 0 0 1px #42616b66;padding:22px}.scene-detail:has(.model-inspector.open) .model-inspector-header{position:sticky;top:-22px;margin-top:-22px;padding-top:22px;background:#10212b;z-index:1}} + @media(max-width:680px) {.presence-brand{display:none}.presence-nav{padding:8px}.presence-nav button[aria-current]::before{left:12px;right:12px;top:auto;bottom:0;width:auto;height:2px}body > header{display:none}main{padding-top:22px}.presence-heading h2{font-size:29px}.scene-columns{grid-template-columns:1fr 1fr;gap:18px;padding:20px;min-height:0}.scene-gateway-column{grid-column:2;grid-row:1}.scene-client-column{grid-column:1;grid-row:1}.scene-machine-column{grid-column:1/-1}.scene-gateway-wrap{min-height:210px;padding:30px 0 0}.scene-clients{padding:25px 0;gap:12px}.scene-diagram{min-height:0}.scene-machines{padding-top:16px}.scene-analytics{grid-template-columns:1fr;gap:22px}.scene-chart + .scene-chart{border-left:0;border-top:1px solid #20323e;padding:20px 0 0}.scene-chart svg{height:74px}.scene-model-row{grid-template-columns:34px minmax(0,1fr) auto;padding:15px 12px;gap:10px}.scene-model-row .scene-row-policy{display:none}.scene-row-status{font-size:10px}.scene-memory-panel{padding:17px 15px 15px}.scene-mem-sticky{position:static}.scene-mem-bar{height:112px}.scene-mem-bar[data-known="false"]{height:60px}.scene-mem-legend{grid-template-columns:1fr}.scene-mem-readouts{gap:14px}.scene-mem-total{font-size:23px}.scene-mem-readout{min-width:72px}.scene-add-capability{padding:18px;flex-wrap:wrap}.scene-add-capability button{width:100%;margin:0}.scene-network{padding:25px 18px;justify-content:flex-start}.scene-machine-list .presence-card{grid-template-columns:40px 1fr;gap:14px;padding:18px}.scene-machine-list .presence-card .actions{grid-column:1/-1}.scene-machine-list .presence-card .machine-memory{grid-column:1/-1;grid-row:auto}.scene-links{opacity:.85}.scene-model-name{font-size:12px}.scene-model-state{font-size:10px}} + @media(max-width:1050px){.scene-detail:has(.model-inspector.open){position:fixed;z-index:80;left:12px;right:12px;bottom:12px;height:auto;max-height:calc(100dvh - 96px);background:linear-gradient(145deg,#122530,#09141c);box-shadow:0 -20px 70px #0009,0 0 0 1px #42616b66;padding:22px}.scene-detail:has(.model-inspector.open) .model-inspector-header{position:sticky;top:-22px;margin-top:-22px;padding-top:22px;background:#10212b;z-index:1}} @media(prefers-reduced-motion:reduce){.scene-gateway{animation:none!important}.scene-links .scene-particle{display:none}.scene-model{transition:none}} + .scene-memory-panel {padding:18px 20px;margin-bottom:20px;background:#0b151d} + .scene-memory-panel h3 {margin-bottom:14px} + .scene-mem-head {margin-bottom:12px}.scene-mem-total {font-size:22px}.scene-mem-total b {font-weight:500} + .scene-mem-bar {height:76px}.scene-mem-block {border-radius:0;transition:width .38s cubic-bezier(.22,1,.36,1),filter .2s,opacity .2s} + .scene-mem-face em {max-width:100%;overflow:hidden;text-overflow:ellipsis;font-size:10px} + .scene-mem-legend {grid-template-columns:repeat(auto-fit,minmax(145px,1fr));gap:5px;margin-top:10px} + .scene-mem-key {padding:7px 9px;font-size:11px;min-height:44px;gap:7px}.scene-mem-key-size {font-size:10px} + .scene-mem-note {font-size:10px;margin:9px 0 0}.scene-mem-forecast {margin-top:10px;min-height:36px;font-size:12px} + .scene-mem-ruler {overflow:hidden}.scene-mem-tick {position:absolute;top:0;bottom:0;width:0}.scene-mem-tick:last-child span {left:auto;right:5px} + .scene-mem-reserve {right:0;background:repeating-linear-gradient(120deg,#edbe6c0d 0 5px,transparent 5px 10px)} + .scene-mem-reserve b {transform:none;left:6px;bottom:6px;font-size:8px;white-space:normal;line-height:1.2} + .scene-mem-ghost {z-index:3}.scene-mem-reserve {z-index:4} + .scene-model-row[data-preview="true"],.scene-model[data-preview="true"] {border-color:#53cada;box-shadow:inset 0 0 25px #36d9e507,0 0 20px #31b6ce0b} + .scene-row-footprint {color:#7dbbc8;font-size:11px;white-space:nowrap} + @media(min-width:1051px){.scene-memory-panel{position:sticky;top:12px;z-index:7}#scene-model-detail{align-self:start;position:sticky;top:12px;max-height:calc(100dvh - 24px);height:auto;min-height:440px}.scene-model-row{scroll-margin-top:390px}} + @media(max-width:680px){.scene-memory-panel{padding:16px 12px}.scene-mem-bar{height:64px}.scene-mem-legend{grid-template-columns:repeat(2,minmax(0,1fr))}.scene-mem-key{min-width:0}.scene-mem-head{gap:8px}.scene-mem-readouts{gap:10px}.scene-mem-total{font-size:20px}.scene-mem-forecast{font-size:11px}.scene-mem-reserve b{font-size:7px}.scene-mem-block.compact .scene-mem-face{display:none}.scene-row-footprint{display:block;margin-top:4px}} + @media(prefers-reduced-motion:reduce){.scene-mem-block,.scene-mem-ghost{transition:none!important}.scene-mem-ghost-fill{animation:none!important}} + `; export const sceneScript = String.raw` @@ -182,24 +284,88 @@ export const sceneScript = String.raw` function sceneDeviceIcon(node){return node.local || /apple|mac/i.test(node.profile?.platformId || node.profile?.cpuBrand || '')?'laptop':'server';} document.querySelector('.presence-brand').innerHTML=sceneLogo+'LLooMby Enntity'; const scene=document.createElement('section');scene.id='presence-scene';scene.dataset.presencePanel='live'; - scene.innerHTML='

Clients

Waiting for telemetry

LLooM Gateway

Routes and balances

'+sceneLogo+'
GatewayConnecting…

Models on your machines

Requests

Observed during this session

Response time

Recent completed requests · includes generation time

Machine memory Used / total

Waiting for gateway telemetry
'; + scene.innerHTML='

Clients

Waiting for telemetry

LLooM Gateway

Routes and balances

'+sceneLogo+'
GatewayConnecting…

Models on your machines

Requests

Observed during this session

Response time

Recent completed requests · includes generation time

Machine memory Used / total

Waiting for gateway telemetry
'; $('.topology').before(scene); $('.topology').dataset.presencePanel='diagnostic';$('.topology').hidden=true; const modelLeft=document.createElement('div');modelLeft.className='scene-model-main'; const modelLayout=document.createElement('div');modelLayout.className='scene-model-layout';$('#view-models').append(modelLayout);modelLayout.append(modelLeft); - const memoryPanel=document.createElement('section');memoryPanel.className='scene-memory-panel';memoryPanel.innerHTML='

Your memory

Downloads stay on disk when models leave memory.

'; + const memoryPanel=document.createElement('section');memoryPanel.className='scene-memory-panel';memoryPanel.innerHTML='

Room for your AI

Point to a model to preview its memory.

'; modelLeft.append(memoryPanel); const modelTabs=document.createElement('div');modelTabs.className='scene-model-tabs';modelTabs.innerHTML='Installed'; modelLeft.append(modelTabs,$('.presence-toolbar')); const chips=document.createElement('div');chips.className='scene-chips';chips.innerHTML=[['','All'],['chat','Chat & code'],['image','Images'],['audio','Voice'],['embedding','Search'],['video','Video']].map(([id,name])=>'').join(''); modelLeft.append(chips,$('#presence-models'));$('#presence-models').className='scene-model-list';$('#presence-kind').hidden=true; const capability=document.createElement('div');capability.className='scene-add-capability';capability.innerHTML=sceneIcon('plus')+'
Add a capability

Get more done with another model.

';modelLeft.append(capability); - const modelDetail=document.createElement('aside');modelDetail.id='scene-model-detail';modelDetail.className='scene-detail';modelDetail.innerHTML='
'+sceneIcon('model')+'

Make room for more.

Select a model to choose how it uses memory. Your downloaded files stay on disk.

'; + const modelDetail=document.createElement('aside');modelDetail.id='scene-model-detail';modelDetail.className='scene-detail';modelDetail.innerHTML='
'+sceneIcon('model')+'

Ready when you are.

Choose a model to try it or connect an app. LLooM takes care of getting it ready.

'; modelLayout.append(modelLeft,modelDetail);$('#view-models').append(modelLayout); const network=document.createElement('div');network.id='scene-network';network.className='scene-network';$('#presence-machines').before(network);$('#presence-machines').className='scene-machine-list'; - const inspectorMemory=document.createElement('div');inspectorMemory.className='scene-inspector-memory';inspectorMemory.id='scene-inspector-memory';$('#presence-policy').before(inspectorMemory); + const inspectorMemory=document.createElement('div');inspectorMemory.className='scene-inspector-memory';inspectorMemory.id='scene-inspector-memory';$('#presence-availability').after(inspectorMemory); let sceneClientKey='',sceneMachineKey='',sceneModelKey='',sceneNodeKey='',sceneLinkKey='',sceneSamples=[],sceneSampleAt=0; let sceneMemoryNode=null,sceneFollowing=false; + let sceneMemPointer=null,sceneMemFocus=null,sceneMemShape='',sceneMemPaint=false; + const sceneMemColors=['#25bbd8','#53d5b6','#709eec','#a58ceb','#e1b176','#dc93bd','#79c5ce','#a0bb78']; + function sceneMemColor(segment){return segment.kind==='system'?'#304955':segment.kind==='available'?'#123337':sceneMemColors[Math.abs(segment.colorIndex||0)%sceneMemColors.length];} + function sceneMemModelNode(id){ + const model=(state.physicalModels||[]).find(m=>m.id===id),rt=model&&presenceRuntime(model); + if(!model?.runtime)return null; + return rt?.node||rt?.placement?.node||(!rt?.remote?sceneNodes().find(n=>n.local)?.id:null); + } + function sceneMemSnapshot(id=sceneMemPointer||sceneMemFocus||state.selectedModelId,ownNode=false){ + const nodes=sceneNodes(),follow=sceneMemPointer||sceneMemFocus; + const target=ownNode?sceneMemModelNode(id):follow&&sceneMemModelNode(follow); + const node=nodes.find(n=>n.id===(target||sceneMemoryNode))||nodes.find(n=>n.local)||nodes[0]; + return buildMemoryMap({node,runtimes:state.status?.runtimeManager?.runtimes||{},models:state.physicalModels||[],previewModelId:id,memorySafety:state.status?.runtimeManager?.memorySafety}); + } + function sceneMemFormat(value){return typeof value==='number'&&Number.isFinite(value)?formatBytes(Math.max(0,value)):'—';} + function sceneMemText(memory){ + const p=memory.preview; + if(!p)return 'Point to a model to see where it would fit. Nothing starts until you use it.'; + const status={fits:'Expected to fit',tight:'Little room to spare',blocked:'Needs more room',resident:'Already available',external:'Runs elsewhere','other-node':'Runs on another machine',unknown:'Footprint not yet known',paused:'Paused'}[p.status]||'Checking room'; + const delta=p.additionalBytes>0?' · about '+sceneMemFormat(p.additionalBytes)+' more':''; + const remaining=p.additionalBytes>0&&p.remainingBytes!=null?' · '+(p.remainingBytes<0?sceneMemFormat(-p.remainingBytes)+' over capacity':sceneMemFormat(p.remainingBytes)+' available after'):''; + return p.label+' · '+status+delta+remaining; + } + function sceneMemSchedule(){if(sceneMemPaint)return;sceneMemPaint=true;requestAnimationFrame(()=>{sceneMemPaint=false;sceneMemRender();});} + function sceneMemRender(){ + const host=$('#scene-memory-bar');if(!host||typeof buildMemoryMap!=='function')return; + const memory=sceneMemSnapshot(),segments=memory.segments||[],p=memory.preview; + const select=$('#scene-memory-machine');if(memory.nodeId)select.value=memory.nodeId; + const shape=JSON.stringify([memory.nodeId,memory.known,segments.map(s=>[s.id,s.kind,s.modelIds])]); + if(shape!==sceneMemShape){ + const focused=host.contains(document.activeElement)?document.activeElement.dataset.memoryFocus:null; + sceneMemShape=shape; + host.innerHTML=memory.known?'
total memory
In use
Available
'+segments.map((s,i)=>{const model=s.modelIds?.[0],tag=model?'button':'div';return '<'+tag+(model?' type="button" data-presence-model="'+escapeHtml(model)+'"':'')+' class="scene-mem-block" data-memory-index="'+i+'" data-memory-focus="block-'+i+'" data-kind="'+escapeHtml(s.kind)+'">';}).join('')+'Protected headroom
'+segments.map((s,i)=>{const model=s.modelIds?.[0],tag=model?'button':'div';return '<'+tag+(model?' type="button" data-presence-model="'+escapeHtml(model)+'"':'')+' class="scene-mem-key" data-memory-key="'+i+'" data-memory-focus="key-'+i+'">';}).join('')+'

':'
Waiting for a memory reading from this machine.
'; + if(focused)host.querySelector('[data-memory-focus="'+CSS.escape(focused)+'"]')?.focus({preventScroll:true}); + } + if(memory.known){ + $('#scene-mem-total').textContent=sceneMemFormat(memory.totalBytes); + $('#scene-mem-used').textContent=sceneMemFormat(memory.usedBytes); + $('#scene-mem-free').textContent=sceneMemFormat(memory.availableBytes); + const ruler=$('#scene-mem-ruler'),rulerKey=String(memory.totalBytes); + if(ruler.dataset.total!==rulerKey){ruler.dataset.total=rulerKey;ruler.innerHTML=Array.from({length:5},(_,i)=>''+escapeHtml(sceneMemFormat(memory.totalBytes*i/4))+'').join('');} + for(const [i,s] of segments.entries()){ + const block=host.querySelector('[data-memory-index="'+i+'"]'),key=host.querySelector('[data-memory-key="'+i+'"]'); + const isPreview=Boolean(p&&s.modelIds?.includes(p.modelId)),selected=Boolean(state.selectedModelId&&s.modelIds?.includes(state.selectedModelId)); + for(const el of [block,key]){el.style.setProperty('--seg-fill',sceneMemColor(s));el.dataset.estimated=String(Boolean(s.estimated));el.dataset.selected=String(selected);el.dataset.preview=String(isPreview);el.setAttribute('aria-label',s.label+', '+sceneMemFormat(s.bytes)+(s.estimated?', approximate':''));el.title=s.label+' · '+sceneMemFormat(s.bytes)+(s.estimated?' (estimate)':'');} + block.style.width=Math.max(0,Math.min(100,s.percent||0))+'%';block.classList.toggle('narrow',s.percent<10);block.classList.toggle('compact',s.percent<20); + block.querySelector('em').textContent=s.label;block.querySelector('b').textContent=sceneMemFormat(s.bytes); + key.querySelector('.scene-mem-key-name').textContent=s.label;key.querySelector('.scene-mem-key-size').textContent=sceneMemFormat(s.bytes); + } + const ghost=$('#scene-mem-ghost'),delta=p?.additionalBytes; + ghost.hidden=!(delta>0);ghost.style.left=memory.usedBytes/memory.totalBytes*100+'%';ghost.style.width=Math.max(0,Math.min(delta||0,memory.availableBytes))/memory.totalBytes*100+'%';ghost.dataset.overflow=String(p?.status==='blocked'); + $('#scene-mem-track').dataset.preview=String(Boolean(p)); + const reserve=$('#scene-mem-reserve');reserve.hidden=!(memory.reserveBytes>0);reserve.style.width=memory.reserveBytes/memory.totalBytes*100+'%';reserve.title=sceneMemFormat(memory.reserveBytes)+' protected for your machine';reserve.querySelector('b').textContent=sceneMemFormat(memory.reserveBytes)+' reserved'; + $('#scene-mem-note').textContent=memory.attributionNote||'Live memory use. Hover previews are estimates.'; + } + const forecast=$('#scene-memory-forecast');forecast.dataset.status=p?.status||'idle';const forecastText=memory.known?sceneMemText(memory):'Memory is not available yet. LLooM will check before preparing a model.';if(forecast.querySelector('p').textContent!==forecastText)forecast.querySelector('p').textContent=forecastText; + for(const el of document.querySelectorAll('.scene-model-row,.scene-model'))el.dataset.preview=String(el.dataset.presenceModel===p?.modelId); + if(state.selectedModelId){const own=sceneMemSnapshot(state.selectedModelId,true).preview;$('#scene-inspector-memory').innerHTML=''+(own?.additionalBytes>0?'Expected extra memory':'Memory')+''+escapeHtml(own?.additionalBytes>0?'About '+sceneMemFormat(own.additionalBytes):own?.status==='resident'?'Already available':own?.status==='external'?'Runs elsewhere':'Checked when needed')+'';} + } + const sceneMemTarget=target=>target?.closest?.('[data-presence-model]')?.dataset.presenceModel||null; + document.addEventListener('pointerover',event=>{if(event.pointerType==='touch')return;const id=sceneMemTarget(event.target);if(id&&id!==sceneMemPointer){sceneMemPointer=id;sceneMemSchedule();}}); + document.addEventListener('pointerout',event=>{if(event.pointerType==='touch')return;if(sceneMemTarget(event.target)&&sceneMemTarget(event.relatedTarget)!==sceneMemPointer){sceneMemPointer=sceneMemTarget(event.relatedTarget);sceneMemSchedule();}}); + document.addEventListener('focusin',event=>{const id=sceneMemTarget(event.target);if(id)sceneMemPointer=null;if(id!==sceneMemFocus){sceneMemFocus=id;sceneMemSchedule();}}); + document.addEventListener('focusout',event=>{sceneMemFocus=sceneMemTarget(event.relatedTarget);sceneMemSchedule();}); function sceneMoveInspector(){ const slot=presenceView==='models'?modelDetail:$('#scene-live-detail'); if($('#model-inspector').parentElement!==slot)slot.append($('#model-inspector')); @@ -207,15 +373,19 @@ export const sceneScript = String.raw` if(state.selectedModelId){const model=state.physicalModels.find(m=>m.id===state.selectedModelId),rt=model&&presenceRuntime(model);inspectorMemory.innerHTML='Memory estimate'+(rt?.memoryGb!=null?escapeHtml(rt.memoryGb)+' GB':'Not reported')+'';} } const scenePreviousView=presenceSetView; - presenceSetView=function(name){scenePreviousView(name);sceneMoveInspector();renderScene();}; + presenceSetView=function(name){sceneMemPointer=null;sceneMemFocus=null;scenePreviousView(name);sceneMoveInspector();renderScene();}; const scenePreviousInspector=renderModelInspector; - renderModelInspector=function(){scenePreviousInspector();sceneMoveInspector();}; + renderModelInspector=function(){scenePreviousInspector();sceneMoveInspector();sceneMemSchedule();}; renderPresenceModels=function(){ const search=$('#presence-search').value.trim().toLowerCase(),kind=$('#presence-kind').value; const models=(state.physicalModels||[]).filter(m=>(!search||(m.name+' '+m.id).toLowerCase().includes(search))&&(!kind||(m.kind||'chat').startsWith(kind))).sort((a,b)=>Number(Boolean(b.runtime))-Number(Boolean(a.runtime))||Number(Boolean(presenceRuntime(b)?.healthy))-Number(Boolean(presenceRuntime(a)?.healthy))); $('#presence-model-count').textContent=models.length+(models.length===1?' model':' models'); const key=JSON.stringify(models.map(m=>[m.id,m.name,sceneKind(m),presenceModelLabel(m),presencePolicy(presenceRuntime(m)),state.selectedModelId===m.id]));if(key===sceneModelKey)return;sceneModelKey=key; + const catalog=$('#presence-models'),focused=catalog.contains(document.activeElement)?document.activeElement.closest('[data-presence-model]')?.dataset.presenceModel:null; $('#presence-models').innerHTML=models.map(model=>{const rt=presenceRuntime(model);return '';}).join('')||'
No models match. Choose another filter or add a model.
'; + if(focused)catalog.querySelector('[data-presence-model="'+CSS.escape(focused)+'"]')?.focus({preventScroll:true}); + if(sceneMemPointer&&!models.some(m=>m.id===sceneMemPointer))sceneMemPointer=null; + sceneMemSchedule(); }; renderPresenceMachines=function(){ const nodes=sceneNodes(),key=JSON.stringify(nodes.map(n=>[n.id,n.name,n.local,n.reachable,n.profile?.cpuBrand,sceneMemory(n)]));if(key===sceneNodeKey)return;sceneNodeKey=key; @@ -254,7 +424,7 @@ export const sceneScript = String.raw` $('#scene-memory-list').innerHTML=nodes.slice(0,4).map(node=>{const m=sceneMemory(node);return '
'+escapeHtml(sceneNodeName(node))+''+(m.known?escapeHtml(formatBytes(m.used))+' / '+escapeHtml(formatBytes(m.total)):'Unknown')+'
';}).join('')||'

Memory telemetry unavailable

'; $('#scene-health').textContent=state.status?.error?'Gateway telemetry needs attention':nodes.length?'Live gateway telemetry · '+(summary.recentErrors||0)+' errors in the last minute':'Connecting to your gateway…'; const machineSelect=$('#scene-memory-machine');const optionsKey=nodes.map(n=>n.id).join('|');if(machineSelect.dataset.nodes!==optionsKey){machineSelect.dataset.nodes=optionsKey;machineSelect.innerHTML=nodes.map(n=>'').join('');if(nodes.some(n=>n.id===sceneMemoryNode))machineSelect.value=sceneMemoryNode;} - sceneMemoryNode=machineSelect.value;const memory=sceneMemory(nodes.find(n=>n.id===sceneMemoryNode));$('#scene-memory-bar').innerHTML=memory.known?'
In use'+escapeHtml(formatBytes(memory.used))+'
Available'+escapeHtml(formatBytes(memory.total-memory.used))+'
':'
Waiting for a physical machine memory reading.
'; + sceneMemRender(); sceneMoveInspector();requestAnimationFrame(sceneDrawLinks); } function sceneDrawLinks(){ @@ -272,7 +442,7 @@ export const sceneScript = String.raw` $('#scene-memory-machine').addEventListener('change',()=>{sceneMemoryNode=$('#scene-memory-machine').value;renderScene();}); $('#scene-follow').addEventListener('change',()=>{sceneFollowing=$('#scene-follow').checked;sceneMachineKey='';renderScene();}); $('#scene-diagnostic').addEventListener('click',()=>{const detail=$('.topology');detail.hidden=!detail.hidden;if(!detail.hidden)detail.scrollIntoView({behavior:window.matchMedia('(prefers-reduced-motion: reduce)').matches?'auto':'smooth',block:'start'});}); - document.addEventListener('click',event=>{const kind=event.target.closest('[data-scene-kind]');if(kind){$('#presence-kind').value=kind.dataset.sceneKind;document.querySelectorAll('[data-scene-kind]').forEach(b=>b.setAttribute('aria-pressed',String(b===kind)));renderPresenceModels();}if(event.target.closest('[data-scene-all]'))presenceSetView('models');const selected=event.target.closest('[data-presence-model]');if(selected){sceneModelKey='';sceneMachineKey='';renderPresenceModels();renderScene();}}); + document.addEventListener('click',event=>{const kind=event.target.closest('[data-scene-kind]');if(kind){$('#presence-kind').value=kind.dataset.sceneKind;document.querySelectorAll('[data-scene-kind]').forEach(b=>b.setAttribute('aria-pressed',String(b===kind)));renderPresenceModels();}if(event.target.closest('[data-scene-all]'))presenceSetView('models');const selected=event.target.closest('[data-presence-model]');if(selected){sceneMemoryNode=sceneMemModelNode(selected.dataset.presenceModel)||sceneMemoryNode;sceneModelKey='';sceneMachineKey='';renderPresenceModels();renderScene();}}); $('#scene-machines').addEventListener('scroll',()=>{sceneLinkKey='';requestAnimationFrame(sceneDrawLinks);}); window.addEventListener('resize',()=>{sceneLinkKey='';requestAnimationFrame(sceneDrawLinks);}); document.addEventListener('visibilitychange',()=>{if(!document.hidden){sceneLinkKey='';renderScene();}}); diff --git a/src/dashboard.mjs b/src/dashboard.mjs index 258723e..9867929 100644 --- a/src/dashboard.mjs +++ b/src/dashboard.mjs @@ -1,3 +1,4 @@ +import { buildMemoryMap } from './dashboard-memory.mjs'; import { presenceStyles, presenceNav, presenceViews } from './dashboard-presence.mjs'; import { sceneStyles, sceneScript } from './dashboard-scene.mjs'; import { presenceScript } from './dashboard-presence-client.mjs'; @@ -2744,6 +2745,12 @@ export function renderDashboardPage() { ) .replace( ' refresh();\n refreshActivity();', - () => presenceScript + sceneScript + '\n refresh();\n refreshActivity();' + () => + 'const buildMemoryMap = (' + + buildMemoryMap.toString() + + ');\n' + + presenceScript + + sceneScript + + '\n refresh();\n refreshActivity();' ); } diff --git a/src/runtime-manager.mjs b/src/runtime-manager.mjs index dcc93e7..4183d66 100644 --- a/src/runtime-manager.mjs +++ b/src/runtime-manager.mjs @@ -16,6 +16,7 @@ import { } from './cluster.mjs'; import { cleanupPortListener, terminateProcessTree } from './process-control.mjs'; import { memorySafetyPolicy, createMemorySafetyGuard, RuntimeMemorySafetyError } from './runtime-memory-safety.mjs'; +import { createRuntimeMemoryUsageSampler } from './runtime-memory-usage.mjs'; import { maintenanceBlocksRouting, assertMaintenanceStartAllowed, maintenanceError } from './model-maintenance.mjs'; @@ -667,11 +668,17 @@ export function effectiveRuntimeArgs(runtimeId, runtime) { } export class RuntimeManager { - constructor(config, { logger = console, captureOutput = true, clusterCoordinator = null, memorySampler } = {}) { + constructor( + config, + { logger = console, captureOutput = true, clusterCoordinator = null, memorySampler, memoryUsageSampler } = {} + ) { this.config = config; this.logger = logger; this.captureOutput = captureOutput; this.memorySampler = memorySampler; + this.memoryUsageSampler = + memoryUsageSampler ?? + createRuntimeMemoryUsageSampler({ nodeId: currentNodeId(config), platform: process.platform }); this.memorySafetyFailures = new Map(); this.processes = new Map(); this.state = new Map(); @@ -1049,11 +1056,12 @@ export class RuntimeManager { return aborted; } - async status({ localOnly = false } = {}) { + async status({ localOnly = false, includeMemoryUsage = false } = {}) { const runtimes = {}; const keepWarm = new Set(this.keepWarmRuntimeIds()); const preferredWarm = new Set(this.preferredWarmRuntimeIds()); const remoteNodes = new Map(); + const localRuntimeIds = []; for (const [runtimeId, runtime] of Object.entries(this.config.runtimes ?? {})) { const placement = runtimePlacement(runtime, this.config); if (placement.mode === 'distributed') continue; @@ -1062,6 +1070,7 @@ export class RuntimeManager { ? !runtime.node || nodeId === currentNodeId(this.config) : this.clusterCoordinator.isLocalNode(nodeId); if (localOnly && !isLocal) continue; + if (isLocal) localRuntimeIds.push(runtimeId); if (!isLocal && this.clusterCoordinator) { if (!remoteNodes.has(nodeId)) remoteNodes.set(nodeId, this.clusterCoordinator.nodeStatus(nodeId)); const node = await remoteNodes.get(nodeId); @@ -1187,6 +1196,16 @@ export class RuntimeManager { }; } } + if (includeMemoryUsage && this.memoryUsageSampler && localRuntimeIds.length) { + try { + const usage = await this.memoryUsageSampler.sample({ runtimes, runtimeIds: localRuntimeIds }); + for (const runtimeId of localRuntimeIds) { + if (usage[runtimeId]) runtimes[runtimeId].memoryUsage = usage[runtimeId]; + } + } catch { + // Memory telemetry is observational only and must not change runtime health or lifecycle. + } + } return { runtimes, memorySafety: memorySafetyPolicy(this.config), diff --git a/src/runtime-memory-usage.mjs b/src/runtime-memory-usage.mjs new file mode 100644 index 0000000..2a6f486 --- /dev/null +++ b/src/runtime-memory-usage.mjs @@ -0,0 +1,395 @@ +import { execFile } from 'node:child_process'; +import { promisify } from 'node:util'; +import { fileURLToPath } from 'node:url'; + +const execFileAsync = promisify(execFile); +const LSOF = process.platform === 'darwin' ? '/usr/sbin/lsof' : 'lsof'; +const MAX_MODEL_BODY_BYTES = 64 * 1024; + +function positiveInteger(value) { + const number = Number(value); + return Number.isInteger(number) && number > 0 ? number : null; +} + +function commandBasename(value) { + const first = Array.isArray(value) ? value[0] : value; + return ( + String(first ?? '') + .trim() + .split(/\s+/)[0] + .split(/[\\/]/) + .pop() ?? '' + ); +} + +function loopbackHostname(hostname) { + return ['127.0.0.1', 'localhost', '[::1]', '::1'].includes(String(hostname ?? '').toLowerCase()); +} + +function loopbackUrl(value) { + try { + const url = new URL(String(value)); + return url.protocol === 'http:' || url.protocol === 'https:' ? (loopbackHostname(url.hostname) ? url : null) : null; + } catch { + return null; + } +} + +function endpointUrl(runtime, path = '/') { + const configured = loopbackUrl(runtime?.healthUrl); + if (runtime?.healthUrl && !configured) return null; + if (configured) { + if (path === '/') return configured; + const url = new URL(configured); + url.pathname = path; + url.search = ''; + url.hash = ''; + return url; + } + const port = positiveInteger(runtime?.port); + return port ? new URL(`http://127.0.0.1:${port}${path.startsWith('/') ? path : `/${path}`}`) : null; +} + +function parseProcessRows(text) { + const rows = new Map(); + for (const match of String(text ?? '').matchAll(/^\s*(\d+)\s+(\d+)\s+(\d+)\s*$/gm)) { + const pid = Number(match[1]); + const ppid = Number(match[2]); + if (pid > 0 && ppid >= 0) rows.set(pid, { pid, ppid, rss: Number(match[3]) * 1024 }); + } + return [...rows.values()]; +} + +function parseLoopbackListeners(text) { + const listeners = new Map(); + for (const line of String(text ?? '').split(/\r?\n/)) { + const match = line.match(/^\s*\S+\s+(\d+)\s+.*?\sTCP\s+(\S+)\s*(?:\(\s*LISTEN\s*\))?$/i); + if (!match) continue; + const pid = Number(match[1]); + const address = match[2]; + const separator = address.lastIndexOf(':'); + if (!Number.isInteger(pid) || pid <= 0 || separator < 1) continue; + const host = address.slice(0, separator); + const port = Number(address.slice(separator + 1)); + if (!loopbackHostname(host) || !positiveInteger(port)) continue; + if (!listeners.has(port)) listeners.set(port, new Set()); + listeners.get(port).add(pid); + } + return listeners; +} + +function expandProcessTree(rootPids, rowsByPid) { + const selected = new Set(rootPids); + let changed = true; + while (changed) { + changed = false; + for (const row of rowsByPid ?? []) { + if (selected.has(row.ppid) && !selected.has(row.pid)) { + selected.add(row.pid); + changed = true; + } + } + } + return [...selected]; +} + +function boundedText(response, maxBytes) { + const stream = response?.body; + if (stream && typeof stream.getReader === 'function') { + return (async () => { + const reader = stream.getReader(); + const chunks = []; + let bytes = 0; + try { + for (;;) { + const { done, value } = await reader.read(); + if (done) break; + bytes += value?.byteLength ?? value?.length ?? 0; + if (bytes > maxBytes) throw new Error('memory usage response too large'); + chunks.push(value); + } + } finally { + await reader.cancel().catch(() => {}); + reader.releaseLock(); + } + return Buffer.concat(chunks.map((chunk) => Buffer.from(chunk))).toString('utf8'); + })(); + } + return response.text().then((text) => { + if (Buffer.byteLength(text) > maxBytes) throw new Error('memory usage response too large'); + return text; + }); +} + +function extractLoadedModelIds(payload, kind) { + if (payload == null || typeof payload !== 'object') throw new Error('unexpected model response'); + if (kind === 'ollama') { + if (!Array.isArray(payload.models)) throw new Error('missing model residency list'); + return (Array.isArray(payload.models) ? payload.models : []) + .map((item) => (typeof item === 'string' ? item : (item?.model ?? item?.name ?? item?.id))) + .filter((value) => typeof value === 'string' && value.trim()) + .map((value) => value.trim()); + } + if (!Array.isArray(payload.tts_loaded) || !Array.isArray(payload.stt_loaded)) + throw new Error('missing audio residency lists'); + const values = [ + ...(Array.isArray(payload.tts_loaded) ? payload.tts_loaded : []), + ...(Array.isArray(payload.stt_loaded) ? payload.stt_loaded : []) + ]; + return values + .map((value) => (typeof value === 'string' ? value : (value?.id ?? value?.name ?? value?.model))) + .filter((value) => typeof value === 'string' && value.trim()) + .map((value) => value.trim()); +} + +export function createRuntimeMemoryUsageSampler({ + nodeId = null, + platform = process.platform, + psReader = async () => { + const result = await execFileAsync('/bin/ps', ['-axo', 'pid=,ppid=,rss='], { + timeout: 1000, + maxBuffer: 2 * 1024 * 1024 + }); + return result.stdout; + }, + listenerReader = async () => { + const result = await execFileAsync(LSOF, ['-nP', '-iTCP', '-sTCP:LISTEN'], { + timeout: 1000, + maxBuffer: 2 * 1024 * 1024 + }); + return result.stdout; + }, + footprintReader = async (pids) => { + const result = await execFileAsync( + 'python3', + [fileURLToPath(new URL('./darwin-memory-usage.py', import.meta.url)), ...pids.map(String)], + { timeout: 1000, maxBuffer: 2 * 1024 * 1024 } + ); + return JSON.parse(result.stdout); + }, + fetchImpl = globalThis.fetch, + clock = Date.now, + nowIso = () => new Date().toISOString(), + scheduleTimeout = (callback, milliseconds) => { + const timer = setTimeout(callback, milliseconds); + timer.unref?.(); + return timer; + }, + cacheTtlMs = 5000, + modelTimeoutMs = 500 +} = {}) { + const now = typeof clock?.now === 'function' ? clock.now.bind(clock) : (clock ?? Date.now); + let processInFlight = null; + let processCache = null; + const modelInFlight = new Map(); + const modelCache = new Map(); + + async function readProcessSnapshot() { + const at = now(); + if (processCache && at < processCache.expiresAt) return processCache.value; + if (processInFlight) return processInFlight; + processInFlight = Promise.allSettled([psReader(), listenerReader()]) + .then(async ([ps, lsof]) => { + const value = { + rows: ps.status === 'fulfilled' ? parseProcessRows(ps.value) : [], + listeners: lsof.status === 'fulfilled' ? parseLoopbackListeners(lsof.value) : new Map(), + processOk: ps.status === 'fulfilled', + listenerOk: lsof.status === 'fulfilled' + }; + if (platform === 'darwin' && value.rows.length) { + try { + const footprints = await footprintReader(value.rows.map((row) => row.pid)); + for (const row of value.rows) { + const bytes = footprints?.[row.pid]; + if (typeof bytes === 'number' && Number.isFinite(bytes) && bytes >= 0) row.footprint = bytes; + } + } catch { + /* Python or per-process accounting may be unavailable; retain RSS. */ + } + } + processCache = { value, expiresAt: at + cacheTtlMs }; + return value; + }) + .finally(() => { + processInFlight = null; + }); + return processInFlight; + } + + async function readModelIds(url, kind) { + const key = url.toString(); + const at = now(); + const cached = modelCache.get(key); + if (cached && at < cached.expiresAt) return cached.value; + const inFlight = modelInFlight.get(key); + if (inFlight) return inFlight; + const request = (async () => { + let timer = null; + let signal = null; + if (typeof AbortController === 'function' && modelTimeoutMs > 0) { + const controller = new AbortController(); + signal = controller.signal; + timer = scheduleTimeout(() => controller.abort(new Error('model residency request timed out')), modelTimeoutMs); + } + try { + const response = await fetchImpl(url, { signal, redirect: 'error' }); + if (!response?.ok) throw new Error(`model residency request failed (${response?.status ?? 'unknown'})`); + const body = await boundedText(response, MAX_MODEL_BODY_BYTES); + return extractLoadedModelIds(JSON.parse(body), kind); + } finally { + if (timer) clearTimeout(timer); + } + })() + .then( + (value) => { + const result = { value, expiresAt: now() + cacheTtlMs }; + modelCache.set(key, result); + return value; + }, + () => { + const result = { value: null, expiresAt: now() + cacheTtlMs }; + modelCache.set(key, result); + return null; + } + ) + .finally(() => { + modelInFlight.delete(key); + }); + modelInFlight.set(key, request); + return request; + } + + function runtimeRoots(runtime, snapshot) { + const roots = []; + const adapter = String(runtime?.adapter ?? '').toLowerCase(); + if (adapter === 'docker') { + if (platform !== 'linux') return []; + const pid = positiveInteger(runtime?.containerPid ?? runtime?.container?.pid); + if (pid) roots.push(pid); + } else { + const pid = positiveInteger(runtime?.pid); + if (pid) roots.push(pid); + } + const url = endpointUrl(runtime); + const port = url ? positiveInteger(url.port || (url.protocol === 'https:' ? 443 : 80)) : null; + if (port) for (const pid of snapshot.listeners.get(port) ?? []) roots.push(pid); + return [...new Set(roots)]; + } + + return { + nodeId, + async sample({ runtimes = {}, runtimeIds = [] } = {}) { + if (!runtimeIds.length) return {}; + const ids = [...new Set(runtimeIds)].filter((runtimeId) => { + const runtime = runtimes[runtimeId]; + return ( + runtime && + runtime.remote !== true && + runtime.distributed !== true && + runtime.placement?.mode !== 'distributed' && + (!nodeId || + !(runtime.node ?? runtime.placement?.node) || + (runtime.node ?? runtime.placement?.node) === nodeId) && + ['running', 'external', 'starting', 'warming', 'stopping', 'draining'].includes(runtime.status) + ); + }); + if (!ids.length) return {}; + const snapshot = await readProcessSnapshot(); + const expansions = new Map(); + const roots = new Map(); + for (const runtimeId of ids) { + const runtime = runtimes[runtimeId]; + const selectedRoots = runtimeRoots(runtime, snapshot); + roots.set(runtimeId, selectedRoots); + expansions.set(runtimeId, expandProcessTree(selectedRoots, snapshot.rows)); + } + const parent = new Map(); + const find = (pid) => { + if (!parent.has(pid)) return pid; + const root = find(parent.get(pid)); + parent.set(pid, root); + return root; + }; + const union = (left, right) => { + const a = find(left); + const b = find(right); + if (a === b) return; + parent.set(b, a); + }; + for (const pids of expansions.values()) { + if (pids.length > 1) for (const pid of pids.slice(1)) union(pids[0], pid); + } + const groups = new Map(); + for (const runtimeId of ids) { + const pids = expansions.get(runtimeId) ?? []; + const groupKey = pids.length ? `proc:${find(pids[0])}` : `runtime:${runtimeId}`; + if (!groups.has(groupKey)) { + groups.set(groupKey, { runtimeIds: [], pids: new Set(), urls: [] }); + } + const group = groups.get(groupKey); + group.runtimeIds.push(runtimeId); + for (const pid of pids) group.pids.add(pid); + } + const result = {}; + await Promise.all( + [...groups].map(async ([groupKey, group]) => { + group.runtimeIds.sort(); + let residentBytes = null; + const usesFootprint = + platform === 'darwin' && + group.pids.size > 0 && + [...group.pids].every((pid) => snapshot.rows.find((row) => row.pid === pid)?.footprint != null); + let rowsAvailable = snapshot.rows.length > 0; + if (rowsAvailable && group.pids.size) { + let sum = 0; + for (const pid of group.pids) { + const row = snapshot.rows.find((row) => row.pid === pid); + if (!row) { + rowsAvailable = false; + break; + } + sum += usesFootprint ? row.footprint : row.rss; + } + if (rowsAvailable) residentBytes = sum; + } + const loadedByRuntime = new Map(); + const modelCalls = []; + for (const runtimeId of group.runtimeIds) { + const runtime = runtimes[runtimeId]; + loadedByRuntime.set(runtimeId, null); + const base = commandBasename(runtime.command); + const audio = [base, ...(runtime.args ?? []).map(commandBasename)].some((part) => + ['lloom-audio-server', 'lloom_audio_server.py', 'lloom_audio_server'].includes(part) + ); + const kind = base === 'ollama' ? 'ollama' : audio ? 'audio' : null; + if (!kind) continue; + const url = kind === 'ollama' ? endpointUrl(runtime, '/api/ps') : endpointUrl(runtime, '/health'); + if (!url) continue; + modelCalls.push( + readModelIds(url, kind).then((value) => { + loadedByRuntime.set(runtimeId, value); + }) + ); + } + await Promise.all(modelCalls.splice(0, modelCalls.length)); + const knownLists = [...loadedByRuntime.values()].filter((value) => Array.isArray(value)); + const loadedModelIds = + knownLists.length === group.runtimeIds.length + ? [...new Set(knownLists.flat())].sort((left, right) => left.localeCompare(right)) + : null; + const memoryUsage = { + residentBytes, + source: usesFootprint ? 'process-footprint' : 'process-rss', + sampledAt: nowIso(), + groupId: groupKey, + sharedRuntimeIds: group.runtimeIds, + loadedModelIds, + residencyKnown: loadedModelIds !== null + }; + for (const runtimeId of group.runtimeIds) result[runtimeId] = { ...memoryUsage }; + }) + ); + return result; + } + }; +} diff --git a/src/server.mjs b/src/server.mjs index 940c065..25c00df 100644 --- a/src/server.mjs +++ b/src/server.mjs @@ -3866,7 +3866,7 @@ export function createLloomServer(config, { logger = console, runtimeManager = n } if (req.method === 'GET' && url.pathname === '/gateway/status') { - const runtimeStatus = await runtimeManager.status(); + const runtimeStatus = await runtimeManager.status({ includeMemoryUsage: true }); const clustered = Object.keys(config.cluster?.nodes ?? {}).length > 0; const localRuntimeStatus = clustered ? { @@ -3891,7 +3891,7 @@ export function createLloomServer(config, { logger = console, runtimeManager = n if (req.method === 'GET' && url.pathname === '/gateway/node') { sendJson(res, 200, { ok: true, - node: await clusterCoordinator.localNodeStatus() + node: await clusterCoordinator.localNodeStatus({ includeMemoryUsage: true }) }); return; } diff --git a/test/dashboard-memory.test.mjs b/test/dashboard-memory.test.mjs new file mode 100644 index 0000000..3b08d80 --- /dev/null +++ b/test/dashboard-memory.test.mjs @@ -0,0 +1,358 @@ +import assert from 'node:assert/strict'; +import { test } from 'node:test'; +import { buildMemoryMap } from '../src/dashboard-memory.mjs'; + +const GiB = 1024 ** 3; +const policy = { mode: 'enforce', minAvailableMemoryGb: 12, maxMemoryUtilization: 0.9 }; + +function node({ total = 96 * GiB, available = 68 * GiB, reachable = true } = {}) { + return { + id: 'node-a', + local: true, + reachable, + telemetry: { + memory: { + totalBytes: total, + availableBytes: available, + usedBytes: total - available + } + } + }; +} + +function map(overrides = {}) { + const { nodeOverrides, ...rest } = overrides; + return buildMemoryMap({ + node: node(nodeOverrides), + runtimes: {}, + models: [], + memorySafety: policy, + ...rest + }); +} + +test('builds an exact 96 GiB local map with reserve and forecast', () => { + const result = map({ + runtimes: { + main: { + id: 'main', + status: 'running', + healthy: true, + memoryUsage: { + residentBytes: 20 * GiB, + groupId: 'proc:1', + sharedRuntimeIds: ['main'], + loadedModelIds: ['upstream-a'], + residencyKnown: true + } + } + }, + models: [ + { id: 'model-a', name: 'Model A', runtime: 'main', upstreamModel: 'upstream-a', targets: [{ node: 'node-a' }] } + ], + previewModelId: 'model-a' + }); + + assert.equal(result.known, true); + assert.equal(result.totalBytes, 96 * GiB); + assert.equal(result.usedBytes, 28 * GiB); + assert.equal(result.availableBytes, 68 * GiB); + assert.equal(result.reserveBytes, 12 * GiB); + assert.equal(result.usableBytes, 56 * GiB); + assert.equal( + result.segments.reduce((sum, segment) => sum + segment.bytes, 0), + 96 * GiB + ); + assert.deepEqual( + result.segments.map((segment) => segment.kind), + ['runtime', 'system', 'available'] + ); + assert.equal(result.preview.status, 'resident'); + assert.equal(result.preview.additionalBytes, 0); + assert.equal(result.preview.remainingBytes, 68 * GiB); +}); + +test('40 GiB incoming leaves 28 GiB above a 12 GiB reserve', () => { + const runtimes = { + incoming: { id: 'incoming', status: 'stopped', memoryGb: 40 } + }; + const result = map({ + runtimes, + models: [{ id: 'incoming-model', runtime: 'incoming' }], + previewModelId: 'incoming-model' + }); + assert.equal(result.preview.status, 'fits'); + assert.equal(result.preview.additionalBytes, 40 * GiB); + assert.equal(result.preview.remainingBytes, 28 * GiB); + assert.equal(result.preview.projectedUsedBytes, 68 * GiB); + assert.equal(result.preview.percent, 41.666667); + assert.equal(result.preview.message, 'Expected to fit'); +}); + +test('invalid telemetry does not invent segments or fits', () => { + for (const memory of [ + { totalBytes: null, availableBytes: 1, usedBytes: 1 }, + { totalBytes: -1, availableBytes: 1, usedBytes: 1 }, + { totalBytes: 10, availableBytes: 11, usedBytes: 1 }, + { totalBytes: 10, availableBytes: -1, usedBytes: 1 }, + { totalBytes: 10, availableBytes: null, usedBytes: 11 } + ]) { + const result = buildMemoryMap({ + node: { id: 'node-a', local: true, reachable: true, telemetry: { memory } }, + memorySafety: policy + }); + assert.equal(result.known, false); + assert.deepEqual(result.segments, []); + assert.equal(result.preview, null); + } + const unreachable = map({ nodeOverrides: { reachable: false } }); + assert.equal(unreachable.known, false); +}); + +test('RSS is preferred over a configured peak and oversubscription is bounded', () => { + const result = map({ + runtimes: { + a: { + id: 'a', + status: 'running', + memoryGb: 40, + memoryUsage: { residentBytes: 5 * GiB, groupId: 'proc:1', sharedRuntimeIds: ['a'] } + }, + b: { + id: 'b', + status: 'running', + memoryUsage: { residentBytes: 6 * GiB, groupId: 'proc:2', sharedRuntimeIds: ['b'] } + } + } + }); + assert.equal(result.segments.find((segment) => segment.runtimeId === 'a').bytes, 5 * GiB); + assert.equal( + result.segments.reduce((sum, segment) => sum + segment.bytes, 0), + 96 * GiB + ); +}); + +test('oversubscribed estimates are scaled within measured used space', () => { + const result = map({ + nodeOverrides: { total: 10 * GiB, available: 9 * GiB }, + runtimes: { + a: { id: 'a', status: 'starting', memoryGb: 6 }, + b: { id: 'b', status: 'starting', memoryGb: 6 } + } + }); + const runtimeBytes = result.segments + .filter((segment) => segment.kind === 'runtime') + .reduce((sum, segment) => sum + segment.bytes, 0); + assert.ok(runtimeBytes <= result.usedBytes); + assert.equal( + result.segments.reduce((sum, segment) => sum + segment.bytes, 0), + result.totalBytes + ); + assert.ok(result.segments.filter((segment) => segment.kind === 'runtime').every((segment) => segment.estimated)); +}); + +test('shared runtime groups and confirmed empty lazy lists are honest', () => { + const result = map({ + runtimes: { + 'shared-a': { + id: 'shared-a', + status: 'running', + command: 'ollama', + memoryGb: 8, + memoryUsage: { + residentBytes: 8 * GiB, + groupId: 'proc:9', + sharedRuntimeIds: ['shared-a', 'shared-b'], + loadedModelIds: [], + residencyKnown: true + } + }, + 'shared-b': { + id: 'shared-b', + status: 'running', + command: 'ollama', + memoryGb: 8, + memoryUsage: { + residentBytes: 8 * GiB, + groupId: 'proc:9', + sharedRuntimeIds: ['shared-a', 'shared-b'], + loadedModelIds: [], + residencyKnown: true + } + } + }, + models: [ + { id: 'shared-model', runtime: 'shared-a', upstreamModel: 'shared-upstream', targets: [{ node: 'node-a' }] } + ], + previewModelId: 'shared-model' + }); + const runtimeSegments = result.segments.filter((segment) => segment.kind === 'runtime'); + assert.equal(runtimeSegments.length, 1); + assert.equal(runtimeSegments[0].bytes, 8 * GiB); + assert.equal(result.preview.status, 'fits'); + assert.equal(result.preview.additionalBytes, 8 * GiB); +}); + +test('missing cold estimate is unknown rather than zero or fits', () => { + const result = map({ + runtimes: { cold: { id: 'cold', status: 'stopped' } }, + models: [{ id: 'cold-model', runtime: 'cold', targets: [{ node: 'node-a' }] }], + previewModelId: 'cold-model' + }); + assert.equal(result.preview.status, 'unknown'); + assert.equal(result.preview.additionalBytes, null); + assert.equal(result.preview.percent, null); +}); + +test('remote runtimes are excluded and previews remain node-local', () => { + const result = map({ + runtimes: { + remote: { + id: 'remote', + remote: true, + node: 'node-b', + status: 'running', + memoryUsage: { residentBytes: 20 * GiB, groupId: 'proc:remote', sharedRuntimeIds: ['remote'] } + } + }, + models: [{ id: 'remote-model', runtime: 'remote', targets: [{ node: 'node-b' }] }], + previewModelId: 'remote-model' + }); + assert.deepEqual( + result.segments.filter((segment) => segment.kind === 'runtime'), + [] + ); + assert.equal(result.preview.status, 'other-node'); + assert.equal(result.preview.nodeId, 'node-b'); + assert.equal(result.preview.additionalBytes, 0); +}); + +test('distributed wrappers and remote resources do not double count', () => { + const result = map({ + runtimes: { + member: { + id: 'member', + status: 'running', + memoryUsage: { residentBytes: 4 * GiB, groupId: 'proc:member', sharedRuntimeIds: ['member'] } + }, + distributed: { id: 'distributed', distributed: true, members: [{ runtime: 'member' }], status: 'running' } + } + }); + assert.deepEqual( + result.segments.filter((segment) => segment.kind === 'runtime').map((segment) => segment.runtimeId), + ['member'] + ); +}); + +test('remote reserve is not pooled into the local reserve', () => { + const result = map({ + runtimes: { remote: { id: 'remote', remote: true, node: 'node-b', runtimeManager: { memorySafety: policy } } } + }); + assert.equal(result.reserveBytes, 12 * GiB); +}); + +test('runtime colors are stable and order independent', () => { + const first = map({ runtimes: { a: { id: 'a', status: 'running' }, b: { id: 'b', status: 'running' } } }); + const second = map({ runtimes: { b: { id: 'b', status: 'running' }, a: { id: 'a', status: 'running' } } }); + const colors = (result) => + Object.fromEntries( + result.segments + .filter((segment) => segment.kind === 'runtime') + .map((segment) => [segment.runtimeId, segment.colorIndex]) + ); + assert.deepEqual(colors(first), colors(second)); + assert.ok(Object.values(colors(first)).every((value) => value >= 0 && value <= 7)); +}); + +test('the pure builder is serializable into a browser function', () => { + const source = buildMemoryMap.toString(); + const browserFunction = new Function(`return (${source})`)(); + const result = browserFunction({ + node: { + id: 'node-a', + local: true, + reachable: true, + telemetry: { memory: { totalBytes: 10, availableBytes: 7, usedBytes: 3 } } + } + }); + assert.equal(result.known, true); + assert.equal(result.usedBytes, 3); +}); + +test('a peer uses its own observed runtime and reserve, never pooled capacity', () => { + const result = map({ + node: { + id: 'peer', + local: false, + reachable: true, + telemetry: { memory: { totalBytes: 128 * GiB, availableBytes: 32 * GiB } }, + runtimeManager: { + memorySafety: { ...policy, minAvailableMemoryGb: 20 }, + runtimes: { + remote: { + status: 'running', + healthy: true, + memoryUsage: { residentBytes: 16 * GiB, residencyKnown: true, loadedModelIds: ['resident'] } + } + } + } + }, + runtimes: { remote: { remote: true, node: 'peer', memoryGb: 16 } }, + models: [{ id: 'resident', name: 'Resident on peer', runtime: 'remote' }], + previewModelId: 'resident' + }); + assert.equal(result.preview.status, 'resident'); + assert.equal(result.reserveBytes, 20 * GiB); + assert.equal(result.segments.find((s) => s.kind === 'runtime').bytes, 16 * GiB); + assert.equal( + result.segments.reduce((sum, s) => sum + s.percent, 0), + 100 + ); +}); + +test('maintenance and missing runtime state cannot promise a free model', () => { + const base = { models: [{ id: 'paused', runtime: 'model' }], previewModelId: 'paused' }; + assert.equal( + map({ + ...base, + runtimes: { model: { healthy: true, status: 'running', maintenance: { state: 'suspended' }, memoryGb: 32 } } + }).preview.status, + 'paused' + ); + assert.equal(map(base).preview.status, 'unknown'); +}); + +test('a healthy but empty cache still requires incoming memory', () => { + const result = map({ + runtimes: { + a: { + command: 'ollama', + status: 'running', + healthy: true, + memoryGb: 8, + memoryUsage: { residentBytes: GiB, residencyKnown: true, loadedModelIds: [] } + } + }, + models: [{ id: 'cold', runtime: 'a' }], + previewModelId: 'cold' + }); + assert.equal(result.preview.status, 'fits'); + assert.equal(result.preview.additionalBytes, 8 * GiB); +}); + +test('a healthy single-model service may still need to allocate its weights', () => { + const result = map({ + runtimes: { + a: { + status: 'running', + healthy: true, + memoryGb: 32, + memoryUsage: { residentBytes: GiB / 8, residencyKnown: false } + } + }, + models: [{ id: 'image', runtime: 'a' }], + previewModelId: 'image' + }); + assert.equal(result.preview.status, 'fits'); + assert.equal(result.preview.additionalBytes, 31.875 * GiB); +}); diff --git a/test/runtime-memory-usage.test.mjs b/test/runtime-memory-usage.test.mjs new file mode 100644 index 0000000..401a066 --- /dev/null +++ b/test/runtime-memory-usage.test.mjs @@ -0,0 +1,365 @@ +import assert from 'node:assert/strict'; +import { test } from 'node:test'; +import { createRuntimeMemoryUsageSampler } from '../src/runtime-memory-usage.mjs'; + +const KiB = 1024; + +function response(body, { ok = true, status = 200 } = {}) { + return { + ok, + status, + text: async () => JSON.stringify(body) + }; +} + +function fakeClock() { + let value = 0; + return { + now: () => value, + advance(amount) { + value += amount; + } + }; +} + +function sampler(options = {}) { + const calls = { ps: 0, listeners: 0, fetches: [] }; + return { + calls, + sampler: createRuntimeMemoryUsageSampler({ + fetchImpl: options.fetchImpl ?? null, + clock: options.clock, + scheduleTimeout: () => 0, + cacheTtlMs: options.cacheTtlMs ?? 5000, + footprintReader: async () => ({}), + ...options, + nowIso: () => new Date(options.clock?.now?.() ?? 0).toISOString(), + psReader: async () => { + calls.ps++; + return options.psReader ? options.psReader() : ''; + }, + listenerReader: async () => { + calls.listeners++; + return options.listenerReader ? options.listenerReader() : ''; + } + }) + }; +} + +test('parses one unique process tree from listener fallback', async () => { + const { calls, sampler: usage } = sampler({ + psReader: async () => [' 101 1 100', ' 102 101 50', ' 103 102 20'].join('\n'), + listenerReader: async () => 'lloom 101 user 12u IPv4 100 TCP 127.0.0.1:8201 (LISTEN)' + }); + const result = await usage.sample({ + runtimes: { runtime: { id: 'runtime', status: 'running', port: 8201 } }, + runtimeIds: ['runtime'] + }); + assert.equal(result.runtime.residentBytes, 170 * KiB); + assert.equal(result.runtime.groupId, 'proc:101'); + assert.deepEqual(result.runtime.sharedRuntimeIds, ['runtime']); + assert.equal(result.runtime.loadedModelIds, null); + assert.equal(result.runtime.residencyKnown, false); + assert.match(result.runtime.sampledAt, /^\d{4}-\d{2}-\d{2}T/); + assert.equal(calls.ps, 1); + assert.equal(calls.listeners, 1); +}); + +test('shared runtimes are grouped and their process tree is counted once', async () => { + const { sampler: usage } = sampler({ + psReader: async () => [' 101 1 50', ' 102 101 25'].join('\n'), + listenerReader: async () => 'a 101 1u TCP 127.0.0.1:8201 (LISTEN)' + }); + const runtimes = { + a: { id: 'a', status: 'running', port: 8201 }, + b: { id: 'b', status: 'running', port: 8201 } + }; + const result = await usage.sample({ runtimes, runtimeIds: ['b', 'a'] }); + assert.equal(result.a.residentBytes, 75 * KiB); + assert.equal(result.b.residentBytes, 75 * KiB); + assert.equal(result.a.groupId, 'proc:101'); + assert.equal(result.b.groupId, 'proc:101'); + assert.deepEqual(result.a.sharedRuntimeIds, ['a', 'b']); + assert.deepEqual(result.b.sharedRuntimeIds, ['a', 'b']); +}); + +test('shared backend observations use one residency request', async () => { + const fetches = []; + const { sampler: usage } = sampler({ + psReader: async () => ' 101 1 80', + listenerReader: async () => 'ollama 101 1u TCP localhost:11434 (LISTEN)', + fetchImpl: async (url) => { + fetches.push(url.toString()); + return response({ models: [{ model: 'llama3' }, { name: 'qwen' }] }); + } + }); + const result = await usage.sample({ + runtimes: { + a: { id: 'a', status: 'running', command: 'ollama', port: 8201, healthUrl: 'http://localhost:8201' }, + b: { id: 'b', status: 'running', command: 'ollama', port: 8201, healthUrl: 'http://localhost:8201' } + }, + runtimeIds: ['a', 'b'] + }); + assert.deepEqual([...fetches], ['http://localhost:8201/api/ps']); + assert.deepEqual(result.a.loadedModelIds, ['llama3', 'qwen']); + assert.deepEqual(result.b.loadedModelIds, ['llama3', 'qwen']); + assert.equal(result.a.residencyKnown, true); +}); + +test('confirmed empty Ollama residency stays empty and is known', async () => { + const { sampler: usage } = sampler({ + psReader: async () => '', + listenerReader: async () => '', + fetchImpl: async () => response({ models: [] }) + }); + const result = await usage.sample({ + runtimes: { ollama: { id: 'ollama', status: 'running', command: '/usr/local/bin/ollama', port: 8201 } }, + runtimeIds: ['ollama'] + }); + assert.equal(result.ollama.residentBytes, null); + assert.deepEqual(result.ollama.loadedModelIds, []); + assert.equal(result.ollama.residencyKnown, true); +}); + +test('audio server residency reads only its configured loopback health endpoint', async () => { + const fetches = []; + const { sampler: usage } = sampler({ + psReader: async () => '', + listenerReader: async () => '', + fetchImpl: async (url) => { + fetches.push(url.toString()); + return response({ tts_loaded: [{ id: 'voice' }], stt_loaded: ['ears'] }); + } + }); + const result = await usage.sample({ + runtimes: { + audio: { + id: 'audio', + status: 'running', + command: '/usr/local/bin/lloom-audio-server', + healthUrl: 'http://127.0.0.1:8300/' + } + }, + runtimeIds: ['audio'] + }); + assert.deepEqual(fetches, ['http://127.0.0.1:8300/health']); + assert.deepEqual(result.audio.loadedModelIds, ['ears', 'voice']); +}); + +test('sampling failures and failed residency calls remain unknown and do not throw', async () => { + const { sampler: usage } = sampler({ + psReader: async () => { + throw new Error('ps failed'); + }, + listenerReader: async () => { + throw new Error('lsof failed'); + }, + fetchImpl: async () => { + throw new Error('model endpoint failed'); + } + }); + const result = await usage.sample({ + runtimes: { ollama: { id: 'ollama', status: 'running', command: 'ollama', port: 8201 } }, + runtimeIds: ['ollama'] + }); + assert.equal(result.ollama.residentBytes, null); + assert.equal(result.ollama.loadedModelIds, null); + assert.equal(result.ollama.residencyKnown, false); +}); + +test('concurrent samples share the process and residency caches', async () => { + const clock = fakeClock(); + const { calls, sampler: usage } = sampler({ + clock, + psReader: async () => ' 101 1 20', + listenerReader: async () => 'a 101 1u TCP 127.0.0.1:8201 (LISTEN)', + fetchImpl: async () => response({ models: [] }) + }); + const request = { runtimes: { a: { id: 'a', status: 'running', command: 'ollama', port: 8201 } }, runtimeIds: ['a'] }; + const [first, second] = await Promise.all([usage.sample(request), usage.sample(request)]); + assert.deepEqual(first, second); + assert.equal(calls.ps, 1); + assert.equal(calls.listeners, 1); +}); + +test('cached samples expire after the configured TTL', async () => { + const clock = fakeClock(); + const { calls, sampler: usage } = sampler({ + clock, + psReader: async () => ' 101 1 20', + listenerReader: async () => '' + }); + const request = { runtimes: { a: { id: 'a', status: 'running', pid: 101 } }, runtimeIds: ['a'] }; + await usage.sample(request); + clock.advance(4999); + await usage.sample(request); + clock.advance(1); + await usage.sample(request); + assert.equal(calls.ps, 2); + assert.equal(calls.listeners, 2); +}); + +test('remote and distributed runtimes are not attributed locally', async () => { + const { calls, sampler: usage } = sampler({ + psReader: async () => ' 101 1 20', + listenerReader: async () => '' + }); + const result = await usage.sample({ + runtimes: { + remote: { id: 'remote', remote: true, node: 'node-b', status: 'running', pid: 101 }, + distributed: { id: 'distributed', distributed: true, status: 'running', pid: 101 } + }, + runtimeIds: ['remote', 'distributed'] + }); + assert.deepEqual(result, {}); + assert.equal(calls.ps, 0); + assert.equal(calls.listeners, 0); +}); + +test('Docker container PIDs are used only on Linux; macOS VM memory stays unattributed', async () => { + const processRows = ' 999 1 30\n 201 1 10'; + const listeners = 'docker 201 1u TCP 127.0.0.1:8201 (LISTEN)'; + const runtimes = { + docker: { + id: 'docker', + status: 'running', + adapter: 'docker', + container: { pid: 999 }, + port: 8201 + } + }; + const darwin = sampler({ + platform: 'darwin', + psReader: async () => processRows, + listenerReader: async () => listeners + }); + const darwinResult = await darwin.sampler.sample({ runtimes, runtimeIds: ['docker'] }); + assert.equal(darwinResult.docker.residentBytes, null); + + const linux = sampler({ + platform: 'linux', + psReader: async () => processRows, + listenerReader: async () => listeners + }); + const linuxResult = await linux.sampler.sample({ runtimes, runtimeIds: ['docker'] }); + assert.equal(linuxResult.docker.residentBytes, 40 * KiB); +}); + +test('lazy model calls are not made for unconfigured or remote hosts', async () => { + const fetches = []; + const { sampler: usage } = sampler({ + psReader: async () => '', + listenerReader: async () => '', + fetchImpl: async (url) => { + fetches.push(url.toString()); + return response({ models: [] }); + } + }); + const result = await usage.sample({ + runtimes: { + remote: { id: 'remote', remote: true, command: 'ollama', port: 8201 }, + external: { + id: 'external', + status: 'running', + command: 'ollama', + port: 8202, + healthUrl: 'https://example.invalid' + } + }, + runtimeIds: ['remote', 'external'] + }); + assert.deepEqual(fetches, []); + assert.equal(result.external.loadedModelIds, null); +}); + +test('overlapping parent and child ownership is one memory group', async () => { + const { sampler: usage } = sampler({ + psReader: async () => '101 1 20\n102 101 40\n103 102 10' + }); + const result = await usage.sample({ + runtimes: { + parent: { status: 'running', pid: 101 }, + child: { status: 'running', pid: 102 } + }, + runtimeIds: ['child', 'parent'] + }); + assert.equal(result.parent.groupId, result.child.groupId); + assert.equal(result.parent.residentBytes, 70 * KiB); + assert.deepEqual(result.parent.sharedRuntimeIds, ['child', 'parent']); +}); + +test('malformed residency is unknown even when process memory is known', async () => { + const { sampler: usage } = sampler({ + psReader: async () => '101 1 20', + fetchImpl: async (_url, options) => { + assert.equal(options.redirect, 'error'); + return response({ ok: true }); + } + }); + const result = await usage.sample({ + runtimes: { a: { pid: 101, status: 'running', command: 'ollama', port: 8201 } }, + runtimeIds: ['a'] + }); + assert.equal(result.a.residentBytes, 20 * KiB); + assert.equal(result.a.loadedModelIds, null); + assert.equal(result.a.residencyKnown, false); +}); + +test('a slow residency endpoint is aborted and becomes unknown', async () => { + const usage = createRuntimeMemoryUsageSampler({ + psReader: async () => '', + listenerReader: async () => '', + modelTimeoutMs: 20, + scheduleTimeout: (callback, ms) => setTimeout(callback, ms), + fetchImpl: async (_url, { signal }) => + new Promise((_, reject) => signal.addEventListener('abort', () => reject(signal.reason), { once: true })) + }); + const result = await usage.sample({ + runtimes: { a: { status: 'running', command: 'ollama', port: 8201 } }, + runtimeIds: ['a'] + }); + assert.equal(result.a.residencyKnown, false); +}); + +test('routing status avoids dashboard-only sampling', async () => { + const { RuntimeManager } = await import('../src/runtime-manager.mjs'); + let calls = 0; + const manager = new RuntimeManager( + { runtimes: { a: { enabled: false } }, models: [], cluster: {} }, + { + memoryUsageSampler: { + sample: async () => { + calls++; + return { a: { residentBytes: 12 } }; + } + } + } + ); + await manager.status(); + assert.equal(calls, 0); + const status = await manager.status({ includeMemoryUsage: true }); + assert.equal(calls, 1); + assert.equal(status.runtimes.a.memoryUsage.residentBytes, 12); +}); + +test('Darwin footprint includes charged memory omitted by RSS and falls back cleanly', async () => { + const request = { runtimes: { a: { status: 'running', pid: 101 } }, runtimeIds: ['a'] }; + const measured = sampler({ + platform: 'darwin', + psReader: async () => '101 1 20', + footprintReader: async () => ({ 101: 8 * 1024 ** 3 }) + }); + const result = await measured.sampler.sample(request); + assert.equal(result.a.residentBytes, 8 * 1024 ** 3); + assert.equal(result.a.source, 'process-footprint'); + const fallback = sampler({ + platform: 'darwin', + psReader: async () => '101 1 20', + footprintReader: async () => { + throw Error('unavailable'); + } + }); + const absent = await fallback.sampler.sample(request); + assert.equal(absent.a.residentBytes, 20 * 1024); + assert.equal(absent.a.source, 'process-rss'); +}); From 2609b87f84386460e42475a543a88019988631f2 Mon Sep 17 00:00:00 2001 From: Jason McCartney Date: Tue, 22 Sep 2026 00:21:47 -0700 Subject: [PATCH 10/16] Keep memory previews honest across suspended and federated models --- src/dashboard-memory.mjs | 21 ++++++++--- src/dashboard-presence-client.mjs | 2 +- src/dashboard-scene.mjs | 10 +++--- src/runtime-manager.mjs | 8 ++++- src/runtime-memory-usage.mjs | 58 ++++++++++++++++++++---------- test/dashboard-memory.test.mjs | 43 ++++++++++++++++++++++ test/runtime-memory-usage.test.mjs | 32 +++++++++++++++++ 7 files changed, 143 insertions(+), 31 deletions(-) diff --git a/src/dashboard-memory.mjs b/src/dashboard-memory.mjs index 8195cad..125e8fc 100644 --- a/src/dashboard-memory.mjs +++ b/src/dashboard-memory.mjs @@ -65,12 +65,21 @@ function buildMemoryMap({ node = null, runtimes = {}, models = [], previewModelI function targetNodes(model) { return (Array.isArray(model?.targets) ? model.targets : []) .map((target) => { - if (typeof target === 'string') return target; - return target?.node ?? target?.id ?? null; + return typeof target === 'object' ? (target?.node ?? null) : null; }) .filter(Boolean); } + function modelRuntimeId(model) { + if (model?.runtime) return model.runtime; + const ids = uniqueSorted( + (model?.targets ?? []) + .filter((target) => !target.node || target.node === nodeId) + .map((target) => target.remoteRuntime ?? target.runtime) + ); + return ids.length === 1 ? ids[0] : null; + } + function uniqueSorted(values) { return [...new Set(values.filter(Boolean))].sort((left, right) => String(left).localeCompare(String(right))); } @@ -162,7 +171,7 @@ function buildMemoryMap({ node = null, runtimes = {}, models = [], previewModelI const modelIds = new Set(); for (const runtimeId of group.runtimeIds) { for (const model of models ?? []) { - if (model?.runtime === runtimeId) { + if (modelRuntimeId(model) === runtimeId) { if (model.id) modelIds.add(model.id); } } @@ -257,7 +266,7 @@ function buildMemoryMap({ node = null, runtimes = {}, models = [], previewModelI function previewModel(modelId) { const model = (models ?? []).find((item) => item?.id === modelId); if (!model) return null; - const runtimeId = model.runtime ?? null; + const runtimeId = modelRuntimeId(model); const runtime = runtimeId ? runtimes[runtimeId] : null; const label = model.name ?? model.id; const base = { @@ -304,6 +313,8 @@ function buildMemoryMap({ node = null, runtimes = {}, models = [], previewModelI ) { return { ...base, status: 'paused', message: 'Paused; automatic starts are disabled.' }; } + if (!runtimeId && (model.federated || targets.length > 0)) + return { ...base, status: 'unknown', message: 'Waiting for model memory from this machine.' }; if (!runtimeId) { return { ...base, @@ -334,7 +345,7 @@ function buildMemoryMap({ node = null, runtimes = {}, models = [], previewModelI /(?:^|[/\\])(ollama|lloom-audio-server|lloom_audio_server(?:\.py)?)$/.test(String(part)) ); const sharedBackend = usage?.sharedRuntimeIds?.length > 1 || lazyBackend; - const modelsOnRuntime = (models ?? []).filter((item) => item?.runtime === runtimeId); + const modelsOnRuntime = (models ?? []).filter((item) => modelRuntimeId(item) === runtimeId); const runtimeIsSharedGroup = runtimeGroups.get(usage?.groupId ?? `runtime:${runtimeId}`)?.runtimeIds.length > 1; if (residentConfirmed) { return { diff --git a/src/dashboard-presence-client.mjs b/src/dashboard-presence-client.mjs index f2bc681..5289a96 100644 --- a/src/dashboard-presence-client.mjs +++ b/src/dashboard-presence-client.mjs @@ -43,7 +43,7 @@ export const presenceScript = String.raw` const rt=presenceRuntime(model),usage=rt?.memoryUsage; if(!rt?.healthy)return false; if(usage?.residencyKnown&&Array.isArray(usage.loadedModelIds))return usage.loadedModelIds.some(id=>id===model.id||id===model.upstreamModel); - return true; + return false; } function presenceModelLabel(model) { const runtime = presenceRuntime(model); diff --git a/src/dashboard-scene.mjs b/src/dashboard-scene.mjs index f58f99f..b9f33d1 100644 --- a/src/dashboard-scene.mjs +++ b/src/dashboard-scene.mjs @@ -236,7 +236,7 @@ export const sceneStyles = ` @media(max-width:1050px){.scene-detail:has(.model-inspector.open){position:fixed;z-index:80;left:12px;right:12px;bottom:12px;height:auto;max-height:calc(100dvh - 96px);background:linear-gradient(145deg,#122530,#09141c);box-shadow:0 -20px 70px #0009,0 0 0 1px #42616b66;padding:22px}.scene-detail:has(.model-inspector.open) .model-inspector-header{position:sticky;top:-22px;margin-top:-22px;padding-top:22px;background:#10212b;z-index:1}} @media(prefers-reduced-motion:reduce){.scene-gateway{animation:none!important}.scene-links .scene-particle{display:none}.scene-model{transition:none}} .scene-memory-panel {padding:18px 20px;margin-bottom:20px;background:#0b151d} - .scene-memory-panel h3 {margin-bottom:14px} + .scene-memory-panel h3 {margin-bottom:14px}.scene-memory-panel h3 small{display:block;color:#67dbe7;margin-top:5px}.scene-memory-panel h3 small[hidden]{display:none} .scene-mem-head {margin-bottom:12px}.scene-mem-total {font-size:22px}.scene-mem-total b {font-weight:500} .scene-mem-bar {height:76px}.scene-mem-block {border-radius:0;transition:width .38s cubic-bezier(.22,1,.36,1),filter .2s,opacity .2s} .scene-mem-face em {max-width:100%;overflow:hidden;text-overflow:ellipsis;font-size:10px} @@ -289,7 +289,7 @@ export const sceneScript = String.raw` $('.topology').dataset.presencePanel='diagnostic';$('.topology').hidden=true; const modelLeft=document.createElement('div');modelLeft.className='scene-model-main'; const modelLayout=document.createElement('div');modelLayout.className='scene-model-layout';$('#view-models').append(modelLayout);modelLayout.append(modelLeft); - const memoryPanel=document.createElement('section');memoryPanel.className='scene-memory-panel';memoryPanel.innerHTML='

Room for your AI

Point to a model to preview its memory.

'; + const memoryPanel=document.createElement('section');memoryPanel.className='scene-memory-panel';memoryPanel.innerHTML='

Room for your AI

Point to a model to preview its memory.

'; modelLeft.append(memoryPanel); const modelTabs=document.createElement('div');modelTabs.className='scene-model-tabs';modelTabs.innerHTML='Installed'; modelLeft.append(modelTabs,$('.presence-toolbar')); @@ -307,7 +307,7 @@ export const sceneScript = String.raw` function sceneMemColor(segment){return segment.kind==='system'?'#304955':segment.kind==='available'?'#123337':sceneMemColors[Math.abs(segment.colorIndex||0)%sceneMemColors.length];} function sceneMemModelNode(id){ const model=(state.physicalModels||[]).find(m=>m.id===id),rt=model&&presenceRuntime(model); - if(!model?.runtime)return null; + if(!model?.runtime)return model?.targets?.find(target=>target.node)?.node||null; return rt?.node||rt?.placement?.node||(!rt?.remote?sceneNodes().find(n=>n.local)?.id:null); } function sceneMemSnapshot(id=sceneMemPointer||sceneMemFocus||state.selectedModelId,ownNode=false){ @@ -329,7 +329,7 @@ export const sceneScript = String.raw` function sceneMemRender(){ const host=$('#scene-memory-bar');if(!host||typeof buildMemoryMap!=='function')return; const memory=sceneMemSnapshot(),segments=memory.segments||[],p=memory.preview; - const select=$('#scene-memory-machine');if(memory.nodeId)select.value=memory.nodeId; + const select=$('#scene-memory-machine'),nodes=sceneNodes(),base=sceneMemoryNode||nodes.find(n=>n.local)?.id||nodes[0]?.id; if(base)select.value=base;const context=$('#scene-memory-context');context.hidden=memory.nodeId===base;context.textContent=context.hidden?'':'Preview · '+sceneNodeName(nodes.find(n=>n.id===memory.nodeId)||{id:memory.nodeId}); const shape=JSON.stringify([memory.nodeId,memory.known,segments.map(s=>[s.id,s.kind,s.modelIds])]); if(shape!==sceneMemShape){ const focused=host.contains(document.activeElement)?document.activeElement.dataset.memoryFocus:null; @@ -382,7 +382,7 @@ export const sceneScript = String.raw` $('#presence-model-count').textContent=models.length+(models.length===1?' model':' models'); const key=JSON.stringify(models.map(m=>[m.id,m.name,sceneKind(m),presenceModelLabel(m),presencePolicy(presenceRuntime(m)),state.selectedModelId===m.id]));if(key===sceneModelKey)return;sceneModelKey=key; const catalog=$('#presence-models'),focused=catalog.contains(document.activeElement)?document.activeElement.closest('[data-presence-model]')?.dataset.presenceModel:null; - $('#presence-models').innerHTML=models.map(model=>{const rt=presenceRuntime(model);return '';}).join('')||'
No models match. Choose another filter or add a model.
'; + $('#presence-models').innerHTML=models.map(model=>{const rt=presenceRuntime(model);return '';}).join('')||'
No models match. Choose another filter or add a model.
'; if(focused)catalog.querySelector('[data-presence-model="'+CSS.escape(focused)+'"]')?.focus({preventScroll:true}); if(sceneMemPointer&&!models.some(m=>m.id===sceneMemPointer))sceneMemPointer=null; sceneMemSchedule(); diff --git a/src/runtime-manager.mjs b/src/runtime-manager.mjs index 4183d66..a2681aa 100644 --- a/src/runtime-manager.mjs +++ b/src/runtime-manager.mjs @@ -18,7 +18,12 @@ import { cleanupPortListener, terminateProcessTree } from './process-control.mjs import { memorySafetyPolicy, createMemorySafetyGuard, RuntimeMemorySafetyError } from './runtime-memory-safety.mjs'; import { createRuntimeMemoryUsageSampler } from './runtime-memory-usage.mjs'; -import { maintenanceBlocksRouting, assertMaintenanceStartAllowed, maintenanceError } from './model-maintenance.mjs'; +import { + runtimeMaintenance, + maintenanceBlocksRouting, + assertMaintenanceStartAllowed, + maintenanceError +} from './model-maintenance.mjs'; const execFileAsync = promisify(execFile); const packageRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..'); @@ -92,6 +97,7 @@ function compactRuntime(runtimeId, runtime, config) { return { enabled: runtime.enabled === true, keepWarm: runtime.keepWarm === true, + maintenance: runtimeMaintenance(config, runtimeId), memoryGb: runtime.memoryGb ?? runtime.memory?.requiredGb ?? null, maxConcurrency: runtimeMaxConcurrency(runtime), maxQueuedRequests: runtimeMaxQueuedRequests(runtime), diff --git a/src/runtime-memory-usage.mjs b/src/runtime-memory-usage.mjs index 2a6f486..9379d86 100644 --- a/src/runtime-memory-usage.mjs +++ b/src/runtime-memory-usage.mjs @@ -6,6 +6,20 @@ const execFileAsync = promisify(execFile); const LSOF = process.platform === 'darwin' ? '/usr/sbin/lsof' : 'lsof'; const MAX_MODEL_BODY_BYTES = 64 * 1024; +async function withDeadline(work, milliseconds) { + let timer; + try { + return await Promise.race([ + Promise.resolve().then(work), + new Promise((_, reject) => { + timer = setTimeout(() => reject(new Error('Memory observation timed out')), milliseconds); + }) + ]); + } finally { + clearTimeout(timer); + } +} + function positiveInteger(value) { const number = Number(value); return Number.isInteger(number) && number > 0 ? number : null; @@ -188,7 +202,7 @@ export function createRuntimeMemoryUsageSampler({ const at = now(); if (processCache && at < processCache.expiresAt) return processCache.value; if (processInFlight) return processInFlight; - processInFlight = Promise.allSettled([psReader(), listenerReader()]) + processInFlight = Promise.allSettled([withDeadline(psReader, 1000), withDeadline(listenerReader, 1000)]) .then(async ([ps, lsof]) => { const value = { rows: ps.status === 'fulfilled' ? parseProcessRows(ps.value) : [], @@ -198,7 +212,7 @@ export function createRuntimeMemoryUsageSampler({ }; if (platform === 'darwin' && value.rows.length) { try { - const footprints = await footprintReader(value.rows.map((row) => row.pid)); + const footprints = await withDeadline(() => footprintReader(value.rows.map((row) => row.pid)), 1000); for (const row of value.rows) { const bytes = footprints?.[row.pid]; if (typeof bytes === 'number' && Number.isFinite(bytes) && bytes >= 0) row.footprint = bytes; @@ -223,23 +237,29 @@ export function createRuntimeMemoryUsageSampler({ if (cached && at < cached.expiresAt) return cached.value; const inFlight = modelInFlight.get(key); if (inFlight) return inFlight; - const request = (async () => { - let timer = null; - let signal = null; - if (typeof AbortController === 'function' && modelTimeoutMs > 0) { - const controller = new AbortController(); - signal = controller.signal; - timer = scheduleTimeout(() => controller.abort(new Error('model residency request timed out')), modelTimeoutMs); - } - try { - const response = await fetchImpl(url, { signal, redirect: 'error' }); - if (!response?.ok) throw new Error(`model residency request failed (${response?.status ?? 'unknown'})`); - const body = await boundedText(response, MAX_MODEL_BODY_BYTES); - return extractLoadedModelIds(JSON.parse(body), kind); - } finally { - if (timer) clearTimeout(timer); - } - })() + const request = withDeadline( + async () => { + let timer = null; + let signal = null; + if (typeof AbortController === 'function' && modelTimeoutMs > 0) { + const controller = new AbortController(); + signal = controller.signal; + timer = scheduleTimeout( + () => controller.abort(new Error('model residency request timed out')), + modelTimeoutMs + ); + } + try { + const response = await fetchImpl(url, { signal, redirect: 'error' }); + if (!response?.ok) throw new Error(`model residency request failed (${response?.status ?? 'unknown'})`); + const body = await boundedText(response, MAX_MODEL_BODY_BYTES); + return extractLoadedModelIds(JSON.parse(body), kind); + } finally { + if (timer) clearTimeout(timer); + } + }, + Math.max(1, modelTimeoutMs) + 50 + ) .then( (value) => { const result = { value, expiresAt: now() + cacheTtlMs }; diff --git a/test/dashboard-memory.test.mjs b/test/dashboard-memory.test.mjs index 3b08d80..5e14692 100644 --- a/test/dashboard-memory.test.mjs +++ b/test/dashboard-memory.test.mjs @@ -356,3 +356,46 @@ test('a healthy single-model service may still need to allocate its weights', () assert.equal(result.preview.status, 'fits'); assert.equal(result.preview.additionalBytes, 31.875 * GiB); }); + +test('replica identifiers are not machine identifiers', () => { + const result = map({ + models: [{ id: 'cloud', targets: [{ id: 'default', backend: 'provider' }] }], + previewModelId: 'cloud' + }); + assert.equal(result.preview.status, 'external'); +}); + +test('a federated target resolves the observed runtime on its peer', () => { + const result = map({ + node: { + id: 'peer', + reachable: true, + local: false, + telemetry: { memory: { totalBytes: 96 * GiB, availableBytes: 48 * GiB } }, + runtimeManager: { + memorySafety: policy, + runtimes: { + audio: { + status: 'running', + healthy: true, + memoryGb: 8, + memoryUsage: { residentBytes: 4 * GiB, residencyKnown: true, loadedModelIds: ['voice'] } + } + } + } + }, + models: [ + { + id: 'peer/voice', + name: 'Peer voice', + upstreamModel: 'voice', + targets: [{ id: 'replica', node: 'peer', remoteRuntime: 'audio' }], + federated: true + } + ], + previewModelId: 'peer/voice' + }); + assert.equal(result.preview.status, 'resident'); + assert.equal(result.segments[0].runtimeId, 'audio'); + assert.deepEqual(result.segments[0].modelIds, ['peer/voice']); +}); diff --git a/test/runtime-memory-usage.test.mjs b/test/runtime-memory-usage.test.mjs index 401a066..e68881b 100644 --- a/test/runtime-memory-usage.test.mjs +++ b/test/runtime-memory-usage.test.mjs @@ -363,3 +363,35 @@ test('Darwin footprint includes charged memory omitted by RSS and falls back cle assert.equal(absent.a.residentBytes, 20 * 1024); assert.equal(absent.a.source, 'process-rss'); }); + +test('an endpoint that ignores abort cannot hold status forever', async () => { + const usage = createRuntimeMemoryUsageSampler({ + psReader: async () => '', + listenerReader: async () => '', + modelTimeoutMs: 10, + fetchImpl: async () => new Promise(() => {}) + }); + const result = await usage.sample({ + runtimes: { a: { status: 'running', command: 'ollama', port: 8201 } }, + runtimeIds: ['a'] + }); + assert.equal(result.a.residencyKnown, false); +}); + +test('runtime status exposes inherited maintenance to the dashboard', async () => { + const { RuntimeManager } = await import('../src/runtime-manager.mjs'); + const manager = new RuntimeManager({ + cluster: {}, + models: [], + runtimes: { + a: { enabled: false }, + group: { + enabled: false, + placement: { mode: 'distributed', members: [{ runtime: 'a' }] }, + maintenance: { state: 'suspended' } + } + } + }); + const result = await manager.status(); + assert.equal(result.runtimes.a.maintenance.state, 'suspended'); +}); From b67e1a7b42551c540e5591805bfa21095cd80473 Mon Sep 17 00:00:00 2001 From: Jason McCartney Date: Tue, 22 Sep 2026 08:25:47 -0700 Subject: [PATCH 11/16] Enforce configured OpenRouter provider restrictions --- docs/architecture.md | 24 ++ package.json | 4 +- src/protocol/openrouter-provider.mjs | 103 ++++++ src/server.mjs | 3 +- test/openrouter-provider.test.mjs | 468 +++++++++++++++++++++++++++ 5 files changed, 599 insertions(+), 3 deletions(-) create mode 100644 src/protocol/openrouter-provider.mjs create mode 100644 test/openrouter-provider.test.mjs diff --git a/docs/architecture.md b/docs/architecture.md index 52bfe22..d3278b6 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -48,6 +48,30 @@ tool call has no reasoning to replay; a new user turn permits thinking again. This opt-in compatibility setting preserves tool constraints without inventing reasoning history. Ordinary calls outside those cases keep their thinking settings. +### OpenRouter provider restriction + +Set `openrouterProvider` on a dedicated OpenRouter backend to restrict its +upstream providers across Chat Completions, Responses, and Anthropic Messages, +including streaming requests: + +```json +{ + "type": "openai", + "baseUrl": "https://openrouter.ai/api/v1", + "apiKeyEnv": "OPENROUTER_API_KEY", + "openrouterProvider": { "only": ["z-ai"], "allow_fallbacks": false } +} +``` + +Configured provider fields override client requests. Other client provider +preferences remain intact. `only` must contain at least one nonempty provider +slug; `allow_fallbacks` defaults to false. Invalid policy objects fail before +an upstream request is sent. The policy applies only to the exact +`openrouter.ai` host. With the configuration above, an unavailable Z.ai endpoint +returns an error instead of switching to another provider. Models sharing this +backend share its restriction; LLooM alias fallback rules remain separate. +See [OpenRouter provider routing](https://openrouter.ai/docs/guides/routing/provider-selection). + ## Security Defaults | Setting | Default | Meaning | diff --git a/package.json b/package.json index 2df4fe7..9f85856 100644 --- a/package.json +++ b/package.json @@ -64,7 +64,7 @@ "check": "npm run check:cluster && npm run check:js && npm run check:entity-stagger && npm run check:python && npm run check:shell && npm run check:spark-deploy && npm run check:workers", "check:cluster": "node --check src/cluster.mjs && node --check test/cluster.test.mjs", "check:entity-stagger": "node --check scripts/entity-stagger-benchmark.mjs && node --check test/entity-stagger-benchmark.test.mjs", - "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs", + "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs && node --check src/protocol/openrouter-provider.mjs && node --check test/openrouter-provider.test.mjs", "check:python": "python3 -m py_compile backends/dspark-vllm/apply-patch-pack.py backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/vision_exp/*.py backends/qwen38-vllm/apply-ple-fp8-patch.py backends/qwen38-sglang/*.py backends/glm53-exl3/*.py backends/mlx-audio/lloom_audio_server.py backends/hear/lloom_hear_server.py backends/chatterbox/lloom_chatterbox_server.py patches/apply_mtplx_longctx_fix.py backends/comfyui-media/bridge/*.py backends/comfyui-media/graphs/*.py backends/comfyui-media/build/*.py backends/comfyui-media/install.py backends/qwen-image-diffusers/*.py backends/hear/test/test_hear_security.py", "check:shell": "bash -n backends/dspark-vllm/entrypoint.sh backends/dspark-vllm/runtime/build-hardened-image.sh backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/apply-runtime.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.sh backends/qwen38-vllm/entrypoint.sh backends/qwen38-sglang/entrypoint.sh backends/glm53-exl3/entrypoint.sh backends/mlx-audio/install.sh backends/hear/install.sh backends/chatterbox/install.sh clients/examples/lloom-claude clients/examples/lloom-codex clients/examples/lloom-hermes clients/examples/lloom-opencode clients/examples/lloom-zero scripts/remote-install-spark.sh scripts/remote-stage-spark-worker.sh", "check:spark-deploy": "node --check test/spark-deploy.test.mjs && node test/spark-deploy.test.mjs", @@ -73,7 +73,7 @@ "community:check": "node scripts/check-community-deploy.mjs", "package:check": "node scripts/check-package.mjs", "test:adoption": "node test/resource-fit.test.mjs && node test/model-acquisition.test.mjs && node test/dashboard-status.test.mjs", - "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs", + "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/openrouter-provider.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs", "test:cluster": "node test/cluster.test.mjs", "test": "npm run test:cluster && npm run test:unit && npm run test:entity-stagger && npm run smoke && npm run community:check", "test:entity-stagger": "node test/entity-stagger-benchmark.test.mjs", diff --git a/src/protocol/openrouter-provider.mjs b/src/protocol/openrouter-provider.mjs new file mode 100644 index 0000000..c42f3ad --- /dev/null +++ b/src/protocol/openrouter-provider.mjs @@ -0,0 +1,103 @@ +/** + * OpenRouter provider-preference policy. + * + * OpenRouter routes each chat completion across multiple upstream providers and + * accepts a `provider` object on the request body. That object lets a caller + * constrain routing. LLooM exposes an operator-owned policy at + * `backends..openrouterProvider` so a configured lane always routes the way + * the operator intends. + * + * The policy is deliberately narrow: + * + * { "openrouterProvider": { "only": ["z-ai"], "allow_fallbacks": false } } + * + * - `only` is required, non-empty, and every entry must be a non-empty string. + * - `allow_fallbacks` is optional and must be a boolean. + * - Any other key is rejected so the operator cannot believe a setting took + * effect when the gateway did not enforce it. + * + * Enforcement is fail-closed. A malformed policy on an OpenRouter backend + * throws instead of silently sending an unconstrained request. Non-OpenRouter + * backends and backends without a policy leave the body untouched. + */ + +const POLICY_KEY = 'openrouterProvider'; +const ALLOWED_KEYS = new Set(['only', 'allow_fallbacks']); + +function isPlainObject(value) { + return typeof value === 'object' && value !== null && !Array.isArray(value); +} + +/** True only for the real OpenRouter API host, never a lookalike name. */ +export function isOpenRouterBackend(backend = {}) { + const baseUrl = backend?.baseUrl; + if (typeof baseUrl !== 'string' || !baseUrl.trim()) return false; + try { + return new URL(baseUrl).hostname === 'openrouter.ai'; + } catch { + return false; + } +} + +/** + * Validate the configured policy. Returns the normalized policy, `null` when no + * policy is configured, and throws for any malformed configuration. + */ +export function normalizeOpenRouterProviderPolicy(configured, { backendId } = {}) { + if (configured === undefined || configured === null) return null; + const where = backendId ? `backends.${backendId}.${POLICY_KEY}` : POLICY_KEY; + if (!isPlainObject(configured)) { + throw new Error(`${where} must be an object`); + } + for (const key of Object.keys(configured)) { + if (!ALLOWED_KEYS.has(key)) { + throw new Error(`${where} has unsupported key "${key}"; allowed keys are only, allow_fallbacks`); + } + } + if (!Object.hasOwn(configured, 'only')) { + throw new Error(`${where}.only is required`); + } + if (!Array.isArray(configured.only)) { + throw new Error(`${where}.only must be an array of provider slugs`); + } + if (configured.only.length === 0) { + throw new Error(`${where}.only must not be empty`); + } + const only = configured.only.map((entry) => { + if (typeof entry !== 'string' || !entry.trim()) { + throw new Error(`${where}.only entries must be non-empty strings`); + } + return entry.trim(); + }); + if (Object.hasOwn(configured, 'allow_fallbacks') && typeof configured.allow_fallbacks !== 'boolean') { + throw new Error(`${where}.allow_fallbacks must be a boolean`); + } + return { + only, + // Fail closed: an operator who constrains `only` almost never wants + // OpenRouter to silently substitute a different provider. + allow_fallbacks: configured.allow_fallbacks ?? false + }; +} + +/** + * Apply the OpenRouter provider policy to an outbound chat-completions body. + * + * Configured values win over caller-supplied `provider` fields while every + * other caller field is preserved. Bodies that are not plain objects, backends + * on another host, and backends without a policy are returned unchanged. + */ +export function applyOpenRouterProviderPolicy(body, backend = {}) { + if (!isPlainObject(body)) return body; + // Only the real OpenRouter host is ever treated as OpenRouter. Every other + // backend, including lookalike hostnames, is left exactly as the caller sent + // it and never has a policy applied to it. + if (!isOpenRouterBackend(backend)) return body; + const policy = normalizeOpenRouterProviderPolicy(backend?.[POLICY_KEY], { backendId: backend?.id }); + if (!policy) return body; + const callerProvider = isPlainObject(body.provider) ? body.provider : {}; + return { + ...body, + provider: { ...callerProvider, only: [...policy.only], allow_fallbacks: policy.allow_fallbacks } + }; +} diff --git a/src/server.mjs b/src/server.mjs index cc2508b..502a0dc 100644 --- a/src/server.mjs +++ b/src/server.mjs @@ -6,6 +6,7 @@ import { generateProviderVideo } from './video-providers.mjs'; import http from 'node:http'; import { readErrorDiagnostic, streamProviderError } from './protocol/upstream-error.mjs'; import { fetchWithStreamProgress } from './protocol/stream-progress.mjs'; +import { applyOpenRouterProviderPolicy } from './protocol/openrouter-provider.mjs'; import { appendFileSync, existsSync, @@ -1388,7 +1389,7 @@ async function fetchUpstream({ backend, path, body, headers = {}, signal, dispat fetch(upstreamUrl(backend, path), { method: 'POST', headers: backendHeaders(backend, headers), - body: JSON.stringify(body), + body: JSON.stringify(path === '/v1/chat/completions' ? applyOpenRouterProviderPolicy(body, backend) : body), signal: progressSignal, dispatcher }), diff --git a/test/openrouter-provider.test.mjs b/test/openrouter-provider.test.mjs new file mode 100644 index 0000000..fb43fb9 --- /dev/null +++ b/test/openrouter-provider.test.mjs @@ -0,0 +1,468 @@ +import assert from 'node:assert/strict'; +import { MockAgent } from 'undici'; +import { createLloomServer } from '../src/server.mjs'; +import { + applyOpenRouterProviderPolicy, + isOpenRouterBackend, + normalizeOpenRouterProviderPolicy +} from '../src/protocol/openrouter-provider.mjs'; + +// --------------------------------------------------------------------------- +// Pure policy unit coverage +// --------------------------------------------------------------------------- + +const openRouterBackend = (extra = {}) => ({ + id: 'openrouter-lane', + type: 'openai', + baseUrl: 'https://openrouter.ai/api/v1', + ...extra +}); + +function testLookalikeHosts() { + assert.equal(isOpenRouterBackend(openRouterBackend()), true); + assert.equal(isOpenRouterBackend({ baseUrl: 'https://openrouter.ai/api/v1' }), true); + for (const baseUrl of [ + 'https://openrouter.ai.evil.example/api/v1', + 'https://notopenrouter.ai/api/v1', + 'https://openrouter.ai.example.com/api/v1', + 'https://api.openrouter.ai/api/v1', + 'http://127.0.0.1:8200/v1', + 'not a url', + '', + undefined + ]) { + assert.equal(isOpenRouterBackend({ baseUrl }), false, `expected lookalike to be rejected: ${baseUrl}`); + } +} + +function testNoConfigLeavesBodyUntouched() { + const body = { + model: 'z-ai/glm-5.2', + messages: [{ role: 'user', content: 'hi' }], + provider: { order: ['together'], allow_fallbacks: true } + }; + const frozen = JSON.parse(JSON.stringify(body)); + // No policy key at all. + assert.equal(applyOpenRouterProviderPolicy(body, openRouterBackend()), body); + // Policy explicitly null is "not configured", not malformed. + assert.equal(applyOpenRouterProviderPolicy(body, openRouterBackend({ openrouterProvider: null })), body); + // Non-OpenRouter host is untouched. A policy there is never enforced, so a + // plain (not-openrouter) backend must not be rejected for carrying one. + const local = { id: 'local', baseUrl: 'http://127.0.0.1:8201/v1', openrouterProvider: { only: ['z-ai'] } }; + assert.equal(applyOpenRouterProviderPolicy(body, local), body); + assert.deepEqual(body, frozen); +} + +function testValidPolicyEnforced() { + const body = { model: 'z-ai/glm-5.2', messages: [{ role: 'user', content: 'hi' }] }; + const applied = applyOpenRouterProviderPolicy(body, openRouterBackend({ openrouterProvider: { only: ['z-ai'] } })); + assert.deepEqual(applied.provider, { only: ['z-ai'], allow_fallbacks: false }); + // Default allow_fallbacks is false (fail closed) when omitted. + assert.equal(applied.provider.allow_fallbacks, false); + // Original body is not mutated. + assert.equal(body.provider, undefined); + + const explicit = applyOpenRouterProviderPolicy( + body, + openRouterBackend({ openrouterProvider: { only: ['z-ai', 'z-ai-intl'], allow_fallbacks: true } }) + ); + assert.deepEqual(explicit.provider, { only: ['z-ai', 'z-ai-intl'], allow_fallbacks: true }); + + // Whitespace is trimmed and surrounding body fields are preserved. + const trimmed = applyOpenRouterProviderPolicy( + { model: 'z-ai/glm-5.2', temperature: 0.4, provider: { order: ['x'] } }, + openRouterBackend({ openrouterProvider: { only: [' z-ai '] } }) + ); + assert.deepEqual(trimmed.provider.only, ['z-ai']); + assert.equal(trimmed.temperature, 0.4); + // Caller provider fields survive alongside the enforced keys. + assert.deepEqual(trimmed.provider.order, ['x']); +} + +function testCallerOverrideAttemptsLose() { + const backend = openRouterBackend({ openrouterProvider: { only: ['z-ai'], allow_fallbacks: false } }); + const cases = [ + { provider: { only: ['openai'], allow_fallbacks: true } }, + { provider: { allow_fallbacks: true } }, + { provider: { only: [] } }, + { provider: { only: ['any'] } } + ]; + for (const extra of cases) { + const applied = applyOpenRouterProviderPolicy({ model: 'm', messages: [], ...extra }, backend); + assert.deepEqual(applied.provider, { only: ['z-ai'], allow_fallbacks: false }, JSON.stringify(extra)); + } + + const permissive = openRouterBackend({ openrouterProvider: { only: ['z-ai'], allow_fallbacks: true } }); + const applied = applyOpenRouterProviderPolicy({ provider: { allow_fallbacks: false } }, permissive); + assert.deepEqual(applied.provider, { only: ['z-ai'], allow_fallbacks: true }); +} + +function testNonObjectBodyUntouched() { + const backend = openRouterBackend({ openrouterProvider: { only: ['z-ai'] } }); + for (const body of [null, undefined, 'not-an-object', 42, ['array']]) { + assert.equal(applyOpenRouterProviderPolicy(body, backend), body); + } +} + +function testMalformedPoliciesThrow() { + const base = { id: 'openrouter-lane', baseUrl: 'https://openrouter.ai/api/v1' }; + const malformed = [ + {}, // only is required + { only: [] }, + { only: 'z-ai' }, + { only: ['z-ai', ''] }, + { only: ['z-ai', ' '] }, + { only: [null] }, + { only: ['z-ai'], allow_fallbacks: 'no' }, + { only: ['z-ai'], allow_fallbacks: 1 }, + { only: ['z-ai'], order: ['x'] }, // unknown key + { only: ['z-ai'], allow_fallbacks: false, ignore: ['a'] }, + { only: ['z-ai'], dataCollection: 'deny' }, + 'z-ai', + ['z-ai'], + 42 + ]; + for (const openrouterProvider of malformed) { + assert.throws( + () => normalizeOpenRouterProviderPolicy(openrouterProvider, { backendId: 'openrouter-lane' }), + /openrouterProvider/, + `expected malformed policy to throw: ${JSON.stringify(openrouterProvider)}` + ); + assert.throws( + () => applyOpenRouterProviderPolicy({ model: 'm' }, { ...base, openrouterProvider }), + /openrouterProvider/, + `expected apply to fail closed for: ${JSON.stringify(openrouterProvider)}` + ); + } + // No configured value (undefined/null) is not malformed. + assert.equal(normalizeOpenRouterProviderPolicy(undefined), null); + assert.equal(normalizeOpenRouterProviderPolicy(null), null); +} + +// --------------------------------------------------------------------------- +// Gateway integration: the whole /v1/chat/completions path is exercised with a +// mocked undici dispatcher so no network or TLS egress occurs. +// --------------------------------------------------------------------------- + +function gatewayConfig(backend) { + return { + server: { host: '127.0.0.1', port: 0 }, + security: { allowMissingAuth: true, apiKeys: [] }, + defaults: { chatModel: 'z-ai/glm-5.2' }, + backends: { 'openrouter-lane': backend }, + models: [ + { + id: 'z-ai/glm-5.2', + backend: 'openrouter-lane', + upstreamModel: 'z-ai/glm-5.2', + kind: 'chat', + contextWindow: 200000, + maxPromptTokens: 100000 + } + ], + runtimes: {} + }; +} + +async function withMockedDispatcher(fn) { + const { Agent } = await import('undici'); + const originalDispatch = Agent.prototype.dispatch; + const agent = new Agent(); + agent.dispatch = originalDispatch.bind(agent); + const mockAgent = new MockAgent({ agent }); + mockAgent.disableNetConnect(); + Agent.prototype.dispatch = function patchedDispatch(opts, handler) { + return mockAgent.dispatch(opts, handler); + }; + try { + return await fn(mockAgent); + } finally { + Agent.prototype.dispatch = originalDispatch; + await mockAgent.close(); + } +} + +async function readRequestBody(body) { + if (!body || typeof body[Symbol.asyncIterator] !== 'function') return body; + const chunks = []; + for await (const chunk of body) chunks.push(chunk); + return Buffer.concat(chunks).toString('utf8'); +} + +function intercept(mockAgent, { status = 200, contentType = 'application/json', payload, onBody }) { + const pool = mockAgent.get('https://openrouter.ai'); + return pool.intercept({ path: '/api/v1/chat/completions', method: 'POST' }).reply( + status, + async (opts) => { + onBody(await readRequestBody(opts.body)); + return payload; + }, + { headers: { 'content-type': contentType } } + ); +} + +const openAiChatPayload = { + id: 'chatcmpl-1', + object: 'chat.completion', + choices: [{ index: 0, message: { role: 'assistant', content: 'ok' }, finish_reason: 'stop' }], + usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } +}; + +const openAiStreamPayload = [ + 'data: {"id":"chatcmpl-1","object":"chat.completion.chunk","choices":[{"index":0,"delta":{"role":"assistant","content":"ok"},"finish_reason":null}]}', + '', + 'data: {"id":"chatcmpl-1","object":"chat.completion.chunk","choices":[{"index":0,"delta":{},"finish_reason":"stop"}]}', + '', + 'data: [DONE]', + '', + '' +].join('\n'); + +let gatewayServer; +let gatewayPort; + +async function startGateway(backend) { + const app = createLloomServer(gatewayConfig(backend), { logger: { error() {}, warn() {}, info() {}, log() {} } }); + await new Promise((resolve) => app.server.listen(0, '127.0.0.1', resolve)); + gatewayServer = app.server; + gatewayPort = app.server.address().port; + return gatewayPort; +} + +async function stopGateway() { + if (!gatewayServer) return; + const server = gatewayServer; + gatewayServer = null; + await new Promise((resolve) => server.close(resolve)); +} + +async function testGatewayChatBuffered() { + const seen = []; + await startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'], allow_fallbacks: false } })); + try { + await withMockedDispatcher(async (mockAgent) => { + intercept(mockAgent, { payload: openAiChatPayload, onBody: (body) => seen.push(body) }); + const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ + model: 'z-ai/glm-5.2', + messages: [{ role: 'user', content: 'hi' }], + provider: { only: ['openai'], allow_fallbacks: true, order: ['x'] } + }) + }); + assert.equal(res.status, 200); + const json = await res.json(); + assert.equal(json.choices[0].message.content, 'ok'); + }); + assert.equal(seen.length, 1, 'expected exactly one upstream chat call'); + const outbound = JSON.parse(seen[0]); + // Caller override attempt is defeated; other caller provider fields survive. + assert.deepEqual(outbound.provider, { order: ['x'], only: ['z-ai'], allow_fallbacks: false }); + assert.equal(outbound.model, 'z-ai/glm-5.2'); + assert.deepEqual(outbound.messages, [{ role: 'user', content: 'hi' }]); + } finally { + await stopGateway(); + } +} + +async function testGatewayChatStream() { + const seen = []; + await startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'] } })); + try { + await withMockedDispatcher(async (mockAgent) => { + intercept(mockAgent, { + contentType: 'text/event-stream', + payload: openAiStreamPayload, + onBody: (body) => seen.push(body) + }); + const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ + model: 'z-ai/glm-5.2', + messages: [{ role: 'user', content: 'hi' }], + stream: true + }) + }); + assert.equal(res.status, 200); + const text = await res.text(); + assert.match(text, /data: \[DONE\]/); + }); + assert.equal(seen.length, 1); + const outbound = JSON.parse(seen[0]); + assert.deepEqual(outbound.provider, { only: ['z-ai'], allow_fallbacks: false }); + assert.equal(outbound.stream, true); + } finally { + await stopGateway(); + } +} + +async function testGatewayResponsesBridge(stream = false) { + const seen = []; + await startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'] } })); + try { + await withMockedDispatcher(async (mockAgent) => { + intercept(mockAgent, { + payload: stream ? openAiStreamPayload : openAiChatPayload, + contentType: stream ? 'text/event-stream' : 'application/json', + onBody: (body) => seen.push(body) + }); + const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/responses`, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ stream, model: 'z-ai/glm-5.2', input: 'hi', provider: { only: ['openai'] } }) + }); + assert.equal(res.status, 200); + if (stream) assert.match(await res.text(), /response.completed/); + else assert.equal((await res.json()).object, 'response'); + }); + assert.equal(seen.length, 1); + const outbound = JSON.parse(seen[0]); + assert.deepEqual(outbound.provider, { only: ['z-ai'], allow_fallbacks: false }); + } finally { + await stopGateway(); + } +} + +async function testGatewayAnthropicBridge(stream = false) { + const seen = []; + await startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'], allow_fallbacks: true } })); + try { + await withMockedDispatcher(async (mockAgent) => { + intercept(mockAgent, { + payload: stream ? openAiStreamPayload : openAiChatPayload, + contentType: stream ? 'text/event-stream' : 'application/json', + onBody: (body) => seen.push(body) + }); + const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/messages`, { + method: 'POST', + headers: { 'content-type': 'application/json', 'anthropic-version': '2023-06-01' }, + body: JSON.stringify({ + stream, + model: 'z-ai/glm-5.2', + max_tokens: 32, + messages: [{ role: 'user', content: 'hi' }] + }) + }); + assert.equal(res.status, 200); + if (stream) assert.match(await res.text(), /message_stop/); + else assert.equal((await res.json()).type, 'message'); + }); + assert.equal(seen.length, 1); + const outbound = JSON.parse(seen[0]); + assert.deepEqual(outbound.provider, { only: ['z-ai'], allow_fallbacks: true }); + } finally { + await stopGateway(); + } +} + +async function testGatewayMalformedPolicyFailsClosed() { + // The gateway must not silently send an unconstrained request to OpenRouter. + await startGateway(openRouterBackend({ openrouterProvider: { only: [] } })); + try { + await withMockedDispatcher(async () => { + const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ model: 'z-ai/glm-5.2', messages: [{ role: 'user', content: 'hi' }] }) + }); + assert.notEqual(res.status, 200); + await res.text(); + }); + } finally { + await stopGateway(); + } +} + +async function testGatewayNoPolicyUntouched() { + const seen = []; + await startGateway({ id: 'openrouter-lane', type: 'openai', baseUrl: 'https://openrouter.ai/api/v1' }); + try { + await withMockedDispatcher(async (mockAgent) => { + intercept(mockAgent, { payload: openAiChatPayload, onBody: (body) => seen.push(body) }); + const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ + model: 'z-ai/glm-5.2', + messages: [{ role: 'user', content: 'hi' }], + provider: { only: ['openai'], allow_fallbacks: true } + }) + }); + assert.equal(res.status, 200); + await res.text(); + }); + const outbound = JSON.parse(seen[0]); + assert.deepEqual(outbound.provider, { only: ['openai'], allow_fallbacks: true }); + } finally { + await stopGateway(); + } +} + +async function testGatewayLookalikeHostUntouched() { + const seen = []; + await startGateway({ + id: 'openrouter-lane', + type: 'openai', + baseUrl: 'https://notopenrouter.ai/api/v1', + openrouterProvider: { only: ['z-ai'] } + }); + const { Agent } = await import('undici'); + const originalDispatch = Agent.prototype.dispatch; + const agent = new Agent(); + agent.dispatch = originalDispatch.bind(agent); + const mockAgent = new MockAgent({ agent }); + mockAgent.disableNetConnect(); + Agent.prototype.dispatch = function patchedDispatch(opts, handler) { + return mockAgent.dispatch(opts, handler); + }; + try { + mockAgent + .get('https://notopenrouter.ai') + .intercept({ path: '/api/v1/chat/completions', method: 'POST' }) + .reply( + 200, + async (opts) => { + seen.push(await readRequestBody(opts.body)); + return openAiChatPayload; + }, + { headers: { 'content-type': 'application/json' } } + ); + const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ model: 'z-ai/glm-5.2', messages: [{ role: 'user', content: 'hi' }] }) + }); + assert.equal(res.status, 200); + await res.text(); + } finally { + Agent.prototype.dispatch = originalDispatch; + await mockAgent.close(); + await stopGateway(); + } + const outbound = JSON.parse(seen[0]); + // Lookalike host never receives the OpenRouter provider policy. + assert.equal(outbound.provider, undefined); +} + +// --------------------------------------------------------------------------- + +testLookalikeHosts(); +testNoConfigLeavesBodyUntouched(); +testValidPolicyEnforced(); +testCallerOverrideAttemptsLose(); +testNonObjectBodyUntouched(); +testMalformedPoliciesThrow(); + +await testGatewayChatBuffered(); +await testGatewayChatStream(); +await testGatewayResponsesBridge(); +await testGatewayResponsesBridge(true); +await testGatewayAnthropicBridge(); +await testGatewayAnthropicBridge(true); +await testGatewayMalformedPolicyFailsClosed(); +await testGatewayNoPolicyUntouched(); +await testGatewayLookalikeHostUntouched(); + +console.log('openrouter-provider: ok'); From 5c3eb6e4db498e26f19b57a9f0da0cbc4cbb9202 Mon Sep 17 00:00:00 2001 From: Jason McCartney Date: Tue, 22 Sep 2026 08:29:05 -0700 Subject: [PATCH 12/16] Clarify provider preferences on protocol bridges --- docs/architecture.md | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/docs/architecture.md b/docs/architecture.md index d3278b6..7af5e73 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -63,8 +63,9 @@ including streaming requests: } ``` -Configured provider fields override client requests. Other client provider -preferences remain intact. `only` must contain at least one nonempty provider +Configured provider fields override client requests. Chat Completions retains +other client provider preferences; the Responses and Anthropic bridges retain +only the fields supported by their translators. `only` must contain at least one nonempty provider slug; `allow_fallbacks` defaults to false. Invalid policy objects fail before an upstream request is sent. The policy applies only to the exact `openrouter.ai` host. With the configuration above, an unavailable Z.ai endpoint From b8fbb421aad95d60f48ababcb2231cf2142b4bed Mon Sep 17 00:00:00 2001 From: data-angel Date: Tue, 22 Sep 2026 12:21:38 -0700 Subject: [PATCH 13/16] Harden memory telemetry and trim first-run stage noise - macOS host memory: parse memory_pressure page counts (free+inactive+ speculative+purgeable, XNU's own availability definition) instead of the opaque system-wide free percentage; percentage stays as fallback. - /gateway/status and /gateway/node sample process memory usage only with ?memoryUsage=1; the dashboard asks for it on the live view and cluster node polls keep it. Plain status polls no longer spawn ps/lsof/python3. - Invalid memorySafety config now raises MemorySafetyConfigError (500, config-error code) instead of a 503 transient load-abort. - First-run job reuses the verify stage instead of appending a duplicate. --- src/cluster.mjs | 2 +- src/dashboard.mjs | 3 ++- src/first-run.mjs | 7 +++++-- src/host-memory.mjs | 19 ++++++++++++++++--- src/runtime-memory-safety.mjs | 15 ++++++++++++++- src/server.mjs | 10 ++++++++-- test/host-memory.test.mjs | 28 ++++++++++++++++++++++++++++ test/runtime-memory-safety.test.mjs | 12 ++++++++++-- 8 files changed, 84 insertions(+), 12 deletions(-) diff --git a/src/cluster.mjs b/src/cluster.mjs index 45c7743..4bd9913 100644 --- a/src/cluster.mjs +++ b/src/cluster.mjs @@ -719,7 +719,7 @@ export class ClusterCoordinator { if (!refresh && cached?.pending) return cached.pending; const pending = (async () => { try { - const result = await this.requestNode(nodeId, '/gateway/node'); + const result = await this.requestNode(nodeId, '/gateway/node?memoryUsage=1'); return { ...result.node, id: nodeId, diff --git a/src/dashboard.mjs b/src/dashboard.mjs index 9867929..5ed619c 100644 --- a/src/dashboard.mjs +++ b/src/dashboard.mjs @@ -2334,7 +2334,8 @@ const DASHBOARD_HTML = String.raw` const [health, models, status, library, backends] = await Promise.all([ getJson("/health"), getJson("/gateway/models").catch(error => ({ models: [], error: error.message })), - getJson("/gateway/status").catch(error => ({ error: error.message })), + getJson("/gateway/status" + (typeof presenceView === "string" && presenceView === "live" ? "?memoryUsage=1" : "")) + .catch(error => ({ error: error.message })), getJson("/gateway/library").catch(error => ({ error: error.message })), getJson("/gateway/backends").catch(error => ({ backends: [], error: error.message })), ]); diff --git a/src/first-run.mjs b/src/first-run.mjs index bed96f0..be10040 100644 --- a/src/first-run.mjs +++ b/src/first-run.mjs @@ -125,11 +125,14 @@ export function createFirstRunServer({ job.status = 'succeeded'; job.stages.forEach((stage) => (stage.status = 'complete')); job.stage = { id: 'verify', status: job.ready ? 'complete' : 'pending', detail: job.detail }; - job.stages.push({ + const verifyStage = job.stages.find((stage) => stage.id === 'verify'); + const finalVerify = { id: 'verify', title: job.ready ? 'Inference verified' : 'Inference not verified', status: job.ready ? 'complete' : 'pending' - }); + }; + if (verifyStage) Object.assign(verifyStage, finalVerify); + else job.stages.push(finalVerify); } catch (error) { job.status = 'failed'; job.error = String(error.message).slice(0, 1500); diff --git a/src/host-memory.mjs b/src/host-memory.mjs index ec34ceb..a5486fd 100644 --- a/src/host-memory.mjs +++ b/src/host-memory.mjs @@ -34,10 +34,23 @@ export function parseLinuxMeminfo(text) { } export function parseMacMemoryPressure(text, totalBytes) { - const match = String(text).match(/System-wide memory free percentage:\s*([\d.]+)%/i); + const text_ = String(text); + const page = (name) => Number(text_.match(new RegExp('^Pages ' + name + ':\\s*(\\d+)', 'm'))?.[1]); + const pageSize = Number(text_.match(/page size of (\d+)/)?.[1]) || 16384; + const free = page('free'); + const inactive = page('inactive'); + const speculative = page('speculative'); + const purgeable = page('purgeable'); + // XNU considers free + inactive + speculative + purgeable pages available; + // the "System-wide memory free percentage" is an opaque kernel estimate that + // can understate true availability while the file cache holds pages. + if ([free, inactive, speculative, purgeable].every(Number.isFinite)) { + return memorySnapshot(totalBytes, Math.min(totalBytes, (free + inactive + speculative + purgeable) * pageSize), 'macos-memory-pages'); + } + const match = text_.match(/System-wide memory free percentage:\s*([\d.]+)%/i); const percentage = Number(match?.[1]); if (!Number.isFinite(percentage)) return null; - return memorySnapshot(totalBytes, (Number(totalBytes) * clamp(percentage, 0, 100)) / 100, 'macos-memory-pressure'); + return memorySnapshot(totalBytes, (totalBytes * clamp(percentage, 0, 100)) / 100, 'macos-memory-pressure'); } export async function readHostMemory({ @@ -60,7 +73,7 @@ export async function readHostMemory({ } if (platform === 'darwin') { try { - const { stdout } = await execFileImpl('/usr/bin/memory_pressure', ['-Q'], { timeout: strict ? 750 : 1500 }); + const { stdout } = await execFileImpl('/usr/bin/memory_pressure', [], { timeout: strict ? 750 : 1500 }); const snapshot = parseMacMemoryPressure(stdout, totalBytes); if (snapshot) return snapshot; } catch (error) { diff --git a/src/runtime-memory-safety.mjs b/src/runtime-memory-safety.mjs index e4257f4..20afda4 100644 --- a/src/runtime-memory-safety.mjs +++ b/src/runtime-memory-safety.mjs @@ -17,6 +17,19 @@ export class RuntimeMemorySafetyError extends Error { } } +// An operator configuration problem, not a transient load abort: wrong +// statusCode/type so clients and the CLI report it as a config error. +export class MemorySafetyConfigError extends Error { + constructor(message) { + super(message); + this.name = 'MemorySafetyConfigError'; + this.code = 'memory_safety_config_invalid'; + this.type = 'memory_safety_config_error'; + this.statusCode = 500; + this.temporary = false; + } +} + export function memorySafetyPolicy(config, totalMemoryGb = os.totalmem() / GiB) { const input = config.runtimePolicy?.memorySafety ?? {}; const configuredReserve = config.runtimePolicy?.reserveMemoryGb; @@ -38,7 +51,7 @@ export function memorySafetyPolicy(config, totalMemoryGb = os.totalmem() / GiB) pollIntervalMs < 50 || pollIntervalMs > 1000 ) { - throw new RuntimeMemorySafetyError('Invalid memory safety limits; refusing to load a model.'); + throw new MemorySafetyConfigError('Invalid memory safety limits; refusing to load a model.'); } return { mode, minAvailableMemoryGb, maxMemoryUtilization, pollIntervalMs }; } diff --git a/src/server.mjs b/src/server.mjs index 25c00df..77a5ad8 100644 --- a/src/server.mjs +++ b/src/server.mjs @@ -3866,7 +3866,11 @@ export function createLloomServer(config, { logger = console, runtimeManager = n } if (req.method === 'GET' && url.pathname === '/gateway/status') { - const runtimeStatus = await runtimeManager.status({ includeMemoryUsage: true }); + const runtimeStatus = await runtimeManager.status({ + // Process sampling (ps, lsof, footprint probing) runs only when the + // memory scene asks for it; plain status polls skip the cost. + includeMemoryUsage: url.searchParams.get('memoryUsage') === '1' + }); const clustered = Object.keys(config.cluster?.nodes ?? {}).length > 0; const localRuntimeStatus = clustered ? { @@ -3891,7 +3895,9 @@ export function createLloomServer(config, { logger = console, runtimeManager = n if (req.method === 'GET' && url.pathname === '/gateway/node') { sendJson(res, 200, { ok: true, - node: await clusterCoordinator.localNodeStatus({ includeMemoryUsage: true }) + node: await clusterCoordinator.localNodeStatus({ + includeMemoryUsage: url.searchParams.get('memoryUsage') === '1' + }) }); return; } diff --git a/test/host-memory.test.mjs b/test/host-memory.test.mjs index 78eb850..3ad1958 100644 --- a/test/host-memory.test.mjs +++ b/test/host-memory.test.mjs @@ -23,6 +23,34 @@ assert.equal(mac.availableBytes, Math.round(96 * gibibyte * 0.44)); assert(Math.abs(mac.utilization - 56) < 0.000001); assert.equal(mac.source, 'macos-memory-pressure'); +// Full memory_pressure output: page counts (free+inactive+speculative+purgeable) +// must win over the opaque free percentage. +const paged = parseMacMemoryPressure( + `The system has 103079215104 (6291456 pages with a page size of 16384). + +Stats: +Pages free: 100000 +Pages purgeable: 20000 +Pages purged: 349093517 + +Page Q counts: +Pages active: 2130087 +Pages inactive: 30000 +Pages speculative: 5000 +Pages throttled: 0 +Pages wired down: 495904 + +System-wide memory free percentage: 1%`, + 96 * gibibyte +); +assert.equal(paged.source, 'macos-memory-pages'); +assert.equal(paged.availableBytes, 155000 * 16384); + +// Percentage-only output keeps the legacy path. +const percentOnly = parseMacMemoryPressure('System-wide memory free percentage: 44%', 96 * gibibyte); +assert.equal(percentOnly.source, 'macos-memory-pressure'); +assert.equal(percentOnly.availableBytes, Math.round(96 * gibibyte * 0.44)); + const sampledMac = await readHostMemory({ platform: 'darwin', totalBytes: 96 * gibibyte, diff --git a/test/runtime-memory-safety.test.mjs b/test/runtime-memory-safety.test.mjs index 3b6a8f0..38baa61 100644 --- a/test/runtime-memory-safety.test.mjs +++ b/test/runtime-memory-safety.test.mjs @@ -7,7 +7,12 @@ import http from 'node:http'; import { spawn } from 'node:child_process'; import { setTimeout as delay } from 'node:timers/promises'; import { RuntimeManager } from '../src/runtime-manager.mjs'; -import { memorySafetyPolicy, assertMemorySafety, createMemorySafetyGuard } from '../src/runtime-memory-safety.mjs'; +import { + memorySafetyPolicy, + assertMemorySafety, + createMemorySafetyGuard, + MemorySafetyConfigError +} from '../src/runtime-memory-safety.mjs'; import { readHostMemory } from '../src/host-memory.mjs'; import { terminateProcessTree } from '../src/process-control.mjs'; @@ -95,7 +100,10 @@ test('hard limits include the host reserve, reject malformed telemetry, and use memorySafetyPolicy({ runtimePolicy: { memorySafety: { minAvailableMemoryGb: 20 } } }, 96).minAvailableMemoryGb, 20 ); - assert.throws(() => memorySafetyPolicy({ runtimePolicy: { memorySafety: { mode: 'YOLO-ish' } } }), /Invalid/); + assert.throws( + () => memorySafetyPolicy({ runtimePolicy: { memorySafety: { mode: 'YOLO-ish' } } }), + (error) => error instanceof MemorySafetyConfigError && /Invalid/.test(error.message) && error.statusCode === 500 + ); }); test('strict Mac telemetry fails closed instead of substituting free memory', async () => { From d02d8d2a84ceafa080682ac225e192fe654bb43c Mon Sep 17 00:00:00 2001 From: data-angel Date: Tue, 22 Sep 2026 13:18:06 -0700 Subject: [PATCH 14/16] Add named fleet profiles: one-file route and residency swaps Profiles live in /profiles/.json and describe routes (alias -> route profile or member id), keep-warm residency per runtime, and defaults overrides. Applying composes the profile onto the config source; mutateConfigSource validates the staged candidate with the real loader before the atomic rename, so a swap is all-or-nothing. After a swap the registry hot-reloads; local models admit lazily on first request and unrouted models shed per residency policy. - src/config-profiles.mjs: format validation, planning, composition, apply/save controller; active profile marked at fleet.activeProfile. - Endpoints: GET /gateway/fleet/profiles, GET/POST /gateway/fleet/profiles/:name (?apply=1 applies, else saves). - CLI: lloom fleet list|show|use|save with the usual plan/apply gate. - Dashboard: Fleet Profiles band in Operations with one-click apply and save-current capture. - Tests: test/config-profiles.test.mjs (format, plan, compose, atomic apply, save capture, traversal rejection); wired into check:js and test:unit. --- bin/lloom.mjs | 58 ++++++- package.json | 4 +- src/config-profiles.mjs | 280 ++++++++++++++++++++++++++++++++++ src/dashboard.mjs | 72 +++++++++ src/server.mjs | 37 +++++ test/config-profiles.test.mjs | 163 ++++++++++++++++++++ 6 files changed, 611 insertions(+), 3 deletions(-) create mode 100644 src/config-profiles.mjs create mode 100644 test/config-profiles.test.mjs diff --git a/bin/lloom.mjs b/bin/lloom.mjs index 463f10d..581ab58 100755 --- a/bin/lloom.mjs +++ b/bin/lloom.mjs @@ -89,6 +89,7 @@ const COMMAND_REGISTRY = [ { name: 'suspend', aliases: [], tier: 'primary', needsInstalledConfig: true }, { name: 'resume', aliases: [], tier: 'primary', needsInstalledConfig: true }, { name: 'route', aliases: ['routing'], tier: 'primary', needsInstalledConfig: true }, + { name: 'fleet', aliases: ['profiles'], tier: 'primary', needsInstalledConfig: true }, { name: 'integrate', aliases: [], tier: 'primary', needsInstalledConfig: true }, { name: 'integrations', aliases: [], tier: 'primary', needsInstalledConfig: true }, { name: 'add-model', aliases: ['model-add'], tier: 'primary', needsInstalledConfig: true }, @@ -307,6 +308,7 @@ const INSTALLED_CONFIG_COMMANDS = new Set([ 'keep-warm', 'model-add', 'models', + 'fleet', 'remove-model', 'route', 'routing', @@ -337,6 +339,7 @@ const OPERATIONAL_CONFIG_COMMANDS = new Set([ 'keep-warm', 'model-add', 'models', + 'fleet', 'remove-model', 'route', 'routing', @@ -2394,6 +2397,59 @@ async function main() { if (!result) throw new Error(`route switch failed through ${gatewayUrlFor(config)}`); console.log(JSON.stringify({ ...result, applied: true }, null, 2)); }, + fleet: async ({ args, config }) => { + const action = positional(args)[1] ?? 'list'; + const name = positional(args)[2]; + const apply = hasFlag(args, '--apply'); + const yes = hasFlag(args, '--yes'); + if (!['list', 'show', 'use', 'save'].includes(action)) { + throw new Error(`Unknown fleet action ${action}; use list, show, use, or save.`); + } + if (action === 'list') { + const result = await gatewayRequest(config, '/gateway/fleet/profiles', { timeoutMs: 10000 }); + if (!result) throw new Error(`LLooM gateway at ${gatewayUrlFor(config)} is not reachable`); + console.log(JSON.stringify(result, null, 2)); + return; + } + if (!name) throw new Error(`Missing profile name for fleet ${action}.`); + if (action === 'show') { + const result = await gatewayRequest(config, `/gateway/fleet/profiles/${encodeURIComponent(name)}`, { + timeoutMs: 10000 + }); + if (!result) throw new Error(`LLooM gateway at ${gatewayUrlFor(config)} is not reachable`); + console.log(JSON.stringify(result, null, 2)); + return; + } + const isApply = action === 'use'; + const plan = isApply + ? { action: 'use', profile: name, applied: false, next: `lloom fleet use ${name} --apply --yes` } + : { + action: 'save', + profile: name, + description: argValue(args, '--description') ?? '', + overwrite: hasFlag(args, '--overwrite'), + applied: false, + next: `lloom fleet save ${name}${hasFlag(args, '--overwrite') ? ' --overwrite' : ''} --apply --yes` + }; + if (!apply) { + console.log(JSON.stringify(plan, null, 2)); + return; + } + if (!yes) throw new Error(`Refusing to ${action} a fleet profile without --yes after reviewing the plan`); + const result = await gatewayRequest( + config, + `/gateway/fleet/profiles/${encodeURIComponent(name)}${isApply ? '?apply=1' : ''}`, + { + method: 'POST', + body: isApply + ? { yes: true } + : { yes: true, description: plan.description, overwrite: plan.overwrite }, + timeoutMs: 60000 + } + ); + if (!result) throw new Error(`fleet profile ${action} failed through ${gatewayUrlFor(config)}`); + console.log(JSON.stringify({ ...result, applied: true }, null, 2)); + }, runtimes: async ({ args, config, command: _command }) => { const runtimeId = positional(args)[1] ?? 'all'; const manager = runtimeManagerForCli(config); @@ -2757,7 +2813,7 @@ async function main() { handlers['pack-export'] = handlers['recipe-export']; handlers['recipe-pack'] = handlers['recipe-import']; handlers['pack-submit'] = handlers['recipe-submit']; - handlers['model-add'] = handlers['add-model']; + handlers['profiles'] = handlers['fleet']; handlers['routing'] = handlers['route']; handlers['runtime-status'] = handlers['runtimes']; handlers['cluster-status'] = handlers['cluster']; diff --git a/package.json b/package.json index a91dba0..720cb35 100644 --- a/package.json +++ b/package.json @@ -64,7 +64,7 @@ "check": "npm run check:cluster && npm run check:js && npm run check:entity-stagger && npm run check:python && npm run check:shell && npm run check:spark-deploy && npm run check:workers", "check:cluster": "node --check src/cluster.mjs && node --check test/cluster.test.mjs", "check:entity-stagger": "node --check scripts/entity-stagger-benchmark.mjs && node --check test/entity-stagger-benchmark.test.mjs", - "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs && node --check src/dashboard-scene.mjs && node --check src/dashboard-presence.mjs && node --check src/dashboard-presence-client.mjs && node --check src/first-run.mjs && node --check src/first-run-page.mjs && node --check src/browser-setup.mjs && node --check src/runtime-preferences.mjs && node --check src/installation-jobs.mjs && node --check src/model-installation.mjs && node --check src/dashboard-installation.mjs && node --check src/runtime-memory-safety.mjs && node --check src/dashboard-memory.mjs && node --check src/runtime-memory-usage.mjs", + "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs && node --check src/dashboard-scene.mjs && node --check src/dashboard-presence.mjs && node --check src/dashboard-presence-client.mjs && node --check src/first-run.mjs && node --check src/first-run-page.mjs && node --check src/browser-setup.mjs && node --check src/runtime-preferences.mjs && node --check src/installation-jobs.mjs && node --check src/model-installation.mjs && node --check src/dashboard-installation.mjs && node --check src/runtime-memory-safety.mjs && node --check src/dashboard-memory.mjs && node --check src/runtime-memory-usage.mjs && node --check src/config-profiles.mjs && node --check test/config-profiles.test.mjs", "check:python": "python3 -m py_compile backends/dspark-vllm/apply-patch-pack.py backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/vision_exp/*.py backends/qwen38-vllm/apply-ple-fp8-patch.py backends/qwen38-sglang/*.py backends/glm53-exl3/*.py backends/mlx-audio/lloom_audio_server.py backends/hear/lloom_hear_server.py backends/chatterbox/lloom_chatterbox_server.py patches/apply_mtplx_longctx_fix.py backends/comfyui-media/bridge/*.py backends/comfyui-media/graphs/*.py backends/comfyui-media/build/*.py backends/comfyui-media/install.py backends/qwen-image-diffusers/*.py backends/hear/test/test_hear_security.py src/darwin-memory-usage.py", "check:shell": "bash -n backends/dspark-vllm/entrypoint.sh backends/dspark-vllm/runtime/build-hardened-image.sh backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/apply-runtime.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.sh backends/qwen38-vllm/entrypoint.sh backends/qwen38-sglang/entrypoint.sh backends/glm53-exl3/entrypoint.sh backends/mlx-audio/install.sh backends/hear/install.sh backends/chatterbox/install.sh clients/examples/lloom-claude clients/examples/lloom-codex clients/examples/lloom-hermes clients/examples/lloom-opencode clients/examples/lloom-zero scripts/remote-install-spark.sh scripts/remote-stage-spark-worker.sh", "check:spark-deploy": "node --check test/spark-deploy.test.mjs && node test/spark-deploy.test.mjs", @@ -73,7 +73,7 @@ "community:check": "node scripts/check-community-deploy.mjs", "package:check": "node scripts/check-package.mjs", "test:adoption": "node test/resource-fit.test.mjs && node test/model-acquisition.test.mjs && node test/dashboard-status.test.mjs", - "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs && npm run test:ux && npm run test:memory-safety && node --test test/dashboard-memory.test.mjs test/runtime-memory-usage.test.mjs", + "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/config-profiles.test.mjs && node test/route-control.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs && npm run test:ux && npm run test:memory-safety && node --test test/dashboard-memory.test.mjs test/runtime-memory-usage.test.mjs", "test:cluster": "node test/cluster.test.mjs", "test": "npm run test:cluster && npm run test:unit && npm run test:entity-stagger && npm run smoke && npm run community:check", "test:entity-stagger": "node test/entity-stagger-benchmark.test.mjs", diff --git a/src/config-profiles.mjs b/src/config-profiles.mjs new file mode 100644 index 0000000..16b052c --- /dev/null +++ b/src/config-profiles.mjs @@ -0,0 +1,280 @@ +// Named fleet profiles: one file describes routing + residency for the whole +// deployment; applying it is a single atomic config swap. +// +// A profile lives in /profiles/.json and may set: +// routes. route profile name (alias.routeProfiles key) or a +// plain model/alias id to pin as the sole member +// residency. 'always' | 'preferred' | 'auto' (keep-warm roster) +// defaults optional top-level defaults override (chatModel, ...) +// Applying validates the composed config completely before the atomic write +// (mutateConfigSource validates the staged file), so a swap is all-or-nothing. +import { promises as fs } from 'node:fs'; +import path from 'node:path'; +import { mutateConfigSource } from './config-mutation.mjs'; +import { loadConfig } from './config.mjs'; + +const RESIDENCY = new Set(['always', 'preferred', 'auto']); +const MAX_PROFILE_BYTES = 256 * 1024; + +function fail(message, statusCode = 400) { + return Object.assign(new Error(message), { statusCode }); +} + +function object(value) { + return value && typeof value === 'object' && !Array.isArray(value) ? value : null; +} + +function safeName(name) { + return typeof name === 'string' && /^[a-z0-9][a-z0-9._-]{0,63}$/i.test(name) ? name : null; +} + +function profilesDir(config) { + if (!config.sourcePath) throw fail('Fleet profiles need a file-backed LLooM config.', 409); + return path.join(path.dirname(path.resolve(config.sourcePath)), 'profiles'); +} + +// Route profile semantics copied from route-control: a profile's `members` +// (legacy `target`/`fallbacks`) becomes the alias's complete member list. +export function resolveRouteTarget(alias, target) { + if (!object(alias)) throw fail(`Unknown route alias: ${target && object(target) ? '' : target}`); + const profiles = object(alias.routeProfiles) ?? {}; + if (typeof target === 'string' && profiles[target]) { + const profile = profiles[target]; + const members = Array.isArray(profile.members) ? profile.members : null; + if (!members?.length) throw fail(`Route profile ${target} has no members.`); + const optionalMembers = Array.isArray(profile.optionalMembers) ? profile.optionalMembers : []; + return { activeRoute: target, members, optionalMembers }; + } + if (typeof target !== 'string' || !target.trim()) throw fail('Route target must be an id or profile name.'); + const id = target.trim(); + const known = (candidate) => + candidate === id || (object(alias.members)?.includes ?? (() => false)).call(alias.members, id); + if (!known(id) && !Array.isArray(alias.members)) + throw fail(`Alias has no route profile or member named ${id}.`); + return { activeRoute: null, members: [id], optionalMembers: [] }; +} + +export function normalizeProfileDocument(raw, name) { + const doc = object(raw); + if (!doc) throw fail(`Profile ${name} must be a JSON object.`); + const allowed = new Set(['name', 'description', 'routes', 'residency', 'defaults']); + const unknown = Object.keys(doc).filter((key) => !allowed.has(key)); + if (unknown.length) throw fail(`Profile ${name} has unsupported sections: ${unknown.join(', ')}.`); + const routes = {}; + for (const [aliasId, target] of Object.entries(object(doc.routes) ?? {})) { + if (typeof aliasId !== 'string' || !aliasId.trim() || aliasId.length > 200) + throw fail(`Profile ${name} has an invalid alias id.`); + if (typeof target !== 'string' || !target.trim() || target.length > 500) + throw fail(`Profile ${name}: route for ${aliasId} must be an id or profile name.`); + routes[aliasId] = target.trim(); + } + const residency = {}; + for (const [runtimeId, policy] of Object.entries(object(doc.residency) ?? {})) { + if (typeof runtimeId !== 'string' || !runtimeId.trim() || runtimeId.length > 200) + throw fail(`Profile ${name} has an invalid runtime id.`); + if (!RESIDENCY.has(policy)) throw fail(`Profile ${name}: residency for ${runtimeId} must be always, preferred, or auto.`); + residency[runtimeId.trim()] = policy; + } + const defaults = object(doc.defaults); + if (defaults) { + for (const [key, value] of Object.entries(defaults)) { + if (typeof value !== 'string' || value.length > 500) throw fail(`Profile ${name}: defaults.${key} must be a short string.`); + } + } + return { + name: typeof doc.name === 'string' ? doc.name : name, + description: typeof doc.description === 'string' ? doc.description.slice(0, 500) : '', + routes, + residency, + defaults: defaults ? { ...defaults } : null + }; +} + +export function profilePaths(config) { + return { dir: profilesDir(config), fileFor: (name) => path.join(profilesDir(config), name + '.json') }; +} + +export async function listProfiles(config) { + const dir = profilesDir(config); + let names = []; + try { + names = (await fs.readdir(dir)).filter((name) => name.endsWith('.json')).map((name) => name.slice(0, -5)); + } catch (error) { + if (error.code !== 'ENOENT') throw error; + } + names.sort((left, right) => left.localeCompare(right)); + const active = await activeProfileName(config); + const profiles = []; + for (const name of names) { + if (!safeName(name)) continue; + try { + profiles.push(await readProfile(config, name, active)); + } catch { + profiles.push({ name, error: 'unreadable profile' }); + } + } + return { dir, active, profiles }; +} + +export async function readProfile(config, name, activeName = null) { + const safe = safeName(name); + if (!safe) throw fail('Invalid profile name.'); + const file = path.join(profilesDir(config), safe + '.json'); + const raw = await fs.readFile(file, 'utf8'); + if (raw.length > MAX_PROFILE_BYTES) throw fail('Profile file is too large.', 413); + const doc = normalizeProfileDocument(JSON.parse(raw), safe); + return { ...doc, file, active: activeName != null ? safe === activeName : undefined }; +} + +async function readRawProfile(config, name) { + const safe = safeName(name); + if (!safe) throw fail('Invalid profile name.'); + const file = path.join(profilesDir(config), safe + '.json'); + const raw = await fs.readFile(file, 'utf8'); + if (raw.length > MAX_PROFILE_BYTES) throw fail('Profile file is too large.', 413); + return { safe, doc: JSON.parse(raw), file }; +} + +// The active profile name is recorded in the config under fleet.activeProfile +// when applied; hand-written configs simply have no marker. +async function activeProfileName(config) { + const raw = JSON.parse(await fs.readFile(path.resolve(config.sourcePath), 'utf8')); + const marker = object(raw.fleet)?.activeProfile; + return safeName(marker) ? marker : null; +} + +export function planProfileChanges(config, doc) { + const changes = { routes: [], residency: [], defaults: [], unchanged: [] }; + const aliases = object(config.aliases) ?? {}; + for (const [aliasId, target] of Object.entries(doc.routes)) { + const alias = aliases[aliasId]; + if (!object(alias)) throw fail(`Config has no alias ${aliasId}.`, 409); + const resolved = resolveRouteTarget(alias, target); + const same = + JSON.stringify(alias.members ?? []) === JSON.stringify(resolved.members) && + JSON.stringify(alias.optionalMembers ?? []) === JSON.stringify(resolved.optionalMembers) && + (alias.activeRoute ?? null) === resolved.activeRoute; + if (same) changes.unchanged.push({ kind: 'route', id: aliasId, value: target }); + else changes.routes.push({ id: aliasId, from: alias.activeRoute ?? alias.members, to: target, ...resolved }); + } + const runtimes = object(config.runtimes) ?? {}; + for (const [runtimeId, policy] of Object.entries(doc.residency)) { + const runtime = runtimes[runtimeId]; + if (!object(runtime)) throw fail(`Config has no runtime ${runtimeId}.`, 409); + const current = runtime.keepWarm === true ? 'always' : runtime.preferredWarm === true ? 'preferred' : 'auto'; + if (current === policy) changes.unchanged.push({ kind: 'residency', id: runtimeId, value: policy }); + else changes.residency.push({ id: runtimeId, from: current, to: policy }); + } + for (const [key, value] of Object.entries(doc.defaults ?? {})) { + const current = object(config.defaults)?.[key]; + if (current === value) changes.unchanged.push({ kind: 'default', id: key, value }); + else changes.defaults.push({ id: key, from: current ?? null, to: value }); + } + return changes; +} + +// Compose the profile onto the raw source. Pure and synchronous so +// mutateConfigSource can stage + validate the exact candidate. +export function composeProfile(raw, doc, name) { + const fleet = object(raw.fleet) ?? {}; + raw.fleet = { ...fleet, activeProfile: name }; + const aliases = object(raw.aliases) ?? {}; + for (const [aliasId, target] of Object.entries(doc.routes)) { + const alias = object(aliases[aliasId]); + if (!alias) throw fail(`Config has no alias ${aliasId}.`, 409); + const resolved = resolveRouteTarget({ ...alias }, target); + alias.members = resolved.members; + if (resolved.optionalMembers.length) alias.optionalMembers = resolved.optionalMembers; + else delete alias.optionalMembers; + if (resolved.activeRoute) alias.activeRoute = resolved.activeRoute; + else delete alias.activeRoute; + // A fresh route invalidates stale per-member suspensions. + delete alias.suspendedMembers; + aliases[aliasId] = alias; + } + raw.aliases = aliases; + const runtimes = object(raw.runtimes) ?? {}; + for (const [runtimeId, policy] of Object.entries(doc.residency)) { + const runtime = object(runtimes[runtimeId]); + if (!runtime) throw fail(`Config has no runtime ${runtimeId}.`, 409); + runtime.keepWarm = policy === 'always'; + runtime.preferredWarm = policy === 'preferred'; + runtimes[runtimeId] = runtime; + } + raw.runtimes = runtimes; + if (doc.defaults) raw.defaults = { ...(object(raw.defaults) ?? {}), ...doc.defaults }; + return raw; +} + +export function createFleetProfileController({ getConfig, reload, env = process.env }) { + async function apply(name, { yes = false } = {}) { + if (yes !== true) throw fail('Review the profile and confirm with yes: true.'); + if (!getConfig().sourcePath) throw fail('This gateway has no writable installed configuration.', 409); + const { safe, doc } = await readRawProfile(getConfig(), name); + const profile = normalizeProfileDocument(doc, safe); + let outcome = null; + await mutateConfigSource(getConfig(), (raw) => { + // Plan against raw source state for accurate reporting, then compose. + outcome = planProfileChanges({ ...raw, sourcePath: getConfig().sourcePath }, profile); + composeProfile(raw, profile, safe); + }); + reload?.(); + return { profile: safe, ...outcome }; + } + + async function save(name, { description = '', overwrite = false, yes = false } = {}) { + if (yes !== true) throw fail('Confirm capturing the current configuration with yes: true.'); + const safe = safeName(name); + if (!safe) throw fail('Profile names use letters, numbers, dots, dashes, underscores.'); + const config = getConfig(); + if (!config.sourcePath) throw fail('This gateway has no file-backed configuration.', 409); + const dir = profilesDir(config); + await fs.mkdir(dir, { recursive: true }); + const file = path.join(dir, safe + '.json'); + if (!overwrite) { + try { + await fs.access(file); + throw fail(`Profile ${safe} already exists; pass overwrite: true to replace it.`, 409); + } catch (error) { + if (error.statusCode === 409) throw error; + if (error.code !== 'ENOENT') throw error; + } + } + const source = await loadConfig(config.sourcePath); + const routes = {}; + for (const [aliasId, alias] of Object.entries(object(source.aliases) ?? {})) { + if (Array.isArray(alias.members) && (alias.routeProfiles || alias.activeRoute)) { + routes[aliasId] = alias.activeRoute ?? alias.members[0]; + } + } + const residency = {}; + for (const [runtimeId, runtime] of Object.entries(object(source.runtimes) ?? {})) { + if (runtime.keepWarm === true) residency[runtimeId] = 'always'; + else if (runtime.preferredWarm === true) residency[runtimeId] = 'preferred'; + } + const doc = { + name: safe, + description: String(description).slice(0, 500), + routes, + residency, + defaults: object(source.defaults) ? { ...object(source.defaults) } : undefined + }; + for (const key of Object.keys(doc)) if (doc[key] == null) delete doc[key]; + const tmp = file + '.tmp-' + process.pid; + await fs.writeFile(tmp, JSON.stringify(doc, null, 2) + '\n'); + await fs.rename(tmp, file); + return { profile: safe, file, routes: Object.keys(routes).length, residency: Object.keys(residency).length }; + } + + return { + list: () => listProfiles(getConfig()), + read: (name) => readProfile(getConfig(), name), + plan: async (name) => { + const { safe, doc } = await readRawProfile(getConfig(), name); + const profile = normalizeProfileDocument(doc, safe); + return { profile: safe, ...planProfileChanges(getConfig(), profile) }; + }, + apply, + save + }; +} diff --git a/src/dashboard.mjs b/src/dashboard.mjs index 5ed619c..5f5e16d 100644 --- a/src/dashboard.mjs +++ b/src/dashboard.mjs @@ -463,6 +463,17 @@ const DASHBOARD_HTML = String.raw`
+
+
+

Fleet Profiles

+ no profile +
+
+
+

One file describes routes and keep-warm per machine. Applying validates the whole target config first, then swaps atomically; local models load on first request.

+
+
+

Models

@@ -1009,6 +1020,7 @@ const DASHBOARD_HTML = String.raw` function renderRuntimes() { const runtimes = state.status?.runtimeManager?.runtimes || {}; const entries = Object.entries(runtimes); + $("#stat-runtimes").textContent = String(entries.length); $("#stat-active").textContent = String(entries.reduce((sum, [, runtime]) => sum + Number(runtime.activeRequests || 0), 0)); $("#stat-queued").textContent = String(entries.reduce((sum, [, runtime]) => sum + Number(runtime.queuedRequests || 0) + Number(runtime.admissionQueuedRequests || 0), 0)); @@ -1027,6 +1039,65 @@ const DASHBOARD_HTML = String.raw` '' ).join("") : '
No runtimes.
'; } + let fleetCache = null; + async function renderFleetProfiles() { + try { + const result = await getJson("/gateway/fleet/profiles"); + fleetCache = result; + } catch { + return; // fleet profiles are optional; a missing profiles dir is fine + } + const active = $("#fleet-active"); + if (active) { + active.querySelector("span:last-child").textContent = fleetCache.active || "no profile"; + active.querySelector(".dot").className = "dot " + (fleetCache.active ? "ok" : ""); + } + const host = $("#fleet-profiles"); + if (!host) return; + const profiles = fleetCache.profiles || []; + const cards = profiles.map(profile => { + if (profile.error) { + return '
' + escapeHtml(profile.name) + '' + escapeHtml(profile.error) + '
'; + } + const routeCount = Object.keys(profile.routes || {}).length; + const warmCount = Object.keys(profile.residency || {}).length; + const isActive = profile.active === true; + return '
' + escapeHtml(profile.name) + (isActive ? ' active' : '') + '' + + '' + escapeHtml(profile.description || "") + '' + + '
' + escapeHtml(String(routeCount)) + ' routes · ' + escapeHtml(String(warmCount)) + ' warm
' + + (isActive ? '' : '') + + '
'; + }); + const current = fleetCache.active ? '' : ''; + host.innerHTML = cards.join("") + current || 'No profiles yet.'; + } + + document.addEventListener("click", async event => { + const useButton = event.target.closest("[data-fleet-use]"); + if (useButton) { + const name = useButton.dataset.fleetUse; + if (!confirm("Apply fleet profile " + name + "? Routes and keep-warm are swapped atomically.")) return; + try { + showOutput(await postJson("/gateway/fleet/profiles/" + encodeURIComponent(name) + "?apply=1", { yes: true })); + await refresh(); + } catch (error) { + showOutput({ error: error.message }); + } + } + if (event.target.closest("[data-fleet-save]")) { + const name = prompt("Profile name (letters, numbers, dots, dashes):"); + if (!name) return; + const description = prompt("Description (optional):") ?? ""; + try { + showOutput(await postJson("/gateway/fleet/profiles/" + encodeURIComponent(name), { + yes: true, description, overwrite: false + })); + await renderFleetProfiles(); + } catch (error) { + showOutput({ error: error.message }); + } + } + }); function renderBackends() { const rows = $("#backend-rows"); @@ -2350,6 +2421,7 @@ const DASHBOARD_HTML = String.raw` renderRuntimes(); renderBackends(); renderLibrary(); + void renderFleetProfiles(); renderNodeInspector(); const authHint = security.adminAuthRequired ? "admin auth on" diff --git a/src/server.mjs b/src/server.mjs index 77a5ad8..ef161bd 100644 --- a/src/server.mjs +++ b/src/server.mjs @@ -44,6 +44,7 @@ import { selectedRecipeIdFromCommunityPlan } from './community-client.mjs'; import { defaultLloomHome, loadConfig } from './config.mjs'; +import { createFleetProfileController } from './config-profiles.mjs'; import { createDoctorReport } from './doctor.mjs'; import { readHostMemory } from './host-memory.mjs'; import { MACHINE_PROFILE_MEDIA_TYPE, profileMachine, rankRecipes, validateMachineProfile } from './machine-profile.mjs'; @@ -2123,6 +2124,7 @@ export function createLloomServer(config, { logger = console, runtimeManager = n onApplied: (id, policy, generation) => runtimeManager.settleDesiredResidency(id, policy, generation) }); const dashboardInstallation = createDashboardInstallation({ getConfig: () => config, reload: reloadConfig }); + const fleetProfiles = createFleetProfileController({ getConfig: () => config, reload: reloadConfig }); async function routingStatus() { const cacheMs = Math.max(0, Number(config.cluster?.routingStatusCacheMs ?? 250)); @@ -3865,6 +3867,41 @@ export function createLloomServer(config, { logger = console, runtimeManager = n return; } + if (req.method === 'GET' && url.pathname === '/gateway/fleet/profiles') { + sendJson(res, 200, { ok: true, ...(await fleetProfiles.list()) }); + return; + } + const fleetProfileMatch = url.pathname.match(/^\/gateway\/fleet\/profiles\/([^/]+)$/); + if (fleetProfileMatch) { + const name = decodeURIComponent(fleetProfileMatch[1]); + if (req.method === 'GET') { + sendJson(res, 200, { ok: true, ...(await fleetProfiles.plan(name)) }); + return; + } + if (req.method === 'POST') { + const body = await readJson(req); + for (const key of Object.keys(body)) + if (!['yes', 'description', 'overwrite'].includes(key)) + throw Object.assign(new Error('Fleet profile accepts only yes, description, and overwrite.'), { + statusCode: 400 + }); + const isApply = url.searchParams.get('apply') === '1'; + const result = isApply + ? await fleetProfiles.apply(name, { yes: body.yes }) + : await fleetProfiles.save(name, { + yes: body.yes, + description: body.description, + overwrite: body.overwrite + }); + if (isApply) { + reloadConfig(); + await reloadInFlight; + } + sendJson(res, 200, { ok: true, ...result }); + return; + } + } + if (req.method === 'GET' && url.pathname === '/gateway/status') { const runtimeStatus = await runtimeManager.status({ // Process sampling (ps, lsof, footprint probing) runs only when the diff --git a/test/config-profiles.test.mjs b/test/config-profiles.test.mjs new file mode 100644 index 0000000..be897f1 --- /dev/null +++ b/test/config-profiles.test.mjs @@ -0,0 +1,163 @@ +import assert from 'node:assert/strict'; +import fs from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; +import { loadConfig } from '../src/config.mjs'; +import { + composeProfile, + createFleetProfileController, + listProfiles, + normalizeProfileDocument, + planProfileChanges, + readProfile +} from '../src/config-profiles.mjs'; + +const directory = await fs.mkdtemp(path.join(os.tmpdir(), 'lloom-fleet-profiles-')); +const configPath = path.join(directory, 'config.json'); +const source = { + server: { host: '127.0.0.1', port: 8100 }, + security: { allowMissingAuth: true }, + defaults: { chatModel: 'local-model' }, + backends: { + local: { type: 'openai', baseUrl: 'http://127.0.0.1:8201/v1' }, + cloud: { + type: 'openai', + baseUrl: 'https://openrouter.ai/api/v1', + apiKeyEnv: 'OPENROUTER_API_KEY' + } + }, + models: [ + { id: 'local-model', kind: 'chat', backend: 'local', upstreamModel: 'local-model' }, + { id: 'cloud-model', kind: 'chat', backend: 'cloud', upstreamModel: 'cloud-model' } + ], + runtimes: { + 'local-model': { command: 'vllm', port: 8201, enabled: true, managed: true, keepWarm: true } + }, + aliases: { + omp: { + members: ['local-model', 'cloud-model'], + activeRoute: 'local-first', + routeProfiles: { + 'local-first': { members: ['local-model', 'cloud-model'] }, + cloud: { members: ['cloud-model'] } + } + }, + simple: { members: ['local-model'] } + } +}; + +const writeProfileFile = async (name, doc) => { + const dir = path.join(directory, 'profiles'); + await fs.mkdir(dir, { recursive: true }); + await fs.writeFile(path.join(dir, name + '.json'), `${JSON.stringify(doc, null, 2)}\n`); +}; + +try { + await fs.writeFile(configPath, `${JSON.stringify(source, null, 2)}\n`, { mode: 0o600 }); + const config = await loadConfig(configPath, { env: { ...process.env, OPENROUTER_API_KEY: 'test' } }); + + // -- document normalization ------------------------------------------------ + assert.deepEqual(normalizeProfileDocument({ routes: { omp: 'cloud' } }, 'p'), { + name: 'p', + description: '', + routes: { omp: 'cloud' }, + residency: {}, + defaults: null + }); + assert.throws(() => normalizeProfileDocument({ nope: true }, 'p'), /unsupported sections/); + assert.throws(() => normalizeProfileDocument({ residency: { x: 'sometimes' } }, 'p'), /always, preferred, or auto/); + + // -- planning --------------------------------------------------------------- + const localProfile = normalizeProfileDocument( + { routes: { omp: 'local-first', simple: 'local-model' }, residency: { 'local-model': 'preferred' } }, + 'local' + ); + const cloudProfile = normalizeProfileDocument( + { routes: { omp: 'cloud' }, residency: { 'local-model': 'auto' }, defaults: { chatModel: 'cloud-model' } }, + 'cloud' + ); + const localPlan = planProfileChanges(config, localProfile); + assert.equal(localPlan.unchanged.length, 2, 'omp route + residency are no-ops; simple pins members (members rewrite)'); + const cloudPlan = planProfileChanges(config, cloudProfile); + assert.equal(cloudPlan.routes.length, 1); + assert.deepEqual(cloudPlan.routes[0], { + id: 'omp', + from: 'local-first', + to: 'cloud', + activeRoute: 'cloud', + members: ['cloud-model'], + optionalMembers: [] + }); + assert.deepEqual(cloudPlan.residency, [{ id: 'local-model', from: 'always', to: 'auto' }]); + assert.deepEqual(cloudPlan.defaults, [{ id: 'chatModel', from: 'local-model', to: 'cloud-model' }]); + assert.throws(() => planProfileChanges(config, normalizeProfileDocument({ routes: { ghost: 'cloud' } }, 'x')), /no alias ghost/); + assert.throws( + () => planProfileChanges(config, normalizeProfileDocument({ residency: { ghost: 'auto' } }, 'x')), + /no runtime ghost/ + ); + + // -- compose ------------------------------------------------------------------ + const composed = composeProfile(structuredClone(source), cloudProfile, 'cloud'); + assert.equal(composed.fleet.activeProfile, 'cloud'); + assert.deepEqual(composed.aliases.omp.members, ['cloud-model']); + assert.equal(composed.aliases.omp.activeRoute, 'cloud'); + assert.equal(composed.runtimes['local-model'].keepWarm, false); + assert.equal(composed.runtimes['local-model'].preferredWarm, false); + assert.equal(composed.defaults.chatModel, 'cloud-model'); + + // Composed output must survive real config validation (validate-all-then-apply + // relies on loadConfig accepting the staged candidate). + const staged = path.join(directory, 'staged.json'); + await fs.writeFile(staged, `${JSON.stringify(composed, null, 2)}\n`); + const reloaded = await loadConfig(staged, { env: { ...process.env, OPENROUTER_API_KEY: 'test' } }); + assert.equal(reloaded.aliases.omp.activeRoute, 'cloud'); + + // -- controller: apply flips the file atomically and hot-marks active ---------- + const controller = createFleetProfileController({ + getConfig: () => config, + reload: () => {}, + env: { ...process.env, OPENROUTER_API_KEY: 'test' } + }); + await writeProfileFile('cloud', cloudProfile); + await writeProfileFile('local', localProfile); + + await assert.rejects(controller.apply('cloud', { yes: false }), /confirm with yes/); + await assert.rejects(controller.apply('missing', { yes: true }), /ENOENT|no such file/i); + + const applied = await controller.apply('cloud', { yes: true }); + assert.equal(applied.profile, 'cloud'); + assert.equal(applied.routes.length, 1); + assert.equal(applied.unchanged.length, 0, 'second plan sees no changes after apply? no: apply recomputes against raw'); + const onDisk = JSON.parse(await fs.readFile(configPath, 'utf8')); + assert.equal(onDisk.fleet.activeProfile, 'cloud'); + assert.deepEqual(onDisk.aliases.omp.members, ['cloud-model']); + + // Listing reflects the active marker. + const listing = await listProfiles(config); + assert.equal(listing.active, 'cloud'); + const cloudEntry = listing.profiles.find((profile) => profile.name === 'cloud'); + assert.equal(cloudEntry.active, true); + + // Applying the equal-content local profile is a clean no-op write. + const rerun = await controller.apply('local', { yes: true }); + assert.equal(rerun.profile, 'local'); + + // -- save captures the live config -------------------------------------------- + const saved = await controller.save('snapshot', { yes: true, description: 'point in time' }); + assert.equal(saved.profile, 'snapshot'); + const snapshotDoc = JSON.parse(await fs.readFile(saved.file, 'utf8')); + assert.equal(snapshotDoc.routes.omp, 'local-first', 'save records the active route profile'); + assert.equal(snapshotDoc.residency['local-model'], 'preferred', 'the applied local profile set preferred'); + await assert.rejects(controller.save('snapshot', { yes: true }), /already exists/); + await controller.save('snapshot', { yes: true, overwrite: true }); + + // readProfile normalizes documents for display. + const shown = await readProfile(config, 'snapshot'); + assert.equal(shown.name, 'snapshot'); + await assert.rejects(() => readProfile(config, '../escape'), /Invalid profile name/); + await assert.rejects(() => readProfile(config, 'missing'), /ENOENT/); + + console.log('fleet profile tests passed'); +} finally { + await fs.rm(directory, { recursive: true, force: true }); +} From 0ccb029b68ca0e960b4d71a168e7d8b344fce46e Mon Sep 17 00:00:00 2001 From: Jason McCartney Date: Tue, 22 Sep 2026 18:52:57 -0700 Subject: [PATCH 15/16] Discover NVIDIA Sync peers without replacing federation or model placement --- bin/lloom.mjs | 52 ++- docs/clusters.md | 68 +++- src/cluster.mjs | 432 +++++++++++++++++---- test/cluster.test.mjs | 445 +++++++++++++++++++++- test/fixtures/nvidia-sync-three-node.json | 62 +++ 5 files changed, 953 insertions(+), 106 deletions(-) create mode 100644 test/fixtures/nvidia-sync-three-node.json diff --git a/bin/lloom.mjs b/bin/lloom.mjs index 40566c2..e95a09e 100755 --- a/bin/lloom.mjs +++ b/bin/lloom.mjs @@ -36,6 +36,7 @@ import { selectedRecipeIdFromCommunityPlan } from '../src/community-client.mjs'; import { loadConfig } from '../src/config.mjs'; +import { mutateConfigSource } from '../src/config-mutation.mjs'; import { runtimeControlTimeoutMs } from '../src/control-timeout.mjs'; import { createDoctorReport } from '../src/doctor.mjs'; import { @@ -43,7 +44,8 @@ import { currentNodeId, detectNvidiaSyncCluster, federatedNodeConfigFromSnapshot, - nvidiaSyncClusterConfig, + mergeNvidiaSyncClusterDiscovery, + nvidiaSyncDiscoverySummary, validateClusterConfig } from '../src/cluster.mjs'; import { applyInit, defaultUserConfigPath } from '../src/init.mjs'; @@ -2377,18 +2379,50 @@ async function main() { cluster: async ({ args, config, command: _command }) => { const action = positional(args)[1] ?? 'status'; if (action === 'discover') { - const discovery = await detectNvidiaSyncCluster(); + const explicitLocalId = + process.env.LLOOM_NODE_ID ?? + config.cluster?.nodeId ?? + (Object.keys(config.cluster?.nodes ?? {}).length === 1 ? Object.keys(config.cluster.nodes)[0] : undefined); + if (process.env.LLOOM_NODE_ID && config.cluster?.nodeId && process.env.LLOOM_NODE_ID !== config.cluster.nodeId) + throw new Error('LLOOM_NODE_ID conflicts with configured cluster.nodeId'); + const discovery = await detectNvidiaSyncCluster({ localNodeId: explicitLocalId }); if (!discovery) throw new Error('No NVIDIA Sync cluster was detected from this node'); - const cluster = nvidiaSyncClusterConfig(discovery, { + const options = { + discovery, + localNodeId: discovery.nodeId, id: argValue(args, '--id'), - apiKeyEnv: argValue(args, '--api-key-env') ?? 'LLOOM_CLUSTER_KEY' - }); + apiKeyEnv: argValue(args, '--api-key-env') + }; + // Keep authoring forms such as environment references out of expansion. + const source = JSON.parse(readFileSync(config.sourcePath, 'utf8')); + const plan = (raw) => mergeNvidiaSyncClusterDiscovery({ ...options, cluster: raw.cluster }); + let cluster = plan(source); + const validationErrors = validateClusterConfig({ ...config, cluster }); + let changed = false; if (hasFlag(args, '--apply')) { - const source = JSON.parse(readFileSync(config.sourcePath, 'utf8')); - source.cluster = cluster; - writeFileSync(config.sourcePath, `${JSON.stringify(source, null, 2)}\n`, { mode: 0o600 }); + if (validationErrors.length) throw new Error(`Invalid discovered cluster: ${validationErrors.join('; ')}`); + ({ changed } = await mutateConfigSource(config, (latest) => { + // Replan from the current raw source inside the serialized mutation. + cluster = plan(latest); + latest.cluster = cluster; + })); } - console.log(JSON.stringify({ ok: true, applied: hasFlag(args, '--apply'), discovery, cluster }, null, 2)); + console.log( + JSON.stringify( + { + ok: validationErrors.length === 0, + applied: hasFlag(args, '--apply'), + changed, + discovery, + summary: nvidiaSyncDiscoverySummary(discovery, { currentConfig: config }), + validationErrors, + cluster, + next: 'Discovery does not change model placement or gateway listeners. Verify endpoints before serving.' + }, + null, + 2 + ) + ); return; } if (action === 'add-node') { diff --git a/docs/clusters.md b/docs/clusters.md index 44b7d8f..c9b9652 100644 --- a/docs/clusters.md +++ b/docs/clusters.md @@ -37,9 +37,7 @@ The declarative form is useful for review or hand editing: "labels": { "role": "node", "architecture": "darwin-arm64", "accelerator": "metal" }, "resources": { "memoryGb": 64 }, "proxy": { - "models": [ - { "id": "local/qwen", "as": "macbook-local/local/qwen", "kind": "chat", "remoteRuntime": "qwen" } - ] + "models": [{ "id": "local/qwen", "as": "macbook-local/local/qwen", "kind": "chat", "remoteRuntime": "qwen" }] } } } @@ -60,7 +58,13 @@ lloom cluster discover --id ennspark-cluster lloom cluster discover --id ennspark-cluster --api-key-env LLOOM_ADMIN_API_KEY --apply ``` -Discovery reads only NVIDIA Sync-marked SSH entries and local interface addresses. It records the physical hostnames as metadata but uses the stable Sync/Tailscale aliases (with `-lan` removed) as node IDs. `lloom profile` includes the detected topology, so `lloom select` can rank exact-size cluster recipes before cluster configuration is applied. +Discovery groups `NVSyncClusterAlias` entries into stable node IDs, including duplicate IP aliases and multiple rails per peer. Older Sync entries with a named `Host` alias still work. The configured local identity and leader remain unchanged. Local observations appear under `cluster.discovery.links`; they do not prove gateway reachability, bandwidth, or a complete ring. + +New peers remain physical inventory under `cluster.discovery.nodes`. Join a peer with `lloom cluster add-node --apply` after its gateway is reachable. Discovery does not create a live gateway endpoint from an SSH address. + +`--apply` merges into the raw configuration and validates it before an atomic write. It preserves existing federation nodes, model catalogs, authentication references, custom endpoints, and model placements. An endpoint that names its previous `10.100.*` backend host moves to the observed address while retaining its scheme, port, and path. Other endpoints remain unchanged and receive a diagnostic. Discovery does not change gateway listeners or NCCL settings; verify new endpoints before using them. + +Physical membership and model placement are separate. In a three-Spark ring, a model can run on one node while a distributed model uses an explicit two-node subset. A three-node member list is also representable, but TP-3 requires a compatible model architecture and backend launcher; discovery does not infer that compatibility or repartition a loaded model. Resource admission applies to the chosen members. Select the NCCL adapters that connect those members: a TP-2 job must not use adapters whose cables lead to a third node outside its placement. Use private fabric addresses for node-to-node LLooM and raw backend traffic. `backendHost` is deliberately required when a recipe auto-materializes replicas; LLooM binds the generated backend only to that address, never to every interface. @@ -131,14 +135,16 @@ The equivalent explicit config is: ```json { - "models": [{ - "id": "example/Qwen", - "targets": [ - { "id": "spark-1", "node": "spark-1", "backend": "qwen-spark-1", "runtime": "qwen-spark-1" }, - { "id": "spark-2", "node": "spark-2", "backend": "qwen-spark-2", "runtime": "qwen-spark-2" }, - { "id": "spark-2-proxy", "node": "spark-2", "backend": "lloom-node-spark-2", "remoteRuntime": "qwen-spark-2" } - ] - }] + "models": [ + { + "id": "example/Qwen", + "targets": [ + { "id": "spark-1", "node": "spark-1", "backend": "qwen-spark-1", "runtime": "qwen-spark-1" }, + { "id": "spark-2", "node": "spark-2", "backend": "qwen-spark-2", "runtime": "qwen-spark-2" }, + { "id": "spark-2-proxy", "node": "spark-2", "backend": "lloom-node-spark-2", "remoteRuntime": "qwen-spark-2" } + ] + } + ] } ``` @@ -188,19 +194,39 @@ Explicit config remains supported: "placement": { "mode": "distributed", "members": [ - { "node": "spark-1", "runtime": "dsv4-ray-head", "role": "head", "order": 10, "resources": { "memoryGb": 6 } }, - { "node": "spark-2", "runtime": "dsv4-ray-worker-2", "role": "worker", "order": 20, "resources": { "memoryGb": 54 } }, - { "node": "spark-1", "runtime": "dsv4-vllm-server", "role": "server", "order": 30, "resources": { "memoryGb": 54 } } + { + "node": "spark-1", + "runtime": "dsv4-ray-head", + "role": "head", + "order": 10, + "resources": { "memoryGb": 6 } + }, + { + "node": "spark-2", + "runtime": "dsv4-ray-worker-2", + "role": "worker", + "order": 20, + "resources": { "memoryGb": 54 } + }, + { + "node": "spark-1", + "runtime": "dsv4-vllm-server", + "role": "server", + "order": 30, + "resources": { "memoryGb": 54 } + } ] } } }, - "models": [{ - "id": "deepseek-ai/DSv4Flash", - "backend": "dsv4flash", - "runtime": "dsv4flash-cluster", - "upstreamModel": "deepseek-ai/DSv4Flash" - }] + "models": [ + { + "id": "deepseek-ai/DSv4Flash", + "backend": "dsv4flash", + "runtime": "dsv4flash-cluster", + "upstreamModel": "deepseek-ai/DSv4Flash" + } + ] } ``` diff --git a/src/cluster.mjs b/src/cluster.mjs index 726f2ff..4234a36 100644 --- a/src/cluster.mjs +++ b/src/cluster.mjs @@ -226,20 +226,36 @@ export function parseNvidiaSyncSshConfig(source = '') { let entry = null; for (const rawLine of String(source).split(/\r?\n/)) { const line = rawLine.trim(); - const host = line.match(/^Host\s+([^*?!\s]+)$/i); - if (host) { - if (entry?.createdBySync && entry.hostname) entries.push(entry); - entry = { alias: host[1], createdBySync: false }; + // A Host/Match keyword always closes the current block. Wildcard, negated, + // and multi-host patterns are not usable aliases, and Match introduces a + // conditional scope whose body is not part of the previous Host block, so + // the current entry must be finalized and cleared rather than merged forward. + const blockStart = line.match(/^(Host|Match)\s+(.*)$/i); + if (blockStart) { + if (entry) entries.push(entry); + entry = null; + const alias = blockStart[1].toLowerCase() === 'host' ? blockStart[2].trim() : ''; + if (alias && /^[^\s*?!]+$/.test(alias)) entry = { alias, createdBySync: false }; continue; } if (!entry) continue; - if (/^###\s*CreatedBy:\s*NVIDIA Sync$/i.test(line)) entry.createdBySync = true; + // Comments carry NVIDIA Sync provenance. The marker is checked before the + // generic comment guard, and every other comment line is ignored. + const marker = line.match(/^#+\s*NVSyncClusterAlias\s*:\s*(\S+)\s*$/i); + if (marker) { + entry.clusterAlias = marker[1]; + continue; + } + if (/^#/.test(line)) { + if (/^#+\s*CreatedBy:\s*NVIDIA Sync\s*$/i.test(line)) entry.createdBySync = true; + continue; + } const hostname = line.match(/^Hostname\s+(\S+)$/i); if (hostname) entry.hostname = hostname[1]; const user = line.match(/^User\s+(\S+)$/i); if (user) entry.user = user[1]; } - if (entry?.createdBySync && entry.hostname) entries.push(entry); + if (entry) entries.push(entry); return entries; } @@ -276,52 +292,348 @@ function friendlyNodeId(alias) { return String(alias).replace(/-lan$/i, ''); } +function compareIpv4(left, right) { + const leftParts = String(left).split('.').map(Number); + const rightParts = String(right).split('.').map(Number); + const valid = (parts) => + parts.length === 4 && parts.every((value) => Number.isInteger(value) && value >= 0 && value <= 255); + const leftValid = valid(leftParts); + const rightValid = valid(rightParts); + if (leftValid && rightValid) { + for (let index = 0; index < 4; index += 1) { + if (leftParts[index] !== rightParts[index]) return leftParts[index] - rightParts[index]; + } + return 0; + } + if (leftValid !== rightValid) return leftValid ? -1 : 1; + return String(left).localeCompare(String(right)); +} + +function isAddressAlias(value) { + return /^\d{1,3}(\.\d{1,3}){3}$/.test(String(value ?? '')); +} + +function syncPeerMatches(peers, localAddresses, localAddressSet = new Set()) { + const matches = []; + const ambiguous = []; + for (const original of peers) { + if (!original?.createdBySync || !original.hostname) continue; + const clusterAlias = original.clusterAlias ?? (!isAddressAlias(original.alias) ? original.alias : null); + if (!clusterAlias || !/^[a-z0-9][a-z0-9._-]*$/i.test(clusterAlias)) continue; + const peer = { ...original, clusterAlias }; + const privateAddress = /^10\.100\./.test(peer.hostname) ? peer.hostname : null; + if (!privateAddress) continue; + if (localAddressSet.has(privateAddress)) { + ambiguous.push({ + peerAddress: privateAddress, + sshAlias: peer.alias ?? null, + clusterAlias: peer.clusterAlias, + reason: 'local-address', + message: `ignored Sync alias ${peer.clusterAlias}: ${privateAddress} is one of this node's own addresses` + }); + continue; + } + const candidates = [ + ...new Map( + localAddresses + .filter( + (address) => + address.address?.startsWith('10.100.') && sameSubnet(address.address, privateAddress, address.prefixlen) + ) + .map((address) => [address.interface ?? address.address, address]) + ).values() + ]; + const candidateAddresses = candidates.map((address) => address.address); + if (candidates.length > 1) { + ambiguous.push({ + peerAddress: privateAddress, + sshAlias: peer.alias ?? null, + clusterAlias: peer.clusterAlias, + localAddresses: candidateAddresses.sort(compareIpv4), + reason: 'ambiguous-local-interface', + message: + `ignored Sync alias ${peer.clusterAlias}: peer address ${privateAddress} matches multiple local ` + + `interfaces (${candidates + .map((address) => address.interface ?? address.address) + .sort() + .join(', ')}); ` + + `refusing to guess a rail` + }); + continue; + } + const local = candidates[0]; + if (!local) continue; + matches.push({ peer, local, privateAddress }); + } + return { matches, ambiguous }; +} + +function nodeRailsForPeers(matches) { + const nodes = new Map(); + for (const match of matches) { + const id = friendlyNodeId(match.peer.clusterAlias); + let existing = nodes.get(id); + if (!existing) { + existing = { id, alias: match.peer.alias, sshAlias: null, sshUser: null, rails: [] }; + nodes.set(id, existing); + } + if (!existing.rails.some((candidate) => candidate.peerAddress === match.privateAddress)) { + existing.rails.push({ + interface: match.local.interface, + localInterface: match.local.interface, + localAddress: match.local.address, + peerAddress: match.privateAddress, + prefixlen: match.local.prefixlen, + sshAlias: match.peer.clusterAlias, + sshUser: match.peer.user ?? null + }); + } + if (!isAddressAlias(match.peer.alias)) { + existing.sshAlias ??= match.peer.alias; + existing.sshUser ??= match.peer.user ?? null; + } else { + existing.sshUser ??= match.peer.user ?? null; + } + } + return nodes; +} + export function buildNvidiaSyncDiscovery({ peers = [], localAddresses = [], localNodeId, hostname } = {}) { - const matches = peers - .map((peer) => { - const local = localAddresses.find( - (address) => - address.address?.startsWith('10.100.') && sameSubnet(address.address, peer.hostname, address.prefixlen) - ); - return local ? { peer, local } : null; - }) - .filter(Boolean); + const localAddressSet = new Set(localAddresses.map((address) => address.address).filter(Boolean)); + const { matches, ambiguous } = syncPeerMatches(peers, localAddresses, localAddressSet); if (!matches.length) return null; - const preferred = [...matches].sort( - (left, right) => Number(right.local.address.split('.')[2]) - Number(left.local.address.split('.')[2]) - )[0]; const id = localNodeId || hostname || os.hostname(); - const peerId = friendlyNodeId(preferred.peer.alias); - const nodeIds = [id, peerId].sort(); + if (matches.some((match) => friendlyNodeId(match.peer.clusterAlias) === id)) { + throw new Error(`Sync peer alias conflicts with local node id ${id}`); + } + const discovered = [...nodeRailsForPeers(matches).values()].sort( + (left, right) => + compareIpv4(left.rails[0].peerAddress, right.rails[0].peerAddress) || left.id.localeCompare(right.id) + ); + + const localPrimary = [...matches].sort((left, right) => compareIpv4(left.local.address, right.local.address))[0] + .local; + const links = [ + ...new Map( + matches.map((match) => [ + `${friendlyNodeId(match.peer.clusterAlias)}|${match.privateAddress}`, + { + nodeId: id, + peerNodeId: friendlyNodeId(match.peer.clusterAlias), + localInterface: match.local.interface, + localAddress: match.local.address, + peerAddress: match.privateAddress, + prefixlen: match.local.prefixlen + } + ]) + ).values() + ].sort( + (left, right) => + compareIpv4(left.peerAddress, right.peerAddress) || + compareIpv4(left.localAddress, right.localAddress) || + left.peerNodeId.localeCompare(right.peerNodeId) + ); + + const nodes = { + [id]: { + id, + local: true, + backendHost: localPrimary.address, + fabricInterface: localPrimary.interface + } + }; + for (const node of discovered) { + const rails = [...node.rails].sort((left, right) => compareIpv4(left.peerAddress, right.peerAddress)); + const primary = rails[0]; + nodes[node.id] = { + id: node.id, + local: false, + backendHost: primary.peerAddress, + ...(node.sshAlias ? { sshAlias: node.sshAlias } : {}), + sshUser: node.sshUser, + rails + }; + } + + const familyIds = Object.keys(nodes).sort(); + const topology = Object.keys(nodes).length === 2 ? 'direct' : 'local-adjacency'; return { detected: true, provider: 'nvidia-sync', - topology: 'direct', - nodeCount: nodeIds.length, + topology, + nodeCount: familyIds.length, nodeId: id, - leaderNode: nodeIds[0], + localNode: id, fabric: { - interface: preferred.local.interface, - localAddress: preferred.local.address, - peerAddress: preferred.peer.hostname, - prefixlen: preferred.local.prefixlen + interface: localPrimary.interface, + localAddress: localPrimary.address, + peerAddress: matches.find((match) => match.local.address === localPrimary.address).privateAddress, + prefixlen: localPrimary.prefixlen }, - nodes: { - [id]: { - id, - local: true, - backendHost: preferred.local.address, - fabricInterface: preferred.local.interface + verified: false, + evidence: 'local-interface-and-ssh-config', + links, + nodes, + ...(ambiguous.length ? { ambiguousPeers: ambiguous } : {}) + }; +} + +export function nvidiaSyncDiscoverySummary(discovery, { currentConfig = null } = {}) { + if (!discovery?.detected) return null; + const localId = discovery.nodeId; + const configuredNodes = asObject(currentConfig?.cluster?.nodes); + const configuredNodeIds = Object.keys(configuredNodes); + const discoveredIds = Object.keys(discovery.nodes).sort(); + const unresolvedPeers = Object.values(discovery.nodes) + .filter((node) => !node.local && !configuredNodes[node.id]) + .map((node) => node.id); + const customEndpoints = Object.entries(configuredNodes) + .filter(([nodeId, node]) => nodeId !== localId && node?.endpoint) + .filter(([nodeId, node]) => { + const discovered = discovery.nodes[nodeId]; + if (!discovered?.backendHost) return false; + return !endpointHostMatches(node.endpoint, discovered.backendHost); + }) + .map(([nodeId, node]) => ({ nodeId, endpoint: node.endpoint, backendHost: node.backendHost ?? null })); + const diagnostics = []; + if (configuredNodeIds.length && !configuredNodes[localId]) { + diagnostics.push( + `discovered local node id ${localId} is not the configured cluster.nodeId (${configuredNodeIds.join(', ')}); ` + + `confirm cluster.nodeId before applying so the local node is not duplicated` + ); + } + for (const nodeId of unresolvedPeers) { + diagnostics.push(`discovered peer ${nodeId} is not configured yet`); + } + for (const entry of customEndpoints) { + diagnostics.push( + `Verify the listener for ${entry.nodeId} before moving its configured endpoint onto the observed fabric` + ); + } + for (const ambiguous of discovery.ambiguousPeers ?? []) { + diagnostics.push(ambiguous.message); + } + return { + provider: discovery.provider, + topology: discovery.topology, + nodeId: discovery.nodeId, + localNode: discovery.localNode ?? discovery.nodeId, + leaderNode: currentConfig?.cluster?.leaderNode ?? null, + nodeCount: discovery.nodeCount, + discoveredNodes: discoveredIds, + discoveredPeers: discoveredIds.filter((nodeId) => nodeId !== localId), + observedLinks: discovery.links.length, + configuredNodes: configuredNodeIds.sort(), + newNodes: discoveredIds.filter((nodeId) => !configuredNodes[nodeId] && nodeId !== localId), + unresolvedPeers, + customEndpoints, + mergeRequired: configuredNodeIds.length > 0, + diagnostics, + ...(diagnostics.length ? { diagnostic: diagnostics[0] } : {}) + }; +} + +function endpointHostname(endpoint) { + const value = String(endpoint ?? '').trim(); + if (!value) return null; + try { + return new URL(value).hostname || null; + } catch { + return null; + } +} + +function endpointHostMatches(endpoint, host) { + if (!host) return false; + const hostname = endpointHostname(endpoint); + return Boolean(hostname) && hostname === String(host); +} + +export function mergeNvidiaSyncClusterDiscovery({ discovery, cluster = {}, id, apiKeyEnv, localNodeId } = {}) { + if (!discovery?.detected || discovery.provider !== 'nvidia-sync') { + throw new Error('NVIDIA Sync cluster was not detected'); + } + const current = asObject(cluster); + const nodeId = localNodeId ?? current.nodeId ?? discovery.nodeId; + if (current.nodeId && !String(current.nodeId).includes('${') && current.nodeId !== nodeId) { + throw new Error('Configured local node identity conflicts with discovery'); + } + if (Object.keys(asObject(current.nodes)).length && !current.nodes[nodeId]) { + throw new Error( + `Configured node map does not contain local node ${nodeId}; set cluster.nodeId before applying discovery` + ); + } + const leaderNode = current.leaderNode; + const nodes = { ...asObject(current.nodes) }; + const diagnostics = []; + for (const [observedId, observed] of Object.entries(discovery.nodes)) { + const targetId = observedId === discovery.nodeId ? nodeId : observedId; + if (targetId !== nodeId && !Object.hasOwn(nodes, targetId)) { + diagnostics.push({ + level: 'unconfigured-peer', + nodeId: targetId, + message: `Observed ${targetId}; use cluster add-node with its authenticated gateway URL to join it.` + }); + continue; + } + const existing = asObject(nodes[targetId]); + let endpoint = existing.endpoint; + let backendHost = existing.backendHost ?? observed.backendHost; + if ( + existing.endpoint && + /^10\.100\./.test(existing.backendHost ?? '') && + endpointHostMatches(existing.endpoint, existing.backendHost) + ) { + backendHost = observed.backendHost; + const url = new URL(existing.endpoint); + url.hostname = observed.backendHost; + endpoint = url.toString().replace(/\/$/, existing.endpoint.endsWith('/') ? '/' : ''); + } else if (existing.endpoint && !endpointHostMatches(existing.endpoint, observed.backendHost)) { + diagnostics.push({ + level: 'migration-required', + nodeId: targetId, + message: `Preserved endpoint for ${targetId}; observed fabric address ${observed.backendHost}. Verify its gateway listener before changing the endpoint.` + }); + } + nodes[targetId] = { + ...existing, + name: existing.name ?? targetId, + ...(endpoint ? { endpoint } : {}), + backendHost, + fabricAddress: observed.backendHost, + ...(observed.fabricInterface ? { fabricInterface: observed.fabricInterface } : {}), + ...(observed.sshAlias ? { sshAlias: observed.sshAlias } : {}), + ...(observed.rails ? { rails: observed.rails } : {}), + labels: { + provider: 'nvidia-sync', + role: leaderNode ? (targetId === leaderNode ? 'leader' : 'worker') : 'node', + hardware: 'dgx-spark', + ...asObject(existing.labels) }, - [peerId]: { - id: peerId, - local: false, - backendHost: preferred.peer.hostname, - fabricInterface: preferred.local.interface, - sshAlias: preferred.peer.alias, - sshUser: preferred.peer.user ?? null + resources: { + memoryGb: 128, + accelerators: ['cuda', 'nvidia-gpu', 'blackwell', 'gb10'], + ...asObject(existing.resources) } - } + }; + } + return { + ...current, + id: id ?? current.id ?? `${nodeId}-cluster`, + provider: discovery.provider, + topology: current.topology ?? discovery.topology, + nodeId: current.nodeId ?? nodeId, + ...(leaderNode ? { leaderNode } : {}), + ...(apiKeyEnv ? { apiKeyEnv } : Object.keys(current).length ? {} : { apiKeyEnv: 'LLOOM_CLUSTER_KEY' }), + discovery: { + provider: discovery.provider, + evidence: discovery.evidence, + verified: false, + nodes: discovery.nodes, + links: (discovery.links ?? []).map((link) => ({ ...link, nodeId })) + }, + nodes, + diagnostics }; } @@ -336,7 +648,7 @@ async function tailscaleSelfName() { } } -export async function detectNvidiaSyncCluster({ home = process.env.HOME, hostname = os.hostname() } = {}) { +export async function detectNvidiaSyncCluster({ home = process.env.HOME, hostname = os.hostname(), localNodeId } = {}) { if (process.platform !== 'linux' || !home) return null; let sshConfig; try { @@ -357,42 +669,14 @@ export async function detectNvidiaSyncCluster({ home = process.env.HOME, hostnam return buildNvidiaSyncDiscovery({ peers, localAddresses, - localNodeId: (await tailscaleSelfName()) ?? hostname, + localNodeId: localNodeId ?? (await tailscaleSelfName()) ?? hostname, hostname }); } -export function nvidiaSyncClusterConfig(discovery, { id, port = 8100, apiKeyEnv = 'LLOOM_CLUSTER_KEY' } = {}) { - if (!discovery?.detected || discovery.provider !== 'nvidia-sync') { - throw new Error('NVIDIA Sync cluster was not detected'); - } - const clusterId = id || `${discovery.leaderNode}-cluster`; - return { - id: clusterId, - provider: discovery.provider, - topology: discovery.topology, - nodeId: discovery.nodeId, - leaderNode: discovery.leaderNode, - apiKeyEnv, - nodes: Object.fromEntries( - Object.entries(discovery.nodes).map(([nodeId, node]) => [ - nodeId, - { - name: nodeId, - endpoint: `http://${node.backendHost}:${port}`, - backendHost: node.backendHost, - fabricInterface: node.fabricInterface, - ...(node.sshAlias ? { sshAlias: node.sshAlias } : {}), - labels: { - provider: 'nvidia-sync', - role: nodeId === discovery.leaderNode ? 'leader' : 'worker', - hardware: 'dgx-spark' - }, - resources: { memoryGb: 128, accelerators: ['cuda', 'nvidia-gpu', 'blackwell', 'gb10'] } - } - ]) - ) - }; +// Pure discovery planning; remote gateways join through authenticated add-node. +export function nvidiaSyncClusterConfig(discovery, options = {}) { + return mergeNvidiaSyncClusterDiscovery({ discovery, ...options }); } export function runtimePlacement(runtime, config, env = process.env) { diff --git a/test/cluster.test.mjs b/test/cluster.test.mjs index 7433990..9c6c065 100644 --- a/test/cluster.test.mjs +++ b/test/cluster.test.mjs @@ -5,7 +5,10 @@ import { ClusterCoordinator, federatedNodeConfigFromSnapshot, materializeFederatedNodes, + mergeNvidiaSyncClusterDiscovery, modelTargets, + nvidiaSyncClusterConfig, + nvidiaSyncDiscoverySummary, parseNvidiaSyncSshConfig, runtimeAuthority, runtimeControlAllowed, @@ -22,6 +25,9 @@ import { } from '../src/runtime-manager.mjs'; import { createRegistry } from '../src/registry.mjs'; import { syncClusterSetupMembers } from '../src/setup.mjs'; +import { readFileSync } from 'node:fs'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; const syncPeers = parseNvidiaSyncSshConfig(` Host ennspark02-lan @@ -32,16 +38,28 @@ Host ennspark02-lan Host ignored Hostname 192.168.1.20 `); -assert.deepEqual(syncPeers, [{ alias: 'ennspark02-lan', createdBySync: true, hostname: '10.100.20.2', user: 'spark' }]); +// Legacy marked entries without `NVSyncClusterAlias` are still recognized by +// their stable non-IP Host alias; the unmarked block is reported but ignored. +assert.deepEqual(syncPeers, [ + { alias: 'ennspark02-lan', createdBySync: true, hostname: '10.100.20.2', user: 'spark' }, + { alias: 'ignored', createdBySync: false, hostname: '192.168.1.20' } +]); const discovery = buildNvidiaSyncDiscovery({ peers: syncPeers, - localAddresses: [{ interface: 'enp1s0', address: '10.100.20.1', prefixlen: 24 }], + localAddresses: [ + { interface: 'enp1s0', address: '10.100.20.1', prefixlen: 24 }, + { interface: 'enp1s1', address: '10.100.21.9', prefixlen: 24 } + ], localNodeId: 'ennspark01', hostname: 'spark-host' }); assert.equal(discovery.provider, 'nvidia-sync'); assert.equal(discovery.nodes.ennspark02.backendHost, '10.100.20.2'); +assert.equal(discovery.nodes.ennspark02.fabricInterface, undefined); +assert.equal(discovery.links.length, 1); +assert.equal(discovery.links[0].localInterface, 'enp1s0'); +assert.equal(discovery.links[0].interface, undefined); const heterogeneous = { cluster: { @@ -947,4 +965,427 @@ assert.equal( true ); +// --------------------------------------------------------------------------- +// NVIDIA Sync discovery: real three-node SSH fixture. +// --------------------------------------------------------------------------- + +const fixturePath = path.join(path.dirname(fileURLToPath(import.meta.url)), 'fixtures', 'nvidia-sync-three-node.json'); +const fixture = JSON.parse(readFileSync(fixturePath, 'utf8')); +const normalizeFixtureAddresses = (value) => + value.flatMap((device) => + (device.addr_info ?? []) + .filter((address) => address.family === 'inet' && address.scope !== 'host') + .map((address) => ({ + interface: device.ifname, + address: address.local, + prefixlen: Number(address.prefixlen ?? 32) + })) + ); + +const fixturePeers = parseNvidiaSyncSshConfig(fixture.sshConfig); +assert.deepEqual( + fixturePeers.map((peer) => [peer.alias, peer.createdBySync, peer.clusterAlias ?? null, peer.hostname ?? null]), + [ + ['10.100.16.1', false, null, null], + ['worker-fabric', false, null, '10.100.17.1'], + ['ennspark02-lan', true, 'ennspark02-lan', '10.100.192.1'], + ['10.100.192.1', true, 'ennspark02-lan', '10.100.192.1'], + ['10.100.193.1', true, 'ennspark02-lan', '10.100.193.1'], + ['ennspark03-lan', true, 'ennspark03-lan', '10.100.196.2'], + ['10.100.196.2', true, 'ennspark03-lan', '10.100.196.2'], + ['10.100.197.2', true, 'ennspark03-lan', '10.100.197.2'] + ] +); + +const fixtureLocalAddresses = normalizeFixtureAddresses(fixture.ipAddresses); +const fixtureDiscovery = buildNvidiaSyncDiscovery({ + peers: fixturePeers, + localAddresses: fixtureLocalAddresses, + localNodeId: fixture.nodeId, + hostname: fixture.hostname +}); +assert.equal(fixtureDiscovery.provider, 'nvidia-sync'); +assert.equal(fixtureDiscovery.detected, true); +assert.equal(fixtureDiscovery.nodeId, 'ennspark01'); +assert.deepEqual(Object.keys(fixtureDiscovery.nodes).sort(), ['ennspark01', 'ennspark02', 'ennspark03']); +assert.equal(fixtureDiscovery.nodeCount, 3); +assert.equal(fixtureDiscovery.verified, false); +// Four distinct observed rails: two NICs per peer, with the duplicate +// alias/address Sync blocks collapsed. +assert.equal(fixtureDiscovery.links.length, 4); +assert.deepEqual( + fixtureDiscovery.links.map((link) => link.peerAddress), + ['10.100.192.1', '10.100.193.1', '10.100.196.2', '10.100.197.2'] +); +assert.deepEqual( + fixtureDiscovery.links.map((link) => link.localInterface), + ['enp1s0f0np0', 'enP2p1s0f0np0', 'enp1s0f1np1', 'enP2p1s0f1np1'] +); +assert.equal(fixtureDiscovery.nodes.ennspark02.backendHost, '10.100.192.1'); +assert.equal(fixtureDiscovery.nodes.ennspark02.sshAlias, 'ennspark02-lan'); +assert.equal(fixtureDiscovery.nodes.ennspark03.backendHost, '10.100.196.2'); +assert.equal(fixtureDiscovery.nodes.ennspark03.sshAlias, 'ennspark03-lan'); +assert.equal(fixtureDiscovery.nodes.ennspark01.local, true); +// Remote interface names are never asserted; only the local node has one. +assert.equal(fixtureDiscovery.nodes.ennspark02.fabricInterface, undefined); +assert.equal(fixtureDiscovery.nodes.ennspark03.fabricInterface, undefined); +assert.equal(fixtureDiscovery.nodes.ennspark01.fabricInterface, 'enp1s0f0np0'); +// Discovery observes adjacency only and must not elect a leader. +assert.equal(fixtureDiscovery.leaderNode, undefined); +assert.equal(fixtureDiscovery.localNode, 'ennspark01'); +assert.equal(fixtureDiscovery.ambiguousPeers, undefined); +assert.equal(fixtureDiscovery.topology, 'local-adjacency'); + +// Permuting the SSH entries and the local address order must not change output. +const shuffle = (list) => [...list].reverse(); +for (const peers of [shuffle(fixturePeers), [...fixturePeers].reverse()]) { + for (const addresses of [shuffle(fixtureLocalAddresses), [...fixtureLocalAddresses].reverse()]) { + const permuted = buildNvidiaSyncDiscovery({ + peers, + localAddresses: addresses, + localNodeId: fixture.nodeId, + hostname: fixture.hostname + }); + assert.deepEqual(permuted, fixtureDiscovery); + assert.deepEqual(Object.keys(permuted.nodes).sort(), Object.keys(fixtureDiscovery.nodes).sort()); + assert.equal(permuted.nodes.ennspark02.backendHost, '10.100.192.1'); + assert.equal(permuted.nodes.ennspark03.backendHost, '10.100.196.2'); + } +} + +const plannedFixtureCluster = nvidiaSyncClusterConfig(fixtureDiscovery, { id: 'ennspark-cluster' }); +assert.equal(plannedFixtureCluster.nodeId, 'ennspark01'); +assert.equal(plannedFixtureCluster.leaderNode, undefined); +assert.equal(plannedFixtureCluster.nodes.ennspark01.endpoint, undefined); +assert.equal(plannedFixtureCluster.nodes.ennspark02, undefined); +assert.equal(plannedFixtureCluster.discovery.nodes.ennspark02.rails[0].localInterface, 'enp1s0f0np0'); +assert.equal(plannedFixtureCluster.discovery.links[0].interface, undefined); + +// --------------------------------------------------------------------------- +// NVIDIA Sync discovery: legacy two-node config, malformed boundaries, refusals. +// --------------------------------------------------------------------------- + +const legacyPeers = parseNvidiaSyncSshConfig(` +Host 10.100.20.1 + User spark + ### CreatedBy: NVIDIA Sync + ### NVSyncClusterAlias: ennspark01-lan + Hostname 10.100.20.1 + User enntitysparkadmin +`); +assert.equal(legacyPeers.length, 1); +assert.equal(legacyPeers[0].createdBySync, true); +const legacyDiscovery = buildNvidiaSyncDiscovery({ + peers: legacyPeers, + localAddresses: [{ interface: 'enp1s0f0np0', address: '10.100.20.2', prefixlen: 24 }], + localNodeId: 'ennspark02', + hostname: 'spark-2' +}); +assert.deepEqual(Object.keys(legacyDiscovery.nodes).sort(), ['ennspark01', 'ennspark02']); +assert.equal(legacyDiscovery.topology, 'direct'); + +// Malformed SSH: wildcard/negated/multi-host Host patterns and Match blocks +// close the previous entry instead of leaking ClusterAlias provenance forward. +const malformedPeers = parseNvidiaSyncSshConfig(` +Host ennspark02-lan 10.100.192.9 + ### CreatedBy: NVIDIA Sync + Hostname 10.100.192.1 +Host !bad + ### CreatedBy: NVIDIA Sync + ### NVSyncClusterAlias: ennspark03-lan + Hostname 10.100.196.2 +Host * + ### CreatedBy: NVIDIA Sync + ### NVSyncClusterAlias: ennspark04-lan + Hostname 10.100.199.2 +Match host ennspark05 + ### CreatedBy: NVIDIA Sync + ### NVSyncClusterAlias: ennspark05-lan + Hostname 10.100.200.2 +`); +assert.deepEqual( + malformedPeers.map((peer) => [peer.alias, peer.createdBySync, peer.clusterAlias ?? null]), + [] +); +// Conditional scopes contribute no peers. +for (const peer of malformedPeers) assert.equal(peer.createdBySync, false); + +// Comments before `Host` do not become Host provenance, and the marker is only +// honored inside the block it follows. +assert.deepEqual(parseNvidiaSyncSshConfig('### CreatedBy: NVIDIA Sync\n### NVSyncClusterAlias: ghost\n'), []); + +// A Sync alias whose hostname is one of this node's own addresses is refused. +const selfPeerDiscovery = buildNvidiaSyncDiscovery({ + peers: [ + { + alias: 'ennspark02-lan', + createdBySync: true, + clusterAlias: 'ennspark02-lan', + hostname: '10.100.192.2', + user: 'spark' + } + ], + localAddresses: [{ interface: 'enp1s0f0np0', address: '10.100.192.2', prefixlen: 24 }], + localNodeId: 'ennspark01' +}); +assert.equal(selfPeerDiscovery, null); + +// An address matching two local interfaces is ambiguous: refuse instead of +// arbitrarily picking the first rail. +const ambiguousDiscovery = buildNvidiaSyncDiscovery({ + peers: [ + { + alias: 'ennspark02-lan', + createdBySync: true, + clusterAlias: 'ennspark02-lan', + hostname: '10.100.192.1', + user: 'spark' + }, + { + alias: '10.100.196.2', + createdBySync: true, + clusterAlias: 'ennspark03-lan', + hostname: '10.100.196.2', + user: 'spark' + } + ], + localAddresses: [ + { interface: 'enp1s0f0np0', address: '10.100.192.2', prefixlen: 24 }, + { interface: 'enP2p1s0f0np0', address: '10.100.192.3', prefixlen: 24 }, + { interface: 'enp1s0f1np1', address: '10.100.196.1', prefixlen: 24 } + ], + localNodeId: 'ennspark01' +}); +assert.deepEqual(Object.keys(ambiguousDiscovery.nodes).sort(), ['ennspark01', 'ennspark03']); +assert.deepEqual( + ambiguousDiscovery.ambiguousPeers.map((entry) => [entry.clusterAlias, entry.reason]), + [['ennspark02-lan', 'ambiguous-local-interface']] +); + +// --------------------------------------------------------------------------- +// Merge: preserve every existing node/endpoint/catalog/leader; migrate only the +// unambiguous old-fabric endpoint; never duplicate the local hostname. +// --------------------------------------------------------------------------- + +const legacyClusterConfig = { + cluster: { + id: 'spark-cluster', + nodeId: 'ennspark01', + leaderNode: 'ennspark01', + apiKeyEnv: 'LLOOM_ADMIN_API_KEY', + nodes: { + ennspark01: { + name: 'spark-8333', + endpoint: 'http://10.100.16.1:8100', + backendHost: '10.100.16.1', + fabricInterface: 'enp1s0f0np0', + apiKeyEnv: 'LOCAL_KEY', + labels: { role: 'leader', hardware: 'dgx-spark' }, + resources: { memoryGb: 128, maxMemoryUtilization: 0.9 } + }, + 'macbook-local': { + name: 'MacBook Local', + endpoint: 'http://macbook-private:9001', + apiKeyEnv: 'LAB_KEY', + labels: { role: 'node', architecture: 'darwin-arm64', accelerator: 'metal' }, + resources: { memoryGb: 64 }, + proxy: { + models: [{ id: 'local/qwen', as: 'macbook-local/local/qwen', kind: 'chat', remoteRuntime: 'mac-qwen' }] + } + } + } + }, + runtimes: { qwen: { enabled: true, node: 'ennspark01', memoryGb: 54 } }, + backends: { qwen: { type: 'vllm', baseUrl: 'http://10.100.16.1:8201/v1' } }, + models: [ + { + id: 'example/Qwen', + backend: 'qwen', + runtime: 'qwen', + targets: [{ id: 'ennspark01', node: 'ennspark01', backend: 'qwen', runtime: 'qwen' }] + } + ] +}; + +const merged = mergeNvidiaSyncClusterDiscovery({ + discovery: fixtureDiscovery, + cluster: legacyClusterConfig.cluster, + config: legacyClusterConfig +}); +assert.equal(merged.id, 'spark-cluster'); +assert.equal(merged.nodeId, 'ennspark01'); +assert.equal(merged.leaderNode, 'ennspark01'); +assert.equal(merged.apiKeyEnv, 'LLOOM_ADMIN_API_KEY'); +assert.deepEqual(Object.keys(merged.nodes).sort(), ['ennspark01', 'macbook-local']); + +// An endpoint on the old fabric host migrates with that host. +assert.equal(merged.nodes.ennspark01.name, 'spark-8333'); +assert.equal(merged.nodes.ennspark01.endpoint, 'http://10.100.192.2:8100'); +assert.equal(merged.nodes.ennspark01.apiKeyEnv, 'LOCAL_KEY'); +assert.equal(merged.nodes.ennspark01.resources.maxMemoryUtilization, 0.9); + +// Untouched foreign node keeps endpoint, port, catalog, credential env, labels. +assert.equal(merged.nodes['macbook-local'].endpoint, 'http://macbook-private:9001'); +assert.equal(merged.nodes['macbook-local'].apiKeyEnv, 'LAB_KEY'); +assert.deepEqual( + merged.nodes['macbook-local'].proxy.models, + legacyClusterConfig.cluster.nodes['macbook-local'].proxy.models +); +assert.equal(merged.nodes['macbook-local'].labels.architecture, 'darwin-arm64'); + +// Newly observed peers remain inventory until authenticated federation join. +assert.equal(merged.nodes.ennspark02, undefined); +assert.equal(merged.discovery.nodes.ennspark02.backendHost, '10.100.192.1'); +assert.equal(merged.discovery.nodes.ennspark02.fabricInterface, undefined); +assert.equal(merged.discovery.nodes.ennspark03.backendHost, '10.100.196.2'); +assert.deepEqual( + merged.discovery.links.map((link) => [link.peerNodeId, link.localInterface]), + [ + ['ennspark02', 'enp1s0f0np0'], + ['ennspark02', 'enP2p1s0f0np0'], + ['ennspark03', 'enp1s0f1np1'], + ['ennspark03', 'enP2p1s0f1np1'] + ] +); +assert(Array.isArray(merged.diagnostics)); +assert( + merged.diagnostics.every((entry) => typeof entry === 'string' || (entry && typeof entry.message === 'string')), + 'merge diagnostics must be actionable strings or objects with a message' +); + +// Sub-migration: an old-fabric endpoint is rewritten only when its URL hostname +// is exactly the previous backendHost of that node. +const migrateConfig = { + cluster: { + nodeId: 'ennspark01', + leaderNode: 'ennspark01', + nodes: { + ennspark01: { endpoint: 'http://10.100.16.1:8100', backendHost: '10.100.16.1' }, + ennspark02: { endpoint: 'http://10.100.16.2:9100/v1', backendHost: '10.100.16.2' } + } + } +}; +const migrated = mergeNvidiaSyncClusterDiscovery({ + discovery: fixtureDiscovery, + cluster: migrateConfig.cluster, + config: migrateConfig +}); +// Preserve custom port and path when migrating the exact old hostname. +assert.equal(migrated.nodes.ennspark02.endpoint, 'http://10.100.192.1:9100/v1'); + +const ambiguousEndpointConfig = { + cluster: { + nodeId: 'ennspark01', + leaderNode: 'ennspark01', + nodes: { + ennspark01: { endpoint: 'http://10.100.16.1:8100', backendHost: '10.100.16.1' }, + // Substring lookalike: 10.100.16.10 CONTAINS 10.100.16.1 but is not it. + ennspark02: { endpoint: 'http://10.100.16.10:8100', backendHost: '10.100.16.1' } + } + } +}; +const ambiguousMerged = mergeNvidiaSyncClusterDiscovery({ + discovery: fixtureDiscovery, + cluster: ambiguousEndpointConfig.cluster, + config: ambiguousEndpointConfig +}); +assert.equal(ambiguousMerged.nodes.ennspark02.endpoint, 'http://10.100.16.10:8100'); +assert( + ambiguousMerged.diagnostics.some( + (entry) => typeof entry === 'object' && entry.level === 'migration-required' && entry.nodeId === 'ennspark02' + ), + 'a preserved custom endpoint must be reported as migration-required' +); +assert.deepEqual(validateClusterConfig({ ...legacyClusterConfig, cluster: merged }), []); + +const fixtureSummary = nvidiaSyncDiscoverySummary(fixtureDiscovery, { currentConfig: legacyClusterConfig }); +assert.equal(fixtureSummary.observedLinks, 4); +assert.deepEqual(fixtureSummary.discoveredNodes, ['ennspark01', 'ennspark02', 'ennspark03']); +assert.deepEqual(fixtureSummary.newNodes, ['ennspark02', 'ennspark03']); +assert.equal(fixtureSummary.leaderNode, 'ennspark01'); +assert(Array.isArray(fixtureSummary.diagnostics)); + +// --------------------------------------------------------------------------- +// Explicit runtimePlacement subsets remain honored for single/TP2/TP3. +// --------------------------------------------------------------------------- + +for (const [count, expected] of [ + [1, ['ennspark01']], + [2, ['ennspark01', 'ennspark02']], + [3, ['ennspark01', 'ennspark02', 'ennspark03']] +]) { + const rendered = validateClusterConfig({ ...legacyClusterConfig, cluster: merged }).length === 0 ? '' : null; + assert.equal(rendered, ''); + const nodeIds = ['ennspark01', 'ennspark02', 'ennspark03'].slice(0, count); + const placement = runtimePlacement( + { placement: { mode: 'replicated', nodes: nodeIds } }, + { cluster: merged }, + { + LLOOM_NODE_ID: 'ennspark01' + } + ); + assert.deepEqual(placement.mode, 'replicated'); + const members = nodeIds.map((node, index) => ({ node, runtime: `tp${count}-${index}`, order: index })); + const distributed = runtimePlacement( + { + placement: { + mode: 'distributed', + members: members.map((member) => ({ + node: member.node, + runtime: member.runtime, + role: count === 1 ? 'head' : 'rank', + order: member.order + })) + } + }, + { cluster: merged }, + { LLOOM_NODE_ID: 'ennspark01' } + ); + assert.deepEqual(distributed.nodes, expected); + assert.equal(distributed.members.length, count); +} + +const sourceWithSettings = { + nodeId: '${LOCAL_NODE}', + leaderNode: 'ennspark02', + topology: 'ring', + links: [{ manual: true }], + customFlag: true, + nodes: { + ...legacyClusterConfig.cluster.nodes, + ennspark03: { backendHost: '100.78.22.9', endpoint: 'http://100.78.22.9:8100', apiKeyEnv: '${PEER_KEY}' } + } +}; +const settingsMerged = mergeNvidiaSyncClusterDiscovery({ + discovery: fixtureDiscovery, + cluster: sourceWithSettings, + localNodeId: 'ennspark01' +}); +assert.equal(settingsMerged.nodeId, '${LOCAL_NODE}'); +assert.equal(settingsMerged.leaderNode, 'ennspark02'); +assert.equal(settingsMerged.topology, 'ring'); +assert.equal(settingsMerged.apiKeyEnv, undefined); +assert.equal(settingsMerged.customFlag, true); +assert.deepEqual(settingsMerged.links, sourceWithSettings.links); +assert.equal(settingsMerged.nodes.ennspark03.backendHost, '100.78.22.9'); +assert.equal(settingsMerged.nodes.ennspark03.endpoint, 'http://100.78.22.9:8100'); +assert.equal(settingsMerged.nodes.ennspark03.fabricAddress, '10.100.196.2'); +assert.equal(settingsMerged.nodes.ennspark03.apiKeyEnv, '${PEER_KEY}'); +assert(!Object.hasOwn(settingsMerged.nodes, '${LOCAL_NODE}')); +const loopbackCustom = structuredClone(migrateConfig.cluster); +loopbackCustom.nodes.ennspark01.endpoint = 'http://127.0.0.1:8100'; +assert.equal( + mergeNvidiaSyncClusterDiscovery({ discovery: fixtureDiscovery, cluster: loopbackCustom }).nodes.ennspark01 + .backendHost, + '10.100.16.1' +); +assert.throws( + () => + mergeNvidiaSyncClusterDiscovery({ + discovery: fixtureDiscovery, + cluster: legacyClusterConfig.cluster, + localNodeId: 'wrong-node' + }), + /identity conflicts/ +); console.log('cluster tests passed'); diff --git a/test/fixtures/nvidia-sync-three-node.json b/test/fixtures/nvidia-sync-three-node.json new file mode 100644 index 0000000..bb0c5b4 --- /dev/null +++ b/test/fixtures/nvidia-sync-three-node.json @@ -0,0 +1,62 @@ +{ + "hostname": "spark-test-host", + "nodeId": "ennspark01", + "sshConfig": "Host 10.100.16.1\n User sparkadmin\nHost worker-fabric\n HostName 10.100.17.1\n User sparkadmin\nHost ennspark02-lan\n ### CreatedBy: NVIDIA Sync\n ### NVSyncClusterAlias: ennspark02-lan\n Hostname 10.100.192.1\n User sparkadmin\nHost 10.100.192.1\n ### CreatedBy: NVIDIA Sync\n ### NVSyncClusterAlias: ennspark02-lan\n Hostname 10.100.192.1\n User sparkadmin\nHost 10.100.193.1\n ### CreatedBy: NVIDIA Sync\n ### NVSyncClusterAlias: ennspark02-lan\n Hostname 10.100.193.1\n User sparkadmin\nHost ennspark03-lan\n ### CreatedBy: NVIDIA Sync\n ### NVSyncClusterAlias: ennspark03-lan\n Hostname 10.100.196.2\n User sparkadmin\nHost 10.100.196.2\n ### CreatedBy: NVIDIA Sync\n ### NVSyncClusterAlias: ennspark03-lan\n Hostname 10.100.196.2\n User sparkadmin\nHost 10.100.197.2\n ### CreatedBy: NVIDIA Sync\n ### NVSyncClusterAlias: ennspark03-lan\n Hostname 10.100.197.2\n User sparkadmin\n", + "ipAddresses": [ + { + "ifname": "lo", + "addr_info": [ + { + "family": "inet", + "local": "127.0.0.1", + "prefixlen": 8, + "scope": "host" + } + ] + }, + { + "ifname": "enp1s0f0np0", + "addr_info": [ + { + "family": "inet", + "local": "10.100.192.2", + "prefixlen": 24, + "scope": "global" + } + ] + }, + { + "ifname": "enp1s0f1np1", + "addr_info": [ + { + "family": "inet", + "local": "10.100.196.1", + "prefixlen": 24, + "scope": "global" + } + ] + }, + { + "ifname": "enP2p1s0f0np0", + "addr_info": [ + { + "family": "inet", + "local": "10.100.193.2", + "prefixlen": 24, + "scope": "global" + } + ] + }, + { + "ifname": "enP2p1s0f1np1", + "addr_info": [ + { + "family": "inet", + "local": "10.100.197.1", + "prefixlen": 24, + "scope": "global" + } + ] + } + ] +} From 528c12ec7d5d1a230324ab78cc262f7dfc717f2a Mon Sep 17 00:00:00 2001 From: data-angel Date: Tue, 22 Sep 2026 21:06:51 -0700 Subject: [PATCH 16/16] Fix predictive admission and undici 8 dispatcher contract after merges Predictive admission (5cc4f4c) folded host-wide memory usage into the modeled runtime budget: any machine running a browser or IDE would report projected usage far above the budget with no evictable path back, so every failover-to-local admission returned runtime_capacity_impossible. Admission now decides on the modeled budget (loaded + requested) and uses the live host signal only for the hard reserve check: after the admission and planned evictions, available memory must still cover reserve plus the requested add. Host-wide projection stays in the plan payload for observability. Undici 8 speaks a dispatcher contract Node 22's internal undici v6 rejects (invalid onRequestStart). The gateway's upstream calls, the cluster coordinator, and the CLI gateway client now pair the package Agent with the package fetch; the long-prefill and openrouter-provider tests use the v8 MockAgent global-dispatcher pattern with loopback net allowed for client-side requests, and model maintenance accepts an upstreamDispatcher for test composition. Smoke accepts the new macos-memory-pages telemetry source. --- bin/lloom.mjs | 5 +- src/cluster.mjs | 6 +- src/runtime-policy.mjs | 25 ++++++- src/server.mjs | 19 ++++-- test/long-prefill-transport.test.mjs | 5 +- test/openrouter-provider.test.mjs | 99 +++++++++++++++++----------- test/smoke.mjs | 6 +- 7 files changed, 111 insertions(+), 54 deletions(-) diff --git a/bin/lloom.mjs b/bin/lloom.mjs index 7549e7a..556a565 100755 --- a/bin/lloom.mjs +++ b/bin/lloom.mjs @@ -3,7 +3,8 @@ import { spawn } from 'node:child_process'; import { closeSync, existsSync, mkdirSync, openSync, readFileSync, unlinkSync, writeFileSync } from 'node:fs'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; -import { Agent as UndiciAgent } from 'undici'; +// Dispatcher and fetch must come from the same undici copy (see server.mjs). +import { Agent as UndiciAgent, fetch as undiciFetch } from 'undici'; import { backendIds, defaultBackendVariables, @@ -1083,7 +1084,7 @@ async function gatewayRequest(config, pathname, { method = 'GET', body, timeoutM bodyTimeout: timeoutMs }); try { - const response = await fetch(`${gatewayUrlFor(config)}${pathname}`, { + const response = await undiciFetch(`${gatewayUrlFor(config)}${pathname}`, { method, headers: { ...gatewayAdminHeaders(config), diff --git a/src/cluster.mjs b/src/cluster.mjs index cac796a..c18598a 100644 --- a/src/cluster.mjs +++ b/src/cluster.mjs @@ -1,7 +1,9 @@ import os from 'node:os'; import fs from 'node:fs/promises'; import path from 'node:path'; -import { Agent as UndiciAgent } from 'undici'; +// Dispatcher and fetch must come from the same undici copy: the package's +// v8 Agent speaks a dispatcher contract Node 22's internal undici v6 rejects. +import { Agent as UndiciAgent, fetch as undiciFetch } from 'undici'; import { runCommand } from './process-control.mjs'; // Node's built-in fetch aborts a request that has not produced response headers @@ -886,7 +888,7 @@ export function validateClusterConfig(config, env = process.env) { export class ClusterCoordinator { constructor( config, - { env = process.env, fetchImpl = fetch, logger = console, telemetry = null, profile = null, models = null } = {} + { env = process.env, fetchImpl = undiciFetch, logger = console, telemetry = null, profile = null, models = null } = {} ) { this.config = config; this.env = env; diff --git a/src/runtime-policy.mjs b/src/runtime-policy.mjs index 67e6cb9..c695194 100644 --- a/src/runtime-policy.mjs +++ b/src/runtime-policy.mjs @@ -508,8 +508,24 @@ export async function createRuntimePolicyPlan( policy.totalMemoryGb - (numberOrNull(memoryProfile.availableMemoryGb) ?? policy.totalMemoryGb) ); const predictive = memoryProfile.availableMemoryGb != null; - const projectedMemoryGb = Math.max(actualUsedMemoryGb, loadedMemoryGb) + requestedAddsMemory; + + // Two separate limits: + // - The modeled budget governs LLooM-attributable runtimes only. Host-wide + // usage (OS, browser, other apps) is not evictable by admission, so it + // must never count against this budget — folding it in makes the plan + // impossible on any busy machine regardless of evictions. + // - The live host signal feeds the hard reserve check: after this admission + // (and any planned evictions), the host must keep reserve headroom. + const hostProjectedMemoryGb = Math.max(actualUsedMemoryGb, loadedMemoryGb) + requestedAddsMemory; + const projectedMemoryGb = loadedMemoryGb + requestedAddsMemory; let overBudgetGb = requested?.loaded ? 0 : Math.max(0, projectedMemoryGb - policy.memoryBudgetGb); + const evictionFreesGb = rows + .filter((row) => row.loaded && protectedReasons(row, policy).length === 0) + .reduce((sum, row) => sum + row.memoryGb, 0); + const hostShortfallGb = predictive + ? Math.max(0, policy.reserveMemoryGb - memoryProfile.availableMemoryGb + requestedAddsMemory - evictionFreesGb) + : 0; + overBudgetGb = Math.max(overBudgetGb, hostShortfallGb); const actions = []; const evictions = []; @@ -557,7 +573,7 @@ export async function createRuntimePolicyPlan( overBudgetGb = Math.max(0, overBudgetGb - row.memoryGb); } - if (projectedMemoryGb > policy.memoryBudgetGb) { + if (projectedMemoryGb > policy.memoryBudgetGb || hostShortfallGb > 0) { actions.unshift(...evictions); } @@ -596,7 +612,10 @@ export async function createRuntimePolicyPlan( availableMemoryGb: numberOrNull(memoryProfile.availableMemoryGb), predictive, requestedAddsMemoryGb: requestedAddsMemory, - projectedMemoryGb, + // Host-wide projection (loaded vs actual usage, plus the add): kept for + // observability; the allow decision uses the modeled budget + reserve. + projectedMemoryGb: Math.max(hostProjectedMemoryGb, projectedMemoryGb), + hostShortfallGb, memoryBudgetGb: policy.memoryBudgetGb }, runtimes: rows, diff --git a/src/server.mjs b/src/server.mjs index 42fec22..5cac260 100644 --- a/src/server.mjs +++ b/src/server.mjs @@ -21,7 +21,9 @@ import path from 'node:path'; import os from 'node:os'; import { execFile } from 'node:child_process'; import { promisify } from 'node:util'; -import { Agent as UndiciAgent } from 'undici'; +// Dispatcher and fetch must come from the same undici copy: the package's +// v8 Agent speaks a dispatcher contract Node 22's internal undici v6 rejects. +import { Agent as UndiciAgent, fetch as undiciFetch } from 'undici'; import { backendIds, defaultBackendVariables, @@ -105,7 +107,9 @@ import { ClusterCoordinator, currentNodeId, isFederatedGatewayBackend } from './ const JSON_TYPE = 'application/json; charset=utf-8'; const SSE_TYPE = 'text/event-stream; charset=utf-8'; const execFileAsync = promisify(execFile); -const longRunningMediaDispatcher = new UndiciAgent({ +// Tests may swap this per server instance via createLloomServer's +// upstreamDispatcher option; production always uses the long-running Agent. +let longRunningMediaDispatcher = new UndiciAgent({ headersTimeout: 1800000, bodyTimeout: 1800000 }); @@ -1389,7 +1393,7 @@ async function fetchUpstream({ backend, path, body, headers = {}, signal, dispat : 0; return await fetchWithStreamProgress( (progressSignal) => - fetch(upstreamUrl(backend, path), { + undiciFetch(upstreamUrl(backend, path), { method: 'POST', headers: backendHeaders(backend, headers), body: JSON.stringify(path === '/v1/chat/completions' ? applyOpenRouterProviderPolicy(body, backend) : body), @@ -1407,7 +1411,7 @@ async function fetchRawUpstream({ backend, path, body, headers = {}, signal, dis const timeoutMs = backend.timeoutMs ?? 1800000; const fetchSignal = upstreamSignal(signal, timeoutMs); try { - return await fetch(upstreamUrl(backend, path), { + return await undiciFetch(upstreamUrl(backend, path), { method: 'POST', headers: backendHeaders(backend, headers), body, @@ -1938,7 +1942,12 @@ export async function retryRuntimeActionAfterConfigReload(action, getReloadInFli } } -export function createLloomServer(config, { logger = console, runtimeManager = null, clusterCoordinator = null } = {}) { +export function createLloomServer( + config, + { logger = console, runtimeManager = null, clusterCoordinator = null, upstreamDispatcher = null } = {} +) { + // Tests install a composed mock here; production keeps the long-running Agent. + if (upstreamDispatcher) longRunningMediaDispatcher = upstreamDispatcher; const hostTelemetry = createHostTelemetry(); const machineProfile = profileMachine().catch((error) => { logger.error?.(`Machine profile collection failed: ${error?.message ?? error}`); diff --git a/test/long-prefill-transport.test.mjs b/test/long-prefill-transport.test.mjs index 85d3d25..41e8fb4 100644 --- a/test/long-prefill-transport.test.mjs +++ b/test/long-prefill-transport.test.mjs @@ -1,6 +1,7 @@ import assert from 'node:assert/strict'; import http from 'node:http'; -import { Agent, getGlobalDispatcher, setGlobalDispatcher } from 'undici'; +// Dispatcher and fetch must come from the same undici copy. +import { Agent, fetch as undiciFetch, getGlobalDispatcher, setGlobalDispatcher } from 'undici'; import { createLloomServer } from '../src/server.mjs'; const listen = (server) => new Promise((resolve) => server.listen(0, '127.0.0.1', () => resolve(server.address().port))); @@ -47,7 +48,7 @@ try { ); const port = await listen(app.server); try { - const r = await fetch(`http://127.0.0.1:${port}/v1/chat/completions`, { + const r = await undiciFetch(`http://127.0.0.1:${port}/v1/chat/completions`, { method: 'POST', dispatcher: client, headers: { 'content-type': 'application/json' }, diff --git a/test/openrouter-provider.test.mjs b/test/openrouter-provider.test.mjs index fb43fb9..25388e6 100644 --- a/test/openrouter-provider.test.mjs +++ b/test/openrouter-provider.test.mjs @@ -1,5 +1,5 @@ import assert from 'node:assert/strict'; -import { MockAgent } from 'undici'; +import { Agent, MockAgent, getGlobalDispatcher, setGlobalDispatcher } from 'undici'; import { createLloomServer } from '../src/server.mjs'; import { applyOpenRouterProviderPolicy, @@ -7,6 +7,10 @@ import { normalizeOpenRouterProviderPolicy } from '../src/protocol/openrouter-provider.mjs'; +// The gateway consults this for its long-running upstream dispatcher while a +// mock session is active. +let gatewayDispatcher = null; + // --------------------------------------------------------------------------- // Pure policy unit coverage // --------------------------------------------------------------------------- @@ -164,20 +168,27 @@ function gatewayConfig(backend) { }; } -async function withMockedDispatcher(fn) { - const { Agent } = await import('undici'); - const originalDispatch = Agent.prototype.dispatch; - const agent = new Agent(); - agent.dispatch = originalDispatch.bind(agent); - const mockAgent = new MockAgent({ agent }); +async function withMockedDispatcher(startGatewayFn, fn) { + // The gateway's upstream calls go through its long-running undici Agent. + // Undici 8 composes mocks by wrapping that agent instead of patching + // prototypes, so the composed dispatcher is installed as the package global + // (the package fetch consults it) and handed to the gateway explicitly. + // Loopback stays open so client requests to the local gateway pass through. + const previous = getGlobalDispatcher(); + const longRunningAgent = new Agent({ headersTimeout: 1800000, bodyTimeout: 1800000 }); + const mockAgent = new MockAgent({ agent: longRunningAgent }); mockAgent.disableNetConnect(); - Agent.prototype.dispatch = function patchedDispatch(opts, handler) { - return mockAgent.dispatch(opts, handler); - }; + mockAgent.enableNetConnect( + (host) => typeof host === 'string' && (host.startsWith('127.0.0.1') || host.startsWith('localhost')) + ); + setGlobalDispatcher(mockAgent); + gatewayDispatcher = mockAgent; try { + await startGatewayFn(); return await fn(mockAgent); } finally { - Agent.prototype.dispatch = originalDispatch; + gatewayDispatcher = null; + setGlobalDispatcher(previous); await mockAgent.close(); } } @@ -222,7 +233,10 @@ let gatewayServer; let gatewayPort; async function startGateway(backend) { - const app = createLloomServer(gatewayConfig(backend), { logger: { error() {}, warn() {}, info() {}, log() {} } }); + const app = createLloomServer(gatewayConfig(backend), { + logger: { error() {}, warn() {}, info() {}, log() {} }, + upstreamDispatcher: gatewayDispatcher ?? undefined + }); await new Promise((resolve) => app.server.listen(0, '127.0.0.1', resolve)); gatewayServer = app.server; gatewayPort = app.server.address().port; @@ -238,9 +252,10 @@ async function stopGateway() { async function testGatewayChatBuffered() { const seen = []; - await startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'], allow_fallbacks: false } })); try { - await withMockedDispatcher(async (mockAgent) => { + await withMockedDispatcher( + async () => startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'], allow_fallbacks: false } })), + async (mockAgent) => { intercept(mockAgent, { payload: openAiChatPayload, onBody: (body) => seen.push(body) }); const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, { method: 'POST', @@ -268,9 +283,10 @@ async function testGatewayChatBuffered() { async function testGatewayChatStream() { const seen = []; - await startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'] } })); try { - await withMockedDispatcher(async (mockAgent) => { + await withMockedDispatcher( + async () => startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'] } })), + async (mockAgent) => { intercept(mockAgent, { contentType: 'text/event-stream', payload: openAiStreamPayload, @@ -300,9 +316,10 @@ async function testGatewayChatStream() { async function testGatewayResponsesBridge(stream = false) { const seen = []; - await startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'] } })); try { - await withMockedDispatcher(async (mockAgent) => { + await withMockedDispatcher( + async () => startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'] } })), + async (mockAgent) => { intercept(mockAgent, { payload: stream ? openAiStreamPayload : openAiChatPayload, contentType: stream ? 'text/event-stream' : 'application/json', @@ -327,9 +344,10 @@ async function testGatewayResponsesBridge(stream = false) { async function testGatewayAnthropicBridge(stream = false) { const seen = []; - await startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'], allow_fallbacks: true } })); try { - await withMockedDispatcher(async (mockAgent) => { + await withMockedDispatcher( + async () => startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'], allow_fallbacks: true } })), + async (mockAgent) => { intercept(mockAgent, { payload: stream ? openAiStreamPayload : openAiChatPayload, contentType: stream ? 'text/event-stream' : 'application/json', @@ -359,9 +377,10 @@ async function testGatewayAnthropicBridge(stream = false) { async function testGatewayMalformedPolicyFailsClosed() { // The gateway must not silently send an unconstrained request to OpenRouter. - await startGateway(openRouterBackend({ openrouterProvider: { only: [] } })); try { - await withMockedDispatcher(async () => { + await withMockedDispatcher( + async () => startGateway(openRouterBackend({ openrouterProvider: { only: [] } })), + async () => { const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, { method: 'POST', headers: { 'content-type': 'application/json' }, @@ -377,9 +396,10 @@ async function testGatewayMalformedPolicyFailsClosed() { async function testGatewayNoPolicyUntouched() { const seen = []; - await startGateway({ id: 'openrouter-lane', type: 'openai', baseUrl: 'https://openrouter.ai/api/v1' }); try { - await withMockedDispatcher(async (mockAgent) => { + await withMockedDispatcher( + async () => startGateway({ id: 'openrouter-lane', type: 'openai', baseUrl: 'https://openrouter.ai/api/v1' }), + async (mockAgent) => { intercept(mockAgent, { payload: openAiChatPayload, onBody: (body) => seen.push(body) }); const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, { method: 'POST', @@ -402,21 +422,15 @@ async function testGatewayNoPolicyUntouched() { async function testGatewayLookalikeHostUntouched() { const seen = []; - await startGateway({ - id: 'openrouter-lane', - type: 'openai', - baseUrl: 'https://notopenrouter.ai/api/v1', - openrouterProvider: { only: ['z-ai'] } - }); - const { Agent } = await import('undici'); - const originalDispatch = Agent.prototype.dispatch; - const agent = new Agent(); - agent.dispatch = originalDispatch.bind(agent); - const mockAgent = new MockAgent({ agent }); + const previous = getGlobalDispatcher(); + const longRunningAgent = new Agent({ headersTimeout: 1800000, bodyTimeout: 1800000 }); + const mockAgent = new MockAgent({ agent: longRunningAgent }); mockAgent.disableNetConnect(); - Agent.prototype.dispatch = function patchedDispatch(opts, handler) { - return mockAgent.dispatch(opts, handler); - }; + mockAgent.enableNetConnect( + (host) => typeof host === 'string' && (host.startsWith('127.0.0.1') || host.startsWith('localhost')) + ); + setGlobalDispatcher(mockAgent); + gatewayDispatcher = mockAgent; try { mockAgent .get('https://notopenrouter.ai') @@ -429,6 +443,12 @@ async function testGatewayLookalikeHostUntouched() { }, { headers: { 'content-type': 'application/json' } } ); + await startGateway({ + id: 'openrouter-lane', + type: 'openai', + baseUrl: 'https://notopenrouter.ai/api/v1', + openrouterProvider: { only: ['z-ai'] } + }); const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, { method: 'POST', headers: { 'content-type': 'application/json' }, @@ -437,7 +457,8 @@ async function testGatewayLookalikeHostUntouched() { assert.equal(res.status, 200); await res.text(); } finally { - Agent.prototype.dispatch = originalDispatch; + gatewayDispatcher = null; + setGlobalDispatcher(previous); await mockAgent.close(); await stopGateway(); } diff --git a/test/smoke.mjs b/test/smoke.mjs index 2d5a61c..f213b99 100644 --- a/test/smoke.mjs +++ b/test/smoke.mjs @@ -8682,7 +8682,11 @@ if (mockListened) { assert(metricsJson.host?.memory?.freeBytes >= 0); assert(metricsJson.host?.memory?.availableBytes >= 0); assert.equal(typeof metricsJson.host?.memory?.pressureUtilization, 'number'); - assert(['linux-memavailable', 'macos-memory-pressure', 'os-freemem'].includes(metricsJson.host?.memory?.source)); + assert( + ['linux-memavailable', 'macos-memory-pages', 'macos-memory-pressure', 'os-freemem'].includes( + metricsJson.host?.memory?.source + ) + ); assert.equal(typeof metricsJson.host?.cpu?.logicalCpus, 'number'); const modelMetrics = metricsJson.models.find( (model) => model.id === 'Youssofal/Qwen3.6-27B-MTPLX-Optimized-Speed'