diff --git a/.gitignore b/.gitignore index 4456512..8dbcd91 100644 --- a/.gitignore +++ b/.gitignore @@ -1,4 +1,4 @@ -node_modules/ +node_modules .lloom/ .DS_Store .env diff --git a/README.md b/README.md index b61fd4c..08484a1 100644 --- a/README.md +++ b/README.md @@ -3,7 +3,7 @@ [![CI](https://github.com/enntity/lloom/actions/workflows/ci.yml/badge.svg)](https://github.com/enntity/lloom/actions/workflows/ci.yml) [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE) -LLooM is a local-first LLM gateway for people who run serious open models on their own hardware. It treats NVIDIA systems—including DGX Spark / GB10—and Apple Silicon Macs as first-class platforms. LLooM sits in front of vLLM, SGLang, MLX, MTPLX, llama.cpp, Ollama, image generators, and other local runtimes, then exposes stable OpenAI-compatible and Anthropic-compatible APIs to agent tools. +LLooM installs, manages, and serves AI models on your hardware. It treats NVIDIA systems—including DGX Spark / GB10—and Apple Silicon Macs as first-class platforms. LLooM sits in front of vLLM, SGLang, MLX, MTPLX, llama.cpp, Ollama, image generators, and other local runtimes, then exposes stable OpenAI-compatible and Anthropic-compatible APIs to agent tools. The goal is simple: install one bridge, let it inspect the machine, choose the best agentic model recipe from the LLooM community library, install the backend needed for that recipe, download and configure the model, keep it warm, and point Codex, Claude Code, OMP, OpenCode, Hermes, Zero, or any OpenAI-compatible client at one base URL. @@ -15,11 +15,11 @@ The planned public community host is `https://lloom.enntity.com`; source checkou ## First-Class Platforms -| NVIDIA / DGX Spark | Apple Silicon | -| -------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------- | -| CUDA, Blackwell, DGX Spark / GB10, and Linux NVIDIA hosts | M-series Macs with unified memory | -| vLLM and SGLang are the primary high-throughput backends | MLX, MTPLX, OptiQ, and llama.cpp are the primary native backends | -| Managed Docker runtimes, GPU-memory admission, warm/on-demand lanes, and Spark recipes | Native processes, unified-memory-aware recipes, model-root reuse, and Mac recipes | +| NVIDIA / DGX Spark | Apple Silicon | +| --------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------- | +| CUDA, Blackwell, DGX Spark / GB10, and Linux NVIDIA hosts | M-series Macs with unified memory | +| vLLM and SGLang are the primary high-throughput backends | MLX, MTPLX, OptiQ, and llama.cpp are the primary native backends | +| Managed Docker runtimes, GPU-memory admission, warm/on-demand lanes, and Spark recipes | Native processes, unified-memory-aware recipes, model-root reuse, and Mac recipes | | See [`docs/dgx-spark.md`](docs/dgx-spark.md) and [`docs/clusters.md`](docs/clusters.md) | See the bundled `apple-silicon-*` recipes | Both platforms get the same gateway APIs, runtime policy, per-connection telemetry, live dashboard, client integrations, external-provider passthrough, and community recipe/benchmark workflow. Independent LLooM gateways can also form a heterogeneous lab behind one central endpoint; node profiles and optional GPU telemetry degrade cleanly across CUDA, Metal, ROCm, and CPU-only hosts. @@ -34,7 +34,6 @@ cd lloom npm ci npm link lloom -lloom up --go ``` `npm link` installs the same `lloom` and `lloom-host` commands from your checkout. After the first npm release, `npm install -g lloom` will be the supported package install path. @@ -48,7 +47,9 @@ curl -sS http://127.0.0.1:8100/v1/models Dashboard: [http://127.0.0.1:8100/](http://127.0.0.1:8100/) -That is the 1.0 path. A bare `lloom` is a dry run first: it inspects the machine, asks the LLooM community host for the best known recipe pack and backend catalog, shows what will be installed, and refuses writes until you rerun it with `--go`. `up` is the named alias for the same first-run flow. `--go` applies the plan, confirms noninteractive writes, and starts the selected keep-warm runtime after setup. Use `--offline` when you want to ignore the host and select from only the local recipe library. +On a new installation in an interactive terminal, `lloom` opens a local browser setup. It detects your hardware, asks what you want to use AI for, and recommends a compatible vendor recipe from the bundled library. Review the plan and choose **Set up my AI** to install it. The terminal stays open during setup. Chat setup checks a real response through the gateway before showing **Ready**; media installations show when output still needs verification. + +Use `lloom up --browser` to request browser setup explicitly, or `lloom ui` to open an installed gateway. The CLI remains available: `lloom --no-browser` previews the community-based plan; `lloom up --go` installs, integrates, and starts it. Scripted, JSON, offline, and explicit recipe commands keep their CLI behavior. See [the browser experience](docs/browser-experience.md) for the flow and current boundaries. The default gateway endpoint is `127.0.0.1:8100`; managed backend runtimes default to `8201-8299`. This source checkout defaults community lookup to the local development host at `127.0.0.1:8110`, starts it automatically if it is not already running, serves signed seed host data from `community/`, and requires signed recipe packs by default. Local imports still land in `recipes/` and `benchmarks/community/`. A production package should point at the signed public LLooM host. Most users should not need to care. @@ -60,12 +61,12 @@ Use this path when validating the repository before a package release: ```bash npm install -g . -lloom up +lloom up --no-browser lloom up --go lloom doctor --no-runtimes ``` -`lloom up` should show the detected machine profile, the trusted community recommendation, the selected recipe, benchmark evidence, and the exact apply command. +`lloom up --no-browser` should show the detected machine profile, the trusted community recommendation, the selected recipe, benchmark evidence, and the exact apply command. - On NVIDIA Linux, LLooM detects CUDA devices, compute capability, Blackwell, and DGX Spark / GB10 markers. Spark recipes use vLLM or SGLang, managed Docker containers, and explicit GPU-memory/runtime policy. The checked-in Spark deployment demonstrates a warm primary chat model, warm embedding model, and an on-demand alternate chat lane. - On a 96 GB Apple Silicon machine, the bundled development host should recommend `apple-silicon-qwen36-35b-a3b-mtplx` and select `Youssofal/Qwen3.6-35B-A3B-MTPLX-Optimized-Speed-FP16`. Lower-memory Macs should fall back to the 27B MTPLX recipe. @@ -99,14 +100,16 @@ Then open OMP normally. The generated OMP config points at `http://127.0.0.1:810 - Backend recipes for vLLM, SGLang, MTPLX, MLX LM, llama.cpp, Ollama, OptiQ, and stable-diffusion.cpp, with dedicated DGX Spark / GB10 and Apple Silicon recipes. - Community recipe packs and hardware-matched benchmark evidence so machines can select the best known model/backend recipe automatically instead of blindly chasing global tok/s. - Generated client profiles for OMP, OpenCode, Codex-compatible, Claude-compatible, Hermes, Zero, and any OpenAI-compatible client. -- A small dashboard at `/` for local status and guarded setup actions, with a live topology that can switch between the default columnar racks and an action view whose camera and cards follow live models. +- A browser dashboard with Live, Models, Machines, Clients, and Settings. Inspect real topology and memory blocks, preview a model’s expected footprint by pointing or focusing, and use models directly through chat or connected apps. LLooM prepares models automatically; optional readiness and memory controls live under Options & details. The action camera follows serving models while preserving manual zoom. ## Daily Commands Primary ladder (see `lloom help`; full catalog under `lloom help advanced`): ```bash -lloom # preview plan +lloom # browser setup on first interactive run +lloom ui # open the installed gateway +lloom --no-browser # preview the CLI plan lloom up --go # install + integrate + start lloom down # stop the gateway and all managed model backends lloom doctor --no-runtimes @@ -123,7 +126,7 @@ lloom add-model 'openai:http://127.0.0.1:8000/v1#my-model' --default --apply --y lloom serve --config ~/.lloom/config.json ``` -Bare `lloom`, `up`, and `onboard` all route to the same first-run flow. By default, the community request asks for the best known `agentic-coding` recipe with `tools`, `reasoning`, and `long-context`; use repeated `--workload`, `--capability`, or `--tag` flags to target a different kind of local model. `doctor` is the readiness view for humans and automation. `integrate` repairs or writes client configs from the registry. `add-model` imports an ad hoc Hugging Face, local, or Ollama model outside the community recipe library. +The CLI `onboard` flow remains available independently of browser setup. By default, the community request asks for the best known `agentic-coding` recipe with `tools`, `reasoning`, and `long-context`; use repeated `--workload`, `--capability`, or `--tag` flags to target a different kind of local model. `doctor` is the readiness view for humans and automation. `integrate` repairs or writes client configs from the registry. `add-model` imports an ad hoc Hugging Face, local, or Ollama model outside the community recipe library. After `~/.lloom/config.json` exists, operational commands such as `doctor`, `models`, `serve`, `integrate`, `add-model`, and runtime controls automatically read that installed config when `--config` is not supplied. Before that file exists, those commands return a `not-installed` report with the exact `lloom up` command to run, instead of silently operating on bundled model defaults. Read-only planning commands such as `lloom`, `lloom up`, `onboard`, and `integrations` can still preview from the packaged gateway shell plus community data. Use `--config` whenever you want to inspect or operate a different config file. diff --git a/SECURITY.md b/SECURITY.md index 92dbe80..e35b57c 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -24,8 +24,11 @@ The maintainers will acknowledge reports as soon as practical, validate the issu - A signature proves which key signed a recipe pack. Trusting keys served by the same community host is equivalent to trusting that host and its TLS connection; use explicit local trusted keys for stronger publisher pinning. - Development keys and the checked-in public seed key are not production trust roots. - Model weights and external runtimes have their own licenses and security posture. LLooM does not make untrusted model code safe. +- Browser management requests must come from the gateway's own origin. First-run setup additionally requires a local session token and applies only reviewed plans. Keep the setup terminal and tokens private. See [the browser security boundary](docs/browser-experience.md#local-security). - The public community MVP is read-only by design. Proposals arrive through reviewed pull requests; anonymous recipe and benchmark uploads are disabled in production. - Remote community feeds must use HTTPS and a locally pinned public signing key. Do not treat a signing key downloaded from the same remote host as an independent trust root. - The production community deployment is isolated from inference, databases, and Docker control-plane access. See [`deploy/community/README.md`](deploy/community/README.md). See [docs/architecture.md](docs/architecture.md) for route authorization and network-binding defaults. + +Remote admin writes require `security.allowRemoteAdmin=true` and a key in `security.adminApiKeys`. Inference credentials alone do not grant remote process or installation control. diff --git a/bin/lloom.mjs b/bin/lloom.mjs index 40566c2..556a565 100755 --- a/bin/lloom.mjs +++ b/bin/lloom.mjs @@ -3,7 +3,8 @@ import { spawn } from 'node:child_process'; import { closeSync, existsSync, mkdirSync, openSync, readFileSync, unlinkSync, writeFileSync } from 'node:fs'; import path from 'node:path'; import { fileURLToPath } from 'node:url'; -import { Agent as UndiciAgent } from 'undici'; +// Dispatcher and fetch must come from the same undici copy (see server.mjs). +import { Agent as UndiciAgent, fetch as undiciFetch } from 'undici'; import { backendIds, defaultBackendVariables, @@ -36,14 +37,18 @@ import { selectedRecipeIdFromCommunityPlan } from '../src/community-client.mjs'; import { loadConfig } from '../src/config.mjs'; +import { mutateConfigSource } from '../src/config-mutation.mjs'; import { runtimeControlTimeoutMs } from '../src/control-timeout.mjs'; import { createDoctorReport } from '../src/doctor.mjs'; +import { createBrowserSetup } from '../src/browser-setup.mjs'; +import { shouldOpenSetup, openLocalBrowser } from '../src/first-run.mjs'; import { ClusterCoordinator, currentNodeId, detectNvidiaSyncCluster, federatedNodeConfigFromSnapshot, - nvidiaSyncClusterConfig, + mergeNvidiaSyncClusterDiscovery, + nvidiaSyncDiscoverySummary, validateClusterConfig } from '../src/cluster.mjs'; import { applyInit, defaultUserConfigPath } from '../src/init.mjs'; @@ -78,6 +83,7 @@ const __filename = fileURLToPath(import.meta.url); /** Command tiers used for help + installed-config policy. Aliases resolve before dispatch. */ const COMMAND_REGISTRY = [ { name: 'up', aliases: [], tier: 'primary', needsInstalledConfig: false }, + { name: 'ui', aliases: [], tier: 'primary', needsInstalledConfig: true }, { name: 'onboard', aliases: [], tier: 'advanced', needsInstalledConfig: false }, { name: 'down', aliases: [], tier: 'primary', needsInstalledConfig: true }, { name: 'doctor', aliases: [], tier: 'primary', needsInstalledConfig: true }, @@ -86,6 +92,7 @@ const COMMAND_REGISTRY = [ { name: 'suspend', aliases: [], tier: 'primary', needsInstalledConfig: true }, { name: 'resume', aliases: [], tier: 'primary', needsInstalledConfig: true }, { name: 'route', aliases: ['routing'], tier: 'primary', needsInstalledConfig: true }, + { name: 'fleet', aliases: ['profiles'], tier: 'primary', needsInstalledConfig: true }, { name: 'integrate', aliases: [], tier: 'primary', needsInstalledConfig: true }, { name: 'integrations', aliases: [], tier: 'primary', needsInstalledConfig: true }, { name: 'add-model', aliases: ['model-add'], tier: 'primary', needsInstalledConfig: true }, @@ -147,8 +154,11 @@ function usage() { return `LLooM — local-first LLM gateway Primary commands: - lloom / lloom up Start installed LLooM; preview setup on first run + lloom / lloom up Start installed LLooM; guided setup on first run lloom up --go First run: install, integrate, and start the model + lloom up --browser Open guided local setup on first run + lloom up --no-browser Keep setup in the terminal + lloom ui Open the installed gateway dashboard lloom down Stop the gateway and all managed model backends lloom doctor Readiness report (blockers, warnings, next actions) lloom serve Run the gateway (reads ~/.lloom/config.json) @@ -301,6 +311,7 @@ const INSTALLED_CONFIG_COMMANDS = new Set([ 'keep-warm', 'model-add', 'models', + 'fleet', 'remove-model', 'route', 'routing', @@ -331,6 +342,7 @@ const OPERATIONAL_CONFIG_COMMANDS = new Set([ 'keep-warm', 'model-add', 'models', + 'fleet', 'remove-model', 'route', 'routing', @@ -1072,7 +1084,7 @@ async function gatewayRequest(config, pathname, { method = 'GET', body, timeoutM bodyTimeout: timeoutMs }); try { - const response = await fetch(`${gatewayUrlFor(config)}${pathname}`, { + const response = await undiciFetch(`${gatewayUrlFor(config)}${pathname}`, { method, headers: { ...gatewayAdminHeaders(config), @@ -1118,15 +1130,22 @@ function gatewayProcessPaths(configPath) { }; } -async function startGatewayBackground(configPath) { +async function startGatewayBackground(configPath, { requireOwned = false } = {}) { const gatewayConfig = await loadConfig(configPath); const url = gatewayUrlFor(gatewayConfig); const alreadyHealthy = await gatewayHealth(gatewayConfig); const paths = gatewayProcessPaths(configPath); if (alreadyHealthy?.ok) { + const ownedPid = + requireOwned && existsSync(paths.pidPath) ? Number(readFileSync(paths.pidPath, 'utf8').trim()) : null; + if (requireOwned && (!ownedPid || alreadyHealthy.pid !== ownedPid)) + throw new Error( + 'The gateway port is already used by another server. Choose another port or stop that server before retrying setup.' + ); return { status: 'already-running', url, + pid: alreadyHealthy.pid, health: alreadyHealthy, ...paths }; @@ -1158,6 +1177,8 @@ async function startGatewayBackground(configPath) { ...paths }; } + if (requireOwned && health.pid !== child.pid) + throw new Error('Another server answered on the gateway port. The installed gateway has not been verified.'); return { status: 'started', url, @@ -1448,6 +1469,24 @@ async function main() { }, up: async (context) => { const { args, configPath } = context; + if ( + shouldOpenSetup(args, { + installed: Boolean(configPath), + interactive: Boolean(process.stdin.isTTY && process.stdout.isTTY) + }) && + (!wantsOnboardingPlan(args) || hasFlag(args, '--browser')) + ) { + const options = await firstRunCliOptions([...args, '--offline']); + const setup = createBrowserSetup(context.config, options, { + gatewayStarter: (configPath) => startGatewayBackground(configPath, { requireOwned: true }) + }); + await setup.listen(); + const url = setup.bootstrapUrl(); + console.log('LLooM setup: ' + url + '\nKeep this terminal open while setup runs. Press Ctrl+C to stop.'); + const opened = await openLocalBrowser(url); + if (!opened) console.log('Open the local setup URL above in your browser.'); + return; + } if (!configPath || wantsOnboardingPlan(args)) return handlers.onboard(context); const gateway = await startGatewayBackground(configPath); const report = { @@ -1473,6 +1512,16 @@ async function main() { } if (!report.ok) process.exitCode = 1; }, + ui: async ({ args, config }) => { + const url = gatewayUrlFor(config); + if (wantsJson(args)) { + console.log(JSON.stringify({ url })); + return; + } + console.log('LLooM: ' + url); + if (!hasFlag(args, '--no-browser') && !(await openLocalBrowser(url))) + console.log('Open the gateway URL above in your browser.'); + }, onboard: async ({ args, config, command: _command }) => { const go = wantsGo(args); const apply = hasFlag(args, '--apply') || go; @@ -2351,6 +2400,59 @@ async function main() { if (!result) throw new Error(`route switch failed through ${gatewayUrlFor(config)}`); console.log(JSON.stringify({ ...result, applied: true }, null, 2)); }, + fleet: async ({ args, config }) => { + const action = positional(args)[1] ?? 'list'; + const name = positional(args)[2]; + const apply = hasFlag(args, '--apply'); + const yes = hasFlag(args, '--yes'); + if (!['list', 'show', 'use', 'save'].includes(action)) { + throw new Error(`Unknown fleet action ${action}; use list, show, use, or save.`); + } + if (action === 'list') { + const result = await gatewayRequest(config, '/gateway/fleet/profiles', { timeoutMs: 10000 }); + if (!result) throw new Error(`LLooM gateway at ${gatewayUrlFor(config)} is not reachable`); + console.log(JSON.stringify(result, null, 2)); + return; + } + if (!name) throw new Error(`Missing profile name for fleet ${action}.`); + if (action === 'show') { + const result = await gatewayRequest(config, `/gateway/fleet/profiles/${encodeURIComponent(name)}`, { + timeoutMs: 10000 + }); + if (!result) throw new Error(`LLooM gateway at ${gatewayUrlFor(config)} is not reachable`); + console.log(JSON.stringify(result, null, 2)); + return; + } + const isApply = action === 'use'; + const plan = isApply + ? { action: 'use', profile: name, applied: false, next: `lloom fleet use ${name} --apply --yes` } + : { + action: 'save', + profile: name, + description: argValue(args, '--description') ?? '', + overwrite: hasFlag(args, '--overwrite'), + applied: false, + next: `lloom fleet save ${name}${hasFlag(args, '--overwrite') ? ' --overwrite' : ''} --apply --yes` + }; + if (!apply) { + console.log(JSON.stringify(plan, null, 2)); + return; + } + if (!yes) throw new Error(`Refusing to ${action} a fleet profile without --yes after reviewing the plan`); + const result = await gatewayRequest( + config, + `/gateway/fleet/profiles/${encodeURIComponent(name)}${isApply ? '?apply=1' : ''}`, + { + method: 'POST', + body: isApply + ? { yes: true } + : { yes: true, description: plan.description, overwrite: plan.overwrite }, + timeoutMs: 60000 + } + ); + if (!result) throw new Error(`fleet profile ${action} failed through ${gatewayUrlFor(config)}`); + console.log(JSON.stringify({ ...result, applied: true }, null, 2)); + }, runtimes: async ({ args, config, command: _command }) => { const runtimeId = positional(args)[1] ?? 'all'; const manager = runtimeManagerForCli(config); @@ -2377,18 +2479,50 @@ async function main() { cluster: async ({ args, config, command: _command }) => { const action = positional(args)[1] ?? 'status'; if (action === 'discover') { - const discovery = await detectNvidiaSyncCluster(); + const explicitLocalId = + process.env.LLOOM_NODE_ID ?? + config.cluster?.nodeId ?? + (Object.keys(config.cluster?.nodes ?? {}).length === 1 ? Object.keys(config.cluster.nodes)[0] : undefined); + if (process.env.LLOOM_NODE_ID && config.cluster?.nodeId && process.env.LLOOM_NODE_ID !== config.cluster.nodeId) + throw new Error('LLOOM_NODE_ID conflicts with configured cluster.nodeId'); + const discovery = await detectNvidiaSyncCluster({ localNodeId: explicitLocalId }); if (!discovery) throw new Error('No NVIDIA Sync cluster was detected from this node'); - const cluster = nvidiaSyncClusterConfig(discovery, { + const options = { + discovery, + localNodeId: discovery.nodeId, id: argValue(args, '--id'), - apiKeyEnv: argValue(args, '--api-key-env') ?? 'LLOOM_CLUSTER_KEY' - }); + apiKeyEnv: argValue(args, '--api-key-env') + }; + // Keep authoring forms such as environment references out of expansion. + const source = JSON.parse(readFileSync(config.sourcePath, 'utf8')); + const plan = (raw) => mergeNvidiaSyncClusterDiscovery({ ...options, cluster: raw.cluster }); + let cluster = plan(source); + const validationErrors = validateClusterConfig({ ...config, cluster }); + let changed = false; if (hasFlag(args, '--apply')) { - const source = JSON.parse(readFileSync(config.sourcePath, 'utf8')); - source.cluster = cluster; - writeFileSync(config.sourcePath, `${JSON.stringify(source, null, 2)}\n`, { mode: 0o600 }); + if (validationErrors.length) throw new Error(`Invalid discovered cluster: ${validationErrors.join('; ')}`); + ({ changed } = await mutateConfigSource(config, (latest) => { + // Replan from the current raw source inside the serialized mutation. + cluster = plan(latest); + latest.cluster = cluster; + })); } - console.log(JSON.stringify({ ok: true, applied: hasFlag(args, '--apply'), discovery, cluster }, null, 2)); + console.log( + JSON.stringify( + { + ok: validationErrors.length === 0, + applied: hasFlag(args, '--apply'), + changed, + discovery, + summary: nvidiaSyncDiscoverySummary(discovery, { currentConfig: config }), + validationErrors, + cluster, + next: 'Discovery does not change model placement or gateway listeners. Verify endpoints before serving.' + }, + null, + 2 + ) + ); return; } if (action === 'add-node') { @@ -2714,7 +2848,7 @@ async function main() { handlers['pack-export'] = handlers['recipe-export']; handlers['recipe-pack'] = handlers['recipe-import']; handlers['pack-submit'] = handlers['recipe-submit']; - handlers['model-add'] = handlers['add-model']; + handlers['profiles'] = handlers['fleet']; handlers['routing'] = handlers['route']; handlers['runtime-status'] = handlers['runtimes']; handlers['cluster-status'] = handlers['cluster']; diff --git a/config/default.json b/config/default.json index e84e8de..feccb4d 100644 --- a/config/default.json +++ b/config/default.json @@ -31,7 +31,12 @@ "enabled": true, "autoEvict": false, "reserveMemoryGb": 12, - "protectActiveRequests": true + "protectActiveRequests": true, + "memorySafety": { + "mode": "enforce", + "maxMemoryUtilization": 0.9, + "pollIntervalMs": 250 + } }, "community": { "hostUrl": "http://127.0.0.1:8110", diff --git a/docs/architecture.md b/docs/architecture.md index 52bfe22..08c5ac2 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -48,6 +48,31 @@ tool call has no reasoning to replay; a new user turn permits thinking again. This opt-in compatibility setting preserves tool constraints without inventing reasoning history. Ordinary calls outside those cases keep their thinking settings. +### OpenRouter provider restriction + +Set `openrouterProvider` on a dedicated OpenRouter backend to restrict its +upstream providers across Chat Completions, Responses, and Anthropic Messages, +including streaming requests: + +```json +{ + "type": "openai", + "baseUrl": "https://openrouter.ai/api/v1", + "apiKeyEnv": "OPENROUTER_API_KEY", + "openrouterProvider": { "only": ["z-ai"], "allow_fallbacks": false } +} +``` + +Configured provider fields override client requests. Chat Completions retains +other client provider preferences; the Responses and Anthropic bridges retain +only the fields supported by their translators. `only` must contain at least one nonempty provider +slug; `allow_fallbacks` defaults to false. Invalid policy objects fail before +an upstream request is sent. The policy applies only to the exact +`openrouter.ai` host. With the configuration above, an unavailable Z.ai endpoint +returns an error instead of switching to another provider. Models sharing this +backend share its restriction; LLooM alias fallback rules remain separate. +See [OpenRouter provider routing](https://openrouter.ai/docs/guides/routing/provider-selection). + ## Security Defaults | Setting | Default | Meaning | @@ -318,3 +343,12 @@ The current repository includes a static `lloom-host` development server that se Setup composes initialization, backend setup, recipe setup, generated clients, and client integration writes into one audited plan. It does not bypass the lower-level safety gates: dry-run is the default, and real execution requires explicit `--apply --yes`. Bootstrap remains the lower-level backend/model/client phase for an existing config. The default generated gateway port is `8100`; selected backend runtimes occupy the default backend range beginning at `8201`. `setup --port` and `setup --backend-port-range` retarget the generated provider URL, backend base URLs, runtime ports, health URLs, and warmup URLs together so custom port layouts remain internally consistent. + + +### Dashboard memory map + +The Models view divides one machine’s physical memory into running backends, system/other applications, and available space. Protected headroom is an overlay within available memory, not additional capacity. Hover, keyboard focus, and selection preview the incoming model’s configured peak estimate against that machine’s live memory and safety reserve. These interactions are read-only. Normal inference prepares the model through admission; optional manual preparation uses the same guarded admission endpoint. + +Dashboard and node-status requests opt into passive memory attribution. A five-second shared cache reads process memory and loopback listeners with bounded subprocesses. Overlapping process trees are counted once; Docker PIDs are used only on Linux. Known local Ollama and LLooM audio servers also expose their loaded-model lists through bounded, read-only requests. A healthy server with an empty model cache is not a resident model. Routing and admission status do not request this dashboard sampling. + +On macOS, an optional bounded Python helper reads Darwin’s physical-footprint accounting, which includes charged Metal and compressed memory. If unavailable, attribution falls back to RSS. Process memory is approximate and does not describe all shared or cached allocations. Unmeasured running backends use visibly marked estimates. If attribution exceeds measured host use, blocks are reconciled to that measured total. The host’s available-memory reading remains authoritative. Remote nodes use their own observations and safety policy; distributed capacity is never pooled into a single fit promise. Preview estimates never override admission or the live memory safety guard. diff --git a/docs/browser-experience.md b/docs/browser-experience.md new file mode 100644 index 0000000..c0ca77e --- /dev/null +++ b/docs/browser-experience.md @@ -0,0 +1,48 @@ +# The LLooM browser experience + +LLooM chooses and operates vendor recipes. Recipe vendors own model execution, backend tuning, and distributed execution contracts. The dashboard presents the hardware, installation, readiness, and serving decisions that belong to LLooM. + +## First run + +After installing the CLI from the repository, run `lloom` in an interactive terminal. With no installed configuration, a loopback setup server opens a browser. Choose Chat & write, Code, Images, or Voice; review the compatible recipe and installation details; then choose **Set up my AI**. + +The setup page uses the bundled recipe library. Unsupported hardware or workloads produce an explicit explanation. Unknown download sizes, credentials, or license information stay unknown. A recipe metadata license does not cover the model weights. + +Installation runs as a background job in the setup process. Refreshing the page resumes its progress. Keep the terminal open. A failed bootstrap can retry only the exact, unchanged configuration created by that session, within the original 30-minute review window. After a process restart, use `lloom bootstrap --apply --yes` or the installed dashboard to continue. Setup never silently replaces another configuration. + +After installation, LLooM starts the gateway and checks a chat request through its normal inference API. Health alone does not mean the model is ready. Media recipes require checking their actual output; the page identifies that remaining step. + +`lloom --no-browser`, `lloom onboard`, and `lloom up --go` preserve the CLI paths. JSON, offline, and explicit recipe flags do not unexpectedly launch a browser. `lloom ui` opens an installed gateway. + +## Daily use + +- **Live** shows clients, gateway nodes, models, and observed traffic. A luminous gateway connects client cards to model rows grouped by physical machine. Active requests drive the light trails. Follow activity focuses the scene on serving models. Detailed topology retains the diagnostic canvas and its manual camera controls. Both views honor reduced motion. +- **Models** searches and filters the actual gateway catalog. The inspector offers Load, Warm up, Unload, readiness policy, and a small chat trial. Runtime details remain available in a disclosure. +- **Add model** reviews a vendor recipe or custom model reference before starting a background installation. Plan IDs bind the reviewed input, recipe/backend catalog, and configuration version. Concurrent or stale changes fail visibly. Custom Hugging Face imports require an immutable commit link and publish only after staged acquisition verification. Downloads remain reusable. The new installation API does not accept arbitrary config paths or shell commands. +- **Machines** shows configured physical nodes and their individual memory readings. Memory is never presented as one interchangeable pool. Distributed model execution requires a compatible vendor recipe. +- **Clients** supplies the gateway base URL, exact model IDs, an example request, and CLI integration commands. Admin credentials do not belong in client applications. +- **Settings** retains the advanced recipe, backend, runtime, and setup tools. + +Readiness policies express intent: **Auto** loads on demand, **Prefer ready** uses the existing idle residency reconciler, and **Always ready** prevents automatic eviction. All loading still passes through memory admission. Changing readiness does not restart a model or interrupt active work. The API returns a pending job while the current admission completes; the page reports completion or failure. Queued residency starts recheck the saved policy, and pending hard pins protect eviction victims. Use Load when you want to start a cold model immediately. + +### Memory protection + +Admission counts live host memory use, including other applications, even when the policy specifies only a reserve or an absolute budget. Model estimates are planning inputs, not allocation limits. + +Memory protection is enabled by default. A newly started backend is checked before launch and monitored during loading and warmup. If available memory reaches the hard reserve or host utilization reaches the ceiling, LLooM aborts that load, cleans up its processes, and reports the failed threshold. Ordinary Load, forced starts, and disabling automatic eviction do not disable this protection. Automatic retries are blocked until a manual retry or gateway restart. Suspend a model to keep it blocked across restarts. + +The installed config accepts `runtimePolicy.memorySafety` with `mode`, `minAvailableMemoryGb`, `maxMemoryUtilization` (a fraction), and `pollIntervalMs`. The normal mode is `"enforce"`. The default ceiling is 90%; the host reserve can impose a stricter limit. Sampling is a userspace safeguard, not a kernel-enforced allocation quota: a backend can allocate between samples. + +For deliberate manual experiments, `"mode": "yolo"` disables memory admission and the hard load guard. It leaves authentication, runtime ownership, and maintenance gates in place. The dashboard displays a persistent YOLO warning. Restore `"enforce"` before normal operation; YOLO can exhaust the host's memory. + +## Local security + +First-run setup binds only to loopback and uses a random session token passed through the URL fragment. It removes the fragment immediately and keeps the token in that tab's session storage. Setup rejects foreign origins, alternate authorities, oversized bodies, and unreviewed apply inputs. Configuration publication cannot overwrite a file created concurrently. + +The installed dashboard rejects cross-origin management requests and DNS-rebound loopback authorities. Its page cannot be framed. Remote management writes require both explicit remote-admin enablement and a configured admin key; inference keys alone cannot authorize them. Inference compatibility is unchanged. Local same-user processes remain inside the local trust boundary. + +Installation jobs and reviewed plan IDs live in the gateway process. Browser refresh is supported; a gateway restart requires a fresh plan. Existing installer stage state and downloaded files support subsequent retries. The config mutation store detects concurrent writes but is not a cross-process transaction lock. + +## Remaining product work + +The repository is still the distribution source; a signed single-command installer and published npm package are not part of this change. Nearby discovery and approval-based pairing between independently installed gateways need an authenticated node identity protocol. The Machines page does not invent peers or automatically grant access. Existing configured federation remains available through the CLI and gateway. diff --git a/docs/clusters.md b/docs/clusters.md index 44b7d8f..c9b9652 100644 --- a/docs/clusters.md +++ b/docs/clusters.md @@ -37,9 +37,7 @@ The declarative form is useful for review or hand editing: "labels": { "role": "node", "architecture": "darwin-arm64", "accelerator": "metal" }, "resources": { "memoryGb": 64 }, "proxy": { - "models": [ - { "id": "local/qwen", "as": "macbook-local/local/qwen", "kind": "chat", "remoteRuntime": "qwen" } - ] + "models": [{ "id": "local/qwen", "as": "macbook-local/local/qwen", "kind": "chat", "remoteRuntime": "qwen" }] } } } @@ -60,7 +58,13 @@ lloom cluster discover --id ennspark-cluster lloom cluster discover --id ennspark-cluster --api-key-env LLOOM_ADMIN_API_KEY --apply ``` -Discovery reads only NVIDIA Sync-marked SSH entries and local interface addresses. It records the physical hostnames as metadata but uses the stable Sync/Tailscale aliases (with `-lan` removed) as node IDs. `lloom profile` includes the detected topology, so `lloom select` can rank exact-size cluster recipes before cluster configuration is applied. +Discovery groups `NVSyncClusterAlias` entries into stable node IDs, including duplicate IP aliases and multiple rails per peer. Older Sync entries with a named `Host` alias still work. The configured local identity and leader remain unchanged. Local observations appear under `cluster.discovery.links`; they do not prove gateway reachability, bandwidth, or a complete ring. + +New peers remain physical inventory under `cluster.discovery.nodes`. Join a peer with `lloom cluster add-node --apply` after its gateway is reachable. Discovery does not create a live gateway endpoint from an SSH address. + +`--apply` merges into the raw configuration and validates it before an atomic write. It preserves existing federation nodes, model catalogs, authentication references, custom endpoints, and model placements. An endpoint that names its previous `10.100.*` backend host moves to the observed address while retaining its scheme, port, and path. Other endpoints remain unchanged and receive a diagnostic. Discovery does not change gateway listeners or NCCL settings; verify new endpoints before using them. + +Physical membership and model placement are separate. In a three-Spark ring, a model can run on one node while a distributed model uses an explicit two-node subset. A three-node member list is also representable, but TP-3 requires a compatible model architecture and backend launcher; discovery does not infer that compatibility or repartition a loaded model. Resource admission applies to the chosen members. Select the NCCL adapters that connect those members: a TP-2 job must not use adapters whose cables lead to a third node outside its placement. Use private fabric addresses for node-to-node LLooM and raw backend traffic. `backendHost` is deliberately required when a recipe auto-materializes replicas; LLooM binds the generated backend only to that address, never to every interface. @@ -131,14 +135,16 @@ The equivalent explicit config is: ```json { - "models": [{ - "id": "example/Qwen", - "targets": [ - { "id": "spark-1", "node": "spark-1", "backend": "qwen-spark-1", "runtime": "qwen-spark-1" }, - { "id": "spark-2", "node": "spark-2", "backend": "qwen-spark-2", "runtime": "qwen-spark-2" }, - { "id": "spark-2-proxy", "node": "spark-2", "backend": "lloom-node-spark-2", "remoteRuntime": "qwen-spark-2" } - ] - }] + "models": [ + { + "id": "example/Qwen", + "targets": [ + { "id": "spark-1", "node": "spark-1", "backend": "qwen-spark-1", "runtime": "qwen-spark-1" }, + { "id": "spark-2", "node": "spark-2", "backend": "qwen-spark-2", "runtime": "qwen-spark-2" }, + { "id": "spark-2-proxy", "node": "spark-2", "backend": "lloom-node-spark-2", "remoteRuntime": "qwen-spark-2" } + ] + } + ] } ``` @@ -188,19 +194,39 @@ Explicit config remains supported: "placement": { "mode": "distributed", "members": [ - { "node": "spark-1", "runtime": "dsv4-ray-head", "role": "head", "order": 10, "resources": { "memoryGb": 6 } }, - { "node": "spark-2", "runtime": "dsv4-ray-worker-2", "role": "worker", "order": 20, "resources": { "memoryGb": 54 } }, - { "node": "spark-1", "runtime": "dsv4-vllm-server", "role": "server", "order": 30, "resources": { "memoryGb": 54 } } + { + "node": "spark-1", + "runtime": "dsv4-ray-head", + "role": "head", + "order": 10, + "resources": { "memoryGb": 6 } + }, + { + "node": "spark-2", + "runtime": "dsv4-ray-worker-2", + "role": "worker", + "order": 20, + "resources": { "memoryGb": 54 } + }, + { + "node": "spark-1", + "runtime": "dsv4-vllm-server", + "role": "server", + "order": 30, + "resources": { "memoryGb": 54 } + } ] } } }, - "models": [{ - "id": "deepseek-ai/DSv4Flash", - "backend": "dsv4flash", - "runtime": "dsv4flash-cluster", - "upstreamModel": "deepseek-ai/DSv4Flash" - }] + "models": [ + { + "id": "deepseek-ai/DSv4Flash", + "backend": "dsv4flash", + "runtime": "dsv4flash-cluster", + "upstreamModel": "deepseek-ai/DSv4Flash" + } + ] } ``` diff --git a/package-lock.json b/package-lock.json index 31f9a55..9bba72a 100644 --- a/package-lock.json +++ b/package-lock.json @@ -9,7 +9,7 @@ "version": "0.2.3", "license": "MIT", "dependencies": { - "undici": "^7.28.0" + "undici": "^8.10.2" }, "bin": { "lloom": "bin/lloom.mjs", @@ -17,8 +17,8 @@ }, "devDependencies": { "@eslint/js": "^10.0.1", - "eslint": "^10.10.0", - "prettier": "^3.9.6" + "eslint": "^10.11.0", + "prettier": "^3.9.8" }, "engines": { "node": ">=20.0.0" @@ -418,9 +418,9 @@ } }, "node_modules/eslint": { - "version": "10.10.0", - "resolved": "https://registry.npmjs.org/eslint/-/eslint-10.10.0.tgz", - "integrity": "sha512-NPXn6r5zl4uET1DAVPaOwzX3rut4c0wcmw3dWJAfOsTM5+TogXo0DDjz8pwm/hL8cyVNpHqeK4JpN0NjnyFFNw==", + "version": "10.11.0", + "resolved": "https://registry.npmjs.org/eslint/-/eslint-10.11.0.tgz", + "integrity": "sha512-P7a6UEEqb9G95MYAtqkmsTbVXIYyzIfl6NGOIJk162PaahFxFyeGcrlXYFSiagECg4sEm8IseJdZBKR3rx6MsQ==", "dev": true, "license": "MIT", "workspaces": [ @@ -887,9 +887,9 @@ } }, "node_modules/prettier": { - "version": "3.9.6", - "resolved": "https://registry.npmjs.org/prettier/-/prettier-3.9.6.tgz", - "integrity": "sha512-OpN0zzVdiaiAhxpuuj5efpIS4sY9j7bY6uR5mnj5yPzGkdkjNKSJeUThPb60Jw29QuAZgA4o+/iB49kFiaBX6g==", + "version": "3.9.8", + "resolved": "https://registry.npmjs.org/prettier/-/prettier-3.9.8.tgz", + "integrity": "sha512-WRFq3Wn3WId7LLROfMLdH7xaFr2jR62wU8nLO6rQUOLOxNZUviyJQs1M0iIhLexSFy+L+w0ch66wtoO2jRjG0A==", "dev": true, "license": "MIT", "bin": { @@ -969,12 +969,12 @@ } }, "node_modules/undici": { - "version": "7.29.0", - "resolved": "https://registry.npmjs.org/undici/-/undici-7.29.0.tgz", - "integrity": "sha512-IDxfleLmmbSskfWSUATiN1nfn2rDuvnMOqb5CWR92iIfojA0Ud+ulOAAEQ57LPr9rWmsreUyf5lwyao+7GNNVw==", + "version": "8.10.2", + "resolved": "https://registry.npmjs.org/undici/-/undici-8.10.2.tgz", + "integrity": "sha512-/y4/bH9YNU5hi9NIrpOuvGXFcxrj3CMrV+/AYpowAYTpHn8gX/XPFjNy766FPoYY0miQhdW977JFWKGNhBdwyQ==", "license": "MIT", "engines": { - "node": ">=20.18.1" + "node": ">=22.19.0" } }, "node_modules/uri-js": { diff --git a/package.json b/package.json index 2df4fe7..ea65903 100644 --- a/package.json +++ b/package.json @@ -64,8 +64,8 @@ "check": "npm run check:cluster && npm run check:js && npm run check:entity-stagger && npm run check:python && npm run check:shell && npm run check:spark-deploy && npm run check:workers", "check:cluster": "node --check src/cluster.mjs && node --check test/cluster.test.mjs", "check:entity-stagger": "node --check scripts/entity-stagger-benchmark.mjs && node --check test/entity-stagger-benchmark.test.mjs", - "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs", - "check:python": "python3 -m py_compile backends/dspark-vllm/apply-patch-pack.py backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/vision_exp/*.py backends/qwen38-vllm/apply-ple-fp8-patch.py backends/qwen38-sglang/*.py backends/glm53-exl3/*.py backends/mlx-audio/lloom_audio_server.py backends/hear/lloom_hear_server.py backends/chatterbox/lloom_chatterbox_server.py patches/apply_mtplx_longctx_fix.py backends/comfyui-media/bridge/*.py backends/comfyui-media/graphs/*.py backends/comfyui-media/build/*.py backends/comfyui-media/install.py backends/qwen-image-diffusers/*.py backends/hear/test/test_hear_security.py", + "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs && node --check src/dashboard-scene.mjs && node --check src/dashboard-presence.mjs && node --check src/dashboard-presence-client.mjs && node --check src/first-run.mjs && node --check src/first-run-page.mjs && node --check src/browser-setup.mjs && node --check src/runtime-preferences.mjs && node --check src/installation-jobs.mjs && node --check src/model-installation.mjs && node --check src/dashboard-installation.mjs && node --check src/runtime-memory-safety.mjs && node --check src/dashboard-memory.mjs && node --check src/runtime-memory-usage.mjs && node --check src/config-profiles.mjs && node --check test/config-profiles.test.mjs", + "check:python": "python3 -m py_compile backends/dspark-vllm/apply-patch-pack.py backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/vision_exp/*.py backends/qwen38-vllm/apply-ple-fp8-patch.py backends/qwen38-sglang/*.py backends/glm53-exl3/*.py backends/mlx-audio/lloom_audio_server.py backends/hear/lloom_hear_server.py backends/chatterbox/lloom_chatterbox_server.py patches/apply_mtplx_longctx_fix.py backends/comfyui-media/bridge/*.py backends/comfyui-media/graphs/*.py backends/comfyui-media/build/*.py backends/comfyui-media/install.py backends/qwen-image-diffusers/*.py backends/hear/test/test_hear_security.py src/darwin-memory-usage.py", "check:shell": "bash -n backends/dspark-vllm/entrypoint.sh backends/dspark-vllm/runtime/build-hardened-image.sh backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/apply-runtime.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.sh backends/qwen38-vllm/entrypoint.sh backends/qwen38-sglang/entrypoint.sh backends/glm53-exl3/entrypoint.sh backends/mlx-audio/install.sh backends/hear/install.sh backends/chatterbox/install.sh clients/examples/lloom-claude clients/examples/lloom-codex clients/examples/lloom-hermes clients/examples/lloom-opencode clients/examples/lloom-zero scripts/remote-install-spark.sh scripts/remote-stage-spark-worker.sh", "check:spark-deploy": "node --check test/spark-deploy.test.mjs && node test/spark-deploy.test.mjs", "generate:clients": "node scripts/generate-client-configs.mjs", @@ -73,7 +73,7 @@ "community:check": "node scripts/check-community-deploy.mjs", "package:check": "node scripts/check-package.mjs", "test:adoption": "node test/resource-fit.test.mjs && node test/model-acquisition.test.mjs && node test/dashboard-status.test.mjs", - "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs", + "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/openrouter-provider.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/config-profiles.test.mjs && node test/route-control.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs && npm run test:ux && npm run test:memory-safety && node --test test/dashboard-memory.test.mjs test/runtime-memory-usage.test.mjs", "test:cluster": "node test/cluster.test.mjs", "test": "npm run test:cluster && npm run test:unit && npm run test:entity-stagger && npm run smoke && npm run community:check", "test:entity-stagger": "node test/entity-stagger-benchmark.test.mjs", @@ -84,7 +84,9 @@ "test:workers": "node --test test/research-workers.test.mjs test/research-worker-process.test.mjs", "check:workers": "node --check clients/examples/research-workers/runner.mjs && node --check clients/examples/research-workers/process.mjs && node --check clients/examples/research-workers/process-guard.mjs", "test:comfyui": "python3 scripts/test-comfyui-media.py", - "test:hear": "python3 -m unittest discover -s backends/hear/test -v" + "test:hear": "python3 -m unittest discover -s backends/hear/test -v", + "test:ux": "node --test test/first-run.test.mjs test/runtime-preferences.test.mjs test/admin-browser-security.test.mjs test/installation-jobs.test.mjs", + "test:memory-safety": "node --test test/runtime-memory-safety.test.mjs" }, "engines": { "node": ">=20.0.0" @@ -92,10 +94,10 @@ "license": "MIT", "devDependencies": { "@eslint/js": "^10.0.1", - "eslint": "^10.10.0", - "prettier": "^3.9.6" + "eslint": "^10.11.0", + "prettier": "^3.9.8" }, "dependencies": { - "undici": "^7.28.0" + "undici": "^8.10.2" } } diff --git a/src/bootstrap.mjs b/src/bootstrap.mjs index 05d0375..144d940 100644 --- a/src/bootstrap.mjs +++ b/src/bootstrap.mjs @@ -4,7 +4,7 @@ import { selectIntegrationArtifacts, writeGeneratedIntegrationArtifacts } from './client-integrations.mjs'; -import { applyBackend, applyRecipe, defaultInstallStatePathFor } from './installer.mjs'; +import { applyBackend, applyRecipe, pinDownloadCommands, defaultInstallStatePathFor } from './installer.mjs'; import { backendIds, defaultBackendVariables, @@ -175,17 +175,25 @@ export async function createBootstrapPlan( name: recipe.name, backendId: recipe.backend?.id }, + // Frozen executable evidence: the exact backend/recipe step plans plus the + // selected recipe and modelRoot. Retries and reviewed-plan execution must + // replay these untouched rather than re-profiling or re-selecting. + reviewedRecipe: recipe, + reviewedBackend: backend, + modelRoot: selectedModelRoot, backend: await planBackend(backend, { variables: backendVariables, checkCommands: true }), - recipe: planRecipe(recipe, config, { - modelRoot: selectedModelRoot, - backendIds: backendIds(catalog), - benchmarkEvidence, - benchmarksRoot: selectedBenchmarksRoot, - benchmarkValidationErrors: benchmarkErrors - }), + recipe: await pinDownloadCommands( + planRecipe(recipe, config, { + modelRoot: selectedModelRoot, + backendIds: backendIds(catalog), + benchmarkEvidence, + benchmarksRoot: selectedBenchmarksRoot, + benchmarkValidationErrors: benchmarkErrors + }) + ), benchmarks: { root: selectedBenchmarksRoot, validationErrors: benchmarkErrors, @@ -212,6 +220,7 @@ export async function applyBootstrap( generatedRoot, backendVariables = defaultBackendVariables(process.env), _benchmarkDocuments = [], + reviewedPlan, recipesRoot, recipeDocuments = [], backendCatalogPath, @@ -223,11 +232,27 @@ export async function applyBootstrap( throw new Error('Refusing to bootstrap without yes=true. Re-run with --yes after reviewing the dry-run plan.'); } - const profile = await profileMachine(); - const recipes = [...recipeDocuments, ...(await loadRecipes(recipesRoot))]; - const recipe = await selectRecipe({ recipeId, recipes, profile, recipesRoot }); - const catalog = await loadBackendCatalog(backendCatalogPath); - const backend = getBackend(catalog, recipe.backend?.id); + // A reviewed plan carries the exact backend/recipe plans that were shown to + // and approved by the user. When supplied, reuse the frozen evidence so a + // later hardware/catalog change cannot silently swap in different commands. + const reviewed = reviewedPlan ?? null; + let recipe; + let backend; + let backendPlan = null; + if (reviewed) { + if (!reviewed.reviewedRecipe || !reviewed.reviewedBackend || !reviewed.backend?.steps || !reviewed.recipe?.steps) + throw new Error('Reviewed bootstrap plan is incomplete. Review a fresh plan.'); + recipe = reviewed.reviewedRecipe; + backend = reviewed.reviewedBackend; + backendPlan = reviewed.backend; + } else { + const profile = await profileMachine(); + const recipes = [...recipeDocuments, ...(await loadRecipes(recipesRoot))]; + recipe = await selectRecipe({ recipeId, recipes, profile, recipesRoot }); + const catalog = await loadBackendCatalog(backendCatalogPath); + backend = getBackend(catalog, recipe.backend?.id); + } + if (!recipe) throw new Error('Bootstrap requires a reviewed recipe or recipeId.'); if (!backend) throw new Error(`Recipe ${recipe.id} references unknown backend ${recipe.backend?.id}`); const registry = createRegistry(config); @@ -237,7 +262,8 @@ export async function applyBootstrap( yes, statePath, variables: backendVariables, - env: commandEnv + env: commandEnv, + ...(backendPlan ? { reviewedPlan: backendPlan } : {}) }); backendResult.summary = phaseSummary('backend', backendResult); @@ -249,7 +275,10 @@ export async function applyBootstrap( env: commandEnv, onProgress, stdio, - ...(modelRoot ? { modelRoot } : {}) + ...((reviewed?.modelRoot ?? modelRoot) ? { modelRoot: reviewed?.modelRoot ?? modelRoot } : {}), + ...((reviewed?.recipePlan ?? (reviewed?.recipe?.steps ? reviewed.recipe : null)) + ? { reviewedPlan: reviewed.recipePlan ?? reviewed.recipe } + : {}) }) : blockedPhase('recipe', 'backend phase failed', { dryRun }); recipeResult.summary ??= phaseSummary('recipe', recipeResult); diff --git a/src/browser-setup.mjs b/src/browser-setup.mjs new file mode 100644 index 0000000..d27e787 --- /dev/null +++ b/src/browser-setup.mjs @@ -0,0 +1,219 @@ +import fs from 'node:fs/promises'; +import { createHash } from 'node:crypto'; +import { createFirstRunServer } from './first-run.mjs'; +import { createOnboardingPlan } from './onboarding.mjs'; +import { applySetup } from './setup.mjs'; +import { applyBootstrap } from './bootstrap.mjs'; +import { profileMachine, rankRecipes } from './machine-profile.mjs'; +import { loadRecipes, loadRecipeById } from './recipes.mjs'; +import { loadBackendCatalog } from './backend-catalog.mjs'; +import { loadConfig } from './config.mjs'; + +const fingerprint = (value) => createHash('sha256').update(JSON.stringify(value)).digest('hex'); +export function recipeSupportsWorkload(recipe, workload) { + const values = new Set([ + ...(recipe.capabilities || []), + ...(recipe.models || []).flatMap((model) => [model.kind, ...(model.capabilities || [])]) + ]); + if (workload === 'images') return [...values].some((value) => /^image/.test(value)); + if (workload === 'voice') return [...values].some((value) => /^audio|speech|transcription|tts|stt/.test(value)); + const chat = ['chat', 'responses', 'anthropic-messages'].some((value) => values.has(value)); + return workload === 'code' + ? chat && (values.has('tools') || (recipe.keywords || []).some((value) => /cod/.test(value))) + : chat; +} + +export function createBrowserSetup(config, options, { gatewayStarter, serverOptions = {} } = {}) { + const evidence = async (recipeId) => ({ + recipe: await loadRecipeById(recipeId, options.recipesRoot), + backendCatalog: await loadBackendCatalog(options.backendCatalogPath) + }); + return createFirstRunServer({ + ...serverOptions, + gatewayStarter, + async planBuilder({ workloadId, recipeId }) { + const [recipes, profile] = await Promise.all([loadRecipes(options.recipesRoot), profileMachine()]); + const ranked = await rankRecipes( + recipes.filter((recipe) => recipeSupportsWorkload(recipe, workloadId)), + profile, + { checkCommands: true } + ); + const candidates = ranked.filter((candidate) => candidate.selectable); + const selected = recipeId ? candidates.find((candidate) => candidate.recipeId === recipeId) : candidates[0]; + if (!selected) + throw new Error( + 'No compatible ' + + workloadId + + ' recipe is available for this hardware. Choose another use or add a vendor recipe through the CLI.' + ); + const selectedOptions = { + ...options, + recipeId: selected.recipeId, + offline: true, + start: false, + includeRuntimes: false + }; + const plan = await createOnboardingPlan(config, selectedOptions); + const source = await evidence(selected.recipeId); + plan.browserSetup = { options: selectedOptions, fingerprint: fingerprint(source) }; + const recipe = source.recipe; + const option = (candidate) => ({ + id: candidate.recipeId, + name: candidate.name, + reason: candidate.reasons.length + ? candidate.reasons.join('; ') + : 'Compatible with ' + (profile.cpuBrand || profile.platformId) + '.', + memoryRequiredGb: candidate.memoryRequiredGb, + downloadSizeBytes: null, + license: + recipes.find((item) => item.id === candidate.recipeId)?.license?.name || + recipes.find((item) => item.id === candidate.recipeId)?.license?.id || + null, + credentials: null + }); + return { + plan, + view: { + workloadId, + selected: option(selected), + alternatives: candidates + .filter((c) => c.recipeId !== selected.recipeId) + .slice(0, 8) + .map(option), + machine: { + name: profile.cpuBrand || profile.platformId, + platformId: profile.platformId, + totalMemoryGb: profile.totalMemoryGb, + accelerators: profile.accelerators || [] + }, + paths: { configPath: plan.configPath, modelRoot: plan.modelRoot }, + ports: plan.ports, + review: { + stages: plan.stages, + doctorCommands: plan.next?.doctor, + packages: (plan.setup?.phases?.bootstrap?.backend?.steps || []).map((step) => step.label || step.id), + models: (recipe.models || []).map((model) => model.gatewayModel || model.model) + } + } + }; + }, + applyRunner: createReviewedSetupRunner(config, { evidence }), + async gatewayProbe(report, plan, gateway) { + const installed = await loadConfig(report.configPath); + const url = gateway?.url || report.dashboardUrl; + const key = installed.security?.apiKeys?.[0] || installed.security?.adminApiKeys?.[0]; + const headers = { 'content-type': 'application/json', ...(key ? { authorization: 'Bearer ' + key } : {}) }; + const healthReply = await fetch(new URL('/health', url), { signal: AbortSignal.timeout(15000) }); + const healthData = healthReply.ok ? await healthReply.json() : null; + const health = healthData?.ok === true && (!gateway?.pid || healthData.pid === gateway.pid); + if (!health) + return { + healthy: false, + inferenceVerified: false, + endpoint: url, + detail: 'Installed, but the gateway did not pass its health check.' + }; + const model = installed.models.find( + (item) => + (item.kind || 'chat') === 'chat' && + (plan.setup?.phases?.init?.config?.models || []).some((candidate) => candidate.id === item.id) + ); + if (!model) + return { + healthy: true, + inferenceVerified: false, + endpoint: url, + detail: 'Gateway is running. Try the installed media model to verify its output.' + }; + try { + const response = await fetch(new URL('/v1/chat/completions', url), { + method: 'POST', + headers, + body: JSON.stringify({ + model: model.id, + messages: [{ role: 'user', content: 'Reply with Ready.' }], + max_tokens: 16, + stream: false + }), + signal: AbortSignal.timeout(180000) + }); + const body = await response.json(); + const verified = + response.ok && + Array.isArray(body.choices) && + body.choices.some((choice) => typeof choice.message?.content === 'string' && choice.message.content.trim()); + return { + healthy: true, + inferenceVerified: verified, + endpoint: url, + detail: verified + ? 'A model answered through your gateway.' + : body.error?.message || 'Gateway is running, but the model did not return a text response.' + }; + } catch (error) { + return { + healthy: true, + inferenceVerified: false, + endpoint: url, + detail: 'Gateway is running. Model verification needs attention: ' + error.message + }; + } + } + }); +} + +export function createReviewedSetupRunner(config, { evidence, apply = applySetup, bootstrap = applyBootstrap }) { + let ownedConfig = null; + return async (plan, { onProgress }) => { + const details = plan.browserSetup; + if (!details || fingerprint(await evidence(plan.selectedRecipe.id)) !== details.fingerprint) + throw new Error('The recipe or backend catalog changed. Review a fresh plan.'); + // A first-run wizard must never silently replace a newly created or + // concurrently installed configuration. + let existing; + try { + existing = JSON.parse(await fs.readFile(plan.configPath, 'utf8')); + } catch (error) { + if (error.code !== 'ENOENT') throw error; + } + if (existing) { + if ( + !ownedConfig || + ownedConfig.path !== plan.configPath || + ownedConfig.recipeId !== plan.selectedRecipe.id || + fingerprint(existing) !== ownedConfig.fingerprint + ) + throw new Error('A configuration now exists or changed. Open LLooM to manage it; setup will not overwrite it.'); + if (fingerprint(plan.setup.phases.init.config) !== ownedConfig.fingerprint) + throw new Error( + 'The installation plan changed after configuration was created. Continue with lloom bootstrap --apply --yes to resume the installed recipe.' + ); + // Retry only the bootstrap for the exact configuration this session + // successfully created. Never replace it after a partial download. + const installed = await loadConfig(plan.configPath); + const result = await bootstrap(installed, { + ...details.options, + modelRoot: plan.modelRoot, + // Retry replays the exact reviewed bootstrap evidence; never re-plan. + reviewedPlan: plan.setup?.phases?.bootstrap, + dryRun: false, + yes: true, + onProgress + }); + return { ...result, configPath: plan.configPath, dashboardUrl: plan.dashboardUrl }; + } + const result = await apply(config, { + ...details.options, + reviewedPlan: plan.setup, + dryRun: false, + yes: true, + start: false, + onProgress, + exclusiveConfig: true, + onConfigWritten(file, value) { + ownedConfig = { path: file, recipeId: plan.selectedRecipe.id, fingerprint: fingerprint(value) }; + } + }); + return { ...result, dashboardUrl: plan.dashboardUrl }; + }; +} diff --git a/src/cluster.mjs b/src/cluster.mjs index 726f2ff..c18598a 100644 --- a/src/cluster.mjs +++ b/src/cluster.mjs @@ -1,7 +1,9 @@ import os from 'node:os'; import fs from 'node:fs/promises'; import path from 'node:path'; -import { Agent as UndiciAgent } from 'undici'; +// Dispatcher and fetch must come from the same undici copy: the package's +// v8 Agent speaks a dispatcher contract Node 22's internal undici v6 rejects. +import { Agent as UndiciAgent, fetch as undiciFetch } from 'undici'; import { runCommand } from './process-control.mjs'; // Node's built-in fetch aborts a request that has not produced response headers @@ -226,20 +228,36 @@ export function parseNvidiaSyncSshConfig(source = '') { let entry = null; for (const rawLine of String(source).split(/\r?\n/)) { const line = rawLine.trim(); - const host = line.match(/^Host\s+([^*?!\s]+)$/i); - if (host) { - if (entry?.createdBySync && entry.hostname) entries.push(entry); - entry = { alias: host[1], createdBySync: false }; + // A Host/Match keyword always closes the current block. Wildcard, negated, + // and multi-host patterns are not usable aliases, and Match introduces a + // conditional scope whose body is not part of the previous Host block, so + // the current entry must be finalized and cleared rather than merged forward. + const blockStart = line.match(/^(Host|Match)\s+(.*)$/i); + if (blockStart) { + if (entry) entries.push(entry); + entry = null; + const alias = blockStart[1].toLowerCase() === 'host' ? blockStart[2].trim() : ''; + if (alias && /^[^\s*?!]+$/.test(alias)) entry = { alias, createdBySync: false }; continue; } if (!entry) continue; - if (/^###\s*CreatedBy:\s*NVIDIA Sync$/i.test(line)) entry.createdBySync = true; + // Comments carry NVIDIA Sync provenance. The marker is checked before the + // generic comment guard, and every other comment line is ignored. + const marker = line.match(/^#+\s*NVSyncClusterAlias\s*:\s*(\S+)\s*$/i); + if (marker) { + entry.clusterAlias = marker[1]; + continue; + } + if (/^#/.test(line)) { + if (/^#+\s*CreatedBy:\s*NVIDIA Sync\s*$/i.test(line)) entry.createdBySync = true; + continue; + } const hostname = line.match(/^Hostname\s+(\S+)$/i); if (hostname) entry.hostname = hostname[1]; const user = line.match(/^User\s+(\S+)$/i); if (user) entry.user = user[1]; } - if (entry?.createdBySync && entry.hostname) entries.push(entry); + if (entry) entries.push(entry); return entries; } @@ -276,52 +294,348 @@ function friendlyNodeId(alias) { return String(alias).replace(/-lan$/i, ''); } +function compareIpv4(left, right) { + const leftParts = String(left).split('.').map(Number); + const rightParts = String(right).split('.').map(Number); + const valid = (parts) => + parts.length === 4 && parts.every((value) => Number.isInteger(value) && value >= 0 && value <= 255); + const leftValid = valid(leftParts); + const rightValid = valid(rightParts); + if (leftValid && rightValid) { + for (let index = 0; index < 4; index += 1) { + if (leftParts[index] !== rightParts[index]) return leftParts[index] - rightParts[index]; + } + return 0; + } + if (leftValid !== rightValid) return leftValid ? -1 : 1; + return String(left).localeCompare(String(right)); +} + +function isAddressAlias(value) { + return /^\d{1,3}(\.\d{1,3}){3}$/.test(String(value ?? '')); +} + +function syncPeerMatches(peers, localAddresses, localAddressSet = new Set()) { + const matches = []; + const ambiguous = []; + for (const original of peers) { + if (!original?.createdBySync || !original.hostname) continue; + const clusterAlias = original.clusterAlias ?? (!isAddressAlias(original.alias) ? original.alias : null); + if (!clusterAlias || !/^[a-z0-9][a-z0-9._-]*$/i.test(clusterAlias)) continue; + const peer = { ...original, clusterAlias }; + const privateAddress = /^10\.100\./.test(peer.hostname) ? peer.hostname : null; + if (!privateAddress) continue; + if (localAddressSet.has(privateAddress)) { + ambiguous.push({ + peerAddress: privateAddress, + sshAlias: peer.alias ?? null, + clusterAlias: peer.clusterAlias, + reason: 'local-address', + message: `ignored Sync alias ${peer.clusterAlias}: ${privateAddress} is one of this node's own addresses` + }); + continue; + } + const candidates = [ + ...new Map( + localAddresses + .filter( + (address) => + address.address?.startsWith('10.100.') && sameSubnet(address.address, privateAddress, address.prefixlen) + ) + .map((address) => [address.interface ?? address.address, address]) + ).values() + ]; + const candidateAddresses = candidates.map((address) => address.address); + if (candidates.length > 1) { + ambiguous.push({ + peerAddress: privateAddress, + sshAlias: peer.alias ?? null, + clusterAlias: peer.clusterAlias, + localAddresses: candidateAddresses.sort(compareIpv4), + reason: 'ambiguous-local-interface', + message: + `ignored Sync alias ${peer.clusterAlias}: peer address ${privateAddress} matches multiple local ` + + `interfaces (${candidates + .map((address) => address.interface ?? address.address) + .sort() + .join(', ')}); ` + + `refusing to guess a rail` + }); + continue; + } + const local = candidates[0]; + if (!local) continue; + matches.push({ peer, local, privateAddress }); + } + return { matches, ambiguous }; +} + +function nodeRailsForPeers(matches) { + const nodes = new Map(); + for (const match of matches) { + const id = friendlyNodeId(match.peer.clusterAlias); + let existing = nodes.get(id); + if (!existing) { + existing = { id, alias: match.peer.alias, sshAlias: null, sshUser: null, rails: [] }; + nodes.set(id, existing); + } + if (!existing.rails.some((candidate) => candidate.peerAddress === match.privateAddress)) { + existing.rails.push({ + interface: match.local.interface, + localInterface: match.local.interface, + localAddress: match.local.address, + peerAddress: match.privateAddress, + prefixlen: match.local.prefixlen, + sshAlias: match.peer.clusterAlias, + sshUser: match.peer.user ?? null + }); + } + if (!isAddressAlias(match.peer.alias)) { + existing.sshAlias ??= match.peer.alias; + existing.sshUser ??= match.peer.user ?? null; + } else { + existing.sshUser ??= match.peer.user ?? null; + } + } + return nodes; +} + export function buildNvidiaSyncDiscovery({ peers = [], localAddresses = [], localNodeId, hostname } = {}) { - const matches = peers - .map((peer) => { - const local = localAddresses.find( - (address) => - address.address?.startsWith('10.100.') && sameSubnet(address.address, peer.hostname, address.prefixlen) - ); - return local ? { peer, local } : null; - }) - .filter(Boolean); + const localAddressSet = new Set(localAddresses.map((address) => address.address).filter(Boolean)); + const { matches, ambiguous } = syncPeerMatches(peers, localAddresses, localAddressSet); if (!matches.length) return null; - const preferred = [...matches].sort( - (left, right) => Number(right.local.address.split('.')[2]) - Number(left.local.address.split('.')[2]) - )[0]; const id = localNodeId || hostname || os.hostname(); - const peerId = friendlyNodeId(preferred.peer.alias); - const nodeIds = [id, peerId].sort(); + if (matches.some((match) => friendlyNodeId(match.peer.clusterAlias) === id)) { + throw new Error(`Sync peer alias conflicts with local node id ${id}`); + } + const discovered = [...nodeRailsForPeers(matches).values()].sort( + (left, right) => + compareIpv4(left.rails[0].peerAddress, right.rails[0].peerAddress) || left.id.localeCompare(right.id) + ); + + const localPrimary = [...matches].sort((left, right) => compareIpv4(left.local.address, right.local.address))[0] + .local; + const links = [ + ...new Map( + matches.map((match) => [ + `${friendlyNodeId(match.peer.clusterAlias)}|${match.privateAddress}`, + { + nodeId: id, + peerNodeId: friendlyNodeId(match.peer.clusterAlias), + localInterface: match.local.interface, + localAddress: match.local.address, + peerAddress: match.privateAddress, + prefixlen: match.local.prefixlen + } + ]) + ).values() + ].sort( + (left, right) => + compareIpv4(left.peerAddress, right.peerAddress) || + compareIpv4(left.localAddress, right.localAddress) || + left.peerNodeId.localeCompare(right.peerNodeId) + ); + + const nodes = { + [id]: { + id, + local: true, + backendHost: localPrimary.address, + fabricInterface: localPrimary.interface + } + }; + for (const node of discovered) { + const rails = [...node.rails].sort((left, right) => compareIpv4(left.peerAddress, right.peerAddress)); + const primary = rails[0]; + nodes[node.id] = { + id: node.id, + local: false, + backendHost: primary.peerAddress, + ...(node.sshAlias ? { sshAlias: node.sshAlias } : {}), + sshUser: node.sshUser, + rails + }; + } + + const familyIds = Object.keys(nodes).sort(); + const topology = Object.keys(nodes).length === 2 ? 'direct' : 'local-adjacency'; return { detected: true, provider: 'nvidia-sync', - topology: 'direct', - nodeCount: nodeIds.length, + topology, + nodeCount: familyIds.length, nodeId: id, - leaderNode: nodeIds[0], + localNode: id, fabric: { - interface: preferred.local.interface, - localAddress: preferred.local.address, - peerAddress: preferred.peer.hostname, - prefixlen: preferred.local.prefixlen + interface: localPrimary.interface, + localAddress: localPrimary.address, + peerAddress: matches.find((match) => match.local.address === localPrimary.address).privateAddress, + prefixlen: localPrimary.prefixlen }, - nodes: { - [id]: { - id, - local: true, - backendHost: preferred.local.address, - fabricInterface: preferred.local.interface + verified: false, + evidence: 'local-interface-and-ssh-config', + links, + nodes, + ...(ambiguous.length ? { ambiguousPeers: ambiguous } : {}) + }; +} + +export function nvidiaSyncDiscoverySummary(discovery, { currentConfig = null } = {}) { + if (!discovery?.detected) return null; + const localId = discovery.nodeId; + const configuredNodes = asObject(currentConfig?.cluster?.nodes); + const configuredNodeIds = Object.keys(configuredNodes); + const discoveredIds = Object.keys(discovery.nodes).sort(); + const unresolvedPeers = Object.values(discovery.nodes) + .filter((node) => !node.local && !configuredNodes[node.id]) + .map((node) => node.id); + const customEndpoints = Object.entries(configuredNodes) + .filter(([nodeId, node]) => nodeId !== localId && node?.endpoint) + .filter(([nodeId, node]) => { + const discovered = discovery.nodes[nodeId]; + if (!discovered?.backendHost) return false; + return !endpointHostMatches(node.endpoint, discovered.backendHost); + }) + .map(([nodeId, node]) => ({ nodeId, endpoint: node.endpoint, backendHost: node.backendHost ?? null })); + const diagnostics = []; + if (configuredNodeIds.length && !configuredNodes[localId]) { + diagnostics.push( + `discovered local node id ${localId} is not the configured cluster.nodeId (${configuredNodeIds.join(', ')}); ` + + `confirm cluster.nodeId before applying so the local node is not duplicated` + ); + } + for (const nodeId of unresolvedPeers) { + diagnostics.push(`discovered peer ${nodeId} is not configured yet`); + } + for (const entry of customEndpoints) { + diagnostics.push( + `Verify the listener for ${entry.nodeId} before moving its configured endpoint onto the observed fabric` + ); + } + for (const ambiguous of discovery.ambiguousPeers ?? []) { + diagnostics.push(ambiguous.message); + } + return { + provider: discovery.provider, + topology: discovery.topology, + nodeId: discovery.nodeId, + localNode: discovery.localNode ?? discovery.nodeId, + leaderNode: currentConfig?.cluster?.leaderNode ?? null, + nodeCount: discovery.nodeCount, + discoveredNodes: discoveredIds, + discoveredPeers: discoveredIds.filter((nodeId) => nodeId !== localId), + observedLinks: discovery.links.length, + configuredNodes: configuredNodeIds.sort(), + newNodes: discoveredIds.filter((nodeId) => !configuredNodes[nodeId] && nodeId !== localId), + unresolvedPeers, + customEndpoints, + mergeRequired: configuredNodeIds.length > 0, + diagnostics, + ...(diagnostics.length ? { diagnostic: diagnostics[0] } : {}) + }; +} + +function endpointHostname(endpoint) { + const value = String(endpoint ?? '').trim(); + if (!value) return null; + try { + return new URL(value).hostname || null; + } catch { + return null; + } +} + +function endpointHostMatches(endpoint, host) { + if (!host) return false; + const hostname = endpointHostname(endpoint); + return Boolean(hostname) && hostname === String(host); +} + +export function mergeNvidiaSyncClusterDiscovery({ discovery, cluster = {}, id, apiKeyEnv, localNodeId } = {}) { + if (!discovery?.detected || discovery.provider !== 'nvidia-sync') { + throw new Error('NVIDIA Sync cluster was not detected'); + } + const current = asObject(cluster); + const nodeId = localNodeId ?? current.nodeId ?? discovery.nodeId; + if (current.nodeId && !String(current.nodeId).includes('${') && current.nodeId !== nodeId) { + throw new Error('Configured local node identity conflicts with discovery'); + } + if (Object.keys(asObject(current.nodes)).length && !current.nodes[nodeId]) { + throw new Error( + `Configured node map does not contain local node ${nodeId}; set cluster.nodeId before applying discovery` + ); + } + const leaderNode = current.leaderNode; + const nodes = { ...asObject(current.nodes) }; + const diagnostics = []; + for (const [observedId, observed] of Object.entries(discovery.nodes)) { + const targetId = observedId === discovery.nodeId ? nodeId : observedId; + if (targetId !== nodeId && !Object.hasOwn(nodes, targetId)) { + diagnostics.push({ + level: 'unconfigured-peer', + nodeId: targetId, + message: `Observed ${targetId}; use cluster add-node with its authenticated gateway URL to join it.` + }); + continue; + } + const existing = asObject(nodes[targetId]); + let endpoint = existing.endpoint; + let backendHost = existing.backendHost ?? observed.backendHost; + if ( + existing.endpoint && + /^10\.100\./.test(existing.backendHost ?? '') && + endpointHostMatches(existing.endpoint, existing.backendHost) + ) { + backendHost = observed.backendHost; + const url = new URL(existing.endpoint); + url.hostname = observed.backendHost; + endpoint = url.toString().replace(/\/$/, existing.endpoint.endsWith('/') ? '/' : ''); + } else if (existing.endpoint && !endpointHostMatches(existing.endpoint, observed.backendHost)) { + diagnostics.push({ + level: 'migration-required', + nodeId: targetId, + message: `Preserved endpoint for ${targetId}; observed fabric address ${observed.backendHost}. Verify its gateway listener before changing the endpoint.` + }); + } + nodes[targetId] = { + ...existing, + name: existing.name ?? targetId, + ...(endpoint ? { endpoint } : {}), + backendHost, + fabricAddress: observed.backendHost, + ...(observed.fabricInterface ? { fabricInterface: observed.fabricInterface } : {}), + ...(observed.sshAlias ? { sshAlias: observed.sshAlias } : {}), + ...(observed.rails ? { rails: observed.rails } : {}), + labels: { + provider: 'nvidia-sync', + role: leaderNode ? (targetId === leaderNode ? 'leader' : 'worker') : 'node', + hardware: 'dgx-spark', + ...asObject(existing.labels) }, - [peerId]: { - id: peerId, - local: false, - backendHost: preferred.peer.hostname, - fabricInterface: preferred.local.interface, - sshAlias: preferred.peer.alias, - sshUser: preferred.peer.user ?? null + resources: { + memoryGb: 128, + accelerators: ['cuda', 'nvidia-gpu', 'blackwell', 'gb10'], + ...asObject(existing.resources) } - } + }; + } + return { + ...current, + id: id ?? current.id ?? `${nodeId}-cluster`, + provider: discovery.provider, + topology: current.topology ?? discovery.topology, + nodeId: current.nodeId ?? nodeId, + ...(leaderNode ? { leaderNode } : {}), + ...(apiKeyEnv ? { apiKeyEnv } : Object.keys(current).length ? {} : { apiKeyEnv: 'LLOOM_CLUSTER_KEY' }), + discovery: { + provider: discovery.provider, + evidence: discovery.evidence, + verified: false, + nodes: discovery.nodes, + links: (discovery.links ?? []).map((link) => ({ ...link, nodeId })) + }, + nodes, + diagnostics }; } @@ -336,7 +650,7 @@ async function tailscaleSelfName() { } } -export async function detectNvidiaSyncCluster({ home = process.env.HOME, hostname = os.hostname() } = {}) { +export async function detectNvidiaSyncCluster({ home = process.env.HOME, hostname = os.hostname(), localNodeId } = {}) { if (process.platform !== 'linux' || !home) return null; let sshConfig; try { @@ -357,42 +671,14 @@ export async function detectNvidiaSyncCluster({ home = process.env.HOME, hostnam return buildNvidiaSyncDiscovery({ peers, localAddresses, - localNodeId: (await tailscaleSelfName()) ?? hostname, + localNodeId: localNodeId ?? (await tailscaleSelfName()) ?? hostname, hostname }); } -export function nvidiaSyncClusterConfig(discovery, { id, port = 8100, apiKeyEnv = 'LLOOM_CLUSTER_KEY' } = {}) { - if (!discovery?.detected || discovery.provider !== 'nvidia-sync') { - throw new Error('NVIDIA Sync cluster was not detected'); - } - const clusterId = id || `${discovery.leaderNode}-cluster`; - return { - id: clusterId, - provider: discovery.provider, - topology: discovery.topology, - nodeId: discovery.nodeId, - leaderNode: discovery.leaderNode, - apiKeyEnv, - nodes: Object.fromEntries( - Object.entries(discovery.nodes).map(([nodeId, node]) => [ - nodeId, - { - name: nodeId, - endpoint: `http://${node.backendHost}:${port}`, - backendHost: node.backendHost, - fabricInterface: node.fabricInterface, - ...(node.sshAlias ? { sshAlias: node.sshAlias } : {}), - labels: { - provider: 'nvidia-sync', - role: nodeId === discovery.leaderNode ? 'leader' : 'worker', - hardware: 'dgx-spark' - }, - resources: { memoryGb: 128, accelerators: ['cuda', 'nvidia-gpu', 'blackwell', 'gb10'] } - } - ]) - ) - }; +// Pure discovery planning; remote gateways join through authenticated add-node. +export function nvidiaSyncClusterConfig(discovery, options = {}) { + return mergeNvidiaSyncClusterDiscovery({ discovery, ...options }); } export function runtimePlacement(runtime, config, env = process.env) { @@ -602,7 +888,7 @@ export function validateClusterConfig(config, env = process.env) { export class ClusterCoordinator { constructor( config, - { env = process.env, fetchImpl = fetch, logger = console, telemetry = null, profile = null, models = null } = {} + { env = process.env, fetchImpl = undiciFetch, logger = console, telemetry = null, profile = null, models = null } = {} ) { this.config = config; this.env = env; @@ -682,7 +968,7 @@ export class ClusterCoordinator { } } - async localNodeStatus({ runtimeStatus = null } = {}) { + async localNodeStatus({ runtimeStatus = null, includeMemoryUsage = false } = {}) { const profile = typeof this.profile === 'function' ? await this.profile() : this.profile; const models = typeof this.models === 'function' ? await this.models() : this.models; return { @@ -704,7 +990,9 @@ export class ClusterCoordinator { telemetry: this.telemetry ? await this.telemetry.snapshot() : null, runtimeManager: runtimeStatus ?? - (this.runtimeManager ? await this.runtimeManager.status({ localOnly: true }) : { runtimes: {} }) + (this.runtimeManager + ? await this.runtimeManager.status({ localOnly: true, includeMemoryUsage }) + : { runtimes: {} }) }; } @@ -717,7 +1005,7 @@ export class ClusterCoordinator { if (!refresh && cached?.pending) return cached.pending; const pending = (async () => { try { - const result = await this.requestNode(nodeId, '/gateway/node'); + const result = await this.requestNode(nodeId, '/gateway/node?memoryUsage=1'); return { ...result.node, id: nodeId, diff --git a/src/config-profiles.mjs b/src/config-profiles.mjs new file mode 100644 index 0000000..16b052c --- /dev/null +++ b/src/config-profiles.mjs @@ -0,0 +1,280 @@ +// Named fleet profiles: one file describes routing + residency for the whole +// deployment; applying it is a single atomic config swap. +// +// A profile lives in /profiles/.json and may set: +// routes. route profile name (alias.routeProfiles key) or a +// plain model/alias id to pin as the sole member +// residency. 'always' | 'preferred' | 'auto' (keep-warm roster) +// defaults optional top-level defaults override (chatModel, ...) +// Applying validates the composed config completely before the atomic write +// (mutateConfigSource validates the staged file), so a swap is all-or-nothing. +import { promises as fs } from 'node:fs'; +import path from 'node:path'; +import { mutateConfigSource } from './config-mutation.mjs'; +import { loadConfig } from './config.mjs'; + +const RESIDENCY = new Set(['always', 'preferred', 'auto']); +const MAX_PROFILE_BYTES = 256 * 1024; + +function fail(message, statusCode = 400) { + return Object.assign(new Error(message), { statusCode }); +} + +function object(value) { + return value && typeof value === 'object' && !Array.isArray(value) ? value : null; +} + +function safeName(name) { + return typeof name === 'string' && /^[a-z0-9][a-z0-9._-]{0,63}$/i.test(name) ? name : null; +} + +function profilesDir(config) { + if (!config.sourcePath) throw fail('Fleet profiles need a file-backed LLooM config.', 409); + return path.join(path.dirname(path.resolve(config.sourcePath)), 'profiles'); +} + +// Route profile semantics copied from route-control: a profile's `members` +// (legacy `target`/`fallbacks`) becomes the alias's complete member list. +export function resolveRouteTarget(alias, target) { + if (!object(alias)) throw fail(`Unknown route alias: ${target && object(target) ? '' : target}`); + const profiles = object(alias.routeProfiles) ?? {}; + if (typeof target === 'string' && profiles[target]) { + const profile = profiles[target]; + const members = Array.isArray(profile.members) ? profile.members : null; + if (!members?.length) throw fail(`Route profile ${target} has no members.`); + const optionalMembers = Array.isArray(profile.optionalMembers) ? profile.optionalMembers : []; + return { activeRoute: target, members, optionalMembers }; + } + if (typeof target !== 'string' || !target.trim()) throw fail('Route target must be an id or profile name.'); + const id = target.trim(); + const known = (candidate) => + candidate === id || (object(alias.members)?.includes ?? (() => false)).call(alias.members, id); + if (!known(id) && !Array.isArray(alias.members)) + throw fail(`Alias has no route profile or member named ${id}.`); + return { activeRoute: null, members: [id], optionalMembers: [] }; +} + +export function normalizeProfileDocument(raw, name) { + const doc = object(raw); + if (!doc) throw fail(`Profile ${name} must be a JSON object.`); + const allowed = new Set(['name', 'description', 'routes', 'residency', 'defaults']); + const unknown = Object.keys(doc).filter((key) => !allowed.has(key)); + if (unknown.length) throw fail(`Profile ${name} has unsupported sections: ${unknown.join(', ')}.`); + const routes = {}; + for (const [aliasId, target] of Object.entries(object(doc.routes) ?? {})) { + if (typeof aliasId !== 'string' || !aliasId.trim() || aliasId.length > 200) + throw fail(`Profile ${name} has an invalid alias id.`); + if (typeof target !== 'string' || !target.trim() || target.length > 500) + throw fail(`Profile ${name}: route for ${aliasId} must be an id or profile name.`); + routes[aliasId] = target.trim(); + } + const residency = {}; + for (const [runtimeId, policy] of Object.entries(object(doc.residency) ?? {})) { + if (typeof runtimeId !== 'string' || !runtimeId.trim() || runtimeId.length > 200) + throw fail(`Profile ${name} has an invalid runtime id.`); + if (!RESIDENCY.has(policy)) throw fail(`Profile ${name}: residency for ${runtimeId} must be always, preferred, or auto.`); + residency[runtimeId.trim()] = policy; + } + const defaults = object(doc.defaults); + if (defaults) { + for (const [key, value] of Object.entries(defaults)) { + if (typeof value !== 'string' || value.length > 500) throw fail(`Profile ${name}: defaults.${key} must be a short string.`); + } + } + return { + name: typeof doc.name === 'string' ? doc.name : name, + description: typeof doc.description === 'string' ? doc.description.slice(0, 500) : '', + routes, + residency, + defaults: defaults ? { ...defaults } : null + }; +} + +export function profilePaths(config) { + return { dir: profilesDir(config), fileFor: (name) => path.join(profilesDir(config), name + '.json') }; +} + +export async function listProfiles(config) { + const dir = profilesDir(config); + let names = []; + try { + names = (await fs.readdir(dir)).filter((name) => name.endsWith('.json')).map((name) => name.slice(0, -5)); + } catch (error) { + if (error.code !== 'ENOENT') throw error; + } + names.sort((left, right) => left.localeCompare(right)); + const active = await activeProfileName(config); + const profiles = []; + for (const name of names) { + if (!safeName(name)) continue; + try { + profiles.push(await readProfile(config, name, active)); + } catch { + profiles.push({ name, error: 'unreadable profile' }); + } + } + return { dir, active, profiles }; +} + +export async function readProfile(config, name, activeName = null) { + const safe = safeName(name); + if (!safe) throw fail('Invalid profile name.'); + const file = path.join(profilesDir(config), safe + '.json'); + const raw = await fs.readFile(file, 'utf8'); + if (raw.length > MAX_PROFILE_BYTES) throw fail('Profile file is too large.', 413); + const doc = normalizeProfileDocument(JSON.parse(raw), safe); + return { ...doc, file, active: activeName != null ? safe === activeName : undefined }; +} + +async function readRawProfile(config, name) { + const safe = safeName(name); + if (!safe) throw fail('Invalid profile name.'); + const file = path.join(profilesDir(config), safe + '.json'); + const raw = await fs.readFile(file, 'utf8'); + if (raw.length > MAX_PROFILE_BYTES) throw fail('Profile file is too large.', 413); + return { safe, doc: JSON.parse(raw), file }; +} + +// The active profile name is recorded in the config under fleet.activeProfile +// when applied; hand-written configs simply have no marker. +async function activeProfileName(config) { + const raw = JSON.parse(await fs.readFile(path.resolve(config.sourcePath), 'utf8')); + const marker = object(raw.fleet)?.activeProfile; + return safeName(marker) ? marker : null; +} + +export function planProfileChanges(config, doc) { + const changes = { routes: [], residency: [], defaults: [], unchanged: [] }; + const aliases = object(config.aliases) ?? {}; + for (const [aliasId, target] of Object.entries(doc.routes)) { + const alias = aliases[aliasId]; + if (!object(alias)) throw fail(`Config has no alias ${aliasId}.`, 409); + const resolved = resolveRouteTarget(alias, target); + const same = + JSON.stringify(alias.members ?? []) === JSON.stringify(resolved.members) && + JSON.stringify(alias.optionalMembers ?? []) === JSON.stringify(resolved.optionalMembers) && + (alias.activeRoute ?? null) === resolved.activeRoute; + if (same) changes.unchanged.push({ kind: 'route', id: aliasId, value: target }); + else changes.routes.push({ id: aliasId, from: alias.activeRoute ?? alias.members, to: target, ...resolved }); + } + const runtimes = object(config.runtimes) ?? {}; + for (const [runtimeId, policy] of Object.entries(doc.residency)) { + const runtime = runtimes[runtimeId]; + if (!object(runtime)) throw fail(`Config has no runtime ${runtimeId}.`, 409); + const current = runtime.keepWarm === true ? 'always' : runtime.preferredWarm === true ? 'preferred' : 'auto'; + if (current === policy) changes.unchanged.push({ kind: 'residency', id: runtimeId, value: policy }); + else changes.residency.push({ id: runtimeId, from: current, to: policy }); + } + for (const [key, value] of Object.entries(doc.defaults ?? {})) { + const current = object(config.defaults)?.[key]; + if (current === value) changes.unchanged.push({ kind: 'default', id: key, value }); + else changes.defaults.push({ id: key, from: current ?? null, to: value }); + } + return changes; +} + +// Compose the profile onto the raw source. Pure and synchronous so +// mutateConfigSource can stage + validate the exact candidate. +export function composeProfile(raw, doc, name) { + const fleet = object(raw.fleet) ?? {}; + raw.fleet = { ...fleet, activeProfile: name }; + const aliases = object(raw.aliases) ?? {}; + for (const [aliasId, target] of Object.entries(doc.routes)) { + const alias = object(aliases[aliasId]); + if (!alias) throw fail(`Config has no alias ${aliasId}.`, 409); + const resolved = resolveRouteTarget({ ...alias }, target); + alias.members = resolved.members; + if (resolved.optionalMembers.length) alias.optionalMembers = resolved.optionalMembers; + else delete alias.optionalMembers; + if (resolved.activeRoute) alias.activeRoute = resolved.activeRoute; + else delete alias.activeRoute; + // A fresh route invalidates stale per-member suspensions. + delete alias.suspendedMembers; + aliases[aliasId] = alias; + } + raw.aliases = aliases; + const runtimes = object(raw.runtimes) ?? {}; + for (const [runtimeId, policy] of Object.entries(doc.residency)) { + const runtime = object(runtimes[runtimeId]); + if (!runtime) throw fail(`Config has no runtime ${runtimeId}.`, 409); + runtime.keepWarm = policy === 'always'; + runtime.preferredWarm = policy === 'preferred'; + runtimes[runtimeId] = runtime; + } + raw.runtimes = runtimes; + if (doc.defaults) raw.defaults = { ...(object(raw.defaults) ?? {}), ...doc.defaults }; + return raw; +} + +export function createFleetProfileController({ getConfig, reload, env = process.env }) { + async function apply(name, { yes = false } = {}) { + if (yes !== true) throw fail('Review the profile and confirm with yes: true.'); + if (!getConfig().sourcePath) throw fail('This gateway has no writable installed configuration.', 409); + const { safe, doc } = await readRawProfile(getConfig(), name); + const profile = normalizeProfileDocument(doc, safe); + let outcome = null; + await mutateConfigSource(getConfig(), (raw) => { + // Plan against raw source state for accurate reporting, then compose. + outcome = planProfileChanges({ ...raw, sourcePath: getConfig().sourcePath }, profile); + composeProfile(raw, profile, safe); + }); + reload?.(); + return { profile: safe, ...outcome }; + } + + async function save(name, { description = '', overwrite = false, yes = false } = {}) { + if (yes !== true) throw fail('Confirm capturing the current configuration with yes: true.'); + const safe = safeName(name); + if (!safe) throw fail('Profile names use letters, numbers, dots, dashes, underscores.'); + const config = getConfig(); + if (!config.sourcePath) throw fail('This gateway has no file-backed configuration.', 409); + const dir = profilesDir(config); + await fs.mkdir(dir, { recursive: true }); + const file = path.join(dir, safe + '.json'); + if (!overwrite) { + try { + await fs.access(file); + throw fail(`Profile ${safe} already exists; pass overwrite: true to replace it.`, 409); + } catch (error) { + if (error.statusCode === 409) throw error; + if (error.code !== 'ENOENT') throw error; + } + } + const source = await loadConfig(config.sourcePath); + const routes = {}; + for (const [aliasId, alias] of Object.entries(object(source.aliases) ?? {})) { + if (Array.isArray(alias.members) && (alias.routeProfiles || alias.activeRoute)) { + routes[aliasId] = alias.activeRoute ?? alias.members[0]; + } + } + const residency = {}; + for (const [runtimeId, runtime] of Object.entries(object(source.runtimes) ?? {})) { + if (runtime.keepWarm === true) residency[runtimeId] = 'always'; + else if (runtime.preferredWarm === true) residency[runtimeId] = 'preferred'; + } + const doc = { + name: safe, + description: String(description).slice(0, 500), + routes, + residency, + defaults: object(source.defaults) ? { ...object(source.defaults) } : undefined + }; + for (const key of Object.keys(doc)) if (doc[key] == null) delete doc[key]; + const tmp = file + '.tmp-' + process.pid; + await fs.writeFile(tmp, JSON.stringify(doc, null, 2) + '\n'); + await fs.rename(tmp, file); + return { profile: safe, file, routes: Object.keys(routes).length, residency: Object.keys(residency).length }; + } + + return { + list: () => listProfiles(getConfig()), + read: (name) => readProfile(getConfig(), name), + plan: async (name) => { + const { safe, doc } = await readRawProfile(getConfig(), name); + const profile = normalizeProfileDocument(doc, safe); + return { profile: safe, ...planProfileChanges(getConfig(), profile) }; + }, + apply, + save + }; +} diff --git a/src/config.mjs b/src/config.mjs index f496a2f..9ed7402 100644 --- a/src/config.mjs +++ b/src/config.mjs @@ -370,6 +370,30 @@ function validateConfig(config, sourcePath, env) { } } + const memorySafety = config.runtimePolicy?.memorySafety; + if (memorySafety != null) { + if (typeof memorySafety !== 'object' || Array.isArray(memorySafety)) { + errors.push('runtimePolicy.memorySafety must be an object'); + } else { + if (memorySafety.mode != null && !['enforce', 'yolo'].includes(memorySafety.mode)) { + errors.push('runtimePolicy.memorySafety.mode must be enforce or yolo'); + } + for (const key of ['minAvailableMemoryGb', 'maxMemoryUtilization', 'pollIntervalMs']) { + const value = memorySafety[key]; + if (value == null) continue; + if ( + typeof value !== 'number' || + !Number.isFinite(value) || + value <= 0 || + (key === 'maxMemoryUtilization' && value >= 1) || + (key === 'pollIntervalMs' && (value < 50 || value > 1000)) + ) { + errors.push(`runtimePolicy.memorySafety.${key} is outside its safe range`); + } + } + } + } + const distributedMemberGroups = new Map(); for (const [runtimeId, runtime] of Object.entries(config.runtimes ?? {})) { if (runtime?.placement?.mode !== 'distributed') continue; diff --git a/src/darwin-memory-usage.py b/src/darwin-memory-usage.py new file mode 100644 index 0000000..5944cc9 --- /dev/null +++ b/src/darwin-memory-usage.py @@ -0,0 +1,41 @@ +"""Read Darwin's per-process physical footprint; never inspect process contents. + +The rusage_info_v2 ABI is defined by Apple's sys/resource.h (macOS 10.9+). +Physical footprint includes charged memory that RSS omits, notably Metal and +compressed allocations. A failed per-PID observation stays absent. +""" +import ctypes +import json +import sys + + +class RusageInfoV2(ctypes.Structure): + _fields_ = [('uuid', ctypes.c_uint8 * 16)] + [ + (name, ctypes.c_uint64) for name in ( + 'user_time', 'system_time', 'pkg_idle_wkups', 'interrupt_wkups', + 'pageins', 'wired_size', 'resident_size', 'phys_footprint', + 'proc_start_abstime', 'proc_exit_abstime', 'child_user_time', + 'child_system_time', 'child_pkg_idle_wkups', 'child_interrupt_wkups', + 'child_pageins', 'child_elapsed_abstime', 'diskio_bytesread', + 'diskio_byteswritten' + ) + ] + + +def main(): + libproc = ctypes.CDLL('/usr/lib/libproc.dylib') + libproc.proc_pid_rusage.argtypes = [ctypes.c_int, ctypes.c_int, ctypes.c_void_p] + libproc.proc_pid_rusage.restype = ctypes.c_int + result = {} + for value in sys.argv[1:]: + pid = int(value) + if not 0 < pid < 2 ** 31: + continue + usage = RusageInfoV2() + if libproc.proc_pid_rusage(pid, 2, ctypes.byref(usage)) == 0: + result[str(pid)] = usage.phys_footprint + print(json.dumps(result)) + + +if __name__ == '__main__': + main() diff --git a/src/dashboard-installation.mjs b/src/dashboard-installation.mjs new file mode 100644 index 0000000..367b83f --- /dev/null +++ b/src/dashboard-installation.mjs @@ -0,0 +1,155 @@ +import { createHash } from 'node:crypto'; +import fs from 'node:fs/promises'; +import { createInstallationJobs } from './installation-jobs.mjs'; +import { createModelImportPlan } from './model-intake.mjs'; +import { createSetupPlan, applySetup } from './setup.mjs'; +import { loadRecipeById } from './recipes.mjs'; +import { loadBackendCatalog, getBackend, planBackend } from './backend-catalog.mjs'; +import { loadConfig } from './config.mjs'; +import { mutateConfigSource } from './config-mutation.mjs'; +import { pinDownloadCommands } from './installer.mjs'; +import { installImportedModelAssets } from './model-installation.mjs'; +import { validateAcquisitionStep } from './model-acquisition.mjs'; + +const digest = (value) => createHash('sha256').update(JSON.stringify(value)).digest('hex'); +const invalid = (message) => Object.assign(new Error(message), { statusCode: 409 }); +export function createDashboardInstallation({ getConfig, reload, env = process.env }) { + const source = async () => { + if (!getConfig().sourcePath) throw invalid('Installations need a writable LLooM configuration.'); + return loadConfig(getConfig().sourcePath); + }; + const evidence = async (recipeId) => ({ + catalog: await loadBackendCatalog(), + recipe: recipeId ? await loadRecipeById(recipeId) : null + }); + return createInstallationJobs({ + async plan(input) { + if (!input || Object.keys(input).some((key) => !['recipeId', 'modelRef', 'backend', 'name'].includes(key))) + throw invalid('Choose a vendor recipe or model reference.'); + if (Boolean(input.recipeId) === Boolean(input.modelRef)) + throw invalid('Choose exactly one recipe or model reference.'); + for (const value of Object.values(input)) + if (typeof value !== 'string' || value.length > 2000) throw invalid('Invalid model reference.'); + const config = await source(); + const baseline = digest(JSON.parse(await fs.readFile(config.sourcePath, 'utf8'))); + const sources = await evidence(input.recipeId); + const options = { + ...input, + configPath: config.sourcePath, + additive: true, + offline: true, + start: false, + includeRuntimes: false + }; + const plan = input.recipeId ? await createSetupPlan(config, options) : createModelImportPlan(config, options); + if ( + !input.recipeId && + plan.reference.type === 'huggingface' && + !/^[a-f0-9]{40,64}$/i.test(plan.reference.revision || '') + ) + throw invalid( + 'Use a Hugging Face file or repository link pinned to a commit, such as /tree/. Branches and latest references cannot bind a reviewed installation.' + ); + if (!input.recipeId && plan.additions.runtimeId) { + const backend = getBackend(sources.catalog, plan.inference.backend); + if (!backend) throw invalid('Unknown vendor backend.'); + plan.backendPlan = await planBackend(backend); + } + if (!input.recipeId && plan.download?.command?.[0] === 'hf') { + const command = plan.download.command; + const pinned = await pinDownloadCommands( + { + steps: [ + { + action: 'download-model', + provider: 'huggingface', + model: command[2], + revision: plan.reference.revision, + include: plan.reference.filePath ? [plan.reference.filePath] : [], + destination: command[command.indexOf('--local-dir') + 1], + command, + commands: [command] + } + ] + }, + { env, requireAvailable: true } + ); + const errors = validateAcquisitionStep(pinned.steps[0]); + if (plan.reference.filePath?.startsWith('-')) + errors.push('Model file names cannot begin with a command option.'); + if (errors.length) throw invalid(errors.join('; ')); + plan.download.acquisition = pinned.steps[0]; + plan.download.command = pinned.steps[0].command; + delete plan.download.shellCommand; + } + return { + input, + options, + plan, + baseline, + sources, + digest: digest(sources), + view: { + kind: input.recipeId ? 'recipe' : 'model', + summary: + 'Installs the vendor backend and model files, then adds the model to this gateway. Loading uses normal memory admission.', + details: input.recipeId + ? { + recipe: plan.selectedRecipe, + configPath: plan.configPath, + modelRoot: plan.modelRoot, + ports: plan.ports, + backend: plan.phases.bootstrap.backend, + models: plan.phases.bootstrap.recipe, + integrations: plan.phases.bootstrap.integrations + } + : { + reference: plan.reference, + backend: plan.inference, + backendInstallation: plan.backendPlan, + additions: plan.additions, + download: plan.download, + configPath: plan.configPath + } + } + }; + }, + async apply(prepared, onProgress) { + if (digest(await evidence(prepared.input.recipeId)) !== prepared.digest) + throw invalid('The vendor recipe changed. Review a fresh plan.'); + const config = await source(); + if (digest(JSON.parse(await fs.readFile(config.sourcePath, 'utf8'))) !== prepared.baseline) + throw invalid('Your configuration changed. Review a fresh installation plan.'); + const writeConfig = async (file, value) => { + if (file !== config.sourcePath) throw invalid('The installation destination changed.'); + await mutateConfigSource(config, (raw) => { + if (digest(raw) !== prepared.baseline) + throw invalid( + 'Your configuration changed during installation. Files are retained; review a fresh plan to continue.' + ); + for (const key of Object.keys(raw)) delete raw[key]; + Object.assign(raw, value); + }); + }; + let result; + if (prepared.input.recipeId) { + result = await applySetup(config, { + ...prepared.options, + reviewedPlan: prepared.plan, + dryRun: false, + yes: true, + onProgress, + writeConfig + }); + } else { + await installImportedModelAssets(prepared.plan, { onProgress, backendCatalog: prepared.sources.catalog }); + await writeConfig(config.sourcePath, prepared.plan.config); + result = { ok: true }; + } + // Even failed bootstrap can have written configuration. Surface reload + // failures, and make the installed/failed state visible in the dashboard. + await reload(); + return result; + } + }); +} diff --git a/src/dashboard-memory.mjs b/src/dashboard-memory.mjs new file mode 100644 index 0000000..125e8fc --- /dev/null +++ b/src/dashboard-memory.mjs @@ -0,0 +1,433 @@ +function buildMemoryMap({ node = null, runtimes = {}, models = [], previewModelId = null, memorySafety = null } = {}) { + if (node?.local !== true && node?.runtimeManager?.runtimes) { + runtimes = { ...runtimes }; + for (const [id, observed] of Object.entries(node.runtimeManager.runtimes)) { + runtimes[id] = { ...runtimes[id], ...observed, node: node.id, remote: false }; + } + } + const GiB = 1024 * 1024 * 1024; + + function numberOrNull(value) { + return typeof value === 'number' && Number.isFinite(value) ? value : null; + } + + function nonNegativeOrNull(value) { + const number = numberOrNull(value); + return number !== null && number >= 0 ? number : null; + } + + function bounded(value, minimum, maximum) { + const number = numberOrNull(value); + return number === null || number < minimum || number > maximum ? null : number; + } + + function positiveOrNull(value) { + const number = numberOrNull(value); + return number !== null && number > 0 ? number : null; + } + + function sameNode(runtime, currentNode) { + if ( + !runtime || + runtime.remote === true || + runtime.distributed === true || + runtime.placement?.mode === 'distributed' + ) + return false; + const placementNode = runtimeNode(runtime); + if (placementNode) return placementNode === currentNode; + return node?.local === true; + } + + function runtimeNode(runtime) { + return runtime?.node ?? runtime?.placement?.node ?? null; + } + + function safetyReserve(policy, total) { + if (!policy) return null; + if (policy.mode === 'yolo') return 0; + const minAvailable = nonNegativeOrNull(policy.minAvailableMemoryGb); + const maxUtilization = nonNegativeOrNull(policy.maxMemoryUtilization); + if (minAvailable === null || maxUtilization === null || maxUtilization > 1) return null; + return Math.min(total, Math.max(minAvailable * GiB, total * (1 - maxUtilization))); + } + + function colorIndex(value) { + const text = String(value ?? ''); + let hash = 2166136261; + for (let index = 0; index < text.length; index += 1) { + hash ^= text.charCodeAt(index); + hash = (hash * 16777619) >>> 0; + } + return hash % 8; + } + + function targetNodes(model) { + return (Array.isArray(model?.targets) ? model.targets : []) + .map((target) => { + return typeof target === 'object' ? (target?.node ?? null) : null; + }) + .filter(Boolean); + } + + function modelRuntimeId(model) { + if (model?.runtime) return model.runtime; + const ids = uniqueSorted( + (model?.targets ?? []) + .filter((target) => !target.node || target.node === nodeId) + .map((target) => target.remoteRuntime ?? target.runtime) + ); + return ids.length === 1 ? ids[0] : null; + } + + function uniqueSorted(values) { + return [...new Set(values.filter(Boolean))].sort((left, right) => String(left).localeCompare(String(right))); + } + + function roundPercent(value) { + return Number.isFinite(value) ? Math.round(value * 1e6) / 1e6 : null; + } + + const nodeId = node?.id ?? null; + const telemetryMemory = node?.telemetry?.memory; + const totalBytes = nonNegativeOrNull(telemetryMemory?.totalBytes); + const availableBytes = telemetryMemory ? bounded(telemetryMemory.availableBytes, 0, totalBytes) : null; + const reportedUsedBytes = telemetryMemory ? bounded(telemetryMemory.usedBytes, 0, totalBytes) : null; + const hasCapacity = typeof totalBytes === 'number' && totalBytes > 0; + const invalidAvailable = telemetryMemory?.availableBytes != null && availableBytes === null; + const invalidUsed = telemetryMemory?.usedBytes != null && reportedUsedBytes === null; + const hasMemoryFacts = + hasCapacity && !invalidAvailable && !invalidUsed && (availableBytes !== null || reportedUsedBytes !== null); + const known = + (node?.reachable === true || (node?.local === true && node?.reachable !== false)) && hasCapacity && hasMemoryFacts; + let usedBytes; + let availableMemoryBytes; + if (hasCapacity && availableBytes !== null) { + availableMemoryBytes = availableBytes; + usedBytes = totalBytes - availableBytes; + } else if (hasCapacity && reportedUsedBytes !== null) { + usedBytes = reportedUsedBytes; + availableMemoryBytes = totalBytes - usedBytes; + } else { + usedBytes = null; + availableMemoryBytes = null; + } + const activePolicy = node?.local === true ? memorySafety : node?.runtimeManager?.memorySafety; + const reserveBytes = hasMemoryFacts ? safetyReserve(activePolicy, totalBytes) : null; + const usableBytes = hasMemoryFacts && reserveBytes !== null ? Math.max(0, availableMemoryBytes - reserveBytes) : null; + + if (!known) { + return { + known: false, + totalBytes: hasCapacity ? totalBytes : null, + usedBytes: null, + availableBytes: hasCapacity ? availableBytes : null, + reserveBytes: null, + usableBytes: null, + nodeId, + segments: [], + attributionNote: 'Memory telemetry is unavailable; no allocation is inferred.', + preview: null + }; + } + + const runtimeIds = Object.keys(runtimes ?? {}).sort((left, right) => left.localeCompare(right)); + const includedRuntimeIds = []; + for (const runtimeId of runtimeIds) { + const runtime = runtimes[runtimeId]; + const status = String(runtime?.status ?? '').toLowerCase(); + const active = ['running', 'healthy', 'starting', 'warming', 'external', 'stopping', 'draining'].includes(status); + if (!active || runtime?.paused === true) continue; + if (!sameNode(runtime, nodeId)) continue; + includedRuntimeIds.push(runtimeId); + } + + const runtimeGroups = new Map(); + for (const runtimeId of includedRuntimeIds) { + const runtime = runtimes[runtimeId]; + const usage = runtime?.memoryUsage; + const groupId = typeof usage?.groupId === 'string' && usage.groupId ? usage.groupId : `runtime:${runtimeId}`; + if (!runtimeGroups.has(groupId)) { + runtimeGroups.set(groupId, { runtimeIds: [], usage: null }); + } + const group = runtimeGroups.get(groupId); + group.runtimeIds.push(runtimeId); + if (!group.usage) group.usage = usage ?? null; + } + + const attributed = []; + for (const [groupId, group] of runtimeGroups) { + const primaryRuntimeId = group.runtimeIds[0]; + const runtime = runtimes[primaryRuntimeId]; + const usages = group.runtimeIds.map((runtimeId) => runtimes[runtimeId]?.memoryUsage).filter(Boolean); + const rssValues = usages.map((usage) => nonNegativeOrNull(usage.residentBytes)).filter((value) => value !== null); + const rssBytes = rssValues.length ? Math.max(...rssValues) : null; + const estimateValues = group.runtimeIds + .map((runtimeId) => positiveOrNull(runtimes[runtimeId]?.memoryGb)) + .filter((value) => value !== null) + .map((value) => value * GiB); + const bytes = rssBytes !== null ? rssBytes : estimateValues.length ? Math.max(...estimateValues) : null; + if (bytes === null) continue; + const modelIds = new Set(); + for (const runtimeId of group.runtimeIds) { + for (const model of models ?? []) { + if (modelRuntimeId(model) === runtimeId) { + if (model.id) modelIds.add(model.id); + } + } + } + const names = uniqueSorted([...modelIds].map((id) => models.find((model) => model.id === id)?.name || id)); + const label = names.length + ? names.slice(0, 2).join(' + ') + (names.length > 2 ? ' +' + (names.length - 2) : '') + : (runtime?.id ?? primaryRuntimeId); + const estimated = rssBytes === null; + attributed.push({ + id: groupId, + kind: 'runtime', + label, + bytes: Math.round(bytes), + percent: roundPercent((bytes / totalBytes) * 100), + runtimeId: primaryRuntimeId, + modelIds: uniqueSorted([...modelIds]), + estimated + }); + } + + let reconciliationScale = 1; + let attributedBytes = attributed.reduce((sum, segment) => sum + segment.bytes, 0); + if (attributedBytes > usedBytes) { + reconciliationScale = usedBytes / attributedBytes; + for (const segment of attributed) { + segment.bytes = Math.round(segment.bytes * reconciliationScale); + segment.estimated = true; + } + attributedBytes = attributed.reduce((sum, segment) => sum + segment.bytes, 0); + } + attributed.sort((left, right) => String(left.runtimeId).localeCompare(String(right.runtimeId))); + const colors = new Set(); + for (const segment of attributed) { + let index = colorIndex(segment.runtimeId); + if (colors.size < 8) while (colors.has(index)) index = (index + 1) % 8; + colors.add(index); + segment.colorIndex = index; + segment.percent = roundPercent((segment.bytes / totalBytes) * 100); + } + + if (attributedBytes > usedBytes && attributed.length) { + const excess = attributedBytes - usedBytes; + const largest = attributed.reduce((left, right) => (right.bytes > left.bytes ? right : left)); + largest.bytes = Math.max(0, largest.bytes - excess); + attributedBytes = attributed.reduce((sum, segment) => sum + segment.bytes, 0); + } + for (const segment of attributed) segment.percent = roundPercent((segment.bytes / totalBytes) * 100); + const systemBytes = Math.max(0, usedBytes - attributedBytes); + + const segments = [...attributed]; + if (systemBytes > 0 || usedBytes === 0) { + segments.push({ + id: 'system', + kind: 'system', + label: 'System & other apps', + bytes: systemBytes, + percent: roundPercent((systemBytes / totalBytes) * 100), + runtimeId: null, + modelIds: [], + estimated: false, + colorIndex: 8 + }); + } + const availableSegmentBytes = totalBytes - usedBytes; + segments.push({ + id: 'available', + kind: 'available', + label: 'Available', + bytes: availableSegmentBytes, + percent: roundPercent((availableSegmentBytes / totalBytes) * 100), + runtimeId: null, + modelIds: [], + estimated: false, + colorIndex: 9 + }); + + const attributionParts = ['Live process memory is approximate. Shared models use one block. Previews are estimates.']; + if (reconciliationScale !== 1) + attributionParts.push('App measurements overlap; blocks are adjusted to match total memory in use.'); + if ( + [...runtimeGroups.values()].some((group) => + group.runtimeIds.every((runtimeId) => { + const usage = runtimes[runtimeId]?.memoryUsage; + return !usage || nonNegativeOrNull(usage.residentBytes) === null; + }) + ) + ) + attributionParts.push('Striped blocks use an estimate until a live reading is available.'); + + const preview = previewModelId ? previewModel(previewModelId) : null; + function previewModel(modelId) { + const model = (models ?? []).find((item) => item?.id === modelId); + if (!model) return null; + const runtimeId = modelRuntimeId(model); + const runtime = runtimeId ? runtimes[runtimeId] : null; + const label = model.name ?? model.id; + const base = { + modelId, + label, + runtimeId, + status: 'unknown', + additionalBytes: null, + remainingBytes: null, + projectedUsedBytes: null, + percent: null, + nodeId, + shared: false, + message: '' + }; + const targets = targetNodes(model); + const otherTargets = targets.filter((target) => target !== nodeId); + const allTargetsAreOther = targets.length > 0 && otherTargets.length === targets.length; + const runtimePlacementNode = runtime ? runtimeNode(runtime) : null; + const runtimeIsOtherNode = + (runtime?.remote === true && runtimePlacementNode === null) || + (runtimePlacementNode !== null && runtimePlacementNode !== nodeId); + if (runtimeIsOtherNode || allTargetsAreOther) { + return { + ...base, + status: 'other-node', + additionalBytes: 0, + projectedUsedBytes: usedBytes, + remainingBytes: availableMemoryBytes, + percent: roundPercent((usedBytes / totalBytes) * 100), + nodeId: targets[0] ?? runtimePlacementNode, + message: `Runs on node ${targets[0] ?? runtimePlacementNode ?? 'another node'}; no impact on this machine.` + }; + } + if (runtime?.distributed === true || runtime?.placement?.mode === 'distributed' || new Set(targets).size > 1) { + return { ...base, status: 'unknown', shared: true, message: 'Needs room on each machine.' }; + } + if ( + runtime?.maintenance || + runtime?.enabled === false || + runtime?.paused === true || + runtime?.status === 'paused' || + model.paused === true + ) { + return { ...base, status: 'paused', message: 'Paused; automatic starts are disabled.' }; + } + if (!runtimeId && (model.federated || targets.length > 0)) + return { ...base, status: 'unknown', message: 'Waiting for model memory from this machine.' }; + if (!runtimeId) { + return { + ...base, + status: 'external', + additionalBytes: 0, + projectedUsedBytes: usedBytes, + remainingBytes: availableMemoryBytes, + percent: roundPercent((usedBytes / totalBytes) * 100), + message: 'No local model load is expected.' + }; + } + if (!runtime) { + return { ...base, status: 'unknown', message: 'No runtime status is available.' }; + } + const usage = runtime.memoryUsage; + const loadedIds = Array.isArray(usage?.loadedModelIds) ? usage.loadedModelIds : null; + const wantedId = model.upstreamModel ?? model.id; + const residentConfirmed = + runtime.healthy === true && + usage?.residencyKnown === true && + loadedIds !== null && + (loadedIds.includes(wantedId) || loadedIds.includes(model.id)); + const commandParts = String(runtime.command ?? '') + .trim() + .split(/\s+/); + const commandBase = commandParts[0]?.split(/[\\/]/).pop() ?? ''; + const lazyBackend = [commandBase, ...(runtime.args ?? [])].some((part) => + /(?:^|[/\\])(ollama|lloom-audio-server|lloom_audio_server(?:\.py)?)$/.test(String(part)) + ); + const sharedBackend = usage?.sharedRuntimeIds?.length > 1 || lazyBackend; + const modelsOnRuntime = (models ?? []).filter((item) => modelRuntimeId(item) === runtimeId); + const runtimeIsSharedGroup = runtimeGroups.get(usage?.groupId ?? `runtime:${runtimeId}`)?.runtimeIds.length > 1; + if (residentConfirmed) { + return { + ...base, + status: 'resident', + additionalBytes: 0, + projectedUsedBytes: usedBytes, + remainingBytes: availableMemoryBytes, + percent: roundPercent((usedBytes / totalBytes) * 100), + shared: sharedBackend || runtimeIsSharedGroup, + message: + sharedBackend || runtimeIsSharedGroup + ? 'Already available; shared backend footprint can change.' + : 'Already available' + }; + } + if (sharedBackend && runtime.healthy === true && !(usage?.residencyKnown && loadedIds !== null)) { + return { ...base, status: 'unknown', shared: true, message: 'Shared backend residency is unknown.' }; + } + const memoryGb = positiveOrNull(runtime.memoryGb); + if (memoryGb === null) { + return { ...base, status: 'unknown', message: 'Cold start estimate is unknown.' }; + } + // A healthy server may load weights lazily. Only confirmed model-cache + // observations promise residency; otherwise forecast growth toward its peak. + const measured = + !sharedBackend && !runtimeIsSharedGroup && modelsOnRuntime.length === 1 && runtime.healthy === true + ? nonNegativeOrNull(usage?.residentBytes) + : null; + const incoming = Math.max(0, memoryGb * GiB - (measured ?? 0)); + const remaining = availableMemoryBytes - incoming; + const reserve = reserveBytes ?? null; + if (reserve === null) { + return { + ...base, + status: 'unknown', + additionalBytes: incoming, + remainingBytes: remaining, + projectedUsedBytes: usedBytes + incoming, + percent: roundPercent((incoming / totalBytes) * 100), + shared: sharedBackend, + message: 'Memory reserve is unknown; no fit is guaranteed.' + }; + } + if (remaining <= reserve) { + return { + ...base, + status: 'blocked', + additionalBytes: incoming, + remainingBytes: remaining, + projectedUsedBytes: usedBytes + incoming, + percent: roundPercent((incoming / totalBytes) * 100), + shared: sharedBackend, + message: 'Does not fit within the reserve.' + }; + } + const tightLimit = reserve + Math.max(GiB, totalBytes * 0.02); + const tight = remaining <= tightLimit; + return { + ...base, + status: tight ? 'tight' : 'fits', + additionalBytes: incoming, + remainingBytes: remaining, + projectedUsedBytes: usedBytes + incoming, + percent: roundPercent((incoming / totalBytes) * 100), + shared: sharedBackend, + message: tight ? 'Expected to fit, with little reserve headroom.' : 'Expected to fit' + }; + } + + return { + known: true, + totalBytes, + usedBytes, + availableBytes: availableMemoryBytes, + reserveBytes, + usableBytes, + nodeId, + segments, + attributionNote: attributionParts.join(' '), + preview + }; +} + +export { buildMemoryMap }; diff --git a/src/dashboard-presence-client.mjs b/src/dashboard-presence-client.mjs new file mode 100644 index 0000000..5289a96 --- /dev/null +++ b/src/dashboard-presence-client.mjs @@ -0,0 +1,319 @@ +// Runs inside the dashboard's existing script, sharing its authenticated API +// helpers and observed state. It never invents topology or hardware readings. +export const presenceScript = String.raw` + let presenceView = "live"; + let presenceImport = null; + let presencePlanVersion = 0; + let presenceLastFocus = null; + let presenceBusy = false; + let presenceRenderKey = ""; + let presenceMachineKey = ""; + let presenceInstallTimer = null; + let presenceInstallSeen = null; + const presencePolicyNames = { auto:"Automatic", preferred:"Prefer instant replies", always:"Keep ready" }; + function presenceNotice(message, error = false) { + const toast = $("#presence-toast"); + toast.querySelector("span").textContent = message; + toast.dataset.error = String(error); + toast.hidden = false; + } + function presenceSetView(name) { + if (!["live","models","machines","clients","settings"].includes(name)) name = "live"; + presenceView = name; + document.querySelectorAll(".presence-nav [data-view]").forEach(button => { + if (button.dataset.view === name) button.setAttribute("aria-current","page"); + else button.removeAttribute("aria-current"); + }); + document.querySelectorAll("[data-presence-panel]").forEach(panel => panel.hidden = panel.dataset.presencePanel !== name); + for (const id of ["models","machines","clients"]) $("#view-" + id).hidden = id !== name; + $(".operations-dock").open = name === "settings"; + closeModelInspector(); closeNodeInspector(); + if (name === "clients") presenceLoadIntegrations(); + try { history.replaceState(null,"","#" + name); } catch {} + renderPresence(); + window.scrollTo({top:0,behavior:"instant"}); + } + function presenceRuntime(model) { + return model.runtime ? state.status?.runtimeManager?.runtimes?.[model.runtime] : null; + } + function presencePolicy(runtime) { + return runtime?.keepWarm ? "always" : runtime?.preferredWarm ? "preferred" : "auto"; + } + function presenceModelResident(model) { + const rt=presenceRuntime(model),usage=rt?.memoryUsage; + if(!rt?.healthy)return false; + if(usage?.residencyKnown&&Array.isArray(usage.loadedModelIds))return usage.loadedModelIds.some(id=>id===model.id||id===model.upstreamModel); + return false; + } + function presenceModelLabel(model) { + const runtime = presenceRuntime(model); + if (model.alias) return "Route"; + if (!model.runtime) return model.federated ? "Shared model" : "External provider"; + if (runtime?.maintenance) return "Paused"; + const transition={starting:"Getting ready",warming:"Getting ready",queued:"Waiting for room",stopping:"Freeing memory",draining:"Finishing work",failed:"Needs attention",unreachable:"Unavailable",disabled:"Disabled"}[runtime?.status]; + if(transition)return transition; + if (runtime?.activeRequests > 0 && runtime?.healthy) return "Serving"; + if (presenceModelResident(model)) return "Ready to use"; + return "Starts when needed"; + } + function renderPresenceModels() { + const search = $("#presence-search").value.trim().toLowerCase(); + const kind = $("#presence-kind").value; + const models = (state.physicalModels || []).filter(model => + (!search || (model.name + " " + model.id).toLowerCase().includes(search)) && + (!kind || (model.kind || "chat").startsWith(kind)) + ).sort((a,b) => Number(Boolean(b.runtime)) - Number(Boolean(a.runtime)) || Number(Boolean(presenceRuntime(b)?.healthy)) - Number(Boolean(presenceRuntime(a)?.healthy))); + $("#presence-model-count").textContent = models.length + (models.length === 1 ? " model" : " models"); + const key = JSON.stringify(models.map(model => {const rt=presenceRuntime(model);return [model.id,model.name,model.kind,model.contextWindow,model.runtime,presenceModelLabel(model),rt?.memoryGb,rt?.node,rt?.keepWarm,rt?.preferredWarm];})); + if (key === presenceRenderKey) return; + presenceRenderKey = key; + $("#presence-models").innerHTML = models.map(model => { + const rt = presenceRuntime(model), policy = presencePolicy(rt); + const location = rt?.node || (model.targets || []).map(t => t.node).filter(Boolean).join(", ") || (model.runtime ? "This machine" : "Upstream"); + return '
' + escapeHtml(({chat:"Chat & code",audio_speech:"Speech",audio_transcription:"Transcription",audio_generation:"Music & audio",embedding:"Embeddings",image:"Images",video:"Video"})[model.kind] || model.kind || "Chat") + '' + escapeHtml(presenceModelLabel(model)) + '

' + escapeHtml(model.name || model.id) + '

' + escapeHtml(model.id) + '

Runs on
' + escapeHtml(location) + '
Memory estimate
' + (rt?.memoryGb != null ? escapeHtml(rt.memoryGb) + ' GB' : 'Not reported') + '
Availability
' + escapeHtml(rt ? presencePolicyNames[policy] : 'Managed upstream') + '
'; + }).join("") || '
' + (state.models.length ? 'No models match these filters.' : 'No configured models yet. Add a model to get started.') + '
'; + } + function renderPresenceMachines() { + const nodes = Object.values(state.status?.cluster?.nodes || {}); + const rows = nodes.length ? nodes : (state.topologySummary?.host ? [{id:"local",name:"This machine",local:true,telemetry:state.topologySummary.host}] : []); + const key=JSON.stringify(rows.map(node=>[node.id,node.name,node.local,node.reachable,node.profile?.cpuBrand,node.telemetry?.memory?.usedBytes,node.telemetry?.memory?.availableBytes,node.telemetry?.memory?.totalBytes])); + if(key===presenceMachineKey)return; + presenceMachineKey=key; + $("#presence-machines").innerHTML = rows.map(node => { + const memory = node.telemetry?.memory || {}; + const total = Number(memory.totalBytes); + const available = Number(memory.availableBytes); + const used = Number.isFinite(available) ? total - available : Number(memory.usedBytes); + const hasMemory = total > 0 && Number.isFinite(used); + return '
' + (node.local ? 'This machine' : 'Configured peer') + '' + (node.reachable === false ? 'Unavailable' : 'Connected') + '

' + escapeHtml(node.name || node.id) + '

' + escapeHtml(node.profile?.cpuBrand || node.telemetry?.cpu?.model || node.profile?.platformId || 'Hardware details unavailable') + '

' + (hasMemory ? '

' + escapeHtml(formatBytes(Number.isFinite(available) ? available : used)) + (Number.isFinite(available) ? ' available / ' : ' used / ') + escapeHtml(formatBytes(total)) + '

' : '

Memory reading unavailable

') + '
'; + }).join("") || '
Waiting for hardware telemetry.
'; + } + function presenceClientExample() { + const model = $("#presence-client-model").value; + const selected=(state.physicalModels||[]).find(m=>m.id===model),kind=selected?.kind||'chat'; + const templates={chat:['/v1/chat/completions',{model,messages:[{role:'user',content:'Hello'}]}],embedding:['/v1/embeddings',{model,input:'Text to search'}],image:['/v1/images/generations',{model,prompt:'A quiet mountain lake'}],audio_speech:['/v1/audio/speech',{model,input:'Hello there',voice:'default'}],audio_generation:['/v1/audio/generations',{model,prompt:'Gentle piano'}],video:['/v1/videos/generations',{model,prompt:'A quiet mountain lake'}]}; + if(kind==='audio_transcription'){$("#presence-client-example").textContent="POST "+endpoint+"/v1/audio/transcriptions\nAuthorization: Bearer YOUR_LLOOM_KEY\nMultipart form: model="+model+", file=YOUR_AUDIO_FILE";return;} + const [route,body]=templates[kind]||templates.chat; + $("#presence-client-example").textContent = "POST " + endpoint + route + "\nContent-Type: application/json\nAuthorization: Bearer YOUR_LLOOM_KEY\n\n" + JSON.stringify(body,null,2); + } + function renderPresenceClients() { + $("#presence-client-url").value = endpoint + "/v1"; + const select = $("#presence-client-model"), selected = select.value; + const models = state.physicalModels || []; + const key = models.map(m=>m.id).join("\n"); + if (select.dataset.models !== key) { + select.innerHTML = models.map(m=>'').join(""); + if (models.some(m=>m.id === selected)) select.value = selected; + select.dataset.models = key; + } + $("#presence-client-auth").textContent = state.security?.authRequired ? "Use a configured inference key. Keep your admin key out of client applications." : "This loopback gateway allows local clients without a key. Remote access requires authenticated configuration."; + presenceClientExample(); + } + function renderPresence() { + if (!$("#presence-models")) return; + const safety = state.status?.runtimeManager?.memorySafety; + const warning = $("#presence-memory-safety"); + if(warning) warning.hidden = safety?.mode !== "yolo"; + renderPresenceModels(); renderPresenceMachines(); renderPresenceClients(); renderPresencePolicy(); + } + function renderPresencePolicy() { + const model = state.physicalModels.find(m=>m.id === state.selectedModelId); + const runtime = model && presenceRuntime(model); + if(model) $("#model-inspector-state span:last-child").textContent=presenceModelLabel(model); + $("#presence-policy").hidden = !model; + $("#presence-trial").hidden = !model || (model.kind || "chat") !== "chat"; + for (const button of document.querySelectorAll("[data-residency]")) { + button.disabled = presenceBusy || !runtime || runtime.enabled === false || runtime.management === "external" || runtime.remote === true || Boolean(runtime.maintenance); + button.setAttribute("aria-pressed",String(Boolean(runtime) && presencePolicy(runtime) === button.dataset.residency)); + } + const managed=runtime&&runtime.enabled!==false&&runtime.management!=="external"&&!runtime.remote&&!runtime.distributed&&!runtime.maintenance; + const transitioning=runtime&&['starting','warming','queued','draining','stopping'].includes(runtime.status); + for(const id of ['model-start','model-stop'])$("#"+id).disabled=presenceBusy||!managed||transitioning; + $("#model-start").textContent=transitioning?"Getting ready…":"Make ready now"; + $("#model-stop").disabled||=Boolean(runtime?.activeRequests||runtime?.queuedRequests||runtime?.keepWarm); + $("#presence-send").disabled=presenceBusy||Boolean(runtime?.maintenance)||runtime?.enabled===false; + $("#presence-availability").textContent=!runtime?"Available through your connected provider.":runtime.maintenance?"Paused. This model is protected from starting.":transitioning?"LLooM is getting this ready for you.":runtime.status==='failed'?"This model needs attention. Open details to see what happened.":"Just use it. LLooM prepares this automatically when your app asks."; + $("#presence-policy-hint").textContent = !runtime?"Your provider manages availability.":presencePolicy(runtime)==="always"?"Kept ready for quick replies. Choose Automatic if you want LLooM to reclaim its memory.":presencePolicy(runtime)==="preferred"?"Stays ready when there is room. LLooM makes space when another model needs it.":"Recommended: LLooM gets this ready when needed. Your downloaded files stay on disk."; + $("#presence-model-error").textContent=runtime?.lastError||""; + } + async function presenceLoadIntegrations() { + try { + const manifest = await getJson("/gateway/integrations"); + $("#presence-integrations").innerHTML = (manifest.clients || []).map(client=>'
' + escapeHtml(client.name || client.id) + '
' + escapeHtml("lloom integrate " + client.id) + '

Preview first. Add --apply --yes only after reviewing the generated configuration.

').join("") || "No client profiles available."; + } catch(error) { $("#presence-integrations").textContent = error.message; } + } + function presenceInvalidatePlan() { + presenceImport = null; presencePlanVersion++; + $("#presence-add-review").hidden = true; $("#presence-add-apply").hidden = true; + $("#presence-add-plan").hidden = false; + } + function presenceOpenAdd() { + presenceLastFocus = document.activeElement; + presenceInvalidatePlan(); + $("#presence-add-error").textContent = ""; + const selected = state.library?.selected; + $("#presence-recommendation").innerHTML = selected + ? '
Recommended for this hardware

' + escapeHtml(selected.name || selected.recipeId) + '

' + escapeHtml((selected.reasons || []).join(" · ") || "Matched by the local recipe library.") + '

' + : '

No compatible recipe recommendation is available yet.

'; + $("#presence-add").hidden = false; $("#presence-model-ref").focus(); + } + function presenceCloseAdd() { + if (presenceBusy) return; + $("#presence-add").hidden = true; presenceLastFocus?.focus(); + } + async function presenceReviewRecipe() { + const selected = state.library?.selected; + if (!selected) return; + const recipeId = selected.recipeId; + const version = presencePlanVersion; + const plan = await postJson("/gateway/installations/plan",{recipeId}); + if(version!==presencePlanVersion)return; + presenceImport = {planId:plan.planId}; + presenceShowPlan(plan); + } + function presenceShowPlan(plan) { + $("#presence-plan-json").textContent = JSON.stringify(plan.details || plan,null,2); + $("#presence-plan-summary").textContent = plan.summary || "Review the packages, model files, and configuration this installation will use."; + $("#presence-add-review").hidden = false; + $("#presence-add-apply").hidden = plan.ok === false; + $("#presence-add-plan").hidden = plan.ok !== false; + } + async function presenceRun(button, action) { + if (presenceBusy) return; + presenceBusy = true; + if (button) button.disabled = true; + renderPresencePolicy(); + try { return await action(); } + catch(error) { presenceNotice(error.message,true); $("#presence-add-error").textContent = error.message; } + finally { presenceBusy = false; if(button) button.disabled=false; renderPresencePolicy(); } + } + async function presencePollInstallation() { + clearTimeout(presenceInstallTimer); + try { + const {job}=await getJson("/gateway/installations"); + if(!job)return; + if(job.status==="running") { + presenceNotice(job.detail); + presenceInstallTimer=setTimeout(presencePollInstallation,1500); + } else if(presenceInstallSeen!==job.id) { + presenceInstallSeen=job.id; + presenceNotice(job.error || job.detail,job.status==="failed"); + await refresh(); + } + } catch(error) { presenceNotice("Could not read installation progress: "+error.message,true); } + } + // Keep the model inspector available from every view, rather than clipping + // it when the canvas is hidden. + document.body.append($("#model-inspector"),$("#node-inspector")); + const inspectorBody=$("#model-inspector .model-inspector-body"); + const availability=document.createElement("p");availability.id="presence-availability";availability.className="presence-availability";inspectorBody.prepend(availability); + const policy = document.createElement("section");policy.id = "presence-policy"; + policy.innerHTML = '

When should this stay ready?

' + Object.entries(presencePolicyNames).map(([id,name])=>'').join("") + '

'; + const trial = document.createElement("section");trial.id = "presence-trial"; trial.className = "presence-trial"; + trial.innerHTML = '
';
+    inspectorBody.append(trial);
+    const connect=document.createElement("button");connect.type="button";connect.id="presence-connect";connect.textContent="Connect an app";inspectorBody.append(connect);
+    connect.addEventListener("click",()=>{const id=state.selectedModelId;presenceSetView("clients");if([...$("#presence-client-model").options].some(option=>option.value===id))$("#presence-client-model").value=id;presenceClientExample();});
+    const technical=document.createElement("details");technical.className="presence-advanced";
+    technical.innerHTML='Options & details

'; + technical.append(policy,$("#model-inspector .model-inspector-actions"),$("#model-inspector-details"),$("#model-inspector-tags"));inspectorBody.append(technical); + $("#model-start").textContent="Make ready now";$("#model-warm").remove();$("#model-stop").textContent="Free up memory"; + const oldRenderModels = renderModels; + renderModels = function() { oldRenderModels(); renderPresence(); }; + const oldRenderInspector = renderModelInspector; + renderModelInspector = function() { oldRenderInspector(); renderPresencePolicy(); }; + const oldShowOutput = showOutput; + showOutput = function(value) { oldShowOutput(value); presenceNotice(value?.error?.message || value?.error || "Operation complete. Details are in Settings.",Boolean(value?.error)); }; + const keyField = $("#api-key").closest("label"); + keyField.style.cssText = ""; + const settingsHeader = document.createElement("div"); settingsHeader.className="presence-heading"; + settingsHeader.innerHTML='

Settings & details.

Authentication, recipes, runtimes, and installation plans.

'; + $(".operations-content").prepend(settingsHeader,keyField); + $("#api-key").addEventListener("change",()=>{ sessionStorage.setItem("lloom_api_key",$("#api-key").value.trim()); localStorage.removeItem("lloom_api_key"); refresh(); }); + $(".presence-nav").addEventListener("click",event=>{const button=event.target.closest("[data-view]"); if(button) presenceSetView(button.dataset.view);}); + $("#presence-toast button").addEventListener("click",()=>$("#presence-toast").hidden=true); + $("#presence-search").addEventListener("input",()=>renderPresenceModels()); + $("#presence-kind").addEventListener("change",()=>renderPresenceModels()); + $("#presence-client-model").addEventListener("change",presenceClientExample); + $("#presence-copy-url").addEventListener("click",event=>presenceRun(event.currentTarget,async()=>{await navigator.clipboard.writeText(endpoint+"/v1");presenceNotice("Gateway URL copied.");})); + $("#presence-copy-example").addEventListener("click",event=>presenceRun(event.currentTarget,async()=>{await navigator.clipboard.writeText($("#presence-client-example").textContent);presenceNotice("Example request copied. Replace the key placeholder in your client.");})); + $("#presence-add-close").addEventListener("click",presenceCloseAdd); + $("#presence-add-form").addEventListener("input",presenceInvalidatePlan); + $("#presence-add-form").addEventListener("submit",event=>{ + event.preventDefault(); const button=$("#presence-add-plan"); + presenceRun(button,async()=>{ + const version=presencePlanVersion; + const input={modelRef:$("#presence-model-ref").value.trim(),backend:$("#presence-model-backend").value.trim()||undefined,name:$("#presence-model-name").value.trim()||undefined}; + const plan=await postJson("/gateway/installations/plan",input); + if(version!==presencePlanVersion)return; + presenceImport={planId:plan.planId}; presenceShowPlan(plan); + }); + }); + $("#presence-add-apply").addEventListener("click",event=>presenceRun(event.currentTarget,async()=>{ + if(!presenceImport || !ensureAdminKeyIfNeeded()) return; + await postJson("/gateway/installations",{...presenceImport,yes:true}); + presenceBusy=false; presenceCloseAdd(); presencePollInstallation(); + })); + document.addEventListener("click",event=>{ + const add=event.target.closest("[data-add-model]"); + if(add) presenceOpenAdd(); + const model=event.target.closest("[data-presence-model]"); + if(model) {openModelInspector(model.dataset.presenceModel);$("#model-inspector-close").focus();} + const machine=event.target.closest("[data-presence-node]"); + if(machine) openNodeInspector(machine.dataset.presenceNode); + if(event.target.closest("#presence-use-recipe")) presenceRun(event.target.closest("button"),presenceReviewRecipe); + const residency=event.target.closest("[data-residency]"); + if(residency) presenceRun(residency,async()=>{ + const model=state.physicalModels.find(m=>m.id===state.selectedModelId); + if(!model?.runtime || !ensureAdminKeyIfNeeded()) return; + const path="/gateway/runtimes/"+encodeURIComponent(model.runtime)+"/residency"; + const accepted=await postJson(path,{policy:residency.dataset.residency,yes:true}); + presenceNotice("Preference saved. LLooM will take care of it."); + const poll=async()=>{ + try { + const {job}=await getJson(path); + if(!job || job.id!==accepted.id)return; + if(job.status==='pending'){setTimeout(poll,1500);return;} + await refresh(); + presenceNotice(job.status==='failed'?"Could not update availability: "+job.error:job.policy==='auto'?"LLooM will prepare this when your app needs it.":"Availability updated: "+presencePolicyNames[job.policy]+".",job.status==='failed'); + }catch(error){presenceNotice("Could not check readiness: "+error.message,true);} + }; + setTimeout(poll,500); + }); + }); + // Capture legacy runtime actions once, add bounded busy/error handling, and + // keep preparation behind the normal admission and safety checks. + document.addEventListener("click",event=>{ + const button=event.target.closest("button[data-runtime]"); + if(!button) return; + event.stopImmediatePropagation(); + if(button.disabled || !button.dataset.runtime || !ensureAdminKeyIfNeeded()) return; + presenceRun(button,async()=>{ + const load=button.dataset.action==="start"; + const action=load?"admit":button.dataset.action; + const result=await postJson("/gateway/runtimes/"+encodeURIComponent(button.dataset.runtime)+"/"+action,load?{apply:true,yes:true,force:false,warmup:true}:{}); + oldShowOutput(result);await refresh();presenceNotice(load?"Ready for your apps.":"Memory released. The model stays installed."); + }); + },true); + $("#presence-send").addEventListener("click",event=>presenceRun(event.currentTarget,async()=>{ + const prompt=$("#presence-prompt").value.trim(), model=state.selectedModelId; + if(!prompt || !model) return; + $("#presence-answer").textContent="Waiting for "+model+"…"; + try { + const result=await postJson("/v1/chat/completions",{model,messages:[{role:"user",content:prompt}],max_tokens:256,stream:false}); + $("#presence-answer").textContent=result.choices?.[0]?.message?.content || "No text response returned."; + } catch(error) { $("#presence-answer").textContent=error.message; throw error; } + })); + document.addEventListener("keydown",event=>{ + if(event.key==="Escape") {presenceCloseAdd();closeModelInspector();closeNodeInspector();} + const dialog=!$("#presence-add").hidden?$("#presence-add .presence-dialog"):null; + if(event.key==="Tab" && dialog) { + const items=[...dialog.querySelectorAll('button:not([disabled]),input,select,textarea,summary')].filter(el=>el.getClientRects().length); + const first=items[0],last=items.at(-1); + if(event.shiftKey && document.activeElement===first){event.preventDefault();last?.focus();} + else if(!event.shiftKey && document.activeElement===last){event.preventDefault();first?.focus();} + } + }); + presenceSetView(location.hash.slice(1)); + presencePollInstallation(); +`; diff --git a/src/dashboard-presence.mjs b/src/dashboard-presence.mjs new file mode 100644 index 0000000..63a863a --- /dev/null +++ b/src/dashboard-presence.mjs @@ -0,0 +1,139 @@ +export const presenceStyles = ` + :root { color-scheme:dark; --bg:#080d12; --band:#0d151d; --panel:#111c25; --panel-2:#15232e; --line:#243641; --text:#edf5f8; --muted:#9eb3c0; --accent:#38dff5; --accent-2:#85c8ff; --danger:#fb927e; } + body { font-family:Inter,-apple-system,BlinkMacSystemFont,"Segoe UI",sans-serif; font-size:14px; letter-spacing:0; } + body > header { margin-left:190px; min-height:72px; padding:16px 28px; background:var(--bg); border-bottom:1px solid var(--line); } + body > header .brand { display:none; } + .topline { width:100%; justify-content:flex-end; } + main { margin-left:190px; width:auto; padding:24px 28px; max-width:none; min-width:0; } + main > * { min-width:0; } + button.primary { background:#1e4953; border-color:#36717d; } + button,input,select,textarea { font-family:inherit; border-radius:9px; } + button { text-transform:none; font-size:13px; letter-spacing:0; font-weight:500; } + button:focus-visible, a:focus-visible, input:focus-visible, select:focus-visible { outline:2px solid var(--accent); outline-offset:3px; } + h1,h2,h3,strong { font-weight:500; } + .pill { text-transform:none; letter-spacing:0; font-size:12px; border-radius:20px; } + .band,.empty { border-radius:14px; } + .band-head h2,label { text-transform:none; letter-spacing:0; } + .band-body { padding:22px; } + #presence-memory-safety { border:1px solid #b67c37; background:#392915; color:#ffd8a0; border-radius:12px; padding:14px 18px; margin-bottom:20px; } + .presence-nav { position:fixed; top:0; bottom:0; left:0; width:190px; padding:26px 18px; background:#0b1219; border-right:1px solid var(--line); display:flex; flex-direction:column; gap:7px; z-index:20; } + .presence-brand { padding:0 12px 30px; font-size:24px; letter-spacing:-1px; font-weight:500; } + .presence-brand small { display:block; color:var(--muted); font-size:10px; letter-spacing:2px; margin-top:3px; } + .presence-nav button { display:flex; align-items:center; gap:12px; text-align:left; background:transparent; border:1px solid transparent; min-height:43px; color:var(--muted); padding:10px 12px; } + .presence-nav button[aria-current="page"] { color:var(--text); background:#18303a; border-color:#25515e; } + .presence-nav button:last-child { margin-top:auto; } + .presence-nav svg { width:18px; height:18px; fill:none; stroke:currentColor; stroke-width:1.6; } + .presence-heading { display:flex; justify-content:space-between; align-items:center; flex-wrap:wrap; gap:16px; margin-bottom:24px; } + .presence-heading h2 { margin:0 0 7px; font-size:clamp(26px,3vw,38px); letter-spacing:-1.2px; font-weight:500; } + .presence-heading p { margin:0; color:var(--muted); line-height:1.6; } + .presence-view[hidden], [hidden] { display:none!important; } + .presence-toolbar { display:flex; flex-wrap:wrap; gap:12px; align-items:center; margin-bottom:22px; } + .presence-toolbar input { min-width:180px; max-width:380px; flex:1; margin:0; } + .presence-toolbar select { width:auto; min-width:140px; margin:0; } + .presence-grid { display:grid; grid-template-columns:repeat(auto-fill,minmax(265px,1fr)); gap:16px; } + .presence-card { background:linear-gradient(140deg,#14222c,#101a23); border:1px solid var(--line); border-radius:16px; padding:22px; display:flex; flex-direction:column; gap:14px; min-width:0; } + .presence-card h3 { font-size:17px; margin:0; line-height:1.4; overflow-wrap:anywhere; } + .presence-card p { color:var(--muted); margin:0; line-height:1.65; overflow-wrap:anywhere; } + .presence-card .actions { margin-top:auto; } + .presence-eyebrow { font-size:11px; color:var(--muted); letter-spacing:1.1px; text-transform:uppercase; } + .presence-card-top { display:flex; justify-content:space-between; align-items:center; gap:8px; } + .presence-status { font-size:12px; color:var(--muted); display:flex; align-items:center; gap:7px; } + .presence-status::before { content:""; width:6px; height:6px; background:currentColor; border-radius:50%; flex-shrink:0; } + .presence-status[data-ready="true"] { color:var(--accent); } + .presence-card dl { display:grid; grid-template-columns:1fr 1fr; gap:7px; margin:0; font-size:12px; } + .presence-card dt { color:var(--muted); } + .presence-card dd { margin:0; text-align:right; overflow-wrap:anywhere; } + .presence-memory { height:5px; background:#263640; border-radius:4px; overflow:hidden; } + .presence-memory > span { display:block; height:100%; background:var(--accent); } + .presence-section-title { font-size:18px; margin:30px 0 16px; } + .presence-message { color:var(--muted); padding:20px 0; line-height:1.7; max-width:800px; } + .presence-notice { position:fixed; z-index:60; bottom:20px; left:218px; right:28px; max-width:660px; background:#162b35; color:var(--text); border:1px solid #345767; border-radius:12px; padding:16px 48px 16px 18px; box-shadow:0 10px 35px #0008; overflow-wrap:anywhere; } + .presence-notice[data-error="true"] { border-color:#a36050; } + .presence-notice button { position:absolute; right:7px; top:7px; background:none; border:0; } + .presence-connection { display:grid; grid-template-columns:minmax(0,1.3fr) minmax(260px,1fr); gap:20px; } + .presence-connection pre { white-space:pre-wrap; overflow-wrap:anywhere; font-size:12px; background:#091118; border-radius:10px; padding:18px; max-height:360px; overflow:auto; } + .presence-connection label { display:block; font-size:12px; margin-top:12px; } + .presence-connection select { margin-top:8px; } + .presence-readiness { display:flex; flex-wrap:wrap; gap:6px; } + .presence-readiness button { flex:1; font-size:12px; padding:9px 7px; } + .presence-readiness button[aria-pressed="true"] { background:#20414b; border-color:var(--accent); } + .presence-drawer { position:fixed; z-index:45; inset:0; display:grid; place-items:center; padding:20px; background:#02080cc9; } + .presence-dialog { width:min(620px,100%); max-height:90vh; overflow:auto; border:1px solid var(--line); background:var(--panel); border-radius:20px; padding:28px; } + .presence-dialog h2 { margin-top:0; letter-spacing:-.5px; } + .presence-dialog label { display:block; margin:14px 0; } + .presence-dialog pre { max-height:230px; overflow:auto; white-space:pre-wrap; overflow-wrap:anywhere; font-size:12px; } + .presence-dialog details { margin:16px 0; color:var(--muted); } + .presence-dialog .actions { justify-content:flex-end; flex-wrap:wrap; } + .presence-trial { margin-top:18px; } + .presence-trial textarea { width:100%; padding:12px; background:var(--bg); color:var(--text); border:1px solid var(--line); resize:vertical; min-height:90px; } + .presence-trial pre:empty { display:none; } + .presence-trial pre { white-space:pre-wrap; font-size:13px; line-height:1.7; } + .topology { border:1px solid var(--line); border-radius:18px; min-height:600px; background:#080f15; box-shadow:none; } + .topology::before,.topology::after { display:none; } + .topology-canvas { min-height:600px; height:calc(100vh - 230px); background:transparent; } + .topology-hud { top:16px; left:16px; right:16px; flex-wrap:wrap; gap:8px; } + .topology-hud-panel { flex:1 1 220px; } + .topology-hud-right { flex-wrap:wrap; max-width:100%; } + .topology-hud-panel .mono { font-family:inherit; font-size:12px; } + .topology-hud-panel { background:#101b24e8; border-color:var(--line); border-radius:10px; box-shadow:none; } + .fabric-title { font-family:inherit; font-size:13px; letter-spacing:0; font-weight:500; } + .topology-model-filter,.topology-metrics,.topology-zoom { border-radius:9px; } + .fabric-totals { gap:14px; } + .fabric-total strong { font-family:inherit; font-weight:500; } + .model-inspector { position:fixed; z-index:40; top:88px; right:24px; bottom:24px; max-height:calc(100vh - 112px); width:min(390px,calc(100vw - 32px)); border-radius:18px; background:#101b24; border-color:#345461; box-shadow:0 24px 80px #0008; } + .model-inspector:not(.open) { visibility:hidden; } + .model-inspector-title { font-weight:500; font-size:22px; } + .model-detail-grid { grid-template-columns:1fr 1fr; } + .model-detail strong { font-weight:400; } + .model-inspector-actions { grid-template-columns:repeat(2,minmax(0,1fr)); } + .presence-availability {color:#acc8d6;font-size:13px;line-height:1.65;margin:0 0 14px} + .presence-advanced {margin-top:22px;border-top:1px solid var(--line);padding-top:14px} + .presence-advanced summary {color:#8faebd;font-size:12px;cursor:pointer} + .presence-advanced .model-inspector-actions {margin:16px 0} + #presence-connect {width:100%;margin-top:12px} + #presence-send {width:100%;margin-top:8px} + #presence-model-error:empty {display:none} + .operations-dock { margin:0; border-radius:16px; } + .operations-dock > summary { display:none; } + .operations-content { padding:0; } + .operations-content .grid.two { grid-template-columns:1fr; } + @media(max-width:1000px) { .presence-nav { width:155px; padding:24px 10px; } body > header, main { margin-left:155px; } main { padding:22px 18px; } .presence-notice { left:175px; } .topology-hud { flex-wrap:wrap; gap:8px; } .topology-hud-right { flex-wrap:wrap; } .topology-hud-panel > .muted { display:none; } .presence-connection { grid-template-columns:1fr; } } + @media(max-width:640px) { .presence-nav { position:sticky; width:100%; top:0; bottom:auto; flex-direction:row; padding:8px; gap:4px; border-right:0; border-bottom:1px solid var(--line); } .presence-brand { display:none; } .presence-nav button { flex:1; flex-direction:column; justify-content:center; padding:6px 2px; gap:4px; font-size:11px; } .presence-nav button:last-child { margin-top:0; } body > header, main { margin-left:0; } body > header { padding:12px 16px; min-height:0; } .topline { justify-content:space-between; gap:8px; } #endpoint { max-width:65%; overflow:hidden; text-overflow:ellipsis; } main { padding:20px 12px; } .topology-hud { top:10px; left:10px; right:10px; } .topology-canvas { min-height:550px; height:65vh; } .topology { min-height:550px; } .topology-zoom { left:8px; right:auto; } .presence-heading { margin-bottom:20px; } .presence-grid { grid-template-columns:1fr; } .presence-notice { left:12px; right:12px; bottom:12px; } .model-inspector { top:76px; bottom:12px; right:12px; width:calc(100vw - 24px); max-height:calc(100vh - 88px); } .presence-dialog { padding:20px; } .presence-drawer { padding:12px; } } + @media(prefers-reduced-motion:reduce) { *,*::before,*::after { animation:none!important; transition:none!important; scroll-behavior:auto!important; } } +`; + +const icons = { + live: '', + models: '', + machines: '', + clients: '', + settings: '' +}; +export const presenceNav = ``; + +export const presenceViews = ` + + + + + +`; diff --git a/src/dashboard-scene.mjs b/src/dashboard-scene.mjs new file mode 100644 index 0000000..b9f33d1 --- /dev/null +++ b/src/dashboard-scene.mjs @@ -0,0 +1,450 @@ +// The product composition: physical machines around a single gateway, with +// motion driven by observed requests. The old diagnostic canvas stays available. +export const sceneStyles = ` + body { background:radial-gradient(ellipse at 48% 8%,#0a1c24 0,transparent 48%),#050b0f; } + body > header { min-height:38px;padding:8px 26px;background:transparent;border:0; } + body > header .topline {font-size:11px;opacity:.65} + body > header #refresh {padding:4px 10px;font-size:11px} + main {padding-top:8px} + .presence-nav {background:linear-gradient(180deg,#081218,#070d12);padding-top:24px} + .presence-nav button[aria-current] {color:#2be1f6;background:linear-gradient(90deg,#11313c,#102029);position:relative;border-color:#193540} + .presence-nav button[aria-current]::before {content:"";position:absolute;left:-9px;top:8px;bottom:8px;width:3px;border-radius:4px;background:#2be1f6;box-shadow:0 0 14px #2be1f666} + .presence-brand {display:flex;gap:12px;align-items:center;padding:0 8px 30px;font-weight:600;letter-spacing:-.6px} + .presence-brand small {font-weight:400;text-transform:none;letter-spacing:0;font-size:12px} + .presence-brand .scene-logo {width:30px;height:40px;filter:drop-shadow(0 0 8px #29dff344)} + .presence-heading h2 {font-size:36px;font-weight:600;letter-spacing:-1.1px} + .presence-heading {margin-bottom:22px} + button.primary,.scene-primary {background:linear-gradient(115deg,#29def3,#26cfee);border:1px solid #64e5f3;color:#03202a;box-shadow:0 5px 23px #0dcced16;font-weight:600} + button.primary:hover,.scene-primary:hover {background:#72eafb;box-shadow:0 0 22px #2ddaf32b} + .scene-icon {display:inline-flex;align-items:center;justify-content:center;flex-shrink:0} + .scene-icon svg {width:22px;height:22px;fill:none;stroke:currentColor;stroke-width:1.5;stroke-linecap:round;stroke-linejoin:round} + .scene-layout {display:grid;grid-template-columns:minmax(0,1fr) 310px;gap:14px;align-items:stretch} + .scene-diagram,.scene-detail,.scene-analytics,.scene-memory-panel,.scene-network {border:1px solid #24343f;border-radius:16px;background:linear-gradient(135deg,#0b171d88,#060d12b0);box-shadow:inset 0 1px #8ad7f304} + .scene-diagram {position:relative;height:clamp(440px,calc(100vh - 320px),620px);overflow:hidden} + .scene-columns {position:relative;z-index:1;display:grid;grid-template-columns:minmax(130px,.78fr) minmax(100px,.66fr) minmax(240px,1.42fr);gap:18px;height:100%;padding:22px 20px} + .scene-column {min-width:0;min-height:0;display:flex;flex-direction:column} + .scene-column > h3 {font-size:14px;margin:0 0 6px;font-weight:500} + .scene-column > p {font-size:12px;color:#9fb5c7;margin:0} + .scene-clients {display:flex;flex:1;flex-direction:column;justify-content:space-evenly;gap:18px;padding:38px 0} + .scene-client {display:flex;gap:12px;align-items:center;padding:17px 13px;border:1px solid #284553;border-radius:13px;background:linear-gradient(135deg,#162933aa,#08151bec);box-shadow:0 12px 24px #0002;min-height:72px} + .scene-client .scene-icon {width:34px;height:38px;border-radius:9px;background:linear-gradient(135deg,#263a47,#11212c);color:#c9ecf8} + .scene-client strong {display:block;font-size:13px;font-weight:500;overflow-wrap:anywhere} + .scene-client small {display:block;margin-top:6px;color:#9cb5c7;font-size:11px;line-height:1.4} + .scene-client[data-active="true"] {border-color:#2d7283} + .scene-gateway-column {text-align:center} + .scene-gateway-wrap {flex:1;display:flex;flex-direction:column;align-items:center;justify-content:center;padding-bottom:28px} + .scene-gateway {position:relative;width:98px;height:98px;border-radius:50%;border:2px solid #32def7;background:radial-gradient(circle at 32% 25%,#143642,#03131c 70%);display:grid;place-items:center;box-shadow:0 0 0 7px #0a2c3629,0 0 26px #14d3f22b,inset 0 0 24px #10cbea13;isolation:isolate} + .scene-gateway::before,.scene-gateway::after {content:"";position:absolute;inset:-12px;border-radius:50%;border:1px solid #28d4ef18;pointer-events:none} + .scene-gateway::after {inset:-28px;border-color:#27d9f208} + .scene-gateway[data-active="true"] {animation:scene-breathe 3s ease-in-out infinite;box-shadow:0 0 0 7px #0a2c3629,0 0 38px #14d3f258,inset 0 0 28px #10cbea25} + .scene-gateway svg {width:37px;height:48px;filter:drop-shadow(0 0 8px #21dff36b)} + .scene-gateway-wrap > strong {font-size:17px;margin-top:18px;font-weight:500} + .scene-gateway-wrap > small {font-size:12px;color:#a4c0d2;margin-top:8px;text-align:center} + .scene-machines {display:flex;flex-direction:column;justify-content:center;gap:12px;flex:1;min-height:0;overflow:auto;padding-top:18px;justify-content:flex-start} + .scene-machine {border:1px solid #29404d;border-radius:13px;background:linear-gradient(125deg,#10222d80,#0a141cdb);padding:13px;min-width:0} + .scene-machine-header {display:flex;gap:10px;align-items:center;margin-bottom:12px} + .scene-machine-header .scene-icon {color:#b9d8e9} + .scene-machine-header strong {display:block;font-size:13px;font-weight:500} + .scene-machine-header small {display:block;color:#9bb6c8;font-size:11px;margin-top:4px;line-height:1.4} + .scene-machine-header > div:nth-child(2) {flex:1;min-width:0;overflow-wrap:anywhere} + .scene-mini-memory {width:76px;text-align:right;flex-shrink:0} + .scene-mini-memory > i {height:5px;border-radius:5px;background:#1a303e;display:block;overflow:hidden;margin-bottom:5px} + .scene-mini-memory b {height:100%;display:block;border-radius:5px;background:linear-gradient(90deg,#0dc7e7,#4ce5f5);box-shadow:0 0 8px #37d7f85c} + .scene-mini-memory small {font-size:10px!important} + .scene-model {width:100%;display:flex;align-items:center;gap:10px;margin-top:7px;border:1px solid #203440;border-radius:10px;background:linear-gradient(100deg,#11212a8a,#09151ad9);padding:10px 9px;text-align:left;min-height:43px;box-shadow:inset 0 1px #b5eaff03} + .scene-model[data-serving="true"],.scene-model[data-selected="true"] {border-color:#27cbe6;background:linear-gradient(100deg,#07303e,#0a1720);box-shadow:inset 0 0 20px #22dffa07,0 0 15px #19cfea0a} + .scene-model > .scene-icon svg {width:18px;height:18px} + .scene-model-name {flex:1;min-width:0;white-space:nowrap;overflow:hidden;text-overflow:ellipsis;font-size:12px} + .scene-model-state {font-size:10px;color:#a9c2d4;background:#1a2c37;border:1px solid #223d4b;border-radius:15px;padding:4px 8px;white-space:nowrap} + [data-serving="true"] > .scene-model-state {color:#49e5ee;border-color:#087d91;background:#06313b} + .scene-more {border:0;background:none;font-size:11px;color:#9ec2d4;padding:10px 4px 0;display:block} + .scene-links {position:absolute;inset:0;width:100%;height:100%;pointer-events:none;z-index:0;overflow:visible} + .scene-detail {min-width:0;padding:22px;height:clamp(440px,calc(100vh - 320px),620px);overflow:auto} + .scene-placeholder {height:100%;display:flex;flex-direction:column;justify-content:center;text-align:center;align-items:center;gap:18px} + .scene-placeholder .scene-icon {width:72px;height:72px;border:1px solid #28414e;border-radius:50%;color:#7fdfea;background:radial-gradient(circle at 40% 30%,#203d4b,#0b1922)} + .scene-placeholder .scene-icon svg {width:32px;height:32px} + .scene-placeholder h3 {font-size:22px;margin:0} + .scene-placeholder p {font-size:13px;color:#99b2c5;line-height:1.7;margin:0} + .scene-detail .model-inspector {position:static;visibility:visible;width:100%;max-height:none;height:auto;border:0;border-radius:0;background:transparent;box-shadow:none;overflow:visible} + .scene-detail .model-inspector:not(.open) {display:none} + .scene-detail .model-inspector-header {padding:0 0 22px;border-bottom:1px solid #233642} + .scene-detail .model-inspector-body {padding:0;overflow:visible} + .scene-detail .model-inspector-title {font-size:22px;line-height:1.35} + .scene-detail .model-inspector-state {margin-bottom:12px} + .scene-detail #presence-policy {padding:20px 0;border-bottom:1px solid #233642} + .scene-detail #presence-policy h3 {font-size:14px;margin:0 0 14px} + .presence-readiness {gap:0;border:1px solid #2c4757;border-radius:10px;overflow:hidden} + .presence-readiness button {border-radius:0;border:0;border-right:1px solid #243c49;padding:12px 4px;background:transparent;font-size:11px;white-space:nowrap} + .presence-readiness button:last-child {border:0} + .presence-readiness button[aria-pressed="true"] {background:#25d3ee;color:#012333} + #presence-policy-hint {font-size:12px;line-height:1.6;margin:16px 0 0} + .scene-detail .model-inspector-actions {padding:20px 0;gap:7px} + .scene-detail details {border-top:1px solid #233642;padding-top:18px;margin-top:18px} + .scene-detail summary {font-size:13px;cursor:pointer} + .scene-inspector-memory {padding:18px 0;border-bottom:1px solid #233642;display:flex;justify-content:space-between;gap:12px;font-size:12px;color:#a9c1cf} + .scene-inspector-memory strong {color:#e1f2f6;font-weight:400} + .scene-analytics {margin-top:14px;display:grid;grid-template-columns:1fr 1fr 1fr;padding:20px 24px;gap:24px} + .scene-chart {min-width:0} + .scene-chart + .scene-chart {border-left:1px solid #20323e;padding-left:24px} + .scene-chart h3 {display:flex;justify-content:space-between;gap:10px;margin:0 0 16px;font-size:13px;font-weight:500} + .scene-chart h3 span {font-size:11px;color:#9cbed2;font-weight:400} + .scene-chart svg {width:100%;height:80px;overflow:visible} + .scene-chart-caption {font-size:10px;color:#6e899d;margin-top:6px} + .scene-memory-row {display:grid;grid-template-columns:minmax(70px,1fr) 1.4fr auto;align-items:center;gap:12px;font-size:11px;margin:12px 0;color:#b9cedb} + .scene-memory-row > span:first-child {overflow:hidden;text-overflow:ellipsis;white-space:nowrap} + .scene-memory-row i {height:7px;border-radius:4px;background:#182f3c;overflow:hidden} + .scene-memory-row b {display:block;height:100%;border-radius:4px;background:linear-gradient(90deg,#11ccea,#53e7f6)} + .scene-tools {display:flex;align-items:center;justify-content:space-between;gap:12px;margin:17px 0 4px;color:#7998ad;font-size:11px;flex-wrap:wrap} + .scene-tools .actions {align-items:center;gap:8px;margin:0} + .scene-tools button {padding:8px 12px;font-size:11px;background:#0b171f;border-color:#223a48} + .scene-follow {display:flex;gap:8px;align-items:center;font-size:11px;color:#acccdc} + .scene-follow input {accent-color:#22d9f3;width:auto;margin:0} + .scene-memory-panel {position:relative;z-index:6;background:linear-gradient(150deg,#0a1a2299,#060d12d8);padding:22px 24px 20px} + .scene-memory-panel h3 {display:flex;justify-content:space-between;gap:14px;font-size:15px;margin:0 0 6px;align-items:baseline;flex-wrap:wrap} + .scene-memory-panel h3 small {font-size:11px;color:#7d99ac;font-weight:400} + .scene-memory-panel select {width:auto;max-width:190px;font-size:11px;padding:5px 8px;background:#0a161e} + .scene-mem-sub {font-size:11.5px;color:#8ea9bb;margin:0 0 16px;line-height:1.5;max-width:66ch} + .scene-mem-head {display:flex;align-items:flex-end;justify-content:space-between;gap:16px;flex-wrap:wrap;margin:0 0 14px} + .scene-mem-total {display:flex;align-items:baseline;gap:9px;font-size:26px;font-weight:600;letter-spacing:-.6px;line-height:1} + .scene-mem-total span {font-size:12px;font-weight:400;color:#7f9cb0;letter-spacing:0;display:block;line-height:1.15} + .scene-mem-total em {font-style:normal;font-size:13px;color:#9fc4d6;font-weight:400} + .scene-mem-readouts {display:flex;gap:18px;flex-wrap:wrap;justify-content:flex-end} + .scene-mem-readout {text-align:right;min-width:86px} + .scene-mem-readout b {display:block;font-size:15px;font-weight:600;letter-spacing:-.2px;line-height:1.15} + .scene-mem-readout span {font-size:10px;letter-spacing:.7px;text-transform:uppercase;color:#6f8b9e} + .scene-mem-readout.used b {color:#e6f4fa} + .scene-mem-readout.free b {color:#5fe6d0} + .scene-mem-readout span.scene-mem-dot {display:inline-flex;align-items:center;gap:6px} + .scene-mem-readout span.scene-mem-dot::before {content:"";width:6px;height:6px;border-radius:50%;background:currentColor;opacity:.85} + .scene-mem-readout.free span.scene-mem-dot {color:#4fd8c4} + .scene-mem-readout.used span.scene-mem-dot {color:#8fb2c5} + .scene-mem-instrument {position:relative;border:1px solid #1f3a47;border-radius:12px;background:linear-gradient(180deg,#08131a,#060e13);overflow:hidden;isolation:isolate} + .scene-mem-ruler {position:relative;height:20px;border-bottom:1px solid #16303d} + .scene-mem-ticks {position:absolute;inset:0;display:flex;pointer-events:none} + .scene-mem-tick {flex:1 0 0;min-width:0;border-left:1px solid #16303d;position:relative} + .scene-mem-tick span {position:absolute;left:6px;top:5px;font-size:9px;color:#5d7c8e;white-space:nowrap} + .scene-mem-grid {position:absolute;inset:20px 0 0;pointer-events:none;display:flex;opacity:.5} + .scene-mem-grid i {flex:1 0 0;min-width:0;border-left:1px solid #11eaf50a} + .scene-mem-bar {position:relative;display:flex;height:142px;cursor:default;margin-top:2px} + .scene-mem-bar[data-known="false"] {height:70px} + .scene-mem-block {position:relative;flex:0 0 auto;min-width:0;border:0;padding:0;margin:0;background:transparent;color:#dceaf2;font:inherit;text-align:left;overflow:hidden;cursor:pointer;transition:filter .22s ease,opacity .22s ease} + .scene-mem-block > .scene-mem-fill {position:absolute;inset:0;background:var(--seg-fill);opacity:.9;transition:opacity .22s ease,box-shadow .22s ease} + .scene-mem-block > .scene-mem-rim {position:absolute;inset:0;border-right:1px solid #04121a99;background:linear-gradient(180deg,#ffffff14,#ffffff00 42%,#00000038)} + .scene-mem-block > .scene-mem-face {position:relative;z-index:2;display:flex;flex-direction:column;justify-content:space-between;height:100%;padding:11px 12px;gap:6px} + .scene-mem-block em {font-style:normal;font-size:10px;letter-spacing:.5px;color:#eaf7fc;text-shadow:0 1px 3px #04121ad9;white-space:nowrap} + .scene-mem-block em i {font-style:normal;opacity:.72;margin-left:5px;font-size:9px} + .scene-mem-block b {font-size:15px;font-weight:600;letter-spacing:-.2px;text-shadow:0 1px 3px #04121ad9;white-space:nowrap} + .scene-mem-block.wide > .scene-mem-face {padding:11px 14px} + .scene-mem-block.narrow em,.scene-mem-block.narrow b {display:none} + .scene-mem-block[data-kind="available"] {color:#eafffb} + .scene-mem-block[data-kind="available"] > .scene-mem-rim {background:linear-gradient(180deg,#ffffff1f,#ffffff00 40%,#0000001f)} + .scene-mem-block.system > .scene-mem-fill {background-image:repeating-linear-gradient(135deg,#ffffff10 0 7px,#ffffff00 7px 14px)} + .scene-mem-block[data-estimated="true"] > .scene-mem-fill {opacity:.72;background-image:repeating-linear-gradient(115deg,#ffffff12 0 6px,#ffffff00 6px 13px)} + .scene-mem-block:hover > .scene-mem-fill,.scene-mem-block[data-preview="true"] > .scene-mem-fill {opacity:1;box-shadow:inset 0 0 30px #ffffff1f} + .scene-mem-block[data-selected="true"] > .scene-mem-rim {box-shadow:inset 0 0 0 1px #ffffff4d} + .scene-mem-block:focus-visible {outline:2px solid #6ff0ff;outline-offset:-2px} + .scene-mem-ghost {position:absolute;top:0;bottom:0;left:0;width:0;pointer-events:none;transition:width .34s cubic-bezier(.22,1,.36,1),left .34s cubic-bezier(.22,1,.36,1)} + .scene-mem-ghost > .scene-mem-ghost-fill {position:absolute;inset:0;border:1px dashed #7ce8ffd9;border-left:0;border-radius:0 8px 8px 0;background:repeating-linear-gradient(115deg,#7ce8ff2e 0 6px,#7ce8ff0d 6px 12px);box-shadow:0 0 22px #23dcf61f,inset 0 0 24px #23dcf614} + .scene-mem-ghost[data-overflow="true"] > .scene-mem-ghost-fill {border-color:#ffb487ee;background:repeating-linear-gradient(115deg,#ff9d6a3a 0 6px,#ff9d6a12 6px 12px);box-shadow:0 0 22px #ff9d6a26} + .scene-mem-ghost[data-mode="resident"] > .scene-mem-ghost-fill,.scene-mem-ghost[data-mode="external"] > .scene-mem-ghost-fill {border-style:solid;border-color:#7ce8ff77;background:#7ce8ff14} + .scene-mem-ghost-label {position:absolute;right:8px;bottom:8px;font-size:10px;color:#bdf1ff;background:#062028e0;border:1px solid #2a6273;border-radius:7px;padding:4px 8px;white-space:nowrap;pointer-events:none;box-shadow:0 6px 18px #0006} + .scene-mem-ghost[data-overflow="true"] .scene-mem-ghost-label {color:#ffd6bd;border-color:#9b5a39;background:#2a140ce8} + .scene-mem-overflow {position:absolute;inset:0;z-index:3;pointer-events:none;opacity:0;transition:opacity .25s ease;background:repeating-linear-gradient(135deg,#ff9d6a1c 0 8px,#ff9d6a00 8px 16px)} + .scene-mem-overflow[data-on="true"] {opacity:1} + .scene-mem-overflow b {position:absolute;right:8px;top:8px;font-size:10px;font-weight:500;color:#ffd0b4;background:#2a120ae0;border:1px solid #9b5a39;border-radius:7px;padding:4px 8px;white-space:nowrap} + .scene-mem-overlay {position:absolute;top:0;bottom:0;pointer-events:none;border-left:1px dashed #48e9c7aa;background:linear-gradient(90deg,#48e9c729,#48e9c700 75%);transition:opacity .22s ease} + .scene-mem-overlay b {position:absolute;left:7px;top:8px;font-size:9px;letter-spacing:.5px;color:#8bf0da;white-space:nowrap;text-shadow:0 1px 3px #04121ad9} + .scene-mem-reserve {position:absolute;top:0;bottom:0;left:auto;right:0;pointer-events:none} + .scene-mem-reserve::before {content:"";position:absolute;top:0;bottom:0;left:0;width:1px;background:linear-gradient(180deg,#ffd79a00,#ffd79acc 18%,#ffd79acc 82%,#ffd79a00)} + .scene-mem-reserve b {position:absolute;left:0;bottom:6px;font-size:9px;letter-spacing:.5px;color:#e7bd82;white-space:nowrap;transform:translateX(-100%) translateX(-6px);text-shadow:0 1px 3px #04121ad9} + .scene-mem-bar[data-mode="external"] .scene-mem-overlay,.scene-mem-bar[data-mode="unknown"] .scene-mem-overlay {background:linear-gradient(90deg,#8fa8b833,#8fa8b800 75%);border-left-color:#9fb6c4aa} + .scene-mem-forecast {margin:13px 0 0;font-size:12px;line-height:1.55;color:#a9c3d3;display:flex;gap:9px;align-items:flex-start} + .scene-mem-forecast .scene-mem-forecast-icon {flex-shrink:0;width:8px;height:8px;margin-top:5px;border-radius:50%;background:#7f9aab} + .scene-mem-forecast[data-status="fits"] .scene-mem-forecast-icon {background:#3fe3cf;box-shadow:0 0 10px #3fe3cf7a} + .scene-mem-forecast[data-status="tight"] .scene-mem-forecast-icon {background:#f5c86a;box-shadow:0 0 10px #f5c86a7a} + .scene-mem-forecast[data-status="blocked"] .scene-mem-forecast-icon {background:#ff9d6a;box-shadow:0 0 10px #ff9d6a7a} + .scene-mem-forecast[data-status="resident"] .scene-mem-forecast-icon {background:#4fe0f2;box-shadow:0 0 10px #4fe0f27a} + .scene-mem-forecast[data-status="external"] .scene-mem-forecast-icon,.scene-mem-forecast[data-status="other-node"] .scene-mem-forecast-icon {background:#9db7c7} + .scene-mem-forecast strong {color:#eaf6fb;font-weight:500} + .scene-mem-forecast p {margin:0;flex:1} + .scene-mem-forecast .scene-mem-hint {color:#7d99ac;font-size:11px;margin-top:4px;display:block} + .scene-mem-legend {display:grid;grid-template-columns:repeat(auto-fit,minmax(196px,1fr));gap:6px;margin:15px 0 0} + .scene-mem-key {display:flex;align-items:center;gap:10px;width:100%;text-align:left;padding:9px 11px;border:1px solid #1c333f;border-radius:10px;background:#0a151c;color:#d3e4ee;font:inherit;font-size:12px;min-height:38px;transition:border-color .2s ease,background .2s ease,box-shadow .2s ease} + .scene-mem-key:hover {border-color:#2c6072;background:#0d1d26} + .scene-mem-key[data-selected="true"] {border-color:#3ed4ea;box-shadow:0 0 0 1px #28b9d126,0 6px 20px #0ad0f014} + .scene-mem-key[data-preview="true"] {border-color:#2e7f92;background:#0e222c} + .scene-mem-key:focus-visible {outline:2px solid #6ff0ff;outline-offset:2px} + .scene-mem-swatch {width:12px;height:12px;border-radius:4px;flex-shrink:0;background:var(--seg-fill);box-shadow:0 0 0 1px #ffffff1a} + .scene-mem-key-name {flex:1;min-width:0;overflow:hidden;text-overflow:ellipsis;white-space:nowrap} + .scene-mem-key-size {font-size:11px;color:#9db9c9;white-space:nowrap} + .scene-mem-key[data-estimated="true"] .scene-mem-key-size::after {content:" est.";color:#7d99ac} + .scene-mem-key em {font-style:normal;font-size:10px;color:#7d99ac;letter-spacing:.4px} + .scene-mem-rel {font-size:10px;color:#7d99ac;padding:4px 7px;border:1px solid #24404d;border-radius:7px;white-space:nowrap} + .scene-mem-key[data-preview="true"] .scene-mem-rel {color:#8bf0da;border-color:#2b6f7d} + .scene-mem-note {font-size:11px;color:#7d99ac;line-height:1.6;margin:13px 0 0;display:flex;gap:9px;flex-wrap:wrap;align-items:center} + .scene-mem-note span.scene-mem-chip {display:inline-flex;align-items:center;gap:6px;border:1px solid #1f3a47;border-radius:20px;padding:4px 10px;background:#08131a} + .scene-mem-note span.scene-mem-chip::before {content:"";width:8px;height:8px;border-radius:3px;background:#3d5b6b} + .scene-mem-note span.scene-mem-chip.measured::before {background:#2ec9e0} + .scene-mem-note span.scene-mem-chip.estimated::before {background:repeating-linear-gradient(115deg,#f3c268 0 3px,#f3c26800 3px 6px)} + .scene-mem-empty {padding:22px;font-size:12px;color:#93aebe;line-height:1.6} + @keyframes sceneMemBreathe {0%,100%{opacity:.7}50%{opacity:1}} + .scene-mem-bar[data-preview="true"] .scene-mem-ghost > .scene-mem-ghost-fill {animation:sceneMemBreathe 2.6s ease-in-out infinite} + .scene-mem-sticky {position:sticky;top:12px;align-self:start;margin-bottom:16px} + .scene-model-layout {display:grid;grid-template-columns:minmax(0,1fr) 310px;gap:22px} + .scene-model-list {display:flex;flex-direction:column;gap:9px} + .scene-model-row {display:grid;grid-template-columns:44px minmax(130px,1.4fr) minmax(75px,.7fr) auto;gap:16px;align-items:center;border:1px solid #203541;border-radius:13px;padding:19px;background:linear-gradient(120deg,#111d2490,#09141b80);cursor:pointer;text-align:left;min-height:92px} + .scene-model-row[data-selected="true"] {border-color:#29d3eb;background:linear-gradient(100deg,#0a2a37c0,#0c1c2690)} + .scene-model-row > .scene-icon {width:44px;height:48px;background:linear-gradient(135deg,#24374688,#101d2a);border:1px solid #2b414d;border-radius:11px;color:#d6e9f4} + .scene-model-row > .scene-icon svg {width:25px;height:25px} + .scene-model-row h3 {margin:0;font-size:16px;overflow-wrap:anywhere;font-weight:500;line-height:1.35} + .scene-model-row p {font-size:12px;color:#96aebe;margin:6px 0 0} + .scene-row-status {font-size:12px;color:#a6c2d1;display:flex;align-items:center;gap:7px} + .scene-row-status::before {content:"";width:7px;height:7px;border-radius:50%;background:#718b9c;flex-shrink:0} + .scene-model-row[data-ready="true"] .scene-row-status::before {background:#44dccf;box-shadow:0 0 8px #44dccf22} + .scene-row-policy {font-size:10px;color:#a1bbc9;border:1px solid #28414e;border-radius:8px;padding:7px 9px;white-space:nowrap} + .scene-model-tabs {display:flex;align-items:center;gap:25px;border-bottom:1px solid #273b45;margin:0 0 18px} + .scene-model-tabs strong {padding:0 0 13px;font-size:16px;border-bottom:3px solid #1ed5f0} + .scene-model-tabs button {background:none;border:0;color:#8aa5b7;padding:0 0 15px} + .scene-chips {display:flex;gap:8px;flex-wrap:wrap;margin:0 0 18px} + .scene-chips button {border-radius:22px;padding:8px 16px;background:#0b141a;font-size:12px;border-color:#34454e} + .scene-chips button[aria-pressed="true"] {border-color:#29dcee;background:#0d2c36;color:#55e9f5} + .scene-add-capability {display:flex;gap:15px;align-items:center;border:1px dashed #35515d;border-radius:13px;padding:22px;margin-top:18px;background:#09131977} + .scene-add-capability > .scene-icon {width:40px;height:40px;border:1px solid #8aa9b8;border-radius:50%} + .scene-add-capability strong {font-size:16px;font-weight:500;display:block} + .scene-add-capability p {font-size:12px;color:#98b3c4;margin:7px 0 0} + .scene-add-capability button {margin-left:auto;padding:11px 16px} + .scene-network {display:flex;align-items:center;justify-content:center;gap:0;padding:40px 30px;margin-bottom:30px;min-height:190px;overflow:auto} + .scene-device {min-width:110px;text-align:center} + .scene-device .scene-icon {display:flex;filter:drop-shadow(0 0 12px #20d3f72b);color:#a7edff;margin-bottom:12px} + .scene-device svg {width:88px;height:70px;stroke-width:.65} + .scene-device strong {font-size:13px;font-weight:500} + .scene-wire {height:1px;min-width:60px;flex:1;max-width:160px;background:linear-gradient(90deg,#22d3f4,#1c8fa5);box-shadow:0 0 9px #20d5ed60;position:relative;margin:0 16px 22px} + .scene-wire::before,.scene-wire::after {content:"";position:absolute;top:-3px;width:7px;height:7px;border-radius:50%;background:#28def4;box-shadow:0 0 12px #25d5ee} + .scene-wire::after {right:0} + .scene-wire span {position:absolute;top:-22px;left:0;right:0;text-align:center;font-size:10px;color:#72dbe9;white-space:nowrap} + .scene-machine-list {display:flex;flex-direction:column;gap:12px} + .scene-machine-list .presence-card {display:grid;grid-template-columns:54px minmax(120px,1fr) minmax(150px,1fr) auto;gap:24px;align-items:center;padding:23px} + .scene-machine-list .presence-card > .scene-icon {color:#aae9f8;filter:drop-shadow(0 0 9px #1cc5df40)} + .scene-machine-list .presence-card > .scene-icon svg {width:52px;height:45px;stroke-width:.8} + .scene-machine-list .presence-card h3 {font-size:17px} + .scene-machine-list .presence-card p {font-size:12px;margin-top:7px} + @keyframes scene-breathe {50%{box-shadow:0 0 0 10px #0a2c3630,0 0 46px #14d3f262,inset 0 0 28px #10cbea25}} + @media(min-width:1650px) {.scene-layout,.scene-model-layout {grid-template-columns:minmax(0,1fr) 360px}.scene-columns {gap:30px;padding:30px}.scene-diagram,.scene-columns,.scene-detail {min-height:610px}.scene-model-name{font-size:13px}} + @media(max-width:1220px) {.scene-layout,.scene-model-layout {grid-template-columns:minmax(0,1fr) 280px;gap:12px}.scene-columns {grid-template-columns:120px 90px minmax(210px,1fr);gap:8px;padding:20px 15px}.scene-gateway {width:78px;height:78px}.scene-detail {padding:18px}.scene-mini-memory{width:58px}.scene-model-row {grid-template-columns:35px minmax(0,1fr) auto;gap:12px;padding:15px}.scene-row-policy{display:none}.scene-model-row > .scene-icon{width:35px;height:40px}.scene-model-row h3{font-size:14px}.scene-model-state{font-size:9px;padding:4px 6px}.scene-model-name{font-size:11px}} + @media(max-width:1050px) {.scene-layout,.scene-model-layout {grid-template-columns:1fr}.scene-detail {min-height:0}.scene-detail:has(.scene-placeholder:not([hidden])){display:none}.scene-columns{grid-template-columns:minmax(125px,.8fr) minmax(100px,.8fr) minmax(230px,1.4fr);gap:20px}.scene-analytics{padding:20px;gap:18px}.scene-chart + .scene-chart{padding-left:18px}.scene-machine-list .presence-card{grid-template-columns:44px 1fr auto;gap:18px}.scene-machine-list .presence-card .machine-memory{grid-column:2 / -1;grid-row:2}.scene-model-row {grid-template-columns:40px minmax(0,1fr) auto auto}.scene-row-policy{display:block}} + @media(max-width:680px) {.presence-brand{display:none}.presence-nav{padding:8px}.presence-nav button[aria-current]::before{left:12px;right:12px;top:auto;bottom:0;width:auto;height:2px}body > header{display:none}main{padding-top:22px}.presence-heading h2{font-size:29px}.scene-columns{grid-template-columns:1fr 1fr;gap:18px;padding:20px;min-height:0}.scene-gateway-column{grid-column:2;grid-row:1}.scene-client-column{grid-column:1;grid-row:1}.scene-machine-column{grid-column:1/-1}.scene-gateway-wrap{min-height:210px;padding:30px 0 0}.scene-clients{padding:25px 0;gap:12px}.scene-diagram{min-height:0}.scene-machines{padding-top:16px}.scene-analytics{grid-template-columns:1fr;gap:22px}.scene-chart + .scene-chart{border-left:0;border-top:1px solid #20323e;padding:20px 0 0}.scene-chart svg{height:74px}.scene-model-row{grid-template-columns:34px minmax(0,1fr) auto;padding:15px 12px;gap:10px}.scene-model-row .scene-row-policy{display:none}.scene-row-status{font-size:10px}.scene-memory-panel{padding:17px 15px 15px}.scene-mem-sticky{position:static}.scene-mem-bar{height:112px}.scene-mem-bar[data-known="false"]{height:60px}.scene-mem-legend{grid-template-columns:1fr}.scene-mem-readouts{gap:14px}.scene-mem-total{font-size:23px}.scene-mem-readout{min-width:72px}.scene-add-capability{padding:18px;flex-wrap:wrap}.scene-add-capability button{width:100%;margin:0}.scene-network{padding:25px 18px;justify-content:flex-start}.scene-machine-list .presence-card{grid-template-columns:40px 1fr;gap:14px;padding:18px}.scene-machine-list .presence-card .actions{grid-column:1/-1}.scene-machine-list .presence-card .machine-memory{grid-column:1/-1;grid-row:auto}.scene-links{opacity:.85}.scene-model-name{font-size:12px}.scene-model-state{font-size:10px}} + @media(max-width:1050px){.scene-detail:has(.model-inspector.open){position:fixed;z-index:80;left:12px;right:12px;bottom:12px;height:auto;max-height:calc(100dvh - 96px);background:linear-gradient(145deg,#122530,#09141c);box-shadow:0 -20px 70px #0009,0 0 0 1px #42616b66;padding:22px}.scene-detail:has(.model-inspector.open) .model-inspector-header{position:sticky;top:-22px;margin-top:-22px;padding-top:22px;background:#10212b;z-index:1}} + @media(prefers-reduced-motion:reduce){.scene-gateway{animation:none!important}.scene-links .scene-particle{display:none}.scene-model{transition:none}} + .scene-memory-panel {padding:18px 20px;margin-bottom:20px;background:#0b151d} + .scene-memory-panel h3 {margin-bottom:14px}.scene-memory-panel h3 small{display:block;color:#67dbe7;margin-top:5px}.scene-memory-panel h3 small[hidden]{display:none} + .scene-mem-head {margin-bottom:12px}.scene-mem-total {font-size:22px}.scene-mem-total b {font-weight:500} + .scene-mem-bar {height:76px}.scene-mem-block {border-radius:0;transition:width .38s cubic-bezier(.22,1,.36,1),filter .2s,opacity .2s} + .scene-mem-face em {max-width:100%;overflow:hidden;text-overflow:ellipsis;font-size:10px} + .scene-mem-legend {grid-template-columns:repeat(auto-fit,minmax(145px,1fr));gap:5px;margin-top:10px} + .scene-mem-key {padding:7px 9px;font-size:11px;min-height:44px;gap:7px}.scene-mem-key-size {font-size:10px} + .scene-mem-note {font-size:10px;margin:9px 0 0}.scene-mem-forecast {margin-top:10px;min-height:36px;font-size:12px} + .scene-mem-ruler {overflow:hidden}.scene-mem-tick {position:absolute;top:0;bottom:0;width:0}.scene-mem-tick:last-child span {left:auto;right:5px} + .scene-mem-reserve {right:0;background:repeating-linear-gradient(120deg,#edbe6c0d 0 5px,transparent 5px 10px)} + .scene-mem-reserve b {transform:none;left:6px;bottom:6px;font-size:8px;white-space:normal;line-height:1.2} + .scene-mem-ghost {z-index:3}.scene-mem-reserve {z-index:4} + .scene-model-row[data-preview="true"],.scene-model[data-preview="true"] {border-color:#53cada;box-shadow:inset 0 0 25px #36d9e507,0 0 20px #31b6ce0b} + .scene-row-footprint {color:#7dbbc8;font-size:11px;white-space:nowrap} + @media(min-width:1051px){.scene-memory-panel{position:sticky;top:12px;z-index:7}#scene-model-detail{align-self:start;position:sticky;top:12px;max-height:calc(100dvh - 24px);height:auto;min-height:440px}.scene-model-row{scroll-margin-top:390px}} + @media(max-width:680px){.scene-memory-panel{padding:16px 12px}.scene-mem-bar{height:64px}.scene-mem-legend{grid-template-columns:repeat(2,minmax(0,1fr))}.scene-mem-key{min-width:0}.scene-mem-head{gap:8px}.scene-mem-readouts{gap:10px}.scene-mem-total{font-size:20px}.scene-mem-forecast{font-size:11px}.scene-mem-reserve b{font-size:7px}.scene-mem-block.compact .scene-mem-face{display:none}.scene-row-footprint{display:block;margin-top:4px}} + @media(prefers-reduced-motion:reduce){.scene-mem-block,.scene-mem-ghost{transition:none!important}.scene-mem-ghost-fill{animation:none!important}} + +`; + +export const sceneScript = String.raw` + const scenePaths={ + model:'', + chat:'', + code:'', + image:'', + audio:'', + laptop:'', + server:'', + cloud:'', + terminal:'', + plus:'' + }; + function sceneIcon(kind){return '';} + const sceneLogo=''; + function sceneKind(model){return ({chat:'Chat & code',embedding:'Search & retrieval',audio_speech:'Speech',audio_transcription:'Transcription',audio_generation:'Music & audio',image:'Generate & edit',video:'Video'})[model.kind] || model.kind || 'Chat & code';} + function sceneModelIcon(model){return (model.kind||'').startsWith('audio')?'audio':(model.kind||'').startsWith('image')?'image':model.kind==='chat'?'model':'code';} + function sceneMemory(node){ + const memory=node?.telemetry?.memory || {}; + const total=Number(memory.totalBytes), available=Number(memory.availableBytes), rawUsed=Number(memory.usedBytes); + const used=memory.availableBytes!=null&&Number.isFinite(available)?total-available:memory.usedBytes!=null?rawUsed:NaN; + return {total,used,available,known:total>0&&Number.isFinite(used),percent:Math.max(0,Math.min(100,used/total*100))}; + } + function sceneNodes(){const nodes=Object.values(state.status?.cluster?.nodes || {});return nodes.length?nodes:state.topologySummary?.host?[{id:'local',local:true,name:'This machine',telemetry:state.topologySummary.host}]:[];} + function sceneNodeName(node){return node.local?'This machine':node.name || node.id;} + function sceneMiniMemory(node){const m=sceneMemory(node);return m.known?'
'+escapeHtml(formatBytes(m.used))+' / '+escapeHtml(formatBytes(m.total))+'
':'';} + function sceneDeviceIcon(node){return node.local || /apple|mac/i.test(node.profile?.platformId || node.profile?.cpuBrand || '')?'laptop':'server';} + document.querySelector('.presence-brand').innerHTML=sceneLogo+'LLooMby Enntity'; + const scene=document.createElement('section');scene.id='presence-scene';scene.dataset.presencePanel='live'; + scene.innerHTML='

Clients

Waiting for telemetry

LLooM Gateway

Routes and balances

'+sceneLogo+'
GatewayConnecting…

Models on your machines

Requests

Observed during this session

Response time

Recent completed requests · includes generation time

Machine memory Used / total

Waiting for gateway telemetry
'; + $('.topology').before(scene); + $('.topology').dataset.presencePanel='diagnostic';$('.topology').hidden=true; + const modelLeft=document.createElement('div');modelLeft.className='scene-model-main'; + const modelLayout=document.createElement('div');modelLayout.className='scene-model-layout';$('#view-models').append(modelLayout);modelLayout.append(modelLeft); + const memoryPanel=document.createElement('section');memoryPanel.className='scene-memory-panel';memoryPanel.innerHTML='

Room for your AI

Point to a model to preview its memory.

'; + modelLeft.append(memoryPanel); + const modelTabs=document.createElement('div');modelTabs.className='scene-model-tabs';modelTabs.innerHTML='Installed'; + modelLeft.append(modelTabs,$('.presence-toolbar')); + const chips=document.createElement('div');chips.className='scene-chips';chips.innerHTML=[['','All'],['chat','Chat & code'],['image','Images'],['audio','Voice'],['embedding','Search'],['video','Video']].map(([id,name])=>'').join(''); + modelLeft.append(chips,$('#presence-models'));$('#presence-models').className='scene-model-list';$('#presence-kind').hidden=true; + const capability=document.createElement('div');capability.className='scene-add-capability';capability.innerHTML=sceneIcon('plus')+'
Add a capability

Get more done with another model.

';modelLeft.append(capability); + const modelDetail=document.createElement('aside');modelDetail.id='scene-model-detail';modelDetail.className='scene-detail';modelDetail.innerHTML='
'+sceneIcon('model')+'

Ready when you are.

Choose a model to try it or connect an app. LLooM takes care of getting it ready.

'; + modelLayout.append(modelLeft,modelDetail);$('#view-models').append(modelLayout); + const network=document.createElement('div');network.id='scene-network';network.className='scene-network';$('#presence-machines').before(network);$('#presence-machines').className='scene-machine-list'; + const inspectorMemory=document.createElement('div');inspectorMemory.className='scene-inspector-memory';inspectorMemory.id='scene-inspector-memory';$('#presence-availability').after(inspectorMemory); + let sceneClientKey='',sceneMachineKey='',sceneModelKey='',sceneNodeKey='',sceneLinkKey='',sceneSamples=[],sceneSampleAt=0; + let sceneMemoryNode=null,sceneFollowing=false; + let sceneMemPointer=null,sceneMemFocus=null,sceneMemShape='',sceneMemPaint=false; + const sceneMemColors=['#25bbd8','#53d5b6','#709eec','#a58ceb','#e1b176','#dc93bd','#79c5ce','#a0bb78']; + function sceneMemColor(segment){return segment.kind==='system'?'#304955':segment.kind==='available'?'#123337':sceneMemColors[Math.abs(segment.colorIndex||0)%sceneMemColors.length];} + function sceneMemModelNode(id){ + const model=(state.physicalModels||[]).find(m=>m.id===id),rt=model&&presenceRuntime(model); + if(!model?.runtime)return model?.targets?.find(target=>target.node)?.node||null; + return rt?.node||rt?.placement?.node||(!rt?.remote?sceneNodes().find(n=>n.local)?.id:null); + } + function sceneMemSnapshot(id=sceneMemPointer||sceneMemFocus||state.selectedModelId,ownNode=false){ + const nodes=sceneNodes(),follow=sceneMemPointer||sceneMemFocus; + const target=ownNode?sceneMemModelNode(id):follow&&sceneMemModelNode(follow); + const node=nodes.find(n=>n.id===(target||sceneMemoryNode))||nodes.find(n=>n.local)||nodes[0]; + return buildMemoryMap({node,runtimes:state.status?.runtimeManager?.runtimes||{},models:state.physicalModels||[],previewModelId:id,memorySafety:state.status?.runtimeManager?.memorySafety}); + } + function sceneMemFormat(value){return typeof value==='number'&&Number.isFinite(value)?formatBytes(Math.max(0,value)):'—';} + function sceneMemText(memory){ + const p=memory.preview; + if(!p)return 'Point to a model to see where it would fit. Nothing starts until you use it.'; + const status={fits:'Expected to fit',tight:'Little room to spare',blocked:'Needs more room',resident:'Already available',external:'Runs elsewhere','other-node':'Runs on another machine',unknown:'Footprint not yet known',paused:'Paused'}[p.status]||'Checking room'; + const delta=p.additionalBytes>0?' · about '+sceneMemFormat(p.additionalBytes)+' more':''; + const remaining=p.additionalBytes>0&&p.remainingBytes!=null?' · '+(p.remainingBytes<0?sceneMemFormat(-p.remainingBytes)+' over capacity':sceneMemFormat(p.remainingBytes)+' available after'):''; + return p.label+' · '+status+delta+remaining; + } + function sceneMemSchedule(){if(sceneMemPaint)return;sceneMemPaint=true;requestAnimationFrame(()=>{sceneMemPaint=false;sceneMemRender();});} + function sceneMemRender(){ + const host=$('#scene-memory-bar');if(!host||typeof buildMemoryMap!=='function')return; + const memory=sceneMemSnapshot(),segments=memory.segments||[],p=memory.preview; + const select=$('#scene-memory-machine'),nodes=sceneNodes(),base=sceneMemoryNode||nodes.find(n=>n.local)?.id||nodes[0]?.id; if(base)select.value=base;const context=$('#scene-memory-context');context.hidden=memory.nodeId===base;context.textContent=context.hidden?'':'Preview · '+sceneNodeName(nodes.find(n=>n.id===memory.nodeId)||{id:memory.nodeId}); + const shape=JSON.stringify([memory.nodeId,memory.known,segments.map(s=>[s.id,s.kind,s.modelIds])]); + if(shape!==sceneMemShape){ + const focused=host.contains(document.activeElement)?document.activeElement.dataset.memoryFocus:null; + sceneMemShape=shape; + host.innerHTML=memory.known?'
total memory
In use
Available
'+segments.map((s,i)=>{const model=s.modelIds?.[0],tag=model?'button':'div';return '<'+tag+(model?' type="button" data-presence-model="'+escapeHtml(model)+'"':'')+' class="scene-mem-block" data-memory-index="'+i+'" data-memory-focus="block-'+i+'" data-kind="'+escapeHtml(s.kind)+'">';}).join('')+'Protected headroom
'+segments.map((s,i)=>{const model=s.modelIds?.[0],tag=model?'button':'div';return '<'+tag+(model?' type="button" data-presence-model="'+escapeHtml(model)+'"':'')+' class="scene-mem-key" data-memory-key="'+i+'" data-memory-focus="key-'+i+'">';}).join('')+'

':'
Waiting for a memory reading from this machine.
'; + if(focused)host.querySelector('[data-memory-focus="'+CSS.escape(focused)+'"]')?.focus({preventScroll:true}); + } + if(memory.known){ + $('#scene-mem-total').textContent=sceneMemFormat(memory.totalBytes); + $('#scene-mem-used').textContent=sceneMemFormat(memory.usedBytes); + $('#scene-mem-free').textContent=sceneMemFormat(memory.availableBytes); + const ruler=$('#scene-mem-ruler'),rulerKey=String(memory.totalBytes); + if(ruler.dataset.total!==rulerKey){ruler.dataset.total=rulerKey;ruler.innerHTML=Array.from({length:5},(_,i)=>''+escapeHtml(sceneMemFormat(memory.totalBytes*i/4))+'').join('');} + for(const [i,s] of segments.entries()){ + const block=host.querySelector('[data-memory-index="'+i+'"]'),key=host.querySelector('[data-memory-key="'+i+'"]'); + const isPreview=Boolean(p&&s.modelIds?.includes(p.modelId)),selected=Boolean(state.selectedModelId&&s.modelIds?.includes(state.selectedModelId)); + for(const el of [block,key]){el.style.setProperty('--seg-fill',sceneMemColor(s));el.dataset.estimated=String(Boolean(s.estimated));el.dataset.selected=String(selected);el.dataset.preview=String(isPreview);el.setAttribute('aria-label',s.label+', '+sceneMemFormat(s.bytes)+(s.estimated?', approximate':''));el.title=s.label+' · '+sceneMemFormat(s.bytes)+(s.estimated?' (estimate)':'');} + block.style.width=Math.max(0,Math.min(100,s.percent||0))+'%';block.classList.toggle('narrow',s.percent<10);block.classList.toggle('compact',s.percent<20); + block.querySelector('em').textContent=s.label;block.querySelector('b').textContent=sceneMemFormat(s.bytes); + key.querySelector('.scene-mem-key-name').textContent=s.label;key.querySelector('.scene-mem-key-size').textContent=sceneMemFormat(s.bytes); + } + const ghost=$('#scene-mem-ghost'),delta=p?.additionalBytes; + ghost.hidden=!(delta>0);ghost.style.left=memory.usedBytes/memory.totalBytes*100+'%';ghost.style.width=Math.max(0,Math.min(delta||0,memory.availableBytes))/memory.totalBytes*100+'%';ghost.dataset.overflow=String(p?.status==='blocked'); + $('#scene-mem-track').dataset.preview=String(Boolean(p)); + const reserve=$('#scene-mem-reserve');reserve.hidden=!(memory.reserveBytes>0);reserve.style.width=memory.reserveBytes/memory.totalBytes*100+'%';reserve.title=sceneMemFormat(memory.reserveBytes)+' protected for your machine';reserve.querySelector('b').textContent=sceneMemFormat(memory.reserveBytes)+' reserved'; + $('#scene-mem-note').textContent=memory.attributionNote||'Live memory use. Hover previews are estimates.'; + } + const forecast=$('#scene-memory-forecast');forecast.dataset.status=p?.status||'idle';const forecastText=memory.known?sceneMemText(memory):'Memory is not available yet. LLooM will check before preparing a model.';if(forecast.querySelector('p').textContent!==forecastText)forecast.querySelector('p').textContent=forecastText; + for(const el of document.querySelectorAll('.scene-model-row,.scene-model'))el.dataset.preview=String(el.dataset.presenceModel===p?.modelId); + if(state.selectedModelId){const own=sceneMemSnapshot(state.selectedModelId,true).preview;$('#scene-inspector-memory').innerHTML=''+(own?.additionalBytes>0?'Expected extra memory':'Memory')+''+escapeHtml(own?.additionalBytes>0?'About '+sceneMemFormat(own.additionalBytes):own?.status==='resident'?'Already available':own?.status==='external'?'Runs elsewhere':'Checked when needed')+'';} + } + const sceneMemTarget=target=>target?.closest?.('[data-presence-model]')?.dataset.presenceModel||null; + document.addEventListener('pointerover',event=>{if(event.pointerType==='touch')return;const id=sceneMemTarget(event.target);if(id&&id!==sceneMemPointer){sceneMemPointer=id;sceneMemSchedule();}}); + document.addEventListener('pointerout',event=>{if(event.pointerType==='touch')return;if(sceneMemTarget(event.target)&&sceneMemTarget(event.relatedTarget)!==sceneMemPointer){sceneMemPointer=sceneMemTarget(event.relatedTarget);sceneMemSchedule();}}); + document.addEventListener('focusin',event=>{const id=sceneMemTarget(event.target);if(id)sceneMemPointer=null;if(id!==sceneMemFocus){sceneMemFocus=id;sceneMemSchedule();}}); + document.addEventListener('focusout',event=>{sceneMemFocus=sceneMemTarget(event.relatedTarget);sceneMemSchedule();}); + function sceneMoveInspector(){ + const slot=presenceView==='models'?modelDetail:$('#scene-live-detail'); + if($('#model-inspector').parentElement!==slot)slot.append($('#model-inspector')); + document.querySelectorAll('.scene-placeholder').forEach(item=>item.hidden=Boolean(state.selectedModelId)&&item.parentElement===slot); + if(state.selectedModelId){const model=state.physicalModels.find(m=>m.id===state.selectedModelId),rt=model&&presenceRuntime(model);inspectorMemory.innerHTML='Memory estimate'+(rt?.memoryGb!=null?escapeHtml(rt.memoryGb)+' GB':'Not reported')+'';} + } + const scenePreviousView=presenceSetView; + presenceSetView=function(name){sceneMemPointer=null;sceneMemFocus=null;scenePreviousView(name);sceneMoveInspector();renderScene();}; + const scenePreviousInspector=renderModelInspector; + renderModelInspector=function(){scenePreviousInspector();sceneMoveInspector();sceneMemSchedule();}; + renderPresenceModels=function(){ + const search=$('#presence-search').value.trim().toLowerCase(),kind=$('#presence-kind').value; + const models=(state.physicalModels||[]).filter(m=>(!search||(m.name+' '+m.id).toLowerCase().includes(search))&&(!kind||(m.kind||'chat').startsWith(kind))).sort((a,b)=>Number(Boolean(b.runtime))-Number(Boolean(a.runtime))||Number(Boolean(presenceRuntime(b)?.healthy))-Number(Boolean(presenceRuntime(a)?.healthy))); + $('#presence-model-count').textContent=models.length+(models.length===1?' model':' models'); + const key=JSON.stringify(models.map(m=>[m.id,m.name,sceneKind(m),presenceModelLabel(m),presencePolicy(presenceRuntime(m)),state.selectedModelId===m.id]));if(key===sceneModelKey)return;sceneModelKey=key; + const catalog=$('#presence-models'),focused=catalog.contains(document.activeElement)?document.activeElement.closest('[data-presence-model]')?.dataset.presenceModel:null; + $('#presence-models').innerHTML=models.map(model=>{const rt=presenceRuntime(model);return '';}).join('')||'
No models match. Choose another filter or add a model.
'; + if(focused)catalog.querySelector('[data-presence-model="'+CSS.escape(focused)+'"]')?.focus({preventScroll:true}); + if(sceneMemPointer&&!models.some(m=>m.id===sceneMemPointer))sceneMemPointer=null; + sceneMemSchedule(); + }; + renderPresenceMachines=function(){ + const nodes=sceneNodes(),key=JSON.stringify(nodes.map(n=>[n.id,n.name,n.local,n.reachable,n.profile?.cpuBrand,sceneMemory(n)]));if(key===sceneNodeKey)return;sceneNodeKey=key; + network.innerHTML=nodes.map((node,index)=>(index?'
'+(node.reachable===false?'Unavailable':'Configured connection')+'
':'')+'
'+sceneIcon(sceneDeviceIcon(node))+''+escapeHtml(sceneNodeName(node))+'
').join('')||'

Waiting for machine telemetry

'; + $('#presence-machines').innerHTML=nodes.map(node=>{const m=sceneMemory(node);return '
'+sceneIcon(sceneDeviceIcon(node))+'

'+escapeHtml(sceneNodeName(node))+'

'+escapeHtml(node.profile?.cpuBrand||node.id)+' · '+(node.reachable===false?'Unavailable':node.local?'This computer':'Configured peer')+'

'+(m.known?escapeHtml(formatBytes(m.total))+' memory':'Memory unavailable')+'

'+(m.known?escapeHtml(formatBytes(m.used))+' used / '+escapeHtml(formatBytes(m.total)):'No reading')+'

';}).join(''); + }; + function sceneChart(selector,values,color){ + const svg=$(selector),width=300,height=78,max=Math.max(1,...values),points=values.map((value,index)=>[values.length>1?index/(values.length-1)*width:0,height-4-(value/max)*(height-10)]); + const path=points.map((p,i)=>(i?'L':'M')+p[0].toFixed(1)+','+p[1].toFixed(1)).join(' '); + svg.setAttribute('viewBox','0 0 300 80');svg.setAttribute('preserveAspectRatio','none'); + svg.innerHTML=''+[0,26,52,78].map(y=>'').join('')+(values.length>1?'':''); + } + function renderScene(){ + if(!scene.isConnected)return; + const nodes=sceneNodes(),models=state.physicalModels||[],connections=(state.topologyConnections||[]).filter(c=>c.live),summary=state.topologySummary||{}; + const clients=new Map();for(const connection of connections){const id=connection.caller||connection.requester||'API client';if(!clients.has(id))clients.set(id,{name:id,count:0});clients.get(id).count++;} + const clientRows=[...clients.values()].slice(0,4),clientKey=JSON.stringify(clientRows); + if(clientKey!==sceneClientKey||!$('#scene-clients').children.length){sceneClientKey=clientKey;$('#scene-clients').innerHTML=clientRows.map(client=>'
'+sceneIcon(/code|terminal|cli|agent/i.test(client.name)?'terminal':'chat')+'
'+escapeHtml(client.name)+''+client.count+' active request'+(client.count===1?'':'s')+'
').join('')||'
'+sceneIcon('terminal')+'
Ready for a clientConnect an app to see its requests here.
';sceneLinkKey='';} + $('#scene-client-count').textContent=connections.length?clients.size+' active client'+(clients.size===1?'':'s'):'No active requests'; + $('#scene-active').textContent=(summary.active||0)?summary.active+' active request'+(summary.active===1?'':'s'):'Ready when you are';$('#scene-gateway').dataset.active=String(Boolean(summary.active)); + $('#scene-machine-count').textContent=nodes.length+' machine'+(nodes.length===1?'':'s')+' · '+models.length+' models'; + const groups=nodes.map(node=>({node,models:models.filter(model=>{const topology=(state.topologyCatalogModels||[]).find(m=>m.id===model.id);const rt=presenceRuntime(model),ids=[...new Set([...(topology?.nodes||[]),...(model.targets||[]).map(t=>t.node),...(rt?.members||[]).map(m=>m.node),rt?.node].filter(Boolean))];return ids.includes(node.id)||(!ids.length&&Boolean(model.runtime)&&node.local);})})); + const assigned=new Set(groups.flatMap(g=>g.models.map(m=>m.id)));const external=models.filter(m=>!assigned.has(m.id));if(external.length)groups.push({node:{id:'external',name:'External providers'},models:external,external:true}); + const visibleGroups=groups.map(group=>({...group,models:group.models.slice().sort((a,b)=>Number(presenceRuntime(b)?.activeRequests>0)-Number(presenceRuntime(a)?.activeRequests>0)||Number(Boolean(presenceRuntime(b)?.healthy))-Number(Boolean(presenceRuntime(a)?.healthy)))})); + const machineKey=JSON.stringify(visibleGroups.map(g=>[g.node.id,g.node.reachable,g.models.map(m=>[m.id,m.name,presenceModelLabel(m),connections.some(c=>c.model===m.id),m.id===state.selectedModelId]),sceneFollowing])); + if(machineKey!==sceneMachineKey){sceneMachineKey=machineKey;$('#scene-machines').innerHTML=visibleGroups.filter(g=>!sceneFollowing||!summary.active||g.models.some(m=>presenceRuntime(m)?.activeRequests>0||connections.some(c=>c.model===m.id))).map(group=>{ + const ranked=sceneFollowing&&summary.active?group.models.filter(m=>presenceRuntime(m)?.activeRequests>0||connections.some(c=>c.model===m.id)):group.models; + const shown=ranked.slice(0,group.external?2:3); + return '
'+sceneIcon(group.external?'cloud':sceneDeviceIcon(group.node))+'
'+escapeHtml(sceneNodeName(group.node))+''+escapeHtml(group.external?'Through your gateway':group.node.profile?.cpuBrand||group.node.id)+'
'+sceneMiniMemory(group.node)+'
'+shown.map(model=>{const label=presenceModelLabel(model),active=connections.some(c=>c.model===model.id),serving=label==='Serving'||(active&&!model.runtime&&!model.federated);return '';}).join('')+(group.models.length>shown.length?'':'')+'
'; + }).join('')||'
Add a model to bring your hardware to life.
';sceneLinkKey='';} + for(const group of visibleGroups){const box=[...document.querySelectorAll('[data-scene-node]')].find(el=>el.dataset.sceneNode===group.node.id),mini=box?.querySelector('.scene-mini-memory');if(mini){const m=sceneMemory(group.node);mini.querySelector('b').style.width=m.percent+'%';mini.querySelector('small').textContent=formatBytes(m.used)+' / '+formatBytes(m.total);}} + const minute=state.metrics?.rolling?.minute||{};$('#scene-requests-label').textContent=(summary.active||0)+' active'+(minute.requests!=null?' · '+minute.requests+'/min':''); + const now=Date.now();if(now-sceneSampleAt>1900){sceneSamples.push(Number(summary.active||0));if(sceneSamples.length>60)sceneSamples.shift();sceneSampleAt=now;} + sceneChart('#scene-request-chart',sceneSamples,'#29d9f5'); + const durations=(state.metrics?.recent||[]).slice(0,40).reverse().map(r=>Number(r.durationMs)/1000).filter(n=>Number.isFinite(n)&&n>=0);sceneChart('#scene-latency-chart',durations,'#47e7cc');$('#scene-latency-label').textContent=durations.length?(durations.reduce((a,b)=>a+b,0)/durations.length).toFixed(1)+'s avg':'No completed requests'; + $('#scene-memory-list').innerHTML=nodes.slice(0,4).map(node=>{const m=sceneMemory(node);return '
'+escapeHtml(sceneNodeName(node))+''+(m.known?escapeHtml(formatBytes(m.used))+' / '+escapeHtml(formatBytes(m.total)):'Unknown')+'
';}).join('')||'

Memory telemetry unavailable

'; + $('#scene-health').textContent=state.status?.error?'Gateway telemetry needs attention':nodes.length?'Live gateway telemetry · '+(summary.recentErrors||0)+' errors in the last minute':'Connecting to your gateway…'; + const machineSelect=$('#scene-memory-machine');const optionsKey=nodes.map(n=>n.id).join('|');if(machineSelect.dataset.nodes!==optionsKey){machineSelect.dataset.nodes=optionsKey;machineSelect.innerHTML=nodes.map(n=>'').join('');if(nodes.some(n=>n.id===sceneMemoryNode))machineSelect.value=sceneMemoryNode;} + sceneMemRender(); + sceneMoveInspector();requestAnimationFrame(sceneDrawLinks); + } + function sceneDrawLinks(){ + if(scene.hidden||document.hidden)return; + const root=$('.scene-diagram'),bounds=root.getBoundingClientRect(),gateway=$('#scene-gateway').getBoundingClientRect(),gx=gateway.left-bounds.left+gateway.width/2,gy=gateway.top-bounds.top+gateway.height/2; + const curves=[];for(const client of document.querySelectorAll('.scene-client')){const box=client.getBoundingClientRect();curves.push({x:box.right-bounds.left,y:box.top-bounds.top+box.height/2,toX:gx-gateway.width/2,toY:gy,active:client.dataset.active==='true'});} + for(const model of document.querySelectorAll('#scene-machines .scene-model')){const box=model.getBoundingClientRect(),viewport=$('#scene-machines').getBoundingClientRect();if(box.topviewport.bottom)continue;curves.push({x:gx+gateway.width/2,y:gy,toX:box.left-bounds.left,toY:box.top-bounds.top+box.height/2,active:model.dataset.active==='true'});} + const key=JSON.stringify(curves.map(c=>[Math.round(c.x),Math.round(c.y),Math.round(c.toX),Math.round(c.toY),c.active]));if(key===sceneLinkKey)return;sceneLinkKey=key; + const reduced=window.matchMedia('(prefers-reduced-motion: reduce)').matches; + $('#scene-links').setAttribute('viewBox','0 0 '+bounds.width+' '+bounds.height); + $('#scene-links').innerHTML=''+curves.map((c,index)=>{const bend=(c.toX-c.x)*.5;const d='M'+c.x+','+c.y+' C'+(c.x+bend)+','+c.y+' '+(c.toX-bend)+','+c.toY+' '+c.toX+','+c.toY;return (c.active?'':'')+''+(c.active&&!reduced?[0,1,2].map(n=>'').join(''):'')+'';}).join(''); + } + const sceneOriginalPresence=renderPresence;renderPresence=function(){sceneOriginalPresence();renderScene();}; + const sceneOriginalActivity=renderActivity;renderActivity=function(){sceneOriginalActivity();renderScene();}; + $('#scene-memory-machine').addEventListener('change',()=>{sceneMemoryNode=$('#scene-memory-machine').value;renderScene();}); + $('#scene-follow').addEventListener('change',()=>{sceneFollowing=$('#scene-follow').checked;sceneMachineKey='';renderScene();}); + $('#scene-diagnostic').addEventListener('click',()=>{const detail=$('.topology');detail.hidden=!detail.hidden;if(!detail.hidden)detail.scrollIntoView({behavior:window.matchMedia('(prefers-reduced-motion: reduce)').matches?'auto':'smooth',block:'start'});}); + document.addEventListener('click',event=>{const kind=event.target.closest('[data-scene-kind]');if(kind){$('#presence-kind').value=kind.dataset.sceneKind;document.querySelectorAll('[data-scene-kind]').forEach(b=>b.setAttribute('aria-pressed',String(b===kind)));renderPresenceModels();}if(event.target.closest('[data-scene-all]'))presenceSetView('models');const selected=event.target.closest('[data-presence-model]');if(selected){sceneMemoryNode=sceneMemModelNode(selected.dataset.presenceModel)||sceneMemoryNode;sceneModelKey='';sceneMachineKey='';renderPresenceModels();renderScene();}}); + $('#scene-machines').addEventListener('scroll',()=>{sceneLinkKey='';requestAnimationFrame(sceneDrawLinks);}); + window.addEventListener('resize',()=>{sceneLinkKey='';requestAnimationFrame(sceneDrawLinks);}); + document.addEventListener('visibilitychange',()=>{if(!document.hidden){sceneLinkKey='';renderScene();}}); + presenceSetView(location.hash.slice(1)); +`; diff --git a/src/dashboard.mjs b/src/dashboard.mjs index 7f9a34a..5f5e16d 100644 --- a/src/dashboard.mjs +++ b/src/dashboard.mjs @@ -1,3 +1,8 @@ +import { buildMemoryMap } from './dashboard-memory.mjs'; +import { presenceStyles, presenceNav, presenceViews } from './dashboard-presence.mjs'; +import { sceneStyles, sceneScript } from './dashboard-scene.mjs'; +import { presenceScript } from './dashboard-presence-client.mjs'; + const DASHBOARD_HTML = String.raw` @@ -458,6 +463,17 @@ const DASHBOARD_HTML = String.raw` +
+
+

Fleet Profiles

+ no profile +
+
+
+

One file describes routes and keep-warm per machine. Applying validates the whole target config first, then swaps atomically; local models load on first request.

+
+
+

Models

@@ -996,7 +1012,7 @@ const DASHBOARD_HTML = String.raw` state.topologyCamera.frameKey = ""; } else fitTopologyCameraToModels(); } - if (state.selectedModelId && !state.topologyModels.some(model => model.id === state.selectedModelId)) closeModelInspector(); + if (state.selectedModelId && !state.physicalModels.some(model => model.id === state.selectedModelId)) closeModelInspector(); renderTopologyModelFilter(); renderModelInspector(); } @@ -1004,6 +1020,7 @@ const DASHBOARD_HTML = String.raw` function renderRuntimes() { const runtimes = state.status?.runtimeManager?.runtimes || {}; const entries = Object.entries(runtimes); + $("#stat-runtimes").textContent = String(entries.length); $("#stat-active").textContent = String(entries.reduce((sum, [, runtime]) => sum + Number(runtime.activeRequests || 0), 0)); $("#stat-queued").textContent = String(entries.reduce((sum, [, runtime]) => sum + Number(runtime.queuedRequests || 0) + Number(runtime.admissionQueuedRequests || 0), 0)); @@ -1022,6 +1039,65 @@ const DASHBOARD_HTML = String.raw` '' ).join("") : '
No runtimes.
'; } + let fleetCache = null; + async function renderFleetProfiles() { + try { + const result = await getJson("/gateway/fleet/profiles"); + fleetCache = result; + } catch { + return; // fleet profiles are optional; a missing profiles dir is fine + } + const active = $("#fleet-active"); + if (active) { + active.querySelector("span:last-child").textContent = fleetCache.active || "no profile"; + active.querySelector(".dot").className = "dot " + (fleetCache.active ? "ok" : ""); + } + const host = $("#fleet-profiles"); + if (!host) return; + const profiles = fleetCache.profiles || []; + const cards = profiles.map(profile => { + if (profile.error) { + return '
' + escapeHtml(profile.name) + '' + escapeHtml(profile.error) + '
'; + } + const routeCount = Object.keys(profile.routes || {}).length; + const warmCount = Object.keys(profile.residency || {}).length; + const isActive = profile.active === true; + return '
' + escapeHtml(profile.name) + (isActive ? ' active' : '') + '' + + '' + escapeHtml(profile.description || "") + '' + + '
' + escapeHtml(String(routeCount)) + ' routes · ' + escapeHtml(String(warmCount)) + ' warm
' + + (isActive ? '' : '') + + '
'; + }); + const current = fleetCache.active ? '' : ''; + host.innerHTML = cards.join("") + current || 'No profiles yet.'; + } + + document.addEventListener("click", async event => { + const useButton = event.target.closest("[data-fleet-use]"); + if (useButton) { + const name = useButton.dataset.fleetUse; + if (!confirm("Apply fleet profile " + name + "? Routes and keep-warm are swapped atomically.")) return; + try { + showOutput(await postJson("/gateway/fleet/profiles/" + encodeURIComponent(name) + "?apply=1", { yes: true })); + await refresh(); + } catch (error) { + showOutput({ error: error.message }); + } + } + if (event.target.closest("[data-fleet-save]")) { + const name = prompt("Profile name (letters, numbers, dots, dashes):"); + if (!name) return; + const description = prompt("Description (optional):") ?? ""; + try { + showOutput(await postJson("/gateway/fleet/profiles/" + encodeURIComponent(name), { + yes: true, description, overwrite: false + })); + await renderFleetProfiles(); + } catch (error) { + showOutput({ error: error.message }); + } + } + }); function renderBackends() { const rows = $("#backend-rows"); @@ -1700,8 +1776,10 @@ const DASHBOARD_HTML = String.raw` const canvas = $("#topology-canvas"); if (!canvas || !canvas.isConnected) return; const viewportWidth = Math.max(1, canvas.clientWidth), viewportHeight = Math.max(1, canvas.clientHeight); - if (canvas.width !== Math.round(viewportWidth) || canvas.height !== Math.round(viewportHeight)) { canvas.width = Math.round(viewportWidth); canvas.height = Math.round(viewportHeight); } + const pixelRatio = Math.max(1, Math.min(3, Number(window.devicePixelRatio) || 1)); + if (canvas.width !== Math.round(viewportWidth * pixelRatio) || canvas.height !== Math.round(viewportHeight * pixelRatio)) { canvas.width = Math.round(viewportWidth * pixelRatio); canvas.height = Math.round(viewportHeight * pixelRatio); } const ctx = canvas.getContext("2d"); + ctx.setTransform(pixelRatio,0,0,pixelRatio,0,0); ctx.clearRect(0, 0, viewportWidth, viewportHeight); const models = state.topologyModels || []; const clusterNodes = state.status?.cluster?.enabled ? Object.values(state.status?.cluster?.nodes || {}) : []; @@ -1761,7 +1839,7 @@ const DASHBOARD_HTML = String.raw` return hashUnit(aSeed * 29) - hashUnit(bSeed * 29); }); const threadField = { left: 24, right: gate.left - 24, top: 112, bottom: height - 45 }; - ctx.font = '11px "SFMono-Regular",monospace'; + ctx.font = '12px system-ui,sans-serif'; ctx.textAlign = "left"; const connectionLabels = new Map(orderedConnections.map(connection => { const outputRate = smoothRate("connection:" + connection.id + ":out", connection.outputRate, now); @@ -1921,7 +1999,7 @@ const DASHBOARD_HTML = String.raw` if (active) { ctx.shadowColor = "rgba(47,230,200,.55)"; ctx.shadowBlur = 12; } ctx.beginPath(); ctx.roundRect(cardLeft, cardTop, nodeCardWidth, point.cardHeight, 6); ctx.fill(); ctx.stroke(); ctx.shadowBlur = 0; - ctx.textAlign = "left"; ctx.font = '700 10px "SFMono-Regular",monospace'; + ctx.textAlign = "left"; ctx.font = '600 11px system-ui,sans-serif'; ctx.fillStyle = node.reachable === false ? "#ff6f7d" : "#e9fffb"; ctx.fillText(fitCanvasText(ctx, node.name || node.id, clusterEnabled ? 116 : 94), cardLeft + 10, cardTop + 19); ctx.textAlign = "right"; ctx.fillStyle = node.local ? "#8fb4ff" : "rgba(153,163,176,.9)"; @@ -1930,10 +2008,10 @@ const DASHBOARD_HTML = String.raw` if (clusterEnabled) { const platform = node.profile?.platformId || [node.system?.platform, node.system?.arch].filter(Boolean).join("-") || "unknown architecture"; const accelerator = node.profile?.accelerators?.[0] || node.labels?.hardware || "cpu"; - ctx.textAlign = "left"; ctx.font = '8px "SFMono-Regular",monospace'; ctx.fillStyle = "rgba(143,180,255,.72)"; + ctx.textAlign = "left"; ctx.font = '8px system-ui,sans-serif'; ctx.fillStyle = "rgba(143,180,255,.72)"; ctx.fillText(fitCanvasText(ctx, platform + " · " + accelerator, nodeCardWidth - 20), cardLeft + 10, cardTop + 35); } - ctx.font = '9px "SFMono-Regular",monospace'; + ctx.font = '9px system-ui,sans-serif'; point.resources.forEach((row, rowIndex) => { const y = cardTop + 58 + rowIndex * 18; const barLeft = cardLeft + 39; @@ -1992,16 +2070,16 @@ const DASHBOARD_HTML = String.raw` ctx.beginPath(); ctx.roundRect(cardLeft - 5, cardTop - 5, cardWidth + 10, 78, 8); ctx.stroke(); ctx.restore(); } - ctx.beginPath(); ctx.roundRect(cardLeft, cardTop, cardWidth, 68, 5); ctx.fill(); ctx.stroke(); + ctx.beginPath(); ctx.roundRect(cardLeft, cardTop, cardWidth, 68, 12); ctx.fill(); ctx.stroke(); if (processing) { const scanX = cardLeft + ((now * .08) % (cardWidth + 36)) - 18; const scan = ctx.createLinearGradient(scanX - 16, 0, scanX + 16, 0); const scanRgb = externalProcessing ? "192,153,255" : "243,189,79"; scan.addColorStop(0, "rgba(" + scanRgb + ",0)"); scan.addColorStop(.5, "rgba(" + scanRgb + ",.13)"); scan.addColorStop(1, "rgba(" + scanRgb + ",0)"); - ctx.save(); ctx.beginPath(); ctx.roundRect(cardLeft, cardTop, cardWidth, 68, 5); ctx.clip(); ctx.fillStyle = scan; ctx.fillRect(scanX - 16, cardTop, 32, 68); ctx.restore(); + ctx.save(); ctx.beginPath(); ctx.roundRect(cardLeft, cardTop, cardWidth, 68, 12); ctx.clip(); ctx.fillStyle = scan; ctx.fillRect(scanX - 16, cardTop, 32, 68); ctx.restore(); } ctx.fillStyle = serving ? "#42d77d" : externalProcessing ? "#f3bd4f" : external ? "#c099ff" : hot ? "#2fe6c8" : warming ? "#f3bd4f" : evicting ? "#ff7e66" : unavailable ? "#ff6f7d" : "#8fb4ff"; ctx.fillRect(cardLeft, cardTop, 4, 68); - ctx.textAlign = "left"; ctx.font = '11px "SFMono-Regular",monospace'; + ctx.textAlign = "left"; ctx.font = '12px system-ui,sans-serif'; const vendor = modelFamily(point.model.id).toUpperCase(); const vendorText = fitCanvasText(ctx, vendor, 54); const titleWidth = Math.max(24, cardWidth - 12 - 10 - ctx.measureText(vendorText).width - 12); @@ -2059,8 +2137,16 @@ const DASHBOARD_HTML = String.raw` const instantaneousOutputRate = Math.max(0, Number(summary.outputRate || 0)); ctx.textAlign = "center"; if (clusterEnabled) { - ctx.fillStyle = "#e9fffb"; ctx.font = '700 17px "SFMono-Regular",monospace'; ctx.fillText("LLooM", center.x, gate.top + 20); - ctx.fillStyle = "rgba(153,163,176,.9)"; ctx.font = '9px "SFMono-Regular",monospace'; + ctx.save(); + ctx.shadowColor = "rgba(56,223,245,.45)"; + ctx.shadowBlur = summary.active ? 26 : 12; + ctx.fillStyle = "rgba(18,43,54,.92)"; + ctx.strokeStyle = "rgba(56,223,245,.45)"; + ctx.lineWidth = 1; + ctx.beginPath(); ctx.roundRect(center.x - nodeCardWidth / 2, gate.top - 6, nodeCardWidth, 80, 20); ctx.fill(); ctx.stroke(); + ctx.restore(); + ctx.fillStyle = "#e9fffb"; ctx.font = '700 17px system-ui,sans-serif'; ctx.fillText("LLooM", center.x, gate.top + 20); + ctx.fillStyle = "rgba(153,163,176,.9)"; ctx.font = '9px system-ui,sans-serif'; ctx.fillText(fitCanvasText(ctx, (state.status?.cluster?.id || "CLUSTER") + " · ROUTING", nodeCardWidth), center.x, gate.top + 35); const clusterTraffic = [ (summary.active || 0) + " ACTIVE", @@ -2077,13 +2163,13 @@ const DASHBOARD_HTML = String.raw` ctx.beginPath(); ctx.roundRect(gate.left, gate.top, gate.right - gate.left, gate.bottom - gate.top, 9); ctx.fill(); ctx.stroke(); ctx.strokeStyle = "rgba(143,180,255,.35)"; ctx.lineWidth = 1; for (let i = 0; i < 9; i++) { const y = gate.top + 30 + i * 25; ctx.beginPath(); ctx.moveTo(gate.left + 13, y); ctx.lineTo(gate.right - 13, y); ctx.stroke(); } - ctx.fillStyle = "#e9fffb"; ctx.font = '700 18px "SFMono-Regular",monospace'; ctx.fillText("LLooM", center.x, gate.top + 39); - ctx.fillStyle = "rgba(153,163,176,.9)"; ctx.font = '10px "SFMono-Regular",monospace'; ctx.fillText("ROUTING LOOM", center.x, gate.top + 57); + ctx.fillStyle = "#e9fffb"; ctx.font = '700 18px system-ui,sans-serif'; ctx.fillText("LLooM", center.x, gate.top + 39); + ctx.fillStyle = "rgba(153,163,176,.9)"; ctx.font = '10px system-ui,sans-serif'; ctx.fillText("ROUTING LOOM", center.x, gate.top + 57); const statRows = [["ACTIVE", summary.active || 0], ["PROMPT", promptTokens > 0 ? (summary.promptEstimated ? "~" : "") + formatCompact(Math.round(promptTokens)) + " tok" : "—"], ["OUTPUT", formatRate(instantaneousOutputRate) + " ~t/s"], ["ERRORS/1M", summary.recentErrors || 0]]; - ctx.font = '11px "SFMono-Regular",monospace'; + ctx.font = '12px system-ui,sans-serif'; statRows.forEach((row, index) => { const y = gate.top + 94 + index * 27; ctx.textAlign = "left"; ctx.fillStyle = "rgba(153,163,176,.9)"; ctx.fillText(row[0], gate.left + 18, y); ctx.textAlign = "right"; ctx.fillStyle = "rgba(242,245,247,.95)"; ctx.fillText(String(row[1]), gate.right - 18, y); }); const resourceRows = hostResourceRows(summary.host); - ctx.font = '9px "SFMono-Regular",monospace'; + ctx.font = '9px system-ui,sans-serif'; resourceRows.forEach((row, index) => { const y = gate.top + 218 + index * 24, value = row[1]; ctx.textAlign = "left"; ctx.fillStyle = "rgba(153,163,176,.9)"; ctx.fillText(row[0], gate.left + 17, y); @@ -2097,7 +2183,7 @@ const DASHBOARD_HTML = String.raw` } function animateTopology() { - if (!document.hidden) { + if (!document.hidden && !$(".topology").hidden) { const reducedMotion = window.matchMedia("(prefers-reduced-motion: reduce)").matches; // Reduced motion still resolves the layout, it just settles it and // snaps the camera instead of easing both across frames. @@ -2287,6 +2373,7 @@ const DASHBOARD_HTML = String.raw` } async function refreshActivity() { + if (document.hidden) return; try { state.metrics = await getJson("/gateway/metrics?period=" + encodeURIComponent(state.metricsPeriod)); renderActivity(); @@ -2306,6 +2393,7 @@ const DASHBOARD_HTML = String.raw` } async function refresh() { + if (document.hidden) return; const healthPill = $("#health"); healthPill?.classList.add("refreshing"); try { @@ -2317,7 +2405,8 @@ const DASHBOARD_HTML = String.raw` const [health, models, status, library, backends] = await Promise.all([ getJson("/health"), getJson("/gateway/models").catch(error => ({ models: [], error: error.message })), - getJson("/gateway/status").catch(error => ({ error: error.message })), + getJson("/gateway/status" + (typeof presenceView === "string" && presenceView === "live" ? "?memoryUsage=1" : "")) + .catch(error => ({ error: error.message })), getJson("/gateway/library").catch(error => ({ error: error.message })), getJson("/gateway/backends").catch(error => ({ backends: [], error: error.message })), ]); @@ -2332,6 +2421,7 @@ const DASHBOARD_HTML = String.raw` renderRuntimes(); renderBackends(); renderLibrary(); + void renderFleetProfiles(); renderNodeInspector(); const authHint = security.adminAuthRequired ? "admin auth on" @@ -2715,5 +2805,25 @@ const DASHBOARD_HTML = String.raw` `; export function renderDashboardPage() { - return DASHBOARD_HTML; + return DASHBOARD_HTML.replace('', () => presenceStyles + sceneStyles + '\n ') + .replace('', () => '' + presenceNav) + .replace( + '
', + '

Your AI, in motion.

Your hardware. Your models. One gateway.

' + ) + .replace('
', + () => presenceViews + '