Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,7 @@ jobs:
strategy:
fail-fast: false
matrix:
node: [20, 22]
node: [22, 24]

steps:
- name: Check out repository
Expand Down
2 changes: 2 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,8 @@ All notable changes to LLooM will be documented in this file. The format follows

### Changed

- LLooM now requires Node.js 22.19 or newer, the minimum for its undici 8 HTTP client. It already failed to start on Node 20 after that upgrade; the engine range and CI matrix now say so.

- The action view frames only models serving traffic and their live requests. Active cards cluster near the loom; idle models and machine racks no longer pull the camera outward. Manual zoom holds independently of the automatic fit until Reset view restores it.
- Federated nodes now retain sovereign lifecycle control over ordinary local runtimes, while tensor-parallel members explicitly delegate lifecycle authority to their leader and remain non-callable on workers.
- Runtime residency now uses `keepWarm` as the single hard pin, keeps distributed-model pins on the logical runtime, and routes ready alias alternatives without eviction or capacity queuing; embeddings remain non-evicting even when requested by exact model ID.
Expand Down
2 changes: 1 addition & 1 deletion CONTRIBUTING.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ Thank you for helping improve LLooM. The project welcomes focused fixes, new bac

Requirements:

- Node.js 20 or newer
- Node.js 22.19 or newer
- npm
- Python 3 for syntax-checking the optional MLX Audio and MTPLX patch helpers
- macOS, Linux, or another platform capable of running the Node.js test suite
Expand Down
4 changes: 1 addition & 3 deletions bin/lloom.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -2444,9 +2444,7 @@ async function main() {
`/gateway/fleet/profiles/${encodeURIComponent(name)}${isApply ? '?apply=1' : ''}`,
{
method: 'POST',
body: isApply
? { yes: true }
: { yes: true, description: plan.description, overwrite: plan.overwrite },
body: isApply ? { yes: true } : { yes: true, description: plan.description, overwrite: plan.overwrite },
timeoutMs: 60000
}
);
Expand Down
2 changes: 1 addition & 1 deletion package-lock.json

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

2 changes: 1 addition & 1 deletion package.json
Original file line number Diff line number Diff line change
Expand Up @@ -89,7 +89,7 @@
"test:memory-safety": "node --test test/runtime-memory-safety.test.mjs"
},
"engines": {
"node": ">=20.0.0"
"node": ">=22.19.0"
},
"license": "MIT",
"devDependencies": {
Expand Down
9 changes: 8 additions & 1 deletion src/cluster.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -888,7 +888,14 @@ export function validateClusterConfig(config, env = process.env) {
export class ClusterCoordinator {
constructor(
config,
{ env = process.env, fetchImpl = undiciFetch, logger = console, telemetry = null, profile = null, models = null } = {}
{
env = process.env,
fetchImpl = undiciFetch,
logger = console,
telemetry = null,
profile = null,
models = null
} = {}
) {
this.config = config;
this.env = env;
Expand Down
9 changes: 5 additions & 4 deletions src/config-profiles.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -49,8 +49,7 @@
const id = target.trim();
const known = (candidate) =>
candidate === id || (object(alias.members)?.includes ?? (() => false)).call(alias.members, id);
if (!known(id) && !Array.isArray(alias.members))
throw fail(`Alias has no route profile or member named ${id}.`);
if (!known(id) && !Array.isArray(alias.members)) throw fail(`Alias has no route profile or member named ${id}.`);
return { activeRoute: null, members: [id], optionalMembers: [] };
}

Expand All @@ -72,13 +71,15 @@
for (const [runtimeId, policy] of Object.entries(object(doc.residency) ?? {})) {
if (typeof runtimeId !== 'string' || !runtimeId.trim() || runtimeId.length > 200)
throw fail(`Profile ${name} has an invalid runtime id.`);
if (!RESIDENCY.has(policy)) throw fail(`Profile ${name}: residency for ${runtimeId} must be always, preferred, or auto.`);
if (!RESIDENCY.has(policy))
throw fail(`Profile ${name}: residency for ${runtimeId} must be always, preferred, or auto.`);
residency[runtimeId.trim()] = policy;
}
const defaults = object(doc.defaults);
if (defaults) {
for (const [key, value] of Object.entries(defaults)) {
if (typeof value !== 'string' || value.length > 500) throw fail(`Profile ${name}: defaults.${key} must be a short string.`);
if (typeof value !== 'string' || value.length > 500)
throw fail(`Profile ${name}: defaults.${key} must be a short string.`);
}
}
return {
Expand Down Expand Up @@ -206,7 +207,7 @@
return raw;
}

export function createFleetProfileController({ getConfig, reload, env = process.env }) {

Check warning on line 210 in src/config-profiles.mjs

View workflow job for this annotation

GitHub Actions / Node 24

'env' is assigned a value but never used. Allowed unused args must match /^_/u

Check warning on line 210 in src/config-profiles.mjs

View workflow job for this annotation

GitHub Actions / Node 22

'env' is assigned a value but never used. Allowed unused args must match /^_/u
async function apply(name, { yes = false } = {}) {
if (yes !== true) throw fail('Review the profile and confirm with yes: true.');
if (!getConfig().sourcePath) throw fail('This gateway has no writable installed configuration.', 409);
Expand Down
6 changes: 5 additions & 1 deletion src/host-memory.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -45,7 +45,11 @@ export function parseMacMemoryPressure(text, totalBytes) {
// the "System-wide memory free percentage" is an opaque kernel estimate that
// can understate true availability while the file cache holds pages.
if ([free, inactive, speculative, purgeable].every(Number.isFinite)) {
return memorySnapshot(totalBytes, Math.min(totalBytes, (free + inactive + speculative + purgeable) * pageSize), 'macos-memory-pages');
return memorySnapshot(
totalBytes,
Math.min(totalBytes, (free + inactive + speculative + purgeable) * pageSize),
'macos-memory-pages'
);
}
const match = text_.match(/System-wide memory free percentage:\s*([\d.]+)%/i);
const percentage = Number(match?.[1]);
Expand Down
17 changes: 14 additions & 3 deletions test/config-profiles.test.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -77,7 +77,11 @@ try {
'cloud'
);
const localPlan = planProfileChanges(config, localProfile);
assert.equal(localPlan.unchanged.length, 2, 'omp route + residency are no-ops; simple pins members (members rewrite)');
assert.equal(
localPlan.unchanged.length,
2,
'omp route + residency are no-ops; simple pins members (members rewrite)'
);
const cloudPlan = planProfileChanges(config, cloudProfile);
assert.equal(cloudPlan.routes.length, 1);
assert.deepEqual(cloudPlan.routes[0], {
Expand All @@ -90,7 +94,10 @@ try {
});
assert.deepEqual(cloudPlan.residency, [{ id: 'local-model', from: 'always', to: 'auto' }]);
assert.deepEqual(cloudPlan.defaults, [{ id: 'chatModel', from: 'local-model', to: 'cloud-model' }]);
assert.throws(() => planProfileChanges(config, normalizeProfileDocument({ routes: { ghost: 'cloud' } }, 'x')), /no alias ghost/);
assert.throws(
() => planProfileChanges(config, normalizeProfileDocument({ routes: { ghost: 'cloud' } }, 'x')),
/no alias ghost/
);
assert.throws(
() => planProfileChanges(config, normalizeProfileDocument({ residency: { ghost: 'auto' } }, 'x')),
/no runtime ghost/
Expand Down Expand Up @@ -127,7 +134,11 @@ try {
const applied = await controller.apply('cloud', { yes: true });
assert.equal(applied.profile, 'cloud');
assert.equal(applied.routes.length, 1);
assert.equal(applied.unchanged.length, 0, 'second plan sees no changes after apply? no: apply recomputes against raw');
assert.equal(
applied.unchanged.length,
0,
'second plan sees no changes after apply? no: apply recomputes against raw'
);
const onDisk = JSON.parse(await fs.readFile(configPath, 'utf8'));
assert.equal(onDisk.fleet.activeProfile, 'cloud');
assert.deepEqual(onDisk.aliases.omp.members, ['cloud-model']);
Expand Down
178 changes: 92 additions & 86 deletions test/openrouter-provider.test.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -256,20 +256,21 @@ async function testGatewayChatBuffered() {
await withMockedDispatcher(
async () => startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'], allow_fallbacks: false } })),
async (mockAgent) => {
intercept(mockAgent, { payload: openAiChatPayload, onBody: (body) => seen.push(body) });
const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, {
method: 'POST',
headers: { 'content-type': 'application/json' },
body: JSON.stringify({
model: 'z-ai/glm-5.2',
messages: [{ role: 'user', content: 'hi' }],
provider: { only: ['openai'], allow_fallbacks: true, order: ['x'] }
})
});
assert.equal(res.status, 200);
const json = await res.json();
assert.equal(json.choices[0].message.content, 'ok');
});
intercept(mockAgent, { payload: openAiChatPayload, onBody: (body) => seen.push(body) });
const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, {
method: 'POST',
headers: { 'content-type': 'application/json' },
body: JSON.stringify({
model: 'z-ai/glm-5.2',
messages: [{ role: 'user', content: 'hi' }],
provider: { only: ['openai'], allow_fallbacks: true, order: ['x'] }
})
});
assert.equal(res.status, 200);
const json = await res.json();
assert.equal(json.choices[0].message.content, 'ok');
}
);
assert.equal(seen.length, 1, 'expected exactly one upstream chat call');
const outbound = JSON.parse(seen[0]);
// Caller override attempt is defeated; other caller provider fields survive.
Expand All @@ -287,24 +288,25 @@ async function testGatewayChatStream() {
await withMockedDispatcher(
async () => startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'] } })),
async (mockAgent) => {
intercept(mockAgent, {
contentType: 'text/event-stream',
payload: openAiStreamPayload,
onBody: (body) => seen.push(body)
});
const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, {
method: 'POST',
headers: { 'content-type': 'application/json' },
body: JSON.stringify({
model: 'z-ai/glm-5.2',
messages: [{ role: 'user', content: 'hi' }],
stream: true
})
});
assert.equal(res.status, 200);
const text = await res.text();
assert.match(text, /data: \[DONE\]/);
});
intercept(mockAgent, {
contentType: 'text/event-stream',
payload: openAiStreamPayload,
onBody: (body) => seen.push(body)
});
const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, {
method: 'POST',
headers: { 'content-type': 'application/json' },
body: JSON.stringify({
model: 'z-ai/glm-5.2',
messages: [{ role: 'user', content: 'hi' }],
stream: true
})
});
assert.equal(res.status, 200);
const text = await res.text();
assert.match(text, /data: \[DONE\]/);
}
);
assert.equal(seen.length, 1);
const outbound = JSON.parse(seen[0]);
assert.deepEqual(outbound.provider, { only: ['z-ai'], allow_fallbacks: false });
Expand All @@ -320,20 +322,21 @@ async function testGatewayResponsesBridge(stream = false) {
await withMockedDispatcher(
async () => startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'] } })),
async (mockAgent) => {
intercept(mockAgent, {
payload: stream ? openAiStreamPayload : openAiChatPayload,
contentType: stream ? 'text/event-stream' : 'application/json',
onBody: (body) => seen.push(body)
});
const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/responses`, {
method: 'POST',
headers: { 'content-type': 'application/json' },
body: JSON.stringify({ stream, model: 'z-ai/glm-5.2', input: 'hi', provider: { only: ['openai'] } })
});
assert.equal(res.status, 200);
if (stream) assert.match(await res.text(), /response.completed/);
else assert.equal((await res.json()).object, 'response');
});
intercept(mockAgent, {
payload: stream ? openAiStreamPayload : openAiChatPayload,
contentType: stream ? 'text/event-stream' : 'application/json',
onBody: (body) => seen.push(body)
});
const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/responses`, {
method: 'POST',
headers: { 'content-type': 'application/json' },
body: JSON.stringify({ stream, model: 'z-ai/glm-5.2', input: 'hi', provider: { only: ['openai'] } })
});
assert.equal(res.status, 200);
if (stream) assert.match(await res.text(), /response.completed/);
else assert.equal((await res.json()).object, 'response');
}
);
assert.equal(seen.length, 1);
const outbound = JSON.parse(seen[0]);
assert.deepEqual(outbound.provider, { only: ['z-ai'], allow_fallbacks: false });
Expand All @@ -348,25 +351,26 @@ async function testGatewayAnthropicBridge(stream = false) {
await withMockedDispatcher(
async () => startGateway(openRouterBackend({ openrouterProvider: { only: ['z-ai'], allow_fallbacks: true } })),
async (mockAgent) => {
intercept(mockAgent, {
payload: stream ? openAiStreamPayload : openAiChatPayload,
contentType: stream ? 'text/event-stream' : 'application/json',
onBody: (body) => seen.push(body)
});
const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/messages`, {
method: 'POST',
headers: { 'content-type': 'application/json', 'anthropic-version': '2023-06-01' },
body: JSON.stringify({
stream,
model: 'z-ai/glm-5.2',
max_tokens: 32,
messages: [{ role: 'user', content: 'hi' }]
})
});
assert.equal(res.status, 200);
if (stream) assert.match(await res.text(), /message_stop/);
else assert.equal((await res.json()).type, 'message');
});
intercept(mockAgent, {
payload: stream ? openAiStreamPayload : openAiChatPayload,
contentType: stream ? 'text/event-stream' : 'application/json',
onBody: (body) => seen.push(body)
});
const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/messages`, {
method: 'POST',
headers: { 'content-type': 'application/json', 'anthropic-version': '2023-06-01' },
body: JSON.stringify({
stream,
model: 'z-ai/glm-5.2',
max_tokens: 32,
messages: [{ role: 'user', content: 'hi' }]
})
});
assert.equal(res.status, 200);
if (stream) assert.match(await res.text(), /message_stop/);
else assert.equal((await res.json()).type, 'message');
}
);
assert.equal(seen.length, 1);
const outbound = JSON.parse(seen[0]);
assert.deepEqual(outbound.provider, { only: ['z-ai'], allow_fallbacks: true });
Expand All @@ -381,14 +385,15 @@ async function testGatewayMalformedPolicyFailsClosed() {
await withMockedDispatcher(
async () => startGateway(openRouterBackend({ openrouterProvider: { only: [] } })),
async () => {
const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, {
method: 'POST',
headers: { 'content-type': 'application/json' },
body: JSON.stringify({ model: 'z-ai/glm-5.2', messages: [{ role: 'user', content: 'hi' }] })
});
assert.notEqual(res.status, 200);
await res.text();
});
const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, {
method: 'POST',
headers: { 'content-type': 'application/json' },
body: JSON.stringify({ model: 'z-ai/glm-5.2', messages: [{ role: 'user', content: 'hi' }] })
});
assert.notEqual(res.status, 200);
await res.text();
}
);
} finally {
await stopGateway();
}
Expand All @@ -400,19 +405,20 @@ async function testGatewayNoPolicyUntouched() {
await withMockedDispatcher(
async () => startGateway({ id: 'openrouter-lane', type: 'openai', baseUrl: 'https://openrouter.ai/api/v1' }),
async (mockAgent) => {
intercept(mockAgent, { payload: openAiChatPayload, onBody: (body) => seen.push(body) });
const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, {
method: 'POST',
headers: { 'content-type': 'application/json' },
body: JSON.stringify({
model: 'z-ai/glm-5.2',
messages: [{ role: 'user', content: 'hi' }],
provider: { only: ['openai'], allow_fallbacks: true }
})
});
assert.equal(res.status, 200);
await res.text();
});
intercept(mockAgent, { payload: openAiChatPayload, onBody: (body) => seen.push(body) });
const res = await fetch(`http://127.0.0.1:${gatewayPort}/v1/chat/completions`, {
method: 'POST',
headers: { 'content-type': 'application/json' },
body: JSON.stringify({
model: 'z-ai/glm-5.2',
messages: [{ role: 'user', content: 'hi' }],
provider: { only: ['openai'], allow_fallbacks: true }
})
});
assert.equal(res.status, 200);
await res.text();
}
);
const outbound = JSON.parse(seen[0]);
assert.deepEqual(outbound.provider, { only: ['openai'], allow_fallbacks: true });
} finally {
Expand Down
Loading