Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions packages/core/src/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,12 @@ export {
} from './environment-profile.js';
export { ToolSelector, type ToolGroup } from './tool-selector.js';
export { LLMRouter } from './llm/router.js';
export {
CHAT_CAPABILITY,
NON_CHAT_CAPABILITIES,
isChatCapableModel,
type ChatCapabilityShape,
} from './llm/model-capabilities.js';
export type { ChatOptions } from './llm/router.js';
export { LLMLogger, type LLMLogEntry } from './llm/llm-logger.js';
export { AnthropicProvider } from './llm/anthropic.js';
Expand Down
59 changes: 59 additions & 0 deletions packages/core/src/llm/model-capabilities.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,59 @@
/**
* Provider-model capability vocabulary.
*
* `ModelDefinition.capabilities` is a free-form set of tags. Historically
* "is this a chat model?" was inferred from *absence* — `capabilities` empty
* meant chat, anything else meant "not chat". That forced an exclusive choice:
* declaring e.g. `imageGeneration` silently removed a model from every chat
* surface (chat composer picker, text routing, capability suggestions), even
* when the endpoint really does serve `/chat/completions`.
*
* Real models are frequently both: a local diffusion server exposing a chat
* endpoint, `gpt-image-1` (image generation through chat completions),
* `gpt-audio` (a multimodal chat model). The rule is therefore explicit and
* additive:
*
* chat-capable ⇔ capabilities includes 'chat'
* OR (no media tag AND mode is not a non-chat mode)
*
* Backward compatible: an undeclared model is still assumed to be a chat model.
*/

/** Explicit "this endpoint serves /chat/completions" tag. */
export const CHAT_CAPABILITY = 'chat';

/**
* Tags meaning "this is a dedicated media/side endpoint".
* Deliberately excludes `vision` (a property of chat models) and `decision`
* (a different response *shape*, not a non-chat modality).
*/
export const NON_CHAT_CAPABILITIES: ReadonlySet<string> = new Set([
'imageGeneration',
'tts',
'stt',
'videoGeneration',
'audioOutput',
'audioInput',
]);

/** Minimum shape needed to decide chat capability. */
export interface ChatCapabilityShape {
capabilities?: string[];
/** Catalog mode, e.g. 'chat' | 'image_generation' | 'audio_speech'. */
mode?: string;
}

/**
* Can this model serve `/chat/completions`?
*
* Order matters: an explicit `chat` tag wins even alongside media tags, so a
* model can advertise `['imageGeneration', 'chat']` and stay selectable in the
* chat UI while remaining routable as an image endpoint.
*/
export function isChatCapableModel(model: ChatCapabilityShape): boolean {
const caps = model.capabilities ?? [];
if (caps.includes(CHAT_CAPABILITY)) return true;
if (caps.some(c => NON_CHAT_CAPABILITIES.has(c))) return false;
if (model.mode && model.mode !== 'chat') return false;
return true;
}
5 changes: 3 additions & 2 deletions packages/core/src/llm/router.ts
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@ import { discoverProviderModels, PROVIDER_DEFAULT_BASE_URLS } from './model-disc
import { AuthProfileStore } from './auth-profiles.js';
import { OAuthManager } from './oauth-manager.js';
import type { ModelCatalogService } from './model-catalog.js';
import { isChatCapableModel } from './model-capabilities.js';


const log = createLogger('llm-router');
Expand Down Expand Up @@ -603,7 +604,7 @@ export class LLMRouter {
// fetched yet — no key, offline, or first run — would otherwise render an
// empty picker. Seed it with the registry's documented bootstrap model and
// label it 'builtin' so the UI says where it came from.
const hasChatModel = merged.some(m => (m.capabilities?.length ?? 0) === 0 && m.contextWindow > 0);
const hasChatModel = merged.some(m => isChatCapableModel(m) && m.contextWindow > 0);
if (!hasChatModel) {
const bootstrapId = getProviderBootstrapModel(providerName);
if (bootstrapId && !merged.some(m => m.id === bootstrapId)) {
Expand Down Expand Up @@ -1028,7 +1029,7 @@ export class LLMRouter {
// reasoning support has to be resolved from here rather than from a
// hand-written id list that goes stale every release.
const caps = catalogEntry.capabilities;
const isChatModel = !model.capabilities || model.capabilities.length === 0;
const isChatModel = isChatCapableModel(model);
const inputTypes = isChatModel && caps
? (caps.vision ? (['text', 'image'] as Array<'text' | 'image'>) : (['text'] as Array<'text' | 'image'>))
: undefined;
Expand Down
8 changes: 5 additions & 3 deletions packages/core/src/tools/settings.ts
Original file line number Diff line number Diff line change
Expand Up @@ -417,14 +417,16 @@ export function createSettingsTools(ctx: SettingsToolsContext): AgentToolHandler
type: 'array',
items: {
type: 'string',
enum: ['imageGeneration', 'vision', 'tts', 'stt', 'videoGeneration', 'decision'],
enum: ['chat', 'imageGeneration', 'vision', 'tts', 'stt', 'videoGeneration', 'decision'],
},
description:
'Explicit capability declaration for this model (OPTIONAL but recommended for non-chat models). ' +
'Values: imageGeneration (text-to-image), vision (image input), tts, stt, videoGeneration, decision. ' +
'Values: imageGeneration (text-to-image), vision (image input), tts, stt, videoGeneration, decision, chat. ' +
'When declared, capability routing trusts this list instead of guessing from the model id — ' +
'use this for local/self-hosted models (Ollama, vLLM, diffusers servers) whose names do not match ' +
'known commercial naming patterns.',
'known commercial naming patterns. Add "chat" when the model ALSO serves /chat/completions ' +
'(e.g. ["imageGeneration","chat"]): declaring it keeps the model in the chat pickers and text ' +
'routing, which a media-only declaration otherwise removes it from.',
},
},
required: ['provider', 'id', 'name', 'context_window', 'max_output_tokens', 'cost_input', 'cost_output'],
Expand Down
45 changes: 45 additions & 0 deletions packages/core/test/model-capabilities.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
import { describe, it, expect } from 'vitest';
import {
CHAT_CAPABILITY,
NON_CHAT_CAPABILITIES,
isChatCapableModel,
} from '../src/llm/model-capabilities.js';

describe('isChatCapableModel', () => {
it('treats an undeclared model as chat (legacy default, must not regress)', () => {
expect(isChatCapableModel({})).toBe(true);
expect(isChatCapableModel({ capabilities: [] })).toBe(true);
});

it('rejects every dedicated media endpoint', () => {
for (const tag of NON_CHAT_CAPABILITIES) {
expect(isChatCapableModel({ capabilities: [tag] })).toBe(false);
}
});

it('keeps the local qwen-image-2.1 shape out of chat', () => {
// The real declaration that produced `404 not found: /v1/chat/completions`.
expect(isChatCapableModel({ capabilities: ['imageGeneration'] })).toBe(false);
});

it('honours the explicit chat tag alongside media tags', () => {
expect(isChatCapableModel({ capabilities: ['imageGeneration', CHAT_CAPABILITY] })).toBe(true);
expect(isChatCapableModel({ capabilities: [CHAT_CAPABILITY] })).toBe(true);
});

it('lets an explicit chat tag beat a non-chat catalog mode', () => {
expect(isChatCapableModel({ capabilities: [CHAT_CAPABILITY], mode: 'image_generation' })).toBe(true);
});

it('respects a non-chat catalog mode when nothing is declared', () => {
expect(isChatCapableModel({ mode: 'image_generation' })).toBe(false);
expect(isChatCapableModel({ mode: 'audio_speech' })).toBe(false);
expect(isChatCapableModel({ mode: 'chat' })).toBe(true);
});

it('does not treat descriptive tags as media endpoints', () => {
// vision / reasoning describe a chat model; decision is a different shape.
expect(isChatCapableModel({ capabilities: ['vision', 'reasoning'] })).toBe(true);
expect(isChatCapableModel({ capabilities: ['decision'] })).toBe(true);
});
});
11 changes: 6 additions & 5 deletions packages/org-manager/src/api-server.ts
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,7 @@ import {
buildModelsEndpoint,
buildModelsAuthHeaders,
isUsableProviderModelId,
isChatCapableModel,
PROVIDER_DEFAULT_BASE_URLS,
} from '@markus/core';
import type { ChannelMsg } from '@markus/storage';
Expand Down Expand Up @@ -8417,12 +8418,12 @@ EXPLANATION_END`;
for (const capabilityType of ALL_CAPABILITY_TYPES) {
let candidates: CandidateModel[];

const NON_TEXT_CAPS = new Set(['imageGeneration', 'tts', 'stt', 'videoGeneration', 'audioOutput', 'audioInput']);
if (TEXT_CAPABILITIES.has(capabilityType)) {
candidates = allCandidates.filter(m => {
if (m.capabilities && m.capabilities.some(c => NON_TEXT_CAPS.has(c))) return false;
return true;
});
// A model carrying the explicit 'chat' tag stays a chat candidate even
// when it also advertises media capabilities (an image endpoint that
// does serve /chat/completions), so the suggestion is not limited to
// models with no declared capabilities at all.
candidates = allCandidates.filter(m => isChatCapableModel(m));
} else {
const requiredCaps = CAP_MAP[capabilityType] ?? [];
candidates = allCandidates.filter(m =>
Expand Down
24 changes: 21 additions & 3 deletions packages/web-ui/src/components/ChatModelMenu.tsx
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
import { useCallback, useEffect, useMemo, useRef, useState } from 'react';
import { useTranslation } from 'react-i18next';
import { api } from '../api';
import { isChatCapableModel } from '../lib/modelCapabilities';

export interface ChatModelSelection {
provider: string;
Expand All @@ -10,9 +11,17 @@ export interface ChatModelSelection {
interface ProviderModels {
provider: string;
displayName: string;
models: Array<{ id: string; name?: string }>;
models: Array<{ id: string; name?: string; mode?: string; capabilities?: string[] }>;
}

/**
* Media-only endpoints must never reach this picker: binding one (e.g. the local
* `qwen-image-2.1`, which serves `POST /v1/images/generations` but has no chat
* route) made EVERY later turn die at the upstream API with
* `404 not found: /v1/chat/completions`. The rule lives in lib/modelCapabilities
* so the composer, Settings routing, and the server agree on one definition.
*/

/** Optional modifiers for a global-scope selection. */
export interface GlobalScopeOptions {
/**
Expand Down Expand Up @@ -69,7 +78,7 @@ export function ChatModelMenu({ value, onSelect, agentId, disabled }: ChatModelM
enabled?: boolean;
configured?: boolean;
model?: string;
models?: Array<{ id: string; name?: string }>;
models?: Array<{ id: string; name?: string; mode?: string; capabilities?: string[] }>;
}>;
};

Expand All @@ -94,7 +103,11 @@ export function ChatModelMenu({ value, onSelect, agentId, disabled }: ChatModelM
for (const [name, info] of Object.entries(data.providers ?? {})) {
// Only providers that are configured and switched on (same rule as Settings routing).
if (!info.configured || info.enabled === false) continue;
const models = (info.models ?? []).map(m => ({ id: m.id, name: m.name }));
// ...and only models that can actually serve chat — media-only endpoints
// (image/speech/video) belong to capability routing, not to the composer.
const models = (info.models ?? [])
.filter(isChatCapableModel)
.map(m => ({ id: m.id, name: m.name }));
if (models.length === 0) continue;
list.push({
provider: name,
Expand Down Expand Up @@ -296,6 +309,11 @@ export function ChatModelMenu({ value, onSelect, agentId, disabled }: ChatModelM
{t('chatModel.empty', { defaultValue: 'No enabled providers / models' })}
</div>
)}
{!loading && filtered.length > 0 && (
<div className="px-3 pb-1 text-[10px] text-fg-tertiary">
{t('chatModel.chatOnlyHint', { defaultValue: 'Only models with a chat interface are listed' })}
</div>
)}
{!loading && filtered.map(p => (
<div key={p.provider} className="mb-1">
<div className="px-3 py-1 text-[10px] uppercase tracking-wide text-fg-tertiary">
Expand Down
12 changes: 5 additions & 7 deletions packages/web-ui/src/components/ModelRoutingSection.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@ import { useTranslation } from 'react-i18next';
import type { CapabilityRoutingConfigDTO, ModelCapabilityTypeDTO, CapabilityModelAssignmentDTO } from '../api';
import { ModelSelect, type ModelOption } from './ModelSelect';
import { ConfirmModal } from './ConfirmModal.tsx';
import { isChatCapableModel } from '../lib/modelCapabilities';

interface Props {
onSave: (data: { capabilityRouting?: Partial<CapabilityRoutingConfigDTO>; routingDefaultModel?: { provider: string; model: string } | null }) => Promise<void>;
Expand Down Expand Up @@ -573,15 +574,12 @@ function TierBadge({ tier }: { tier: string }) {
);
}

const NON_TEXT_CAPABILITIES = new Set(['imageGeneration', 'tts', 'stt', 'videoGeneration', 'audioOutput', 'audioInput']);

function filterModelsForCapability(models: ModelOption[], capabilityType: ModelCapabilityTypeDTO): ModelOption[] {
if (capabilityType === 'text') {
return models.filter(m => {
if (m.capabilities && m.capabilities.some(c => NON_TEXT_CAPABILITIES.has(c))) return false;
if (m.mode && m.mode !== 'chat') return false;
return true;
});
// Chat-capable = explicit 'chat' tag, or no media tag. Shared rule: a model
// may advertise image generation AND chat (e.g. a local diffusion server
// that also serves /chat/completions) — see lib/modelCapabilities.
return models.filter(m => isChatCapableModel(m));
}

const modeMap: Record<string, string[]> = {
Expand Down
42 changes: 42 additions & 0 deletions packages/web-ui/src/lib/modelCapabilities.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,42 @@
/**
* Browser-side mirror of `packages/core/src/llm/model-capabilities.ts`.
*
* The web bundle cannot import `@markus/core` / `@markus/shared` (see the same
* note in `src/constants/providers.ts`), so the rule is duplicated here. Keep
* both copies in sync — they decide which models appear in the chat composer
* and in Settings → Model Routing.
*/

/** Explicit "this endpoint serves /chat/completions" tag. */
export const CHAT_CAPABILITY = 'chat';

/** Tags meaning "this is a dedicated media/side endpoint". */
export const NON_CHAT_CAPABILITIES: ReadonlySet<string> = new Set([
'imageGeneration',
'tts',
'stt',
'videoGeneration',
'audioOutput',
'audioInput',
]);

export interface ChatCapabilityShape {
capabilities?: string[];
mode?: string;
}

/**
* Can this model serve `/chat/completions`?
*
* A media-only model must never be bindable to an agent or listed in the chat
* composer: every turn would POST to an endpoint that has no chat route and
* fail at the upstream API. An explicit `chat` tag overrides the media tags for
* models that genuinely serve both.
*/
export function isChatCapableModel(model: ChatCapabilityShape): boolean {
const caps = model.capabilities ?? [];
if (caps.includes(CHAT_CAPABILITY)) return true;
if (caps.some(c => NON_CHAT_CAPABILITIES.has(c))) return false;
if (model.mode && model.mode !== 'chat') return false;
return true;
}
Loading