Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions packages/core/src/llm/dashscope.ts
Original file line number Diff line number Diff line change
Expand Up @@ -112,7 +112,7 @@ export class DashScopeProvider extends OpenAIProvider {
method: 'POST',
headers: { 'Content-Type': 'application/json', Authorization: authorization },
body: JSON.stringify(body),
signal: AbortSignal.timeout(120_000),
signal: AbortSignal.timeout(this.imageGenerationTimeoutMs),
});

if (!res.ok) {
Expand Down Expand Up @@ -169,7 +169,7 @@ export class DashScopeProvider extends OpenAIProvider {
method: 'POST',
headers: { 'Content-Type': 'application/json', Authorization: authorization },
body: JSON.stringify(body),
signal: AbortSignal.timeout(60_000),
signal: AbortSignal.timeout(this.ttsTimeoutMs),
});

if (!res.ok) {
Expand Down
2 changes: 1 addition & 1 deletion packages/core/src/llm/fireworks.ts
Original file line number Diff line number Diff line change
Expand Up @@ -64,7 +64,7 @@ export class FireworksProvider extends OpenAIProvider {
Authorization: authorization,
},
body: JSON.stringify(body),
signal: AbortSignal.timeout(120_000),
signal: AbortSignal.timeout(this.imageGenerationTimeoutMs),
});

if (!res.ok) {
Expand Down
30 changes: 25 additions & 5 deletions packages/core/src/llm/markus-provider.ts
Original file line number Diff line number Diff line change
Expand Up @@ -303,6 +303,13 @@ export class MarkusProvider implements MultiModalProviderInterface {
private maxTokens?: number;
private chatTimeoutMs: number;
private streamTimeoutMs: number;
/** Generative-media client timeouts; each independently configurable per
* provider (defaults match the historical hardcoded ceilings). */
private imageGenerationTimeoutMs: number;
private ttsTimeoutMs: number;
private sttTimeoutMs: number;
private videoGenerationTimeoutMs: number;
private decisionTimeoutMs: number;
private cuCache = new CUCache();
/** Hub-served geo-aware model catalog URL. */
private modelsUrl = '';
Expand Down Expand Up @@ -330,6 +337,14 @@ export class MarkusProvider implements MultiModalProviderInterface {
// Stream idle is independent of chat timeoutMs — never inherit a lower chat
// timeout (e.g. 90s) or long reasoning / sparse SSE gaps abort mid-reply.
this.streamTimeoutMs = config?.streamTimeoutMs ?? STREAM_TIMEOUT_MS;
// Generative-media client timeouts; each independently configurable per
// provider. Default: generous 10min — media synthesis through OpenRouter
// and self-hosted servers can take minutes on first use.
this.imageGenerationTimeoutMs = config?.imageGenerationTimeoutMs ?? 600_000;
this.ttsTimeoutMs = config?.ttsTimeoutMs ?? 600_000;
this.sttTimeoutMs = config?.sttTimeoutMs ?? 600_000;
this.videoGenerationTimeoutMs = config?.videoGenerationTimeoutMs ?? 600_000;
this.decisionTimeoutMs = config?.decisionTimeoutMs ?? 60_000;
this.applyRetryConfig(config);
this.modelsUrl = config?.modelsUrl ?? process.env['MARKUS_MODELS_URL'] ?? '';
this.hubUrl = config?.hubUrl ?? process.env['MARKUS_HUB_URL'] ?? '';
Expand Down Expand Up @@ -361,6 +376,11 @@ export class MarkusProvider implements MultiModalProviderInterface {
if (config.hubToken !== undefined) this.hubToken = config.hubToken;
if (config.timeoutMs) this.chatTimeoutMs = config.timeoutMs;
if (config.streamTimeoutMs) this.streamTimeoutMs = config.streamTimeoutMs;
if (config.imageGenerationTimeoutMs !== undefined) this.imageGenerationTimeoutMs = config.imageGenerationTimeoutMs;
if (config.ttsTimeoutMs !== undefined) this.ttsTimeoutMs = config.ttsTimeoutMs;
if (config.sttTimeoutMs !== undefined) this.sttTimeoutMs = config.sttTimeoutMs;
if (config.videoGenerationTimeoutMs !== undefined) this.videoGenerationTimeoutMs = config.videoGenerationTimeoutMs;
if (config.decisionTimeoutMs !== undefined) this.decisionTimeoutMs = config.decisionTimeoutMs;
// Retry knobs follow the same precedence on re-configure (env re-read too).
this.applyRetryConfig(config);
}
Expand Down Expand Up @@ -1549,7 +1569,7 @@ export class MarkusProvider implements MultiModalProviderInterface {
'X-Title': 'Markus',
},
body: JSON.stringify(body),
signal: AbortSignal.timeout(60_000),
signal: AbortSignal.timeout(this.decisionTimeoutMs),
});

if (!res.ok) {
Expand Down Expand Up @@ -1608,7 +1628,7 @@ export class MarkusProvider implements MultiModalProviderInterface {
method: 'POST',
headers: { 'Content-Type': 'application/json', Authorization: this.bearerOpenRouter() },
body: JSON.stringify(body),
signal: AbortSignal.timeout(180_000),
signal: AbortSignal.timeout(this.imageGenerationTimeoutMs),
});
if (!res.ok) {
const errText = await res.text();
Expand Down Expand Up @@ -1705,7 +1725,7 @@ export class MarkusProvider implements MultiModalProviderInterface {
headers: { 'Content-Type': 'application/json', Authorization: this.bearerOpenRouter() },
body: JSON.stringify(body),
// Long narration + MiniMax/Deepgram synthesis often exceeds 60s.
signal: AbortSignal.timeout(180_000),
signal: AbortSignal.timeout(this.ttsTimeoutMs),
});
if (!res.ok) {
const errText = await res.text();
Expand Down Expand Up @@ -1758,7 +1778,7 @@ export class MarkusProvider implements MultiModalProviderInterface {
method: 'POST',
headers: { 'Content-Type': 'application/json', Authorization: this.bearerOpenRouter() },
body: JSON.stringify(body),
signal: AbortSignal.timeout(120_000),
signal: AbortSignal.timeout(this.sttTimeoutMs),
});
if (!res.ok) {
const errText = await res.text();
Expand Down Expand Up @@ -1823,7 +1843,7 @@ export class MarkusProvider implements MultiModalProviderInterface {
method: 'POST',
headers: { 'Content-Type': 'application/json', Authorization: this.bearerOpenRouter() },
body: JSON.stringify(body),
signal: AbortSignal.timeout(120_000),
signal: AbortSignal.timeout(this.videoGenerationTimeoutMs),
});
if (!createRes.ok) {
const errText = await createRes.text();
Expand Down
6 changes: 3 additions & 3 deletions packages/core/src/llm/minimax.ts
Original file line number Diff line number Diff line change
Expand Up @@ -126,7 +126,7 @@ export class MiniMaxProvider extends OpenAIProvider {
method: 'POST',
headers: { 'Content-Type': 'application/json', Authorization: authorization },
body: JSON.stringify(body),
signal: AbortSignal.timeout(120_000),
signal: AbortSignal.timeout(this.imageGenerationTimeoutMs),
});

if (!res.ok) {
Expand Down Expand Up @@ -177,7 +177,7 @@ export class MiniMaxProvider extends OpenAIProvider {
method: 'POST',
headers: { 'Content-Type': 'application/json', Authorization: authorization },
body: JSON.stringify(body),
signal: AbortSignal.timeout(180_000),
signal: AbortSignal.timeout(this.ttsTimeoutMs),
});

if (!res.ok) {
Expand Down Expand Up @@ -224,7 +224,7 @@ export class MiniMaxProvider extends OpenAIProvider {
method: 'POST',
headers: { 'Content-Type': 'application/json', Authorization: authorization },
body: JSON.stringify(body),
signal: AbortSignal.timeout(120_000),
signal: AbortSignal.timeout(this.videoGenerationTimeoutMs),
});

if (!createRes.ok) {
Expand Down
28 changes: 24 additions & 4 deletions packages/core/src/llm/openai.ts
Original file line number Diff line number Diff line change
Expand Up @@ -102,6 +102,11 @@ export class OpenAIProvider implements MultiModalProviderInterface {
protected maxTokens: number;
protected chatTimeoutMs: number;
protected streamTimeoutMs: number;
protected imageGenerationTimeoutMs: number;
protected ttsTimeoutMs: number;
protected sttTimeoutMs: number;
protected videoGenerationTimeoutMs: number;
protected decisionTimeoutMs: number;
protected tokenResolver?: TokenResolver;
/** Explicit decisions endpoint; when unset it is resolved from `baseUrl`. */
protected decisionsUrl?: string;
Expand All @@ -122,6 +127,16 @@ export class OpenAIProvider implements MultiModalProviderInterface {
this.chatTimeoutMs = config?.timeoutMs ?? 90_000;
// Idle gap between chunks (reset on data). Independent of chat timeoutMs.
this.streamTimeoutMs = config?.streamTimeoutMs ?? 180_000;
// Generative media (image/video/audio) is naturally slower than chat and
// each costs real compute — give every modality an independently configured
// client timeout (per provider), instead of one hardcoded ceiling for all.
// Default: generous 10min — local/self-hosted servers (diffusers, vLLM)
// must cold-load weights on the first request and can take minutes.
this.imageGenerationTimeoutMs = config?.imageGenerationTimeoutMs ?? 600_000;
this.ttsTimeoutMs = config?.ttsTimeoutMs ?? 600_000;
this.sttTimeoutMs = config?.sttTimeoutMs ?? 600_000;
this.videoGenerationTimeoutMs = config?.videoGenerationTimeoutMs ?? 600_000;
this.decisionTimeoutMs = config?.decisionTimeoutMs ?? 60_000;
this.decisionsUrl = config?.decisionsUrl;
this.tokenResolver = tokenResolver;
}
Expand All @@ -133,6 +148,11 @@ export class OpenAIProvider implements MultiModalProviderInterface {
if (config.maxTokens) this.maxTokens = config.maxTokens;
if (config.timeoutMs) this.chatTimeoutMs = config.timeoutMs;
if (config.streamTimeoutMs) this.streamTimeoutMs = config.streamTimeoutMs;
if (config.imageGenerationTimeoutMs !== undefined) this.imageGenerationTimeoutMs = config.imageGenerationTimeoutMs;
if (config.ttsTimeoutMs !== undefined) this.ttsTimeoutMs = config.ttsTimeoutMs;
if (config.sttTimeoutMs !== undefined) this.sttTimeoutMs = config.sttTimeoutMs;
if (config.videoGenerationTimeoutMs !== undefined) this.videoGenerationTimeoutMs = config.videoGenerationTimeoutMs;
if (config.decisionTimeoutMs !== undefined) this.decisionTimeoutMs = config.decisionTimeoutMs;
if (config.decisionsUrl) this.decisionsUrl = config.decisionsUrl;
}

Expand Down Expand Up @@ -588,7 +608,7 @@ export class OpenAIProvider implements MultiModalProviderInterface {
'X-Title': 'Markus',
},
body: JSON.stringify(body),
signal: AbortSignal.timeout(60_000),
signal: AbortSignal.timeout(this.decisionTimeoutMs),
});

if (!res.ok) {
Expand Down Expand Up @@ -630,7 +650,7 @@ export class OpenAIProvider implements MultiModalProviderInterface {
method: 'POST',
headers: { 'Content-Type': 'application/json', Authorization: authorization },
body: JSON.stringify(body),
signal: AbortSignal.timeout(120_000),
signal: AbortSignal.timeout(this.imageGenerationTimeoutMs),
});

if (!res.ok) {
Expand Down Expand Up @@ -675,7 +695,7 @@ export class OpenAIProvider implements MultiModalProviderInterface {
method: 'POST',
headers: { 'Content-Type': 'application/json', Authorization: authorization },
body: JSON.stringify(body),
signal: AbortSignal.timeout(180_000),
signal: AbortSignal.timeout(this.ttsTimeoutMs),
});

if (!res.ok) {
Expand Down Expand Up @@ -707,7 +727,7 @@ export class OpenAIProvider implements MultiModalProviderInterface {
method: 'POST',
headers: { Authorization: authorization },
body: formData,
signal: AbortSignal.timeout(120_000),
signal: AbortSignal.timeout(this.sttTimeoutMs),
});

if (!res.ok) {
Expand Down
46 changes: 44 additions & 2 deletions packages/core/src/llm/router.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1333,8 +1333,16 @@ export class LLMRouter {
* Resolve a provider instance for a non-text modality (image_generation, audio_tts, etc.).
* Returns the provider and the assigned model name WITHOUT mutating provider state.
* The caller is responsible for passing `model` into the API call options.
*
* For non-text capabilities this NEVER falls back to the global text routing
* default model id (deepseek/gpt-4o/…): posting a chat id to an image/tts/stt
* endpoint yields a confusing upstream 404 ("model not found" / "not served for
* this capability"). Unassigned capabilities resolve to the provider only; the
* provider's own modality-appropriate default (or an explicit per-call model)
* is used instead.
*/
resolveModalityProvider(capabilityType: ModelCapabilityType): { provider: MultiModalProviderInterface; model?: string } | undefined {
const isText = capabilityType === 'text';
const assignment = this._capabilityRouting.assignments[capabilityType];
if (assignment) {
const provider = this.providers.get(assignment.provider);
Expand All @@ -1349,17 +1357,51 @@ export class LLMRouter {
}
}

// Fallback: try routingDefaultModel, then defaultProvider
// Fallback: try routingDefaultModel, then defaultProvider. Only text routing
// carries the default text model id; non-text capabilities must NOT reuse it
// (see class doc above) — except when the routing default model provably
// serves the capability (declared in its catalog entry), so a sensible
// media default like "image-01" still resolves instead of posting a chat id
// like "deepseek/deepseek-v4-flash-0731" to an image endpoint (upstream 404).
if (this._routingDefaultModel) {
const p = this.providers.get(this._routingDefaultModel.provider);
if (p && this.isAvailable(this._routingDefaultModel.provider)) {
return { provider: p as MultiModalProviderInterface, model: this._routingDefaultModel.model };
const model = isText || this.routingDefaultModelServesCapability(capabilityType)
? this._routingDefaultModel.model
: undefined;
return { provider: p as MultiModalProviderInterface, ...(model ? { model } : {}) };
}
}
const p = this.providers.get(this.defaultProvider);
return p && this.isAvailable(this.defaultProvider) ? { provider: p as MultiModalProviderInterface } : undefined;
}

/**
* True when the routing default model's catalog entry declares the requested
* capability. Used to decide whether a non-text fallback may carry the model
* id — a text/chat model must never be POSTed to a media endpoint.
*/
private routingDefaultModelServesCapability(capabilityType: ModelCapabilityType): boolean {
const r = this._routingDefaultModel;
if (!r) return false;
try {
const entry = this.getProviderModels(r.provider).find(m => m.id === r.model);
if (!entry) return false;
switch (capabilityType) {
case 'image_generation': return !!entry.capabilities?.includes('imageGeneration');
case 'audio_tts': return !!entry.capabilities?.includes('tts');
case 'audio_stt': return !!entry.capabilities?.includes('stt');
case 'video_generation': return !!entry.capabilities?.includes('videoGeneration');
case 'decision': return !!entry.capabilities?.includes('decision');
case 'image_recognition':
return !!entry.inputTypes?.includes('image') || !!entry.capabilities?.includes('vision');
default: return false;
}
} catch {
return false;
}
}

/**
* Look up any registered provider by name for one-shot tool overrides
* (`provider=` + `model=` on generate_image / text_to_speech / …).
Expand Down
26 changes: 25 additions & 1 deletion packages/core/src/tools/multimodal.ts
Original file line number Diff line number Diff line change
Expand Up @@ -238,6 +238,24 @@ function resolveEffectiveCandidates(
};
}
if (agentModel) {
// Qualified "provider/model" ids (e.g. "openai/gpt-image-1") must NOT be
// blindly stamped onto every candidate model — if the caller passes a
// chat id like "deepseek/deepseek-v4-flash-0731" for image_generation, we
// would otherwise POST a text model to an image endpoint and surface a
// confusing upstream 404 ("model not found"). Split the prefix into an
// explicit provider so the modality-support check below applies and the
// error names the real problem (provider does not serve this capability).
const slash = agentModel.indexOf('/');
if (slash > 0) {
const prefix = agentModel.slice(0, slash);
const bare = agentModel.slice(slash + 1);
if (
bare
&& (ctx.resolveProvider?.(prefix) || candidates.some(c => c.name === prefix))
) {
return resolveEffectiveCandidates(candidates, prefix, bare, ctx, method);
}
}
return { candidates: candidates.map(c => ({ ...c, model: agentModel })) };
}
return { candidates };
Expand Down Expand Up @@ -865,7 +883,13 @@ export function createMultiModalTools(ctx: MultiModalToolsContext): AgentToolHan
log.error(`Image generation failed on all ${candidates.length} provider(s)`);
return toolErr(
`Image generation failed (tried: ${tried}): ${lastError instanceof Error ? lastError.message : String(lastError)}`,
{ hint: ROUTING_HINT, tried_models: candidates.map(c => c.model).filter(Boolean) },
{
hint:
`${ROUTING_HINT} ` +
`If the error mentions "model not found" / "not served for this capability", the selected model is a ` +
`text/chat model, not an image model — pick from llm_get_capability_routing.usable_models.image_generation.`,
tried_models: candidates.map(c => c.model).filter(Boolean),
},
);
},
},
Expand Down
Loading
Loading