Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,8 @@ All notable changes to LLooM will be documented in this file. The format follows

- LLooM Hear CPU audio analysis, with structured estimates, optional dashboard images, and opt-in upstream interpretation. File and URL inputs require operator configuration.

- Thirteen standalone NVIDIA ComfyUI media recipes, a public-source backend build, and shared runtime reuse for image, video and music generation.
- Fourteen per-model NVIDIA ComfyUI media recipes and a public-source backend build for image, video and music generation. Each model runs in its own container with read-only mounts of only its files, so LLooM admits, evicts and restores each one independently. The media launcher refuses to start without `LLOOM_MEDIA_MODEL`, and re-applying a recipe moves a model off the retired shared `comfyui-media` runtime, dropping it once unused.
- OpenRouter Lyria audio generation through `/v1/audio/generations`.

- Selective `include` paths or globs on recipe `download-model` steps, so a single-model lane fetches only the files its graph loads instead of every quantization in the model repository. Planned output reports the resolved `--include` command line.
- First-class `audio_generation` models and the `POST /v1/audio/generations` route, with `GET /v1/audio/generations/models` and a `defaults.audioGenerationModel` fallback. Music and other audio-generation lanes no longer have to be registered as `audio_speech` to be reachable: `/v1/audio/speech` stays speech- and clone-only, and each route refuses the other's kind with `wrong_model_kind`. Recipe capabilities `audio-generation`, `music-generation` and `audio-music-generation` materialize as `audio_generation`.
Expand Down
48 changes: 48 additions & 0 deletions backends/comfyui-media/bridge/server.py
Original file line number Diff line number Diff line change
Expand Up @@ -37,6 +37,10 @@
log = logging.getLogger("bridge")

DEFAULT_COMFY_URL = "http://127.0.0.1:8188"
# Optional single-model selector. When set, the bridge serves exactly that one
# registry entry. An empty or unknown value refuses startup rather than falling
# back to advertising every bundled model.
MEDIA_MODEL_ENV = "LLOOM_MEDIA_MODEL"
MAX_SEED = 2**31 - 1
MAX_DURATION_SECONDS = 600
# Text bounds, enforced here regardless of what the graph library does.
Expand Down Expand Up @@ -70,13 +74,50 @@ def _default_data_roots() -> DataRoots:
return DataRoots.from_env()


def _resolve_media_model(select: str | None = None) -> str | None:
"""The exact registry model a single-model runtime serves, if configured.

``select`` defaults to ``LLOOM_MEDIA_MODEL``. An absent variable selects
nothing, which only in-process tests rely on: the container launcher
refuses to start without a selection. A present
but empty or whitespace-only value is a configuration error and refuses
startup. Validation against the registry happens in ``create_app`` so the
rejection also covers explicitly injected ``models``.
"""
value = os.environ.get(MEDIA_MODEL_ENV) if select is None else select
if value is None:
return None
stripped = value.strip()
if not stripped:
raise ValueError(f"{MEDIA_MODEL_ENV} is set but empty; set an exact model ID or unset it.")
if stripped != value:
# Surrounding whitespace is almost certainly a deployment mistake; an
# exact model ID never carries it. Resolve to the trimmed ID, which is
# then validated against the registry like any other selection.
log.warning("%s has surrounding whitespace; using %r", MEDIA_MODEL_ENV, stripped)
return stripped


def _select_models(models: dict, select: str | None) -> dict:
"""Filter ``models`` to ``select``; fail closed on an unknown selection."""
if select is None:
return models
if not isinstance(models, dict) or select not in models:
supported = ", ".join(sorted(models)) if isinstance(models, dict) and models else "(none)"
raise ValueError(
f"{MEDIA_MODEL_ENV}={select!r} is not in this bridge's model registry. Supported models: {supported}."
)
return {select: models[select]}


def create_app(
comfy: ComfyClient | None = None,
*,
models: dict | None = None,
build_graph=None,
start_backend: bool = True,
comfy_url: str | None = None,
media_model: str | None = None,
) -> FastAPI:
app = FastAPI(title="LLooM media bridge", docs_url=None, redoc_url=None, openapi_url=None)
if models is None or build_graph is None:
Expand All @@ -85,11 +126,18 @@ def create_app(
models = default_models
if build_graph is None:
build_graph = default_builder
# A single-model runtime must never advertise a model it cannot serve, so
# the selector is applied here -- after any injected registry is known and
# before the app can accept a request. An unknown or empty value raises out
# of the app factory: startup (including uvicorn import of module-level
# ``app``) fails closed instead of degrading to the full registry.
models = _select_models(models, _resolve_media_model(media_model))
state = {
"comfy": comfy,
"models": models,
"build_graph": build_graph,
"comfy_url": validate_comfy_base_url(comfy_url or _default_comfy_url()),
"media_model": next(iter(models)) if len(models) == 1 else None,
}
runner = SingleFlightRunner(comfy, _default_data_roots()) if comfy is not None else None
state["runner"] = runner
Expand Down
1 change: 1 addition & 0 deletions backends/comfyui-media/bridge/tests/conftest.py
Original file line number Diff line number Diff line change
Expand Up @@ -46,6 +46,7 @@ def make_bridge(fake: FakeComfy, build_graph, *, start_backend: bool = False, **
models=kwargs.pop("models", {"video-model": "video", "audio-model": "audio"}),
build_graph=build_graph,
start_backend=start_backend,
**kwargs,
)
return app, comfy

Expand Down
118 changes: 118 additions & 0 deletions backends/comfyui-media/bridge/tests/test_single_model.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,118 @@
"""Single-model serving mode (``LLOOM_MEDIA_MODEL``).

A dedicated per-model runtime must advertise and serve exactly one registry
entry, and an explicitly set value that is empty or unknown must refuse
startup instead of falling back to the full bundled registry. An absent
variable preserves the previous multi-model behaviour.
"""

from __future__ import annotations

import pytest

from conftest import asgi_client, make_bridge
from fake_comfy import FakeComfy

REGISTRY = {"video-model": "video", "audio-model": "audio"}


def video_builder(model_id, payload, image_filename=None, prefix="lloom"):
return {"9": {"class_type": "SaveVideo", "inputs": {}}}, "9", "video"


async def test_selected_model_is_the_only_advertised_and_served_model():
fake = FakeComfy()
app, comfy = make_bridge(fake, video_builder, models=REGISTRY, media_model="video-model")
assert app.state.bridge["media_model"] == "video-model"
async with asgi_client(app) as client:
await comfy.start()
listed = {m["id"] for m in (await client.get("/v1/models")).json()["data"]}
assert listed == {"video-model"}
health = await client.get("/health")
assert health.status_code == 200, health.text
assert health.json()["backend_ready"] is True
await comfy.aclose()


async def test_other_registered_model_is_rejected_before_any_graph_or_submission():
fake = FakeComfy()
calls: list[str] = []

def builder(model_id, payload, image_filename=None, prefix="lloom"):
calls.append(model_id)
return {"9": {"class_type": "SaveVideo", "inputs": {}}}, "9", "video"

app, comfy = make_bridge(fake, builder, models=REGISTRY, media_model="video-model")
async with asgi_client(app) as client:
await comfy.start()
# ``audio-model`` exists in the injected registry but not in this
# runtime, so it must be an unsupported-model 400 that never reaches
# the video graph builder or ComfyUI's /prompt.
resp = await client.post(
"/v1/videos/generations", json={"model": "audio-model", "prompt": "x"}
)
assert resp.status_code == 400, resp.text
err = resp.json()["error"]
assert err["code"] == "unsupported_model"
assert "video-model" in err["message"]
assert calls == []
assert fake.prompts == {}
assert fake.uploaded == []
await comfy.aclose()


@pytest.mark.parametrize("value", ["", " "])
def test_explicitly_empty_selection_fails_closed(value, monkeypatch):
from server import create_app

monkeypatch.setenv("LLOOM_MEDIA_MODEL", value)
with pytest.raises(ValueError):
create_app(build_graph=video_builder, models=dict(REGISTRY), start_backend=False)


def test_unknown_selection_fails_closed_even_with_injected_registry(monkeypatch):
from server import create_app

monkeypatch.setenv("LLOOM_MEDIA_MODEL", "not-in-registry")
with pytest.raises(ValueError) as excinfo:
# Injected models still go through the selector, so a model the graphs
# module supports cannot be smuggled in by an explicit registry.
create_app(build_graph=video_builder, models=dict(REGISTRY), start_backend=False)
assert "not-in-registry" in str(excinfo.value)


def test_unknown_selection_is_not_silently_ignored(monkeypatch):
monkeypatch.setenv("LLOOM_MEDIA_MODEL", "missing-model")
import importlib

import server as server_mod

# The module-level ``app = create_app()`` runs on import; a bad selector
# would make an import-time startup fail rather than expose all models.
with pytest.raises(ValueError):
importlib.reload(server_mod)
monkeypatch.delenv("LLOOM_MEDIA_MODEL")
importlib.reload(server_mod)


async def test_absent_selection_keeps_full_registry(monkeypatch):
monkeypatch.delenv("LLOOM_MEDIA_MODEL", raising=False)
fake = FakeComfy()
app, comfy = make_bridge(fake, video_builder, models=REGISTRY)
assert app.state.bridge["media_model"] is None
async with asgi_client(app) as client:
await comfy.start()
listed = {m["id"] for m in (await client.get("/v1/models")).json()["data"]}
assert listed == set(REGISTRY)
await comfy.aclose()


async def test_selection_via_environment_is_honoured(monkeypatch):
monkeypatch.setenv("LLOOM_MEDIA_MODEL", "audio-model")
fake = FakeComfy()
app, comfy = make_bridge(fake, video_builder, models=REGISTRY)
async with asgi_client(app) as client:
await comfy.start()
listed = {m["id"] for m in (await client.get("/v1/models")).json()["data"]}
assert listed == {"audio-model"}
await comfy.aclose()
10 changes: 5 additions & 5 deletions backends/comfyui-media/build/launch.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,11 +8,11 @@
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import model_paths # noqa: E402

# Recipe-installed models land in per-repo directories under the LLooM model root.
# ComfyUI reads one extra_model_paths.yaml naming each of them, so installing a
# single-model recipe is enough for the engine to serve it -- no source edit, no
# hand-placed symlink. The aggregate default root stays where it is.
HOST_MODELS_ROOT = os.environ.get("LLOOM_MODELS_ROOT", "/opt/lloom-models")
# Each container serves exactly one model, whose files the recipe bind-mounts
# read-only into ComfyUI's model tree. There is no shared multi-model runtime.
if not os.environ.get("LLOOM_MEDIA_MODEL", "").strip():
raise SystemExit("LLOOM_MEDIA_MODEL is required: each media runtime serves exactly one model")
HOST_MODELS_ROOT = os.environ.get("LLOOM_MODELS_ROOT", "/opt/ComfyUI/models")
EXTRA_PATHS_FILE = "/data/extra_model_paths.yaml"
CACHE_MODE = os.environ.get("LLOOM_COMFY_CACHE_MODE", "none")
if CACHE_MODE not in ("none", "classic"):
Expand Down
15 changes: 15 additions & 0 deletions backends/comfyui-media/build/test_launch.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
import os
import subprocess
import sys

LAUNCH = os.path.join(os.path.dirname(os.path.abspath(__file__)), "launch.py")


def test_launch_refuses_to_serve_without_a_single_model():
for value in (None, "", " "):
env = {k: v for k, v in os.environ.items() if k != "LLOOM_MEDIA_MODEL"}
if value is not None:
env["LLOOM_MEDIA_MODEL"] = value
result = subprocess.run([sys.executable, LAUNCH], env=env, capture_output=True, text=True, timeout=30)
assert result.returncode != 0
assert "LLOOM_MEDIA_MODEL is required" in result.stderr
33 changes: 33 additions & 0 deletions docs/audio-providers.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,33 @@
# Cloud music through OpenRouter

LLooM accepts music requests at `POST /v1/audio/generations`. Set an OpenAI-compatible backend's `audioProvider` to `openrouter` to use OpenRouter's streaming chat audio API. Ordinary audio backends keep their existing binary proxy behavior.

```json
{
"backends": {
"openrouter-music": {
"type": "openai",
"baseUrl": "https://openrouter.ai/api/v1",
"audioProvider": "openrouter",
"apiKeyEnv": "OPENROUTER_API_KEY",
"timeoutMs": 600000
}
},
"models": [{
"id": "google/lyria-3-pro-preview",
"backend": "openrouter-music",
"upstreamModel": "google/lyria-3-pro-preview",
"kind": "audio_generation"
}],
"aliases": {"music": {"members": ["google/lyria-3-pro-preview"]}},
"defaults": {"audioGenerationModel": "music"}
}
```

Send `prompt` or `instructions`, optionally `lyrics`, `input`, and `duration` in seconds. Duration is a prompt instruction, not a guaranteed output length. `response_format` (or `format`) accepts `wav` or `mp3`, defaulting to WAV. One inline PNG/JPEG `image` or `reference_image` is supported. Remote image URLs, seeds, steps and other unsupported controls are rejected. Configured OpenRouter provider restrictions are preserved.

The adapter validates and buffers the provider SSE stream before returning audio bytes. It requires a completion marker or successful stop followed by clean EOF, bounds individual events and total output, and never automatically retries a billable generation. Cancellation and timeout cover both generation and conversion.

Install `ffmpeg` on the gateway host and ensure it is on the managed service's PATH. Some providers return MP3 even when WAV is requested; LLooM detects the actual format and converts it locally when necessary. Conversion takes buffered audio through pipes with network protocols disabled. No music model or GPU runtime is loaded on the gateway for cloud generation. The API responds with the requested audio format and matching Content-Type.

See the [OpenRouter audio documentation](https://openrouter.ai/docs/guides/overview/multimodal/audio) for the upstream contract.
Loading
Loading