diff --git a/CHANGELOG.md b/CHANGELOG.md index 36ed5cd..60c2314 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,7 +8,8 @@ All notable changes to LLooM will be documented in this file. The format follows - LLooM Hear CPU audio analysis, with structured estimates, optional dashboard images, and opt-in upstream interpretation. File and URL inputs require operator configuration. -- Thirteen standalone NVIDIA ComfyUI media recipes, a public-source backend build, and shared runtime reuse for image, video and music generation. +- Fourteen per-model NVIDIA ComfyUI media recipes and a public-source backend build for image, video and music generation. Each model runs in its own container with read-only mounts of only its files, so LLooM admits, evicts and restores each one independently. The media launcher refuses to start without `LLOOM_MEDIA_MODEL`, and re-applying a recipe moves a model off the retired shared `comfyui-media` runtime, dropping it once unused. +- OpenRouter Lyria audio generation through `/v1/audio/generations`. - Selective `include` paths or globs on recipe `download-model` steps, so a single-model lane fetches only the files its graph loads instead of every quantization in the model repository. Planned output reports the resolved `--include` command line. - First-class `audio_generation` models and the `POST /v1/audio/generations` route, with `GET /v1/audio/generations/models` and a `defaults.audioGenerationModel` fallback. Music and other audio-generation lanes no longer have to be registered as `audio_speech` to be reachable: `/v1/audio/speech` stays speech- and clone-only, and each route refuses the other's kind with `wrong_model_kind`. Recipe capabilities `audio-generation`, `music-generation` and `audio-music-generation` materialize as `audio_generation`. diff --git a/backends/comfyui-media/bridge/server.py b/backends/comfyui-media/bridge/server.py index d740198..215083e 100644 --- a/backends/comfyui-media/bridge/server.py +++ b/backends/comfyui-media/bridge/server.py @@ -37,6 +37,10 @@ log = logging.getLogger("bridge") DEFAULT_COMFY_URL = "http://127.0.0.1:8188" +# Optional single-model selector. When set, the bridge serves exactly that one +# registry entry. An empty or unknown value refuses startup rather than falling +# back to advertising every bundled model. +MEDIA_MODEL_ENV = "LLOOM_MEDIA_MODEL" MAX_SEED = 2**31 - 1 MAX_DURATION_SECONDS = 600 # Text bounds, enforced here regardless of what the graph library does. @@ -70,6 +74,42 @@ def _default_data_roots() -> DataRoots: return DataRoots.from_env() +def _resolve_media_model(select: str | None = None) -> str | None: + """The exact registry model a single-model runtime serves, if configured. + + ``select`` defaults to ``LLOOM_MEDIA_MODEL``. An absent variable selects + nothing, which only in-process tests rely on: the container launcher + refuses to start without a selection. A present + but empty or whitespace-only value is a configuration error and refuses + startup. Validation against the registry happens in ``create_app`` so the + rejection also covers explicitly injected ``models``. + """ + value = os.environ.get(MEDIA_MODEL_ENV) if select is None else select + if value is None: + return None + stripped = value.strip() + if not stripped: + raise ValueError(f"{MEDIA_MODEL_ENV} is set but empty; set an exact model ID or unset it.") + if stripped != value: + # Surrounding whitespace is almost certainly a deployment mistake; an + # exact model ID never carries it. Resolve to the trimmed ID, which is + # then validated against the registry like any other selection. + log.warning("%s has surrounding whitespace; using %r", MEDIA_MODEL_ENV, stripped) + return stripped + + +def _select_models(models: dict, select: str | None) -> dict: + """Filter ``models`` to ``select``; fail closed on an unknown selection.""" + if select is None: + return models + if not isinstance(models, dict) or select not in models: + supported = ", ".join(sorted(models)) if isinstance(models, dict) and models else "(none)" + raise ValueError( + f"{MEDIA_MODEL_ENV}={select!r} is not in this bridge's model registry. Supported models: {supported}." + ) + return {select: models[select]} + + def create_app( comfy: ComfyClient | None = None, *, @@ -77,6 +117,7 @@ def create_app( build_graph=None, start_backend: bool = True, comfy_url: str | None = None, + media_model: str | None = None, ) -> FastAPI: app = FastAPI(title="LLooM media bridge", docs_url=None, redoc_url=None, openapi_url=None) if models is None or build_graph is None: @@ -85,11 +126,18 @@ def create_app( models = default_models if build_graph is None: build_graph = default_builder + # A single-model runtime must never advertise a model it cannot serve, so + # the selector is applied here -- after any injected registry is known and + # before the app can accept a request. An unknown or empty value raises out + # of the app factory: startup (including uvicorn import of module-level + # ``app``) fails closed instead of degrading to the full registry. + models = _select_models(models, _resolve_media_model(media_model)) state = { "comfy": comfy, "models": models, "build_graph": build_graph, "comfy_url": validate_comfy_base_url(comfy_url or _default_comfy_url()), + "media_model": next(iter(models)) if len(models) == 1 else None, } runner = SingleFlightRunner(comfy, _default_data_roots()) if comfy is not None else None state["runner"] = runner diff --git a/backends/comfyui-media/bridge/tests/conftest.py b/backends/comfyui-media/bridge/tests/conftest.py index 4f12c96..a7eab9f 100644 --- a/backends/comfyui-media/bridge/tests/conftest.py +++ b/backends/comfyui-media/bridge/tests/conftest.py @@ -46,6 +46,7 @@ def make_bridge(fake: FakeComfy, build_graph, *, start_backend: bool = False, ** models=kwargs.pop("models", {"video-model": "video", "audio-model": "audio"}), build_graph=build_graph, start_backend=start_backend, + **kwargs, ) return app, comfy diff --git a/backends/comfyui-media/bridge/tests/test_single_model.py b/backends/comfyui-media/bridge/tests/test_single_model.py new file mode 100644 index 0000000..112aca7 --- /dev/null +++ b/backends/comfyui-media/bridge/tests/test_single_model.py @@ -0,0 +1,118 @@ +"""Single-model serving mode (``LLOOM_MEDIA_MODEL``). + +A dedicated per-model runtime must advertise and serve exactly one registry +entry, and an explicitly set value that is empty or unknown must refuse +startup instead of falling back to the full bundled registry. An absent +variable preserves the previous multi-model behaviour. +""" + +from __future__ import annotations + +import pytest + +from conftest import asgi_client, make_bridge +from fake_comfy import FakeComfy + +REGISTRY = {"video-model": "video", "audio-model": "audio"} + + +def video_builder(model_id, payload, image_filename=None, prefix="lloom"): + return {"9": {"class_type": "SaveVideo", "inputs": {}}}, "9", "video" + + +async def test_selected_model_is_the_only_advertised_and_served_model(): + fake = FakeComfy() + app, comfy = make_bridge(fake, video_builder, models=REGISTRY, media_model="video-model") + assert app.state.bridge["media_model"] == "video-model" + async with asgi_client(app) as client: + await comfy.start() + listed = {m["id"] for m in (await client.get("/v1/models")).json()["data"]} + assert listed == {"video-model"} + health = await client.get("/health") + assert health.status_code == 200, health.text + assert health.json()["backend_ready"] is True + await comfy.aclose() + + +async def test_other_registered_model_is_rejected_before_any_graph_or_submission(): + fake = FakeComfy() + calls: list[str] = [] + + def builder(model_id, payload, image_filename=None, prefix="lloom"): + calls.append(model_id) + return {"9": {"class_type": "SaveVideo", "inputs": {}}}, "9", "video" + + app, comfy = make_bridge(fake, builder, models=REGISTRY, media_model="video-model") + async with asgi_client(app) as client: + await comfy.start() + # ``audio-model`` exists in the injected registry but not in this + # runtime, so it must be an unsupported-model 400 that never reaches + # the video graph builder or ComfyUI's /prompt. + resp = await client.post( + "/v1/videos/generations", json={"model": "audio-model", "prompt": "x"} + ) + assert resp.status_code == 400, resp.text + err = resp.json()["error"] + assert err["code"] == "unsupported_model" + assert "video-model" in err["message"] + assert calls == [] + assert fake.prompts == {} + assert fake.uploaded == [] + await comfy.aclose() + + +@pytest.mark.parametrize("value", ["", " "]) +def test_explicitly_empty_selection_fails_closed(value, monkeypatch): + from server import create_app + + monkeypatch.setenv("LLOOM_MEDIA_MODEL", value) + with pytest.raises(ValueError): + create_app(build_graph=video_builder, models=dict(REGISTRY), start_backend=False) + + +def test_unknown_selection_fails_closed_even_with_injected_registry(monkeypatch): + from server import create_app + + monkeypatch.setenv("LLOOM_MEDIA_MODEL", "not-in-registry") + with pytest.raises(ValueError) as excinfo: + # Injected models still go through the selector, so a model the graphs + # module supports cannot be smuggled in by an explicit registry. + create_app(build_graph=video_builder, models=dict(REGISTRY), start_backend=False) + assert "not-in-registry" in str(excinfo.value) + + +def test_unknown_selection_is_not_silently_ignored(monkeypatch): + monkeypatch.setenv("LLOOM_MEDIA_MODEL", "missing-model") + import importlib + + import server as server_mod + + # The module-level ``app = create_app()`` runs on import; a bad selector + # would make an import-time startup fail rather than expose all models. + with pytest.raises(ValueError): + importlib.reload(server_mod) + monkeypatch.delenv("LLOOM_MEDIA_MODEL") + importlib.reload(server_mod) + + +async def test_absent_selection_keeps_full_registry(monkeypatch): + monkeypatch.delenv("LLOOM_MEDIA_MODEL", raising=False) + fake = FakeComfy() + app, comfy = make_bridge(fake, video_builder, models=REGISTRY) + assert app.state.bridge["media_model"] is None + async with asgi_client(app) as client: + await comfy.start() + listed = {m["id"] for m in (await client.get("/v1/models")).json()["data"]} + assert listed == set(REGISTRY) + await comfy.aclose() + + +async def test_selection_via_environment_is_honoured(monkeypatch): + monkeypatch.setenv("LLOOM_MEDIA_MODEL", "audio-model") + fake = FakeComfy() + app, comfy = make_bridge(fake, video_builder, models=REGISTRY) + async with asgi_client(app) as client: + await comfy.start() + listed = {m["id"] for m in (await client.get("/v1/models")).json()["data"]} + assert listed == {"audio-model"} + await comfy.aclose() diff --git a/backends/comfyui-media/build/launch.py b/backends/comfyui-media/build/launch.py index b2ac6ba..0a6eb88 100644 --- a/backends/comfyui-media/build/launch.py +++ b/backends/comfyui-media/build/launch.py @@ -8,11 +8,11 @@ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import model_paths # noqa: E402 -# Recipe-installed models land in per-repo directories under the LLooM model root. -# ComfyUI reads one extra_model_paths.yaml naming each of them, so installing a -# single-model recipe is enough for the engine to serve it -- no source edit, no -# hand-placed symlink. The aggregate default root stays where it is. -HOST_MODELS_ROOT = os.environ.get("LLOOM_MODELS_ROOT", "/opt/lloom-models") +# Each container serves exactly one model, whose files the recipe bind-mounts +# read-only into ComfyUI's model tree. There is no shared multi-model runtime. +if not os.environ.get("LLOOM_MEDIA_MODEL", "").strip(): + raise SystemExit("LLOOM_MEDIA_MODEL is required: each media runtime serves exactly one model") +HOST_MODELS_ROOT = os.environ.get("LLOOM_MODELS_ROOT", "/opt/ComfyUI/models") EXTRA_PATHS_FILE = "/data/extra_model_paths.yaml" CACHE_MODE = os.environ.get("LLOOM_COMFY_CACHE_MODE", "none") if CACHE_MODE not in ("none", "classic"): diff --git a/backends/comfyui-media/build/test_launch.py b/backends/comfyui-media/build/test_launch.py new file mode 100644 index 0000000..b29561f --- /dev/null +++ b/backends/comfyui-media/build/test_launch.py @@ -0,0 +1,15 @@ +import os +import subprocess +import sys + +LAUNCH = os.path.join(os.path.dirname(os.path.abspath(__file__)), "launch.py") + + +def test_launch_refuses_to_serve_without_a_single_model(): + for value in (None, "", " "): + env = {k: v for k, v in os.environ.items() if k != "LLOOM_MEDIA_MODEL"} + if value is not None: + env["LLOOM_MEDIA_MODEL"] = value + result = subprocess.run([sys.executable, LAUNCH], env=env, capture_output=True, text=True, timeout=30) + assert result.returncode != 0 + assert "LLOOM_MEDIA_MODEL is required" in result.stderr diff --git a/docs/audio-providers.md b/docs/audio-providers.md new file mode 100644 index 0000000..613686d --- /dev/null +++ b/docs/audio-providers.md @@ -0,0 +1,33 @@ +# Cloud music through OpenRouter + +LLooM accepts music requests at `POST /v1/audio/generations`. Set an OpenAI-compatible backend's `audioProvider` to `openrouter` to use OpenRouter's streaming chat audio API. Ordinary audio backends keep their existing binary proxy behavior. + +```json +{ + "backends": { + "openrouter-music": { + "type": "openai", + "baseUrl": "https://openrouter.ai/api/v1", + "audioProvider": "openrouter", + "apiKeyEnv": "OPENROUTER_API_KEY", + "timeoutMs": 600000 + } + }, + "models": [{ + "id": "google/lyria-3-pro-preview", + "backend": "openrouter-music", + "upstreamModel": "google/lyria-3-pro-preview", + "kind": "audio_generation" + }], + "aliases": {"music": {"members": ["google/lyria-3-pro-preview"]}}, + "defaults": {"audioGenerationModel": "music"} +} +``` + +Send `prompt` or `instructions`, optionally `lyrics`, `input`, and `duration` in seconds. Duration is a prompt instruction, not a guaranteed output length. `response_format` (or `format`) accepts `wav` or `mp3`, defaulting to WAV. One inline PNG/JPEG `image` or `reference_image` is supported. Remote image URLs, seeds, steps and other unsupported controls are rejected. Configured OpenRouter provider restrictions are preserved. + +The adapter validates and buffers the provider SSE stream before returning audio bytes. It requires a completion marker or successful stop followed by clean EOF, bounds individual events and total output, and never automatically retries a billable generation. Cancellation and timeout cover both generation and conversion. + +Install `ffmpeg` on the gateway host and ensure it is on the managed service's PATH. Some providers return MP3 even when WAV is requested; LLooM detects the actual format and converts it locally when necessary. Conversion takes buffered audio through pipes with network protocols disabled. No music model or GPU runtime is loaded on the gateway for cloud generation. The API responds with the requested audio format and matching Content-Type. + +See the [OpenRouter audio documentation](https://openrouter.ai/docs/guides/overview/multimodal/audio) for the upstream contract. diff --git a/docs/comfyui-media.md b/docs/comfyui-media.md index 501969b..4991e12 100644 --- a/docs/comfyui-media.md +++ b/docs/comfyui-media.md @@ -1,15 +1,15 @@ # ComfyUI image, video and music recipes Each `linux-nvidia-comfyui-*` recipe installs one model and the files its workflow -uses. The recipes share a single LLooM-managed `comfyui-media` runtime. The first -installation builds the backend from public source; later installations reuse -that image, container, backend URL and concurrency limit. +uses. Each recipe creates its own LLooM-managed runtime, container, backend URL +and concurrency limit. Installations reuse the immutable backend image on disk. ## Install Use Linux on NVIDIA hardware with Docker, NVIDIA Container Toolkit, Python 3 -with venv support, and a CUDA 13-compatible driver. The recipes conservatively -reserve 95 GiB of host/model memory for the shared runtime. Consult each recipe's +with venv support, and a CUDA 13-compatible driver. Each recipe carries a +provisional memory estimate for admission, not a hard allocation. Validate peak +memory for the intended request size before tightening that estimate. Consult each recipe's `diskGb` estimate before downloading. Model repositories can require separately accepted access terms and Hugging Face authentication; credentials stay in your local Hugging Face environment and are never included in recipes or images. @@ -33,18 +33,15 @@ resolved Python environment in `/opt/package-lock.txt`. These are source-pinned builds, not a claim of bit-for-bit reproducibility across dependency indexes. Add another model with the same command and its recipe ID. Use `--additive` to -preserve the rest of your catalog. The backend mounts `${modelRoot}` read-only; -its extra model paths cover repository roots, `split_files`, and root-level -LoRAs. Paths for all bundled models are registered when it starts, so subsequent -model installations become visible without changing its launch configuration. -Run setup while the media lane is idle: acquisition can temporarily stage a -shared model directory while verifying newly requested files. - -The shared engine has one generation slot across all modalities. LLooM queues -requests through that runtime, so adding models does not start duplicate GPU -engines. This reuse applies to the managed runtime created by these recipes; -an unrelated ComfyUI installation is not silently adopted. Docker backend ports -remain bound to loopback. Clients use the authenticated LLooM gateway. +preserve the rest of your catalog. Each container receives read-only bind mounts +for only its model's exact checkpoint, encoder, VAE and adapter files. Files can +be shared on disk between recipes without combining process or memory residency. +Run setup while the selected lane is idle so acquisition and verification do not +race a running model's file access. + +Each runtime has one generation slot. LLooM admits and unloads it independently +using that model's estimate and current host headroom. Docker backend ports stay +on loopback; clients use the authenticated LLooM gateway. Qwen-Image 2.1 is the one image family here that generates and edits from a single checkpoint. It samples at its own shift with classifier-free guidance off, @@ -61,22 +58,22 @@ existing backend must support `/v1/audio/generations`. ## Models -| Recipe suffix | Gateway model | Endpoint | -| --- | --- | --- | -| `flux-2-klein-4b` | `black-forest-labs/FLUX.2-klein-4B` | `/v1/images/generations` | -| `qwen-image-2512` | `Qwen/Qwen-Image-2512` | `/v1/images/generations` | -| `qwen-image-2512-lightning` | `Qwen/Qwen-Image-2512-Lightning` | `/v1/images/generations` | -| `qwen-image-edit-2511` | `Qwen/Qwen-Image-Edit-2511` | `/v1/images/generations` with inline image | -| `qwen-image-2-1` | `Qwen/Qwen-Image-2.1` | `/v1/images/generations`, with inline image to edit | -| `ideogram-4` | `Comfy-Org/Ideogram-4` | `/v1/images/generations` | -| `krea-2-turbo` | `Comfy-Org/Krea-2-Turbo` | `/v1/images/generations` | -| `minimax-h3` | `MiniMaxAI/MiniMax-H3` | `/v1/videos/generations` | -| `minimax-h3-turbo` | `MiniMaxAI/MiniMax-H3-Turbo` | `/v1/videos/generations` | -| `ltx-2-5` | `Lightricks/LTX-2.5` | `/v1/videos/generations` | -| `minimax-music3` | `MiniMaxAI/MiniMax-Music3` | `/v1/audio/generations` | -| `ace-step-1-5-xl-sft` | `ACE-Step/ACE-Step-1.5-XL-SFT` | `/v1/audio/generations` | -| `ace-step-1-5-xl-turbo` | `ACE-Step/ACE-Step-1.5-XL-Turbo` | `/v1/audio/generations` | -| `yue2-3b` | `Comfy-Org/YuE2-3B` | `/v1/audio/generations` | +| Recipe suffix | Gateway model | Endpoint | +| --------------------------- | ----------------------------------- | --------------------------------------------------- | +| `flux-2-klein-4b` | `black-forest-labs/FLUX.2-klein-4B` | `/v1/images/generations` | +| `qwen-image-2512` | `Qwen/Qwen-Image-2512` | `/v1/images/generations` | +| `qwen-image-2512-lightning` | `Qwen/Qwen-Image-2512-Lightning` | `/v1/images/generations` | +| `qwen-image-edit-2511` | `Qwen/Qwen-Image-Edit-2511` | `/v1/images/generations` with inline image | +| `qwen-image-2-1` | `Qwen/Qwen-Image-2.1` | `/v1/images/generations`, with inline image to edit | +| `ideogram-4` | `Comfy-Org/Ideogram-4` | `/v1/images/generations` | +| `krea-2-turbo` | `Comfy-Org/Krea-2-Turbo` | `/v1/images/generations` | +| `minimax-h3` | `MiniMaxAI/MiniMax-H3` | `/v1/videos/generations` | +| `minimax-h3-turbo` | `MiniMaxAI/MiniMax-H3-Turbo` | `/v1/videos/generations` | +| `ltx-2-5` | `Lightricks/LTX-2.5` | `/v1/videos/generations` | +| `minimax-music3` | `MiniMaxAI/MiniMax-Music3` | `/v1/audio/generations` | +| `ace-step-1-5-xl-sft` | `ACE-Step/ACE-Step-1.5-XL-SFT` | `/v1/audio/generations` | +| `ace-step-1-5-xl-turbo` | `ACE-Step/ACE-Step-1.5-XL-Turbo` | `/v1/audio/generations` | +| `yue2-3b` | `Comfy-Org/YuE2-3B` | `/v1/audio/generations` | Prefix each suffix with `linux-nvidia-comfyui-` for the recipe ID. Recipe metadata is MIT licensed; model weights retain their own licenses. Each recipe links its @@ -142,3 +139,58 @@ allow headroom for retained workflow tensors in its `memoryGb` estimate. Each request uses a unique output-save prefix so cached graphs cannot reuse an artifact already deleted by bridge cleanup. Model-loader inputs remain stable. + +## Per-model runtimes + +The recipes configure independent runtimes by default. Every model is admitted, +unloaded and memory-accounted independently. Each runtime needs its own backend URL, container name and memory estimate. + +Set the bridge environment variable `LLOOM_MEDIA_MODEL` on a per-model runtime +and point it at one exact backend model ID (the same string advertised by the bridge in +`/v1/models`, for example `ACE-Step/ACE-Step-1.5-XL-Turbo`): + +```sh +LLOOM_MEDIA_MODEL=ACE-Step/ACE-Step-1.5-XL-Turbo +``` + +The bridge then advertises and serves only that registry entry. A request for +any other model is rejected with `unsupported_model` before the graph builder +runs and before anything is submitted to ComfyUI, so a request can never pull an +unmounted checkpoint into a single-model engine. The selection is applied while +the app is built, so it also holds for a registry injected by a caller. + +Startup fails closed on a bad selector: an explicitly set but empty (or +whitespace-only) value and an unknown model ID both abort startup rather than +degrade to exposing every bundled model. Unset the variable entirely to keep the +previous multi-model behaviour; an absent variable is never treated as a +selection. Startup stays weight-lazy — the bridge only probes ComfyUI's +`/system_stats` and the registry wiring, it never loads a checkpoint — and +`/health` performs no inference. + +Mount only the files the one model needs, so the runtime's estimate matches what +it can actually load: + +- the specific checkpoint (or diffusion/UNet file) for that model; +- its text encoder and VAE; +- its LoRAs and any tokenizer or dependency files the workflow references; +- nothing for the other bundled models. + +Recipes mount individual files from `${modelRoot}` and set +`LLOOM_MODELS_ROOT=/opt/ComfyUI/models` to avoid scanning an aggregate model root. A single-model runtime that can see only its own files cannot +accidentally load a checkpoint it has no memory budget for. + +The `LLOOM_COMFY_CACHE_MODE` tradeoff applies per runtime and is the reason to +consider both modes here: + +- `classic` keeps ComfyUI's last-workflow node cache, including the reusable + model loaders. On a dedicated single-model runtime the cached graph always + targets the same weights, so repeated requests skip re-loading the checkpoint + and can be noticeably faster. It retains workflow tensors, so allow headroom + in that runtime's `memoryGb` estimate. +- `none` (the default) preserves the uncached behaviour and holds no workflow + tensors between requests, at the cost of re-loading the checkpoint on each + request. Choose it when free memory matters more than repeated-request speed. + +Sharing the immutable Docker image or read-only weight files on disk does not +combine runtime residency. Per-model processes remain independently admitted +and evicted. diff --git a/docs/recipes.md b/docs/recipes.md index 292851a..8815ddf 100644 --- a/docs/recipes.md +++ b/docs/recipes.md @@ -340,4 +340,4 @@ Only the stable active file participates in planning and automatic recommendatio LLooM intentionally does not use stale model fallback aliases to make an index pass. Recipe `model` and `gatewayModel` values must be exact advertised IDs. -Standalone image, video and music recipes with a shared ComfyUI backend are documented in [ComfyUI media](comfyui-media.md). +Per-model image, video and music recipes, each in its own ComfyUI runtime, are documented in [ComfyUI media](comfyui-media.md). diff --git a/package.json b/package.json index 6470aed..1788363 100644 --- a/package.json +++ b/package.json @@ -64,7 +64,7 @@ "check": "npm run check:cluster && npm run check:js && npm run check:entity-stagger && npm run check:python && npm run check:shell && npm run check:spark-deploy && npm run check:workers", "check:cluster": "node --check src/cluster.mjs && node --check test/cluster.test.mjs", "check:entity-stagger": "node --check scripts/entity-stagger-benchmark.mjs && node --check test/entity-stagger-benchmark.test.mjs", - "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs && node --check src/dashboard-scene.mjs && node --check src/dashboard-presence.mjs && node --check src/dashboard-presence-client.mjs && node --check src/first-run.mjs && node --check src/first-run-page.mjs && node --check src/browser-setup.mjs && node --check src/runtime-preferences.mjs && node --check src/installation-jobs.mjs && node --check src/model-installation.mjs && node --check src/dashboard-installation.mjs && node --check src/runtime-memory-safety.mjs && node --check src/dashboard-memory.mjs && node --check src/runtime-memory-usage.mjs && node --check src/config-profiles.mjs && node --check test/config-profiles.test.mjs && node --check src/runtime-policy-config.mjs && node --check test/runtime-policy-config.test.mjs", + "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs && node --check src/dashboard-scene.mjs && node --check src/dashboard-presence.mjs && node --check src/dashboard-presence-client.mjs && node --check src/first-run.mjs && node --check src/first-run-page.mjs && node --check src/browser-setup.mjs && node --check src/runtime-preferences.mjs && node --check src/installation-jobs.mjs && node --check src/model-installation.mjs && node --check src/dashboard-installation.mjs && node --check src/runtime-memory-safety.mjs && node --check src/dashboard-memory.mjs && node --check src/runtime-memory-usage.mjs && node --check src/config-profiles.mjs && node --check test/config-profiles.test.mjs && node --check src/audio-providers.mjs && node --check test/audio-providers.test.mjs && node --check src/runtime-policy-config.mjs && node --check test/runtime-policy-config.test.mjs", "check:python": "python3 -m py_compile backends/dspark-vllm/apply-patch-pack.py backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/vision_exp/*.py backends/qwen38-vllm/apply-ple-fp8-patch.py backends/qwen38-sglang/*.py backends/glm53-exl3/*.py backends/mlx-audio/lloom_audio_server.py backends/hear/lloom_hear_server.py backends/chatterbox/lloom_chatterbox_server.py patches/apply_mtplx_longctx_fix.py backends/comfyui-media/bridge/*.py backends/comfyui-media/graphs/*.py backends/comfyui-media/build/*.py backends/comfyui-media/install.py backends/qwen-image-diffusers/*.py backends/hear/test/test_hear_security.py src/darwin-memory-usage.py", "check:shell": "bash -n backends/dspark-vllm/entrypoint.sh backends/dspark-vllm/runtime/build-hardened-image.sh backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/apply-runtime.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.sh backends/qwen38-vllm/entrypoint.sh backends/qwen38-sglang/entrypoint.sh backends/glm53-exl3/entrypoint.sh backends/mlx-audio/install.sh backends/hear/install.sh backends/chatterbox/install.sh clients/examples/lloom-claude clients/examples/lloom-codex clients/examples/lloom-hermes clients/examples/lloom-opencode clients/examples/lloom-zero scripts/remote-install-spark.sh scripts/remote-stage-spark-worker.sh", "check:spark-deploy": "node --check test/spark-deploy.test.mjs && node test/spark-deploy.test.mjs", @@ -73,7 +73,7 @@ "community:check": "node scripts/check-community-deploy.mjs", "package:check": "node scripts/check-package.mjs", "test:adoption": "node test/resource-fit.test.mjs && node test/model-acquisition.test.mjs && node test/dashboard-status.test.mjs", - "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/openrouter-provider.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/config-profiles.test.mjs && node test/route-control.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs && npm run test:ux && npm run test:memory-safety && node --test test/dashboard-memory.test.mjs test/runtime-memory-usage.test.mjs", + "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/openrouter-provider.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/config-profiles.test.mjs && node test/route-control.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs && npm run test:ux && npm run test:memory-safety && node --test test/dashboard-memory.test.mjs test/runtime-memory-usage.test.mjs && node --test test/audio-providers.test.mjs", "test:cluster": "node test/cluster.test.mjs", "test": "npm run test:cluster && npm run test:unit && npm run test:entity-stagger && npm run smoke && npm run community:check", "test:entity-stagger": "node test/entity-stagger-benchmark.test.mjs", diff --git a/recipes/index.json b/recipes/index.json index b613189..c502bb3 100644 --- a/recipes/index.json +++ b/recipes/index.json @@ -840,7 +840,7 @@ } ], "name": "ACE-Step/ACE-Step-1.5-XL-SFT (ComfyUI, NVIDIA)", - "summary": "Install ACE-Step/ACE-Step-1.5-XL-SFT with a shared, locally built ComfyUI media backend.", + "summary": "Install ACE-Step/ACE-Step-1.5-XL-SFT in an independent, single-model ComfyUI runtime.", "tags": ["ace-step-1.5-xl-sft", "cuda", "fp8", "nvfp4", "nvidia", "single-model"], "capabilities": ["audio-generation", "music-generation"], "source": { @@ -860,7 +860,7 @@ } ], "name": "ACE-Step/ACE-Step-1.5-XL-Turbo (ComfyUI, NVIDIA)", - "summary": "Install ACE-Step/ACE-Step-1.5-XL-Turbo with a shared, locally built ComfyUI media backend.", + "summary": "Install ACE-Step/ACE-Step-1.5-XL-Turbo in an independent, single-model ComfyUI runtime.", "tags": ["ace-step-1.5-xl-turbo", "cuda", "fp8", "nvfp4", "nvidia", "single-model"], "capabilities": ["audio-generation", "music-generation"], "source": { @@ -880,7 +880,7 @@ } ], "name": "black-forest-labs/FLUX.2-klein-4B (ComfyUI, NVIDIA)", - "summary": "Install black-forest-labs/FLUX.2-klein-4B with a shared, locally built ComfyUI media backend.", + "summary": "Install black-forest-labs/FLUX.2-klein-4B in an independent, single-model ComfyUI runtime.", "tags": ["cuda", "flux.2-klein-4b", "fp8", "nvfp4", "nvidia", "single-model"], "capabilities": ["image-generation", "image-editing"], "source": { @@ -900,7 +900,7 @@ } ], "name": "Comfy-Org/Ideogram-4 (ComfyUI, NVIDIA)", - "summary": "Install Comfy-Org/Ideogram-4 with a shared, locally built ComfyUI media backend.", + "summary": "Install Comfy-Org/Ideogram-4 in an independent, single-model ComfyUI runtime.", "tags": ["cuda", "fp8", "ideogram-4", "nvfp4", "nvidia", "single-model"], "capabilities": ["image-generation", "typography", "layout-control", "2k-output"], "source": { @@ -920,7 +920,7 @@ } ], "name": "Comfy-Org/Krea-2-Turbo (ComfyUI, NVIDIA)", - "summary": "Install Comfy-Org/Krea-2-Turbo with a shared, locally built ComfyUI media backend.", + "summary": "Install Comfy-Org/Krea-2-Turbo in an independent, single-model ComfyUI runtime.", "tags": ["cuda", "fp8", "krea-2-turbo", "nvfp4", "nvidia", "single-model"], "capabilities": ["image-generation", "fast-generation"], "source": { @@ -940,7 +940,7 @@ } ], "name": "Lightricks/LTX-2.5 (ComfyUI, NVIDIA)", - "summary": "Install Lightricks/LTX-2.5 with a shared, locally built ComfyUI media backend.", + "summary": "Install Lightricks/LTX-2.5 in an independent, single-model ComfyUI runtime.", "tags": ["cuda", "fp8", "ltx-2.5", "nvfp4", "nvidia", "single-model"], "capabilities": ["video-generation", "text-to-video", "image-to-video"], "source": { @@ -960,7 +960,7 @@ } ], "name": "MiniMaxAI/MiniMax-H3-Turbo (ComfyUI, NVIDIA)", - "summary": "Install MiniMaxAI/MiniMax-H3-Turbo with a shared, locally built ComfyUI media backend.", + "summary": "Install MiniMaxAI/MiniMax-H3-Turbo in an independent, single-model ComfyUI runtime.", "tags": ["cuda", "fp8", "minimax-h3-turbo", "nvfp4", "nvidia", "single-model"], "capabilities": ["video-generation", "text-to-video"], "source": { @@ -980,7 +980,7 @@ } ], "name": "MiniMaxAI/MiniMax-H3 (ComfyUI, NVIDIA)", - "summary": "Install MiniMaxAI/MiniMax-H3 with a shared, locally built ComfyUI media backend.", + "summary": "Install MiniMaxAI/MiniMax-H3 in an independent, single-model ComfyUI runtime.", "tags": ["cuda", "fp8", "minimax-h3", "nvfp4", "nvidia", "single-model"], "capabilities": ["video-generation", "text-to-video", "image-to-video"], "source": { @@ -1000,7 +1000,7 @@ } ], "name": "MiniMaxAI/MiniMax-Music3 (ComfyUI, NVIDIA)", - "summary": "Install MiniMaxAI/MiniMax-Music3 with a shared, locally built ComfyUI media backend.", + "summary": "Install MiniMaxAI/MiniMax-Music3 in an independent, single-model ComfyUI runtime.", "tags": ["cuda", "fp8", "minimax-music3", "nvfp4", "nvidia", "single-model"], "capabilities": ["audio-generation", "music-generation"], "source": { @@ -1020,7 +1020,7 @@ } ], "name": "Qwen/Qwen-Image-2.1 (ComfyUI, NVIDIA)", - "summary": "Install Qwen/Qwen-Image-2.1, one DiT for generation and reference editing, with the shared ComfyUI media backend.", + "summary": "Install Qwen/Qwen-Image-2.1 in an independent, single-model ComfyUI runtime.", "tags": ["cuda", "int8", "nvidia", "qwen-image-2.1", "image-editing", "transparency", "single-model"], "capabilities": ["image-generation", "image-editing"], "source": { @@ -1040,7 +1040,7 @@ } ], "name": "Qwen/Qwen-Image-2512-Lightning (ComfyUI, NVIDIA)", - "summary": "Install Qwen/Qwen-Image-2512-Lightning with a shared, locally built ComfyUI media backend.", + "summary": "Install Qwen/Qwen-Image-2512-Lightning in an independent, single-model ComfyUI runtime.", "tags": ["cuda", "fp8", "nvfp4", "nvidia", "qwen-image-2512-lightning", "single-model"], "capabilities": ["image-generation"], "source": { @@ -1060,7 +1060,7 @@ } ], "name": "Qwen/Qwen-Image-2512 (ComfyUI, NVIDIA)", - "summary": "Install Qwen/Qwen-Image-2512 with a shared, locally built ComfyUI media backend.", + "summary": "Install Qwen/Qwen-Image-2512 in an independent, single-model ComfyUI runtime.", "tags": ["cuda", "fp8", "nvfp4", "nvidia", "qwen-image-2512", "single-model"], "capabilities": ["image-generation"], "source": { @@ -1080,7 +1080,7 @@ } ], "name": "Qwen/Qwen-Image-Edit-2511 (ComfyUI, NVIDIA)", - "summary": "Install Qwen/Qwen-Image-Edit-2511 with a shared, locally built ComfyUI media backend.", + "summary": "Install Qwen/Qwen-Image-Edit-2511 in an independent, single-model ComfyUI runtime.", "tags": ["cuda", "fp8", "nvfp4", "nvidia", "qwen-image-edit-2511", "single-model"], "capabilities": ["image-generation", "image-editing"], "source": { @@ -1100,7 +1100,7 @@ } ], "name": "Comfy-Org/YuE2-3B (ComfyUI, NVIDIA)", - "summary": "Install Comfy-Org/YuE2-3B with a shared, locally built ComfyUI media backend.", + "summary": "Install Comfy-Org/YuE2-3B in an independent, single-model ComfyUI runtime.", "tags": ["cuda", "fp8", "nvfp4", "nvidia", "single-model", "yue2-3b"], "capabilities": ["audio-generation", "music-generation", "symbolic-score"], "source": { @@ -1134,15 +1134,7 @@ ], "name": "LLooM Hear - local sound and music perception", "summary": "CPU audio analysis with loudness, tempo, key, timbre, signal-change estimates and an optional dashboard image. Generated interpretation is opt-in and uses a separately configured upstream model.", - "tags": [ - "cpu", - "audio-analysis", - "audio-understanding", - "music", - "dsp", - "measurement", - "perception" - ], + "tags": ["cpu", "audio-analysis", "audio-understanding", "music", "dsp", "measurement", "perception"], "recommendedFor": [ "agents that natively understand only text and images but must reason about sound or music", "any host with ffmpeg; no accelerator required", diff --git a/recipes/linux-nvidia-comfyui-ace-step-1-5-xl-sft.json b/recipes/linux-nvidia-comfyui-ace-step-1-5-xl-sft.json index 5914e8c..c820670 100644 --- a/recipes/linux-nvidia-comfyui-ace-step-1-5-xl-sft.json +++ b/recipes/linux-nvidia-comfyui-ace-step-1-5-xl-sft.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-ace-step-1-5-xl-sft", "name": "ACE-Step/ACE-Step-1.5-XL-SFT (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install ACE-Step/ACE-Step-1.5-XL-SFT with a shared, locally built ComfyUI media backend.", + "summary": "Install ACE-Step/ACE-Step-1.5-XL-SFT in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights retain their upstream licenses" @@ -21,53 +21,28 @@ "href": "https://huggingface.co/Comfy-Org/ace_step_1.5_ComfyUI_files/tree/694a9723ff772285c73f0700caacf944d3f02f8d" } ], - "keywords": [ - "ace-step-1.5-xl-sft", - "cuda", - "fp8", - "nvfp4", - "nvidia", - "single-model" - ], - "capabilities": [ - "audio-generation", - "music-generation" - ], + "keywords": ["ace-step-1.5-xl-sft", "cuda", "fp8", "nvfp4", "nvidia", "single-model"], + "capabilities": ["audio-generation", "music-generation"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; subsequent recipes share one managed ComfyUI runtime.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with CUDA 13-compatible host driver required." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], - "memoryGb": 95, + "platforms": ["linux-x64", "linux-arm64"], + "memoryGb": 35, "diskGb": 39, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -76,9 +51,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-ace_step_1.5_comfyui_files", @@ -126,21 +99,14 @@ "role": "ace-step-1.5-xl-sft", "model": "ACE-Step/ACE-Step-1.5-XL-SFT", "gatewayModel": "ACE-Step/ACE-Step-1.5-XL-SFT", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-ace-step-1-5-xl-sft", + "backendConfig": "comfyui-ace-step-1-5-xl-sft", "kind": "audio_generation", - "input": [ - "text" - ], - "output": [ - "audio" - ], - "capabilities": [ - "audio-generation", - "music-generation" - ], + "input": ["text"], + "output": ["audio"], + "capabilities": ["audio-generation", "music-generation"], "settings": { - "memoryGb": 95, + "memoryGb": 35, "priority": 20, "keepWarm": false, "maxActiveRequests": 1, @@ -148,11 +114,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-ace-step-1-5-xl-sft", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -163,12 +129,24 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--ace_step_1.5_ComfyUI_files/split_files/diffusion_models/acestep_v1.5_xl_sft_bf16.safetensors,dst=/opt/ComfyUI/models/diffusion_models/acestep_v1.5_xl_sft_bf16.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--ace_step_1.5_ComfyUI_files/split_files/text_encoders/qwen_0.6b_ace15.safetensors,dst=/opt/ComfyUI/models/text_encoders/qwen_0.6b_ace15.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--ace_step_1.5_ComfyUI_files/split_files/text_encoders/qwen_4b_ace15.safetensors,dst=/opt/ComfyUI/models/text_encoders/qwen_4b_ace15.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--ace_step_1.5_ComfyUI_files/split_files/vae/ace_1.5_vae.safetensors,dst=/opt/ComfyUI/models/vae/ace_1.5_vae.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-ace-step-1-5-xl-sft-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=ACE-Step/ACE-Step-1.5-XL-SFT" ] } }, diff --git a/recipes/linux-nvidia-comfyui-ace-step-1-5-xl-turbo.json b/recipes/linux-nvidia-comfyui-ace-step-1-5-xl-turbo.json index d433e69..dc76f4d 100644 --- a/recipes/linux-nvidia-comfyui-ace-step-1-5-xl-turbo.json +++ b/recipes/linux-nvidia-comfyui-ace-step-1-5-xl-turbo.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-ace-step-1-5-xl-turbo", "name": "ACE-Step/ACE-Step-1.5-XL-Turbo (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install ACE-Step/ACE-Step-1.5-XL-Turbo with a shared, locally built ComfyUI media backend.", + "summary": "Install ACE-Step/ACE-Step-1.5-XL-Turbo in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights retain their upstream licenses" @@ -21,53 +21,28 @@ "href": "https://huggingface.co/Comfy-Org/ace_step_1.5_ComfyUI_files/tree/694a9723ff772285c73f0700caacf944d3f02f8d" } ], - "keywords": [ - "ace-step-1.5-xl-turbo", - "cuda", - "fp8", - "nvfp4", - "nvidia", - "single-model" - ], - "capabilities": [ - "audio-generation", - "music-generation" - ], + "keywords": ["ace-step-1.5-xl-turbo", "cuda", "fp8", "nvfp4", "nvidia", "single-model"], + "capabilities": ["audio-generation", "music-generation"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; subsequent recipes share one managed ComfyUI runtime.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with CUDA 13-compatible host driver required." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], - "memoryGb": 95, + "platforms": ["linux-x64", "linux-arm64"], + "memoryGb": 35, "diskGb": 39, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -76,9 +51,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-ace_step_1.5_comfyui_files", @@ -126,21 +99,14 @@ "role": "ace-step-1.5-xl-turbo", "model": "ACE-Step/ACE-Step-1.5-XL-Turbo", "gatewayModel": "ACE-Step/ACE-Step-1.5-XL-Turbo", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-ace-step-1-5-xl-turbo", + "backendConfig": "comfyui-ace-step-1-5-xl-turbo", "kind": "audio_generation", - "input": [ - "text" - ], - "output": [ - "audio" - ], - "capabilities": [ - "audio-generation", - "music-generation" - ], + "input": ["text"], + "output": ["audio"], + "capabilities": ["audio-generation", "music-generation"], "settings": { - "memoryGb": 95, + "memoryGb": 35, "priority": 20, "keepWarm": false, "maxActiveRequests": 1, @@ -148,11 +114,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-ace-step-1-5-xl-turbo", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -163,12 +129,24 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--ace_step_1.5_ComfyUI_files/split_files/diffusion_models/acestep_v1.5_xl_turbo_bf16.safetensors,dst=/opt/ComfyUI/models/diffusion_models/acestep_v1.5_xl_turbo_bf16.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--ace_step_1.5_ComfyUI_files/split_files/text_encoders/qwen_0.6b_ace15.safetensors,dst=/opt/ComfyUI/models/text_encoders/qwen_0.6b_ace15.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--ace_step_1.5_ComfyUI_files/split_files/text_encoders/qwen_4b_ace15.safetensors,dst=/opt/ComfyUI/models/text_encoders/qwen_4b_ace15.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--ace_step_1.5_ComfyUI_files/split_files/vae/ace_1.5_vae.safetensors,dst=/opt/ComfyUI/models/vae/ace_1.5_vae.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-ace-step-1-5-xl-turbo-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=ACE-Step/ACE-Step-1.5-XL-Turbo" ] } }, diff --git a/recipes/linux-nvidia-comfyui-flux-2-klein-4b.json b/recipes/linux-nvidia-comfyui-flux-2-klein-4b.json index 95e2ef1..fd56324 100644 --- a/recipes/linux-nvidia-comfyui-flux-2-klein-4b.json +++ b/recipes/linux-nvidia-comfyui-flux-2-klein-4b.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-flux-2-klein-4b", "name": "black-forest-labs/FLUX.2-klein-4B (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install black-forest-labs/FLUX.2-klein-4B with a shared, locally built ComfyUI media backend.", + "summary": "Install black-forest-labs/FLUX.2-klein-4B in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights retain their upstream licenses" @@ -33,53 +33,28 @@ "href": "https://huggingface.co/black-forest-labs/FLUX.2-klein-4b-nvfp4/tree/1db2b2f776c24b76f1122e5f69ab1949fc620068" } ], - "keywords": [ - "cuda", - "flux.2-klein-4b", - "fp8", - "nvfp4", - "nvidia", - "single-model" - ], - "capabilities": [ - "image-generation", - "image-editing" - ], + "keywords": ["cuda", "flux.2-klein-4b", "fp8", "nvfp4", "nvidia", "single-model"], + "capabilities": ["image-generation", "image-editing"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; subsequent recipes share one managed ComfyUI runtime.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with CUDA 13-compatible host driver required." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], - "memoryGb": 95, + "platforms": ["linux-x64", "linux-arm64"], + "memoryGb": 20, "diskGb": 26, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -88,9 +63,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-flux2-dev", @@ -99,9 +72,7 @@ "provider": "huggingface", "model": "Comfy-Org/flux2-dev", "revision": "ab9055628ea245000e610f2aa2c96f4746093546", - "include": [ - "split_files/vae/flux2-vae.safetensors" - ], + "include": ["split_files/vae/flux2-vae.safetensors"], "downloadSizeBytes": 336213556, "integrity": { "files": [ @@ -120,9 +91,7 @@ "provider": "huggingface", "model": "Comfy-Org/z_image_turbo", "revision": "08d04455279082882deaabc8d0d09fc914c071e1", - "include": [ - "split_files/text_encoders/qwen_3_4b_fp4_mixed.safetensors" - ], + "include": ["split_files/text_encoders/qwen_3_4b_fp4_mixed.safetensors"], "downloadSizeBytes": 3479416193, "integrity": { "files": [ @@ -141,9 +110,7 @@ "provider": "huggingface", "model": "black-forest-labs/FLUX.2-klein-4b-nvfp4", "revision": "1db2b2f776c24b76f1122e5f69ab1949fc620068", - "include": [ - "flux-2-klein-4b-nvfp4.safetensors" - ], + "include": ["flux-2-klein-4b-nvfp4.safetensors"], "downloadSizeBytes": 2460413488, "integrity": { "files": [ @@ -162,22 +129,14 @@ "role": "flux2-klein-4b-nvfp4", "model": "black-forest-labs/FLUX.2-klein-4B", "gatewayModel": "black-forest-labs/FLUX.2-klein-4B", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-flux-2-klein-4b", + "backendConfig": "comfyui-flux-2-klein-4b", "kind": "image", - "input": [ - "text", - "image" - ], - "output": [ - "image" - ], - "capabilities": [ - "image-generation", - "image-editing" - ], + "input": ["text", "image"], + "output": ["image"], + "capabilities": ["image-generation", "image-editing"], "settings": { - "memoryGb": 95, + "memoryGb": 20, "priority": 30, "keepWarm": false, "maxActiveRequests": 1, @@ -185,11 +144,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-flux-2-klein-4b", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -200,12 +159,22 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--flux2-dev/split_files/vae/flux2-vae.safetensors,dst=/opt/ComfyUI/models/vae/flux2-vae.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--z_image_turbo/split_files/text_encoders/qwen_3_4b_fp4_mixed.safetensors,dst=/opt/ComfyUI/models/text_encoders/qwen_3_4b_fp4_mixed.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/black-forest-labs--FLUX.2-klein-4b-nvfp4/flux-2-klein-4b-nvfp4.safetensors,dst=/opt/ComfyUI/models/diffusion_models/flux-2-klein-4b-nvfp4.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-flux-2-klein-4b-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=black-forest-labs/FLUX.2-klein-4B" ] } }, diff --git a/recipes/linux-nvidia-comfyui-ideogram-4.json b/recipes/linux-nvidia-comfyui-ideogram-4.json index 26f8afd..9d15514 100644 --- a/recipes/linux-nvidia-comfyui-ideogram-4.json +++ b/recipes/linux-nvidia-comfyui-ideogram-4.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-ideogram-4", "name": "Comfy-Org/Ideogram-4 (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install Comfy-Org/Ideogram-4 with a shared, locally built ComfyUI media backend.", + "summary": "Install Comfy-Org/Ideogram-4 in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights retain their upstream licenses" @@ -33,55 +33,28 @@ "href": "https://huggingface.co/Comfy-Org/flux2-dev/tree/ab9055628ea245000e610f2aa2c96f4746093546" } ], - "keywords": [ - "cuda", - "fp8", - "ideogram-4", - "nvfp4", - "nvidia", - "single-model" - ], - "capabilities": [ - "image-generation", - "typography", - "layout-control", - "2k-output" - ], + "keywords": ["cuda", "fp8", "ideogram-4", "nvfp4", "nvidia", "single-model"], + "capabilities": ["image-generation", "typography", "layout-control", "2k-output"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; subsequent recipes share one managed ComfyUI runtime.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with CUDA 13-compatible host driver required." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], - "memoryGb": 95, + "platforms": ["linux-x64", "linux-arm64"], + "memoryGb": 64, "diskGb": 48, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -90,9 +63,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-ideogram-4", @@ -128,9 +99,7 @@ "provider": "huggingface", "model": "Comfy-Org/Qwen3-VL", "revision": "5529a3c630b649351fb72d8c251577b5962371d8", - "include": [ - "text_encoders/qwen3vl_8b_fp8_scaled.safetensors" - ], + "include": ["text_encoders/qwen3vl_8b_fp8_scaled.safetensors"], "downloadSizeBytes": 10588637512, "integrity": { "files": [ @@ -149,9 +118,7 @@ "provider": "huggingface", "model": "Comfy-Org/flux2-dev", "revision": "ab9055628ea245000e610f2aa2c96f4746093546", - "include": [ - "split_files/vae/flux2-vae.safetensors" - ], + "include": ["split_files/vae/flux2-vae.safetensors"], "downloadSizeBytes": 336213556, "integrity": { "files": [ @@ -170,23 +137,14 @@ "role": "ideogram-4", "model": "Comfy-Org/Ideogram-4", "gatewayModel": "Comfy-Org/Ideogram-4", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-ideogram-4", + "backendConfig": "comfyui-ideogram-4", "kind": "image", - "input": [ - "text" - ], - "output": [ - "image" - ], - "capabilities": [ - "image-generation", - "typography", - "layout-control", - "2k-output" - ], + "input": ["text"], + "output": ["image"], + "capabilities": ["image-generation", "typography", "layout-control", "2k-output"], "settings": { - "memoryGb": 95, + "memoryGb": 64, "priority": 30, "keepWarm": false, "maxActiveRequests": 1, @@ -194,11 +152,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-ideogram-4", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -209,12 +167,24 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Ideogram-4/diffusion_models/ideogram4_int8_convrot.safetensors,dst=/opt/ComfyUI/models/diffusion_models/ideogram4_int8_convrot.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Ideogram-4/diffusion_models/ideogram4_unconditional_int8_convrot.safetensors,dst=/opt/ComfyUI/models/diffusion_models/ideogram4_unconditional_int8_convrot.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen3-VL/text_encoders/qwen3vl_8b_fp8_scaled.safetensors,dst=/opt/ComfyUI/models/text_encoders/qwen3vl_8b_fp8_scaled.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--flux2-dev/split_files/vae/flux2-vae.safetensors,dst=/opt/ComfyUI/models/vae/flux2-vae.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-ideogram-4-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=Comfy-Org/Ideogram-4" ] } }, diff --git a/recipes/linux-nvidia-comfyui-krea-2-turbo.json b/recipes/linux-nvidia-comfyui-krea-2-turbo.json index c01359c..4c1e131 100644 --- a/recipes/linux-nvidia-comfyui-krea-2-turbo.json +++ b/recipes/linux-nvidia-comfyui-krea-2-turbo.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-krea-2-turbo", "name": "Comfy-Org/Krea-2-Turbo (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install Comfy-Org/Krea-2-Turbo with a shared, locally built ComfyUI media backend.", + "summary": "Install Comfy-Org/Krea-2-Turbo in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights retain their upstream licenses" @@ -25,53 +25,28 @@ "href": "https://huggingface.co/Comfy-Org/Qwen-Image_ComfyUI/tree/7beb7b647f04469fbe64ba8adc2bb0d7e5e9f73f" } ], - "keywords": [ - "cuda", - "fp8", - "krea-2-turbo", - "nvfp4", - "nvidia", - "single-model" - ], - "capabilities": [ - "image-generation", - "fast-generation" - ], + "keywords": ["cuda", "fp8", "krea-2-turbo", "nvfp4", "nvidia", "single-model"], + "capabilities": ["image-generation", "fast-generation"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; subsequent recipes share one managed ComfyUI runtime.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with CUDA 13-compatible host driver required." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], - "memoryGb": 95, + "platforms": ["linux-x64", "linux-arm64"], + "memoryGb": 40, "diskGb": 38, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -80,9 +55,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-krea-2", @@ -118,9 +91,7 @@ "provider": "huggingface", "model": "Comfy-Org/Qwen-Image_ComfyUI", "revision": "7beb7b647f04469fbe64ba8adc2bb0d7e5e9f73f", - "include": [ - "split_files/vae/qwen_image_vae.safetensors" - ], + "include": ["split_files/vae/qwen_image_vae.safetensors"], "downloadSizeBytes": 253806246, "integrity": { "files": [ @@ -139,21 +110,14 @@ "role": "krea-2-turbo", "model": "Comfy-Org/Krea-2-Turbo", "gatewayModel": "Comfy-Org/Krea-2-Turbo", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-krea-2-turbo", + "backendConfig": "comfyui-krea-2-turbo", "kind": "image", - "input": [ - "text" - ], - "output": [ - "image" - ], - "capabilities": [ - "image-generation", - "fast-generation" - ], + "input": ["text"], + "output": ["image"], + "capabilities": ["image-generation", "fast-generation"], "settings": { - "memoryGb": 95, + "memoryGb": 40, "priority": 30, "keepWarm": false, "maxActiveRequests": 1, @@ -161,11 +125,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-krea-2-turbo", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -176,12 +140,22 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Krea-2/diffusion_models/krea2_turbo_int8_convrot.safetensors,dst=/opt/ComfyUI/models/diffusion_models/krea2_turbo_int8_convrot.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Krea-2/text_encoders/qwen3vl_4b_fp8_scaled.safetensors,dst=/opt/ComfyUI/models/text_encoders/qwen3vl_4b_fp8_scaled.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen-Image_ComfyUI/split_files/vae/qwen_image_vae.safetensors,dst=/opt/ComfyUI/models/vae/qwen_image_vae.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-krea-2-turbo-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=Comfy-Org/Krea-2-Turbo" ] } }, diff --git a/recipes/linux-nvidia-comfyui-ltx-2-5.json b/recipes/linux-nvidia-comfyui-ltx-2-5.json index 8a9df73..c37c9c3 100644 --- a/recipes/linux-nvidia-comfyui-ltx-2-5.json +++ b/recipes/linux-nvidia-comfyui-ltx-2-5.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-ltx-2-5", "name": "Lightricks/LTX-2.5 (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install Lightricks/LTX-2.5 with a shared, locally built ComfyUI media backend.", + "summary": "Install Lightricks/LTX-2.5 in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights retain their upstream licenses" @@ -25,54 +25,28 @@ "href": "https://huggingface.co/Lightricks/LTX-2.5/tree/5e6e71018ee1756ed329b697a7b4aedc934dfce9" } ], - "keywords": [ - "cuda", - "fp8", - "ltx-2.5", - "nvfp4", - "nvidia", - "single-model" - ], - "capabilities": [ - "video-generation", - "text-to-video", - "image-to-video" - ], + "keywords": ["cuda", "fp8", "ltx-2.5", "nvfp4", "nvidia", "single-model"], + "capabilities": ["video-generation", "text-to-video", "image-to-video"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; subsequent recipes share one managed ComfyUI runtime.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with CUDA 13-compatible host driver required." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], - "memoryGb": 95, + "platforms": ["linux-x64", "linux-arm64"], + "memoryGb": 70, "diskGb": 57, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -81,9 +55,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-ltx-2.5", @@ -137,24 +109,14 @@ "role": "ltx-2.5", "model": "Lightricks/LTX-2.5", "gatewayModel": "Lightricks/LTX-2.5", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-ltx-2-5", + "backendConfig": "comfyui-ltx-2-5", "kind": "video", - "input": [ - "text", - "image" - ], - "output": [ - "video", - "audio" - ], - "capabilities": [ - "video-generation", - "text-to-video", - "image-to-video" - ], + "input": ["text", "image"], + "output": ["video", "audio"], + "capabilities": ["video-generation", "text-to-video", "image-to-video"], "settings": { - "memoryGb": 95, + "memoryGb": 70, "priority": 20, "keepWarm": false, "maxActiveRequests": 1, @@ -162,11 +124,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-ltx-2-5", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -177,12 +139,26 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Lightricks--LTX-2.5/diffusion_models/ltx-2.5-22b-distilled-transformer-comfy-int8-convrot.safetensors,dst=/opt/ComfyUI/models/diffusion_models/ltx-2.5-22b-distilled-transformer-comfy-int8-convrot.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Lightricks--LTX-2.5/latent_upscale_models/ltx-2.5-latent-spatial-upscaler-x2-bf16-1.0.safetensors,dst=/opt/ComfyUI/models/latent_upscale_models/ltx-2.5-latent-spatial-upscaler-x2-bf16-1.0.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Lightricks--LTX-2.5/text_encoders/gemma4-12b-with-proj-ltx-2.5-comfy-int8-convrot.safetensors,dst=/opt/ComfyUI/models/text_encoders/gemma4-12b-with-proj-ltx-2.5-comfy-int8-convrot.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Lightricks--LTX-2.5/vae/ltx-2.5-audio-vae-bf16.safetensors,dst=/opt/ComfyUI/models/vae/ltx-2.5-audio-vae-bf16.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Lightricks--LTX-2.5/vae/ltx-2.5-video-vae-bf16.safetensors,dst=/opt/ComfyUI/models/vae/ltx-2.5-video-vae-bf16.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-ltx-2-5-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=Lightricks/LTX-2.5" ] } }, diff --git a/recipes/linux-nvidia-comfyui-minimax-h3-turbo.json b/recipes/linux-nvidia-comfyui-minimax-h3-turbo.json index 25606ab..35588b8 100644 --- a/recipes/linux-nvidia-comfyui-minimax-h3-turbo.json +++ b/recipes/linux-nvidia-comfyui-minimax-h3-turbo.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-minimax-h3-turbo", "name": "MiniMaxAI/MiniMax-H3-Turbo (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install MiniMaxAI/MiniMax-H3-Turbo with a shared, locally built ComfyUI media backend.", + "summary": "Install MiniMaxAI/MiniMax-H3-Turbo in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights retain their upstream licenses" @@ -25,53 +25,28 @@ "href": "https://huggingface.co/lightx2v/Minimax-h3-Turbo/tree/3ec17a324ced54151364f24f8b5fb6bf7e26414f" } ], - "keywords": [ - "cuda", - "fp8", - "minimax-h3-turbo", - "nvfp4", - "nvidia", - "single-model" - ], - "capabilities": [ - "video-generation", - "text-to-video" - ], + "keywords": ["cuda", "fp8", "minimax-h3-turbo", "nvfp4", "nvidia", "single-model"], + "capabilities": ["video-generation", "text-to-video"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; subsequent recipes share one managed ComfyUI runtime.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with CUDA 13-compatible host driver required." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], + "platforms": ["linux-x64", "linux-arm64"], "memoryGb": 95, "diskGb": 61, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -80,9 +55,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-minimax-h3", @@ -130,9 +103,7 @@ "provider": "huggingface", "model": "lightx2v/Minimax-h3-Turbo", "revision": "3ec17a324ced54151364f24f8b5fb6bf7e26414f", - "include": [ - "minimax_h3_fl2v_turbo_8step_v1.0_comfyui_bf16.safetensors" - ], + "include": ["minimax_h3_fl2v_turbo_8step_v1.0_comfyui_bf16.safetensors"], "downloadSizeBytes": 1956193000, "integrity": { "files": [ @@ -151,20 +122,12 @@ "role": "minimax-h3-turbo", "model": "MiniMaxAI/MiniMax-H3-Turbo", "gatewayModel": "MiniMaxAI/MiniMax-H3-Turbo", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-minimax-h3-turbo", + "backendConfig": "comfyui-minimax-h3-turbo", "kind": "video", - "input": [ - "text" - ], - "output": [ - "video", - "audio" - ], - "capabilities": [ - "video-generation", - "text-to-video" - ], + "input": ["text"], + "output": ["video", "audio"], + "capabilities": ["video-generation", "text-to-video"], "settings": { "memoryGb": 95, "priority": 20, @@ -174,11 +137,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-minimax-h3-turbo", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -189,12 +152,26 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--MiniMax-H3/diffusion_models/minimax_h3_fl2va_pruned_int8_convrot.safetensors,dst=/opt/ComfyUI/models/diffusion_models/minimax_h3_fl2va_pruned_int8_convrot.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--MiniMax-H3/text_encoders/qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors,dst=/opt/ComfyUI/models/text_encoders/qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--MiniMax-H3/vae/minimax_h3_audio_vae_fp32.safetensors,dst=/opt/ComfyUI/models/vae/minimax_h3_audio_vae_fp32.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--MiniMax-H3/vae/minimax_h3_video_vae_fp16.safetensors,dst=/opt/ComfyUI/models/vae/minimax_h3_video_vae_fp16.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/lightx2v--Minimax-h3-Turbo/minimax_h3_fl2v_turbo_8step_v1.0_comfyui_bf16.safetensors,dst=/opt/ComfyUI/models/loras/minimax_h3_fl2v_turbo_8step_v1.0_comfyui_bf16.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-minimax-h3-turbo-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=MiniMaxAI/MiniMax-H3-Turbo" ] } }, diff --git a/recipes/linux-nvidia-comfyui-minimax-h3.json b/recipes/linux-nvidia-comfyui-minimax-h3.json index 88d7c90..b7931ba 100644 --- a/recipes/linux-nvidia-comfyui-minimax-h3.json +++ b/recipes/linux-nvidia-comfyui-minimax-h3.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-minimax-h3", "name": "MiniMaxAI/MiniMax-H3 (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install MiniMaxAI/MiniMax-H3 with a shared, locally built ComfyUI media backend.", + "summary": "Install MiniMaxAI/MiniMax-H3 in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights retain their upstream licenses" @@ -25,54 +25,28 @@ "href": "https://huggingface.co/Comfy-Org/MiniMax-H3/tree/7e75982b97cd5a41d2dcfa1904ee88d0686d6fd1" } ], - "keywords": [ - "cuda", - "fp8", - "minimax-h3", - "nvfp4", - "nvidia", - "single-model" - ], - "capabilities": [ - "video-generation", - "text-to-video", - "image-to-video" - ], + "keywords": ["cuda", "fp8", "minimax-h3", "nvfp4", "nvidia", "single-model"], + "capabilities": ["video-generation", "text-to-video", "image-to-video"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; subsequent recipes share one managed ComfyUI runtime.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with CUDA 13-compatible host driver required." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], + "platforms": ["linux-x64", "linux-arm64"], "memoryGb": 95, "diskGb": 60, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -81,9 +55,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-minimax-h3", @@ -131,22 +103,12 @@ "role": "minimax-h3", "model": "MiniMaxAI/MiniMax-H3", "gatewayModel": "MiniMaxAI/MiniMax-H3", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-minimax-h3", + "backendConfig": "comfyui-minimax-h3", "kind": "video", - "input": [ - "text", - "image" - ], - "output": [ - "video", - "audio" - ], - "capabilities": [ - "video-generation", - "text-to-video", - "image-to-video" - ], + "input": ["text", "image"], + "output": ["video", "audio"], + "capabilities": ["video-generation", "text-to-video", "image-to-video"], "settings": { "memoryGb": 95, "priority": 20, @@ -156,11 +118,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-minimax-h3", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -171,12 +133,24 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--MiniMax-H3/diffusion_models/minimax_h3_fl2va_pruned_int8_convrot.safetensors,dst=/opt/ComfyUI/models/diffusion_models/minimax_h3_fl2va_pruned_int8_convrot.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--MiniMax-H3/text_encoders/qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors,dst=/opt/ComfyUI/models/text_encoders/qwen3vl_32b_minimax_h3_nvfp4_awq.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--MiniMax-H3/vae/minimax_h3_audio_vae_fp32.safetensors,dst=/opt/ComfyUI/models/vae/minimax_h3_audio_vae_fp32.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--MiniMax-H3/vae/minimax_h3_video_vae_fp16.safetensors,dst=/opt/ComfyUI/models/vae/minimax_h3_video_vae_fp16.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-minimax-h3-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=MiniMaxAI/MiniMax-H3" ] } }, diff --git a/recipes/linux-nvidia-comfyui-minimax-music3.json b/recipes/linux-nvidia-comfyui-minimax-music3.json index cab6449..d73f8ff 100644 --- a/recipes/linux-nvidia-comfyui-minimax-music3.json +++ b/recipes/linux-nvidia-comfyui-minimax-music3.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-minimax-music3", "name": "MiniMaxAI/MiniMax-Music3 (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install MiniMaxAI/MiniMax-Music3 with a shared, locally built ComfyUI media backend.", + "summary": "Install MiniMaxAI/MiniMax-Music3 in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights retain their upstream licenses" @@ -25,53 +25,28 @@ "href": "https://huggingface.co/Comfy-Org/MiniMax-Music-3/tree/6baad88896848433857c170ba4f05d2ea9d5f218" } ], - "keywords": [ - "cuda", - "fp8", - "minimax-music3", - "nvfp4", - "nvidia", - "single-model" - ], - "capabilities": [ - "audio-generation", - "music-generation" - ], + "keywords": ["cuda", "fp8", "minimax-music3", "nvfp4", "nvidia", "single-model"], + "capabilities": ["audio-generation", "music-generation"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; subsequent recipes share one managed ComfyUI runtime.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with CUDA 13-compatible host driver required." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], - "memoryGb": 95, + "platforms": ["linux-x64", "linux-arm64"], + "memoryGb": 35, "diskGb": 33, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -80,9 +55,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-minimax-music-3", @@ -124,21 +97,14 @@ "role": "minimax-music3", "model": "MiniMaxAI/MiniMax-Music3", "gatewayModel": "MiniMaxAI/MiniMax-Music3", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-minimax-music3", + "backendConfig": "comfyui-minimax-music3", "kind": "audio_generation", - "input": [ - "text" - ], - "output": [ - "audio" - ], - "capabilities": [ - "audio-generation", - "music-generation" - ], + "input": ["text"], + "output": ["audio"], + "capabilities": ["audio-generation", "music-generation"], "settings": { - "memoryGb": 95, + "memoryGb": 35, "priority": 20, "keepWarm": false, "maxActiveRequests": 1, @@ -146,11 +112,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-minimax-music3", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -161,12 +127,22 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--MiniMax-Music-3/diffusion_models/minimax_music3_dit_fp16.safetensors,dst=/opt/ComfyUI/models/diffusion_models/minimax_music3_dit_fp16.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--MiniMax-Music-3/text_encoders/minimax_music3_text_encoder_pruned_int8_convrot.safetensors,dst=/opt/ComfyUI/models/text_encoders/minimax_music3_text_encoder_pruned_int8_convrot.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--MiniMax-Music-3/vae/minimax_music3_dav.safetensors,dst=/opt/ComfyUI/models/vae/minimax_music3_dav.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-minimax-music3-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=MiniMaxAI/MiniMax-Music3" ] } }, diff --git a/recipes/linux-nvidia-comfyui-qwen-image-2-1.json b/recipes/linux-nvidia-comfyui-qwen-image-2-1.json index 732cd75..2966123 100644 --- a/recipes/linux-nvidia-comfyui-qwen-image-2-1.json +++ b/recipes/linux-nvidia-comfyui-qwen-image-2-1.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-qwen-image-2-1", "name": "Qwen/Qwen-Image-2.1 (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install Qwen/Qwen-Image-2.1, one DiT for generation and reference editing, with the shared ComfyUI media backend.", + "summary": "Install Qwen/Qwen-Image-2.1 in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights are under the Qwen Research license" @@ -33,55 +33,29 @@ "href": "https://github.com/Comfy-Org/workflow_templates/blob/main/templates/image_qwen_image_2_1_image_edit.json" } ], - "keywords": [ - "cuda", - "int8", - "nvidia", - "qwen-image-2.1", - "image-editing", - "transparency", - "single-model" - ], - "capabilities": [ - "image-generation", - "image-editing" - ], + "keywords": ["cuda", "int8", "nvidia", "qwen-image-2.1", "image-editing", "transparency", "single-model"], + "capabilities": ["image-generation", "image-editing"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; shares the one managed ComfyUI media runtime with the other image, video and music recipes.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with a CUDA 13-compatible host driver required.", "Weight format is int8 convrot for the DiT and text encoder, which the pinned engine runs with its own quantized kernels. No separate LoRA is needed: 2.1 samples at its own shift and cfg 1." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], - "memoryGb": 95, + "platforms": ["linux-x64", "linux-arm64"], + "memoryGb": 48, "diskGb": 26, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -90,9 +64,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-qwen-image-2.1", @@ -134,22 +106,14 @@ "role": "qwen-image-2.1-int8", "model": "Qwen/Qwen-Image-2.1", "gatewayModel": "Qwen/Qwen-Image-2.1", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-qwen-image-2-1", + "backendConfig": "comfyui-qwen-image-2-1", "kind": "image", - "input": [ - "text", - "image" - ], - "output": [ - "image" - ], - "capabilities": [ - "image-generation", - "image-editing" - ], + "input": ["text", "image"], + "output": ["image"], + "capabilities": ["image-generation", "image-editing"], "settings": { - "memoryGb": 95, + "memoryGb": 48, "priority": 30, "keepWarm": false, "maxActiveRequests": 1, @@ -157,11 +121,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-qwen-image-2-1", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -172,12 +136,22 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen-Image-2.1/diffusion_models/qwen_image_2.1_int8_convrot.safetensors,dst=/opt/ComfyUI/models/diffusion_models/qwen_image_2.1_int8_convrot.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen-Image-2.1/text_encoders/qwen3vl_8b_int8_convrot.safetensors,dst=/opt/ComfyUI/models/text_encoders/qwen3vl_8b_int8_convrot.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen-Image-2.1/vae/qwen_image_2.1_vae_bf16.safetensors,dst=/opt/ComfyUI/models/vae/qwen_image_2.1_vae_bf16.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-qwen-image-2-1-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=Qwen/Qwen-Image-2.1" ] } }, diff --git a/recipes/linux-nvidia-comfyui-qwen-image-2512-lightning.json b/recipes/linux-nvidia-comfyui-qwen-image-2512-lightning.json index f522757..efe90da 100644 --- a/recipes/linux-nvidia-comfyui-qwen-image-2512-lightning.json +++ b/recipes/linux-nvidia-comfyui-qwen-image-2512-lightning.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-qwen-image-2512-lightning", "name": "Qwen/Qwen-Image-2512-Lightning (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install Qwen/Qwen-Image-2512-Lightning with a shared, locally built ComfyUI media backend.", + "summary": "Install Qwen/Qwen-Image-2512-Lightning in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights retain their upstream licenses" @@ -25,52 +25,28 @@ "href": "https://huggingface.co/lightx2v/Qwen-Image-2512-Lightning/tree/a52649c9d0f6e1a248bff13f0df33bb8a2abdb52" } ], - "keywords": [ - "cuda", - "fp8", - "nvfp4", - "nvidia", - "qwen-image-2512-lightning", - "single-model" - ], - "capabilities": [ - "image-generation" - ], + "keywords": ["cuda", "fp8", "nvfp4", "nvidia", "qwen-image-2512-lightning", "single-model"], + "capabilities": ["image-generation"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; subsequent recipes share one managed ComfyUI runtime.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with CUDA 13-compatible host driver required." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], - "memoryGb": 95, + "platforms": ["linux-x64", "linux-arm64"], + "memoryGb": 48, "diskGb": 46, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -79,9 +55,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-qwen-image_comfyui", @@ -123,9 +97,7 @@ "provider": "huggingface", "model": "lightx2v/Qwen-Image-2512-Lightning", "revision": "a52649c9d0f6e1a248bff13f0df33bb8a2abdb52", - "include": [ - "Qwen-Image-2512-Lightning-4steps-V1.0-bf16.safetensors" - ], + "include": ["Qwen-Image-2512-Lightning-4steps-V1.0-bf16.safetensors"], "downloadSizeBytes": 849608296, "integrity": { "files": [ @@ -144,20 +116,14 @@ "role": "qwen-image-2512-lightning", "model": "Qwen/Qwen-Image-2512-Lightning", "gatewayModel": "Qwen/Qwen-Image-2512-Lightning", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-qwen-image-2512-lightning", + "backendConfig": "comfyui-qwen-image-2512-lightning", "kind": "image", - "input": [ - "text" - ], - "output": [ - "image" - ], - "capabilities": [ - "image-generation" - ], + "input": ["text"], + "output": ["image"], + "capabilities": ["image-generation"], "settings": { - "memoryGb": 95, + "memoryGb": 48, "priority": 30, "keepWarm": false, "maxActiveRequests": 1, @@ -165,11 +131,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-qwen-image-2512-lightning", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -180,12 +146,24 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen-Image_ComfyUI/split_files/diffusion_models/qwen_image_2512_fp8_e4m3fn.safetensors,dst=/opt/ComfyUI/models/diffusion_models/qwen_image_2512_fp8_e4m3fn.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen-Image_ComfyUI/split_files/text_encoders/qwen_2.5_vl_7b_nvfp4.safetensors,dst=/opt/ComfyUI/models/text_encoders/qwen_2.5_vl_7b_nvfp4.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen-Image_ComfyUI/split_files/vae/qwen_image_vae.safetensors,dst=/opt/ComfyUI/models/vae/qwen_image_vae.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/lightx2v--Qwen-Image-2512-Lightning/Qwen-Image-2512-Lightning-4steps-V1.0-bf16.safetensors,dst=/opt/ComfyUI/models/loras/Qwen-Image-2512-Lightning-4steps-V1.0-bf16.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-qwen-image-2512-lightning-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=Qwen/Qwen-Image-2512-Lightning" ] } }, diff --git a/recipes/linux-nvidia-comfyui-qwen-image-2512.json b/recipes/linux-nvidia-comfyui-qwen-image-2512.json index 8aac3bb..1ed6c25 100644 --- a/recipes/linux-nvidia-comfyui-qwen-image-2512.json +++ b/recipes/linux-nvidia-comfyui-qwen-image-2512.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-qwen-image-2512", "name": "Qwen/Qwen-Image-2512 (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install Qwen/Qwen-Image-2512 with a shared, locally built ComfyUI media backend.", + "summary": "Install Qwen/Qwen-Image-2512 in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights retain their upstream licenses" @@ -25,52 +25,28 @@ "href": "https://huggingface.co/Comfy-Org/Qwen-Image_ComfyUI/tree/7beb7b647f04469fbe64ba8adc2bb0d7e5e9f73f" } ], - "keywords": [ - "cuda", - "fp8", - "nvfp4", - "nvidia", - "qwen-image-2512", - "single-model" - ], - "capabilities": [ - "image-generation" - ], + "keywords": ["cuda", "fp8", "nvfp4", "nvidia", "qwen-image-2512", "single-model"], + "capabilities": ["image-generation"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; subsequent recipes share one managed ComfyUI runtime.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with CUDA 13-compatible host driver required." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], - "memoryGb": 95, + "platforms": ["linux-x64", "linux-arm64"], + "memoryGb": 48, "diskGb": 45, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -79,9 +55,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-qwen-image_comfyui", @@ -123,20 +97,14 @@ "role": "qwen-image-2512-fp8", "model": "Qwen/Qwen-Image-2512", "gatewayModel": "Qwen/Qwen-Image-2512", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-qwen-image-2512", + "backendConfig": "comfyui-qwen-image-2512", "kind": "image", - "input": [ - "text" - ], - "output": [ - "image" - ], - "capabilities": [ - "image-generation" - ], + "input": ["text"], + "output": ["image"], + "capabilities": ["image-generation"], "settings": { - "memoryGb": 95, + "memoryGb": 48, "priority": 30, "keepWarm": false, "maxActiveRequests": 1, @@ -144,11 +112,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-qwen-image-2512", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -159,12 +127,22 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen-Image_ComfyUI/split_files/diffusion_models/qwen_image_2512_fp8_e4m3fn.safetensors,dst=/opt/ComfyUI/models/diffusion_models/qwen_image_2512_fp8_e4m3fn.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen-Image_ComfyUI/split_files/text_encoders/qwen_2.5_vl_7b_nvfp4.safetensors,dst=/opt/ComfyUI/models/text_encoders/qwen_2.5_vl_7b_nvfp4.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen-Image_ComfyUI/split_files/vae/qwen_image_vae.safetensors,dst=/opt/ComfyUI/models/vae/qwen_image_vae.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-qwen-image-2512-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=Qwen/Qwen-Image-2512" ] } }, diff --git a/recipes/linux-nvidia-comfyui-qwen-image-edit-2511.json b/recipes/linux-nvidia-comfyui-qwen-image-edit-2511.json index 3689aa5..615ecfa 100644 --- a/recipes/linux-nvidia-comfyui-qwen-image-edit-2511.json +++ b/recipes/linux-nvidia-comfyui-qwen-image-edit-2511.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-qwen-image-edit-2511", "name": "Qwen/Qwen-Image-Edit-2511 (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install Qwen/Qwen-Image-Edit-2511 with a shared, locally built ComfyUI media backend.", + "summary": "Install Qwen/Qwen-Image-Edit-2511 in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights retain their upstream licenses" @@ -29,53 +29,28 @@ "href": "https://huggingface.co/Comfy-Org/Qwen-Image_ComfyUI/tree/7beb7b647f04469fbe64ba8adc2bb0d7e5e9f73f" } ], - "keywords": [ - "cuda", - "fp8", - "nvfp4", - "nvidia", - "qwen-image-edit-2511", - "single-model" - ], - "capabilities": [ - "image-generation", - "image-editing" - ], + "keywords": ["cuda", "fp8", "nvfp4", "nvidia", "qwen-image-edit-2511", "single-model"], + "capabilities": ["image-generation", "image-editing"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; subsequent recipes share one managed ComfyUI runtime.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with CUDA 13-compatible host driver required." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], - "memoryGb": 95, + "platforms": ["linux-x64", "linux-arm64"], + "memoryGb": 48, "diskGb": 45, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -84,9 +59,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-qwen-image-edit_comfyui", @@ -95,9 +68,7 @@ "provider": "huggingface", "model": "Comfy-Org/Qwen-Image-Edit_ComfyUI", "revision": "7d41107b653d3039be20972fb82398b01b3213eb", - "include": [ - "split_files/diffusion_models/qwen_image_edit_2511_fp8mixed.safetensors" - ], + "include": ["split_files/diffusion_models/qwen_image_edit_2511_fp8mixed.safetensors"], "downloadSizeBytes": 20533762817, "integrity": { "files": [ @@ -143,22 +114,14 @@ "role": "qwen-image-edit-2511-fp8mixed", "model": "Qwen/Qwen-Image-Edit-2511", "gatewayModel": "Qwen/Qwen-Image-Edit-2511", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-qwen-image-edit-2511", + "backendConfig": "comfyui-qwen-image-edit-2511", "kind": "image", - "input": [ - "text", - "image" - ], - "output": [ - "image" - ], - "capabilities": [ - "image-generation", - "image-editing" - ], + "input": ["text", "image"], + "output": ["image"], + "capabilities": ["image-generation", "image-editing"], "settings": { - "memoryGb": 95, + "memoryGb": 48, "priority": 30, "keepWarm": false, "maxActiveRequests": 1, @@ -166,11 +129,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-qwen-image-edit-2511", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -181,12 +144,22 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen-Image-Edit_ComfyUI/split_files/diffusion_models/qwen_image_edit_2511_fp8mixed.safetensors,dst=/opt/ComfyUI/models/diffusion_models/qwen_image_edit_2511_fp8mixed.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen-Image_ComfyUI/split_files/text_encoders/qwen_2.5_vl_7b_nvfp4.safetensors,dst=/opt/ComfyUI/models/text_encoders/qwen_2.5_vl_7b_nvfp4.safetensors,readonly", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--Qwen-Image_ComfyUI/split_files/vae/qwen_image_vae.safetensors,dst=/opt/ComfyUI/models/vae/qwen_image_vae.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-qwen-image-edit-2511-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=Qwen/Qwen-Image-Edit-2511" ] } }, diff --git a/recipes/linux-nvidia-comfyui-yue2-3b.json b/recipes/linux-nvidia-comfyui-yue2-3b.json index 0668474..100229a 100644 --- a/recipes/linux-nvidia-comfyui-yue2-3b.json +++ b/recipes/linux-nvidia-comfyui-yue2-3b.json @@ -5,7 +5,7 @@ "id": "linux-nvidia-comfyui-yue2-3b", "name": "Comfy-Org/YuE2-3B (ComfyUI, NVIDIA)", "version": 1, - "summary": "Install Comfy-Org/YuE2-3B with a shared, locally built ComfyUI media backend.", + "summary": "Install Comfy-Org/YuE2-3B in an independent, single-model ComfyUI runtime.", "license": { "id": "MIT", "name": "Recipe metadata only; model weights retain their upstream licenses" @@ -21,54 +21,28 @@ "href": "https://huggingface.co/Comfy-Org/YuE2/tree/8e6fcf0f23252ed188b634bd50d44f4b01fba890" } ], - "keywords": [ - "cuda", - "fp8", - "nvfp4", - "nvidia", - "single-model", - "yue2-3b" - ], - "capabilities": [ - "audio-generation", - "music-generation", - "symbolic-score" - ], + "keywords": ["cuda", "fp8", "nvfp4", "nvidia", "single-model", "yue2-3b"], + "capabilities": ["audio-generation", "music-generation", "symbolic-score"], "hardware": { "family": "Linux NVIDIA CUDA", "testedMachine": "DGX Spark / GB10 class NVIDIA system", "notes": [ - "Standalone single-GPU serving; subsequent recipes share one managed ComfyUI runtime.", + "Independent single-model serving with per-model admission and unloading.", "NVIDIA Container Toolkit with CUDA 13-compatible host driver required." ] }, "requirements": { - "platforms": [ - "linux-x64", - "linux-arm64" - ], - "memoryGb": 95, + "platforms": ["linux-x64", "linux-arm64"], + "memoryGb": 16, "diskGb": 24, - "commands": [ - "docker", - "python3" - ], - "accelerators": [ - "cuda", - "nvidia-gpu" - ] + "commands": ["docker", "python3"], + "accelerators": ["cuda", "nvidia-gpu"] }, "backend": { "id": "comfyui-media", "name": "ComfyUI media", "type": "openai-compatible-server", - "features": [ - "cuda", - "audio-generation", - "video-generation", - "image-generation", - "image-editing" - ] + "features": ["cuda", "audio-generation", "video-generation", "image-generation", "image-editing"] }, "setup": { "steps": [ @@ -77,9 +51,7 @@ "title": "Check Docker CLI", "action": "check-command", "command": "docker", - "args": [ - "--version" - ] + "args": ["--version"] }, { "id": "download-1-yue2", @@ -88,9 +60,7 @@ "provider": "huggingface", "model": "Comfy-Org/YuE2", "revision": "8e6fcf0f23252ed188b634bd50d44f4b01fba890", - "include": [ - "checkpoints/yue2_3b_int8_convrot.safetensors" - ], + "include": ["checkpoints/yue2_3b_int8_convrot.safetensors"], "downloadSizeBytes": 3960938800, "integrity": { "files": [ @@ -109,21 +79,14 @@ "role": "yue2-3b", "model": "Comfy-Org/YuE2-3B", "gatewayModel": "Comfy-Org/YuE2-3B", - "runtime": "comfyui-media", - "backendConfig": "comfyui-media", + "runtime": "comfyui-yue2-3b", + "backendConfig": "comfyui-yue2-3b", "kind": "audio_generation", - "input": [ - "text" - ], - "output": [ - "audio" - ], - "capabilities": [ - "audio-generation", - "music-generation" - ], + "input": ["text"], + "output": ["audio"], + "capabilities": ["audio-generation", "music-generation"], "settings": { - "memoryGb": 95, + "memoryGb": 16, "priority": 20, "keepWarm": false, "maxActiveRequests": 1, @@ -131,11 +94,11 @@ "runtime": { "adapter": "docker", "management": "managed", - "containerName": "lloom-comfyui-media", + "containerName": "lloom-comfyui-yue2-3b", "healthUrl": "http://127.0.0.1:${port}/health", "bootstrap": { "adapter": "docker", - "image": "lloom/comfyui-media:source-b812f8cb1852f0360dd98b439f2e2ffc3039ae7d3d98482c07508a40319d9f34", + "image": "lloom/comfyui-media:source-381d99b95f2c4afb4e5c07ff12d2fd1c541cde69dc16f00db48506b9e27a7b83", "pull": false, "createArgs": [ "--restart", @@ -146,12 +109,18 @@ "host", "-p", "127.0.0.1:${port}:8000", + "--mount", + "type=bind,src=${modelRoot}/Comfy-Org--YuE2/checkpoints/yue2_3b_int8_convrot.safetensors,dst=/opt/ComfyUI/models/checkpoints/yue2_3b_int8_convrot.safetensors,readonly", "-v", - "${modelRoot}:/opt/lloom-models:ro", - "-v", - "lloom-comfyui-media-data:/data", + "lloom-comfyui-yue2-3b-data:/data", + "-e", + "LLOOM_MEDIA_DATA_ROOT=/data", + "-e", + "LLOOM_MODELS_ROOT=/opt/ComfyUI/models", + "-e", + "LLOOM_COMFY_CACHE_MODE=classic", "-e", - "LLOOM_MEDIA_DATA_ROOT=/data" + "LLOOM_MEDIA_MODEL=Comfy-Org/YuE2-3B" ] } }, diff --git a/src/audio-providers.mjs b/src/audio-providers.mjs new file mode 100644 index 0000000..ffbc51f --- /dev/null +++ b/src/audio-providers.mjs @@ -0,0 +1,455 @@ +/** + * OpenRouter audio-generation (Lyria) adapter. + * + * `POST /v1/audio/generations` normally proxies an OpenAI-compatible media + * backend that streams finished audio bytes. A backend configured with + * `audioProvider: "openrouter"` instead runs the provider's SSE chat-style + * audio generation inside LLooM and returns the assembled audio artifact. + * + * Official contract: + * POST https://openrouter.ai/api/v1/chat/completions + * { model, messages: [{ role: 'user', content }], modalities: ['text', 'audio'], + * audio: { format: 'wav' | 'mp3' }, stream: true } + * + * Audio arrives as SSE `choices[0].delta.audio.data` base64 chunks, terminated + * by `data: [DONE]` (a successful `stop` followed by clean EOF is also accepted). The + * provider API origin is fixed, redirects are rejected, credentials come from + * the backend only, and unknown caller fields are refused rather than + * forwarded. Nothing is returned until a complete, validated stream has been + * aggregated, and billable generations are never retried automatically. + */ + +import { spawn } from 'node:child_process'; +import { applyOpenRouterProviderPolicy } from './protocol/openrouter-provider.mjs'; + +const OPENROUTER_ORIGIN = 'https://openrouter.ai'; +const CHAT_COMPLETIONS_URL = `${OPENROUTER_ORIGIN}/api/v1/chat/completions`; +const AUDIO_FORMATS = new Set(['wav', 'mp3']); +// Caller fields that map to the provider contract. Everything else is refused +// so a caller cannot believe local generation controls (steps, cfg, seeds) +// were honored when the provider never received them. +const ALLOWED_FIELDS = new Set([ + 'model', + 'prompt', + 'instructions', + 'lyrics', + 'input', + 'duration', + 'format', + 'response_format', + 'image', + 'reference_image' +]); +const MAX_TEXT_CHARS = 8000; +const MAX_IMAGE_CHARS = 8 * 1024 * 1024; +const MAX_DURATION_SECONDS = 600; +const DEFAULT_TIMEOUT_MS = 600000; +const DEFAULT_MAX_BYTES = 100 * 1024 * 1024; +const DEFAULT_MAX_EVENT_BYTES = 16 * 1024 * 1024; +const BASE64_CHUNK = /^[A-Za-z0-9+/]+={0,2}$/; +const IMAGE_DATA_URI = /^data:image\/(png|jpeg);base64,[A-Za-z0-9+/]+={0,2}$/; + +function boundedText(value, label) { + if (value === undefined || value === null) return null; + if (typeof value !== 'string') throw invalid(`Audio generation ${label} must be a string`); + const text = value.trim(); + if (!text) return null; + if (text.length > MAX_TEXT_CHARS) throw invalid(`Audio generation ${label} is too long`); + return text; +} + +/** Drain the provider body without echoing it into the error message. */ +async function closeBody(body) { + try { + await boundedCleanup(() => body?.cancel()); + } catch { + // The stream may already be errored or released; nothing to cancel. + } +} + +export async function generateProviderAudio({ + backend, + body, + signal, + fetchFn = fetch, + dispatcher, + timeoutMs = DEFAULT_TIMEOUT_MS, + maxBytes = DEFAULT_MAX_BYTES, + maxEventBytes = DEFAULT_MAX_EVENT_BYTES +}) { + if (backend?.audioProvider !== 'openrouter') throw new Error('Unsupported audio provider'); + const key = backend.apiKeyEnv ? process.env[backend.apiKeyEnv] : backend.apiKey; + if (!key) throw new Error('Audio backend credential is not configured'); + if (!Number.isFinite(timeoutMs) || timeoutMs <= 0) throw new Error('Invalid audio provider timeout'); + if (!Number.isFinite(maxBytes) || maxBytes <= 0) throw new Error('Invalid audio provider size limit'); + if (!Number.isFinite(maxEventBytes) || maxEventBytes <= 0) throw new Error('Invalid audio provider event limit'); + if (!body || typeof body !== 'object' || Array.isArray(body)) + throw invalid('Audio generation request body is required'); + + for (const field of Object.keys(body)) { + if (!ALLOWED_FIELDS.has(field)) throw invalid(`Audio provider does not support field "${field}"`); + } + const model = typeof body.model === 'string' ? body.model.trim() : ''; + if (!model || model.length > 200 || /\s/.test(model)) throw invalid('A valid audio generation model is required'); + if (body.format && body.response_format && body.format !== body.response_format) + throw invalid('Conflicting audio formats'); + const format = body.response_format ?? body.format ?? 'wav'; + if (typeof format !== 'string' || !AUDIO_FORMATS.has(format)) throw invalid('Audio format must be wav or mp3'); + + const prompt = boundedText(body.prompt, 'prompt'); + const instructions = boundedText(body.instructions, 'instructions'); + if (!prompt && !instructions) throw invalid('Audio generation requires a prompt or instructions'); + const lyrics = boundedText(body.lyrics, 'lyrics'); + const input = boundedText(body.input, 'input'); + + let durationLine = null; + if (body.duration !== undefined && body.duration !== null) { + if (!Number.isFinite(body.duration) || body.duration <= 0 || body.duration > MAX_DURATION_SECONDS) + throw invalid(`Audio duration must be between 0 and ${MAX_DURATION_SECONDS} seconds`); + durationLine = `Duration: ${body.duration} seconds.`; + } + + const referenceImage = body.image ?? body.reference_image; + let image = null; + if (referenceImage !== undefined && referenceImage !== null && referenceImage !== '') { + if ( + typeof referenceImage !== 'string' || + referenceImage.length > MAX_IMAGE_CHARS || + !IMAGE_DATA_URI.test(referenceImage) + ) + throw invalid('Reference image must be an inline PNG or JPEG data URI'); + image = referenceImage; + } + + // Local music controls have no provider equivalent, so they are folded into + // the prompt text instead of being faked as provider parameters. + const text = [prompt, instructions, lyrics, input, durationLine].filter(Boolean).join('\n'); + const content = image + ? [ + { type: 'text', text }, + { type: 'image_url', image_url: { url: image } } + ] + : text; + + const payload = applyOpenRouterProviderPolicy( + { + model, + messages: [{ role: 'user', content }], + modalities: ['text', 'audio'], + audio: { format }, + stream: true + }, + { ...backend, baseUrl: OPENROUTER_ORIGIN } + ); + + const boundedSignal = AbortSignal.any([...(signal ? [signal] : []), AbortSignal.timeout(timeoutMs)]); + let response; + try { + response = await fetchFn(CHAT_COMPLETIONS_URL, { + method: 'POST', + signal: boundedSignal, + redirect: 'error', + ...(dispatcher ? { dispatcher } : {}), + headers: { + Authorization: `Bearer ${key}`, + 'Content-Type': 'application/json', + Accept: 'text/event-stream', + 'User-Agent': 'LLooM' + }, + body: JSON.stringify(payload) + }); + } catch { + if (boundedSignal.aborted) throw new Error('Audio generation timed out or was cancelled'); + throw new Error('Audio provider request failed'); + } + if (boundedSignal.aborted) { + await closeBody(response?.body); + throw new Error('Audio generation timed out or was cancelled'); + } + + if (!response.ok) { + await closeBody(response.body); + const error = new Error(`Audio provider openrouter returned HTTP ${response.status}`); + error.statusCode = response.status; + throw error; + } + const contentType = response.headers.get('content-type') ?? ''; + if (contentType && !contentType.toLowerCase().includes('text/event-stream')) { + await closeBody(response.body); + throw new Error('Audio provider returned a non-SSE response'); + } + + let audio; + try { + audio = await collectAudio(response.body, boundedSignal, maxBytes, maxEventBytes); + } catch (error) { + await closeBody(response.body); + throw error; + } + const actualFormat = identifyAudio(audio); + validateAudio(audio, actualFormat); + if (!actualFormat) throw new Error('Audio provider returned unrecognized audio'); + if (actualFormat !== format) audio = await convertAudio(audio, format, boundedSignal, maxBytes); + return new Response(audio, { + status: 200, + headers: { + 'content-type': format === 'mp3' ? 'audio/mpeg' : 'audio/wav', + 'content-length': String(audio.length), + 'x-lloom-provider': 'openrouter', + 'x-lloom-provider-origin': OPENROUTER_ORIGIN, + 'x-lloom-audio-format': format, + 'x-lloom-upstream-model': model + } + }); +} + +function invalid(message) { + return Object.assign(new Error(message), { statusCode: 400 }); +} + +function identifyAudio(audio) { + if (audio.length >= 44 && audio.toString('ascii', 0, 4) === 'RIFF' && audio.toString('ascii', 8, 12) === 'WAVE') + return 'wav'; + if ( + audio.length >= 10 && + (audio.toString('ascii', 0, 3) === 'ID3' || (audio[0] === 0xff && (audio[1] & 0xe0) === 0xe0)) + ) + return 'mp3'; + return null; +} + +async function collectAudio(body, signal, maxBytes, maxEventBytes) { + if (!body) throw new Error('Audio provider returned no audio'); + const chunks = []; + let decodedBytes = 0, + rawBytes = 0, + pending = '', + sawDone = false, + sawFinish = false; + const decoder = new TextDecoder('utf-8', { fatal: true }); + const ingest = (data) => { + if (data.trim() === '[DONE]') { + sawDone = true; + return; + } + let event; + try { + event = JSON.parse(data); + } catch { + throw new Error('Audio provider sent a malformed event'); + } + if (event?.error) throw new Error('Audio provider returned an error event'); + const choice = event?.choices?.find((value) => value.index === 0); + if (choice?.finish_reason != null) { + if (choice.finish_reason !== 'stop') throw new Error('Audio provider generation did not finish successfully'); + sawFinish = true; + } + const data64 = choice?.delta?.audio?.data; + if (data64 === undefined || data64 === '') return; + if (typeof data64 !== 'string' || data64.length % 4 || !BASE64_CHUNK.test(data64)) + throw new Error('Audio provider sent an invalid audio chunk'); + const chunk = Buffer.from(data64, 'base64'); + if (chunk.toString('base64') !== data64) throw new Error('Audio provider sent an invalid audio chunk'); + decodedBytes += chunk.length; + if (decodedBytes > maxBytes) throw new Error('Audio generation exceeded the size limit'); + chunks.push(chunk); + }; + // Race reads against cancellation even if an upstream stream stalls forever. + const reader = body.getReader?.(); + const iterator = reader ? null : body[Symbol.asyncIterator](); + let abort; + const aborted = new Promise((_, reject) => { + abort = () => reject(new Error('Audio generation timed out or was cancelled')); + signal.addEventListener('abort', abort, { once: true }); + }); + try { + while (!sawDone) { + if (signal.aborted) throw new Error('Audio generation timed out or was cancelled'); + let next; + try { + next = await Promise.race([reader ? reader.read() : iterator.next(), aborted]); + } catch { + throw new Error( + signal.aborted ? 'Audio generation timed out or was cancelled' : 'Audio provider stream failed' + ); + } + if (next.done) break; + rawBytes += next.value.byteLength; + if (rawBytes > maxBytes * 2 + maxEventBytes) throw new Error('Audio provider stream exceeded the size limit'); + try { + pending += decoder.decode(next.value, { stream: true }); + } catch { + throw new Error('Audio provider sent a malformed event'); + } + // Match CRLF without rewriting chunks: a split CR/LF remains intact. + let match; + while ((match = /\r?\n\r?\n/.exec(pending))) { + const block = pending.slice(0, match.index); + pending = pending.slice(match.index + match[0].length); + if (Buffer.byteLength(block) > maxEventBytes) throw new Error('Audio provider event exceeded the size limit'); + const data = block + .split(/\r?\n/) + .filter((line) => line.startsWith('data:')) + .map((line) => line.slice(5).replace(/^ /, '')) + .join('\n'); + if (data) ingest(data); + if (sawDone) break; + } + if (Buffer.byteLength(pending) > maxEventBytes) throw new Error('Audio provider event exceeded the size limit'); + } + try { + pending += decoder.decode(); + } catch { + throw new Error('Audio provider stream ended mid-event'); + } + if (!sawDone && pending.trim()) throw new Error('Audio provider stream ended mid-event'); + if (!sawDone && !sawFinish) throw new Error('Audio provider stream ended without a completion marker'); + if (!decodedBytes) throw new Error('Audio provider returned no audio'); + return Buffer.concat(chunks, decodedBytes); + } finally { + signal.removeEventListener('abort', abort); + await boundedCleanup(() => (reader ? reader.cancel() : (body.cancel?.() ?? iterator.return?.()))); + reader?.releaseLock(); + } +} + +async function convertAudio(input, format, signal, maxBytes) { + // Provider formats are best effort (Lyria currently returns MP3 for WAV). + // Decode only buffered audio, with no shell, URLs, files or network protocols. + const output = await new Promise((resolve, reject) => { + const child = spawn( + 'ffmpeg', + [ + '-hide_banner', + '-loglevel', + 'error', + '-xerror', + '-err_detect', + 'explode', + '-protocol_whitelist', + 'pipe', + '-i', + 'pipe:0', + '-map', + '0:a:0', + '-vn', + '-acodec', + format === 'wav' ? 'pcm_s16le' : 'libmp3lame', + '-f', + format, + 'pipe:1' + ], + { stdio: ['pipe', 'pipe', 'ignore'], signal } + ); + const parts = []; + let length = 0, + failure; + child.on('error', () => { + failure = new Error('Audio conversion requires a working ffmpeg executable'); + }); + child.stdin.on('error', () => {}); + child.stdout.on('data', (part) => { + length += part.length; + if (length > maxBytes) { + failure = new Error('Audio conversion exceeded the size limit'); + child.kill('SIGKILL'); + } else parts.push(part); + }); + child.on('close', (code) => { + if (signal.aborted) reject(new Error('Audio generation timed out or was cancelled')); + else if (failure) reject(failure); + else if (code !== 0) reject(new Error('Audio conversion failed')); + else resolve(Buffer.concat(parts, length)); + }); + child.stdin.end(input); + }); + if (identifyAudio(output) !== format) throw new Error('Audio conversion returned an invalid format'); + if (format === 'wav') { + // ffmpeg writes unknown sizes to pipes; make the completed WAV seekable. + output.writeUInt32LE(output.length - 8, 4); + for (let at = 12; at + 8 <= output.length;) { + const size = output.readUInt32LE(at + 4); + if (output.toString('ascii', at, at + 4) === 'data') { + output.writeUInt32LE(output.length - at - 8, at + 4); + break; + } + at += 8 + size + (size % 2); + } + } + return output; +} + +// Resource cleanup is bounded even for injected or broken upstream bodies. +async function boundedCleanup(cleanup) { + let timer; + try { + await Promise.race([ + Promise.resolve() + .then(cleanup) + .catch(() => {}), + new Promise((resolve) => { + timer = setTimeout(resolve, 250); + }) + ]); + } finally { + clearTimeout(timer); + } +} + +function validateAudio(audio, format) { + const malformed = () => { + throw new Error('Audio provider returned truncated or malformed audio'); + }; + if (format === 'wav') { + if (audio.readUInt32LE(4) + 8 !== audio.length) malformed(); + let at = 12, + data = false, + fmt = false; + while (at + 8 <= audio.length) { + const size = audio.readUInt32LE(at + 4), + end = at + 8 + size; + if (end > audio.length) malformed(); + const id = audio.toString('ascii', at, at + 4); + if (id === 'fmt ') { + if (size < 16) malformed(); + fmt = true; + } + if (id === 'data') { + if (size === 0) malformed(); + data = true; + } + at = end + (size % 2); + } + if (at !== audio.length || !fmt || !data) malformed(); + } else if (format === 'mp3') { + let at = 0, + frames = 0; + if (audio.toString('ascii', 0, 3) === 'ID3') { + if ([6, 7, 8, 9].some((i) => audio[i] & 0x80)) malformed(); + at = 10 + ((audio[6] << 21) | (audio[7] << 14) | (audio[8] << 7) | audio[9]) + (audio[5] & 0x10 ? 10 : 0); + } + while (at < audio.length) { + if (audio.length - at === 128 && audio.toString('ascii', at, at + 3) === 'TAG') { + at += 128; + break; + } + if (at + 4 > audio.length || audio[at] !== 255 || (audio[at + 1] & 0xe0) !== 0xe0) malformed(); + const version = (audio[at + 1] >> 3) & 3, + layer = (audio[at + 1] >> 1) & 3; + const rateIndex = (audio[at + 2] >> 2) & 3, + bitrateIndex = audio[at + 2] >> 4; + if (version === 1 || layer !== 1 || rateIndex === 3 || bitrateIndex === 0 || bitrateIndex === 15) malformed(); + const bitrate = ( + version === 3 + ? [0, 32, 40, 48, 56, 64, 80, 96, 112, 128, 160, 192, 224, 256, 320] + : [0, 8, 16, 24, 32, 40, 48, 56, 64, 80, 96, 112, 128, 144, 160] + )[bitrateIndex]; + const sampleRate = [44100, 48000, 32000][rateIndex] / (version === 3 ? 1 : version === 2 ? 2 : 4); + const size = Math.floor(((version === 3 ? 144 : 72) * bitrate * 1000) / sampleRate) + ((audio[at + 2] >> 1) & 1); + if (at + size > audio.length) malformed(); + at += size; + frames++; + } + if (!frames || at !== audio.length) malformed(); + } +} diff --git a/src/init.mjs b/src/init.mjs index eade39b..214ce4c 100644 --- a/src/init.mjs +++ b/src/init.mjs @@ -874,8 +874,13 @@ function ensureRecipeConfigEntries(config, recipe, { modelRoot, sessionCacheRoot finishRecipeModelConfig(config, recipeModel, materializedModel, modelId); continue; } - const runtimeId = existingModel?.runtime ?? recipeModel.runtime ?? `${backendId}-${modelSlug}`; - const backendConfigId = existingModel?.backend ?? recipeModel.backendConfig ?? `${backendId}-${modelSlug}`; + // Media models leave a retired shared runtime for their own; the recipe's + // placement replaces the inherited route. + const leavingSharedMedia = backendId === 'comfyui-media' && isSharedMediaRuntime(config, existingModel?.runtime); + const inherited = leavingSharedMedia ? null : existingModel; + const retired = leavingSharedMedia ? { runtime: existingModel.runtime, backend: existingModel.backend } : null; + const runtimeId = inherited?.runtime ?? recipeModel.runtime ?? `${backendId}-${modelSlug}`; + const backendConfigId = inherited?.backend ?? recipeModel.backendConfig ?? `${backendId}-${modelSlug}`; const port = Number(config.runtimes?.[runtimeId]?.port) || nextBackendPort(config); const modelPath = modelPathForRecipeModel(recipeModel, backendId, modelRoot); @@ -914,8 +919,8 @@ function ensureRecipeConfigEntries(config, recipe, { modelRoot, sessionCacheRoot } else if (config.runtimes[runtimeId]?.recipe?.id === recipe.id) { Object.assign(existingModel, materializedModel); } else if (backendId === 'comfyui-media') { - // Media recipes refresh their API contract when reusing an existing - // shared engine. Preserve the operator's upstream route. + // Media recipes refresh their API contract when reusing a runtime + // another recipe installed. Preserve the operator's upstream route. for (const key of ['kind', 'input', 'output', 'capabilities', 'reasoning', 'supportsTools', 'tts', 'stt']) { if (Object.hasOwn(materializedModel, key)) existingModel[key] = materializedModel[key]; else delete existingModel[key]; @@ -925,6 +930,7 @@ function ensureRecipeConfigEntries(config, recipe, { modelRoot, sessionCacheRoot config.defaults.audioGenerationModel ??= modelId; } } + if (retired) dropUnreferencedRoute(config, retired); finishRecipeModelConfig(config, recipeModel, materializedModel, modelId); } @@ -954,6 +960,18 @@ function materializedRecipeModel(recipe, recipeModel, modelId, placement) { return model; } +/** A media runtime without a single-model selector is the retired shared engine. */ +function isSharedMediaRuntime(config, runtimeId) { + const createArgs = config.runtimes?.[runtimeId]?.bootstrap?.createArgs; + return Array.isArray(createArgs) && !createArgs.some((arg) => String(arg).startsWith('LLOOM_MEDIA_MODEL=')); +} + +function dropUnreferencedRoute(config, { runtime, backend }) { + const routes = config.models.flatMap((model) => [model, ...asArray(model.targets)]); + if (runtime && !routes.some((route) => route.runtime === runtime)) delete config.runtimes[runtime]; + if (backend && !routes.some((route) => route.backend === backend)) delete config.backends[backend]; +} + function finishRecipeModelConfig(config, recipeModel, materializedModel, modelId) { if (!config.clientCatalog.modelOrder.includes(modelId)) { config.clientCatalog.modelOrder.push(modelId); diff --git a/src/server.mjs b/src/server.mjs index 7aba96c..7dff74a 100644 --- a/src/server.mjs +++ b/src/server.mjs @@ -3,6 +3,7 @@ import runtimeCapabilities from './runtime-capabilities.json' with { type: 'json import { executeWebFunction, webFunctionStatus } from './web-functions.mjs'; import { createPerformanceSampler } from './performance-sampler.mjs'; import { generateProviderVideo } from './video-providers.mjs'; +import { generateProviderAudio } from './audio-providers.mjs'; import http from 'node:http'; import { readErrorDiagnostic, streamProviderError } from './protocol/upstream-error.mjs'; import { fetchWithStreamProgress } from './protocol/stream-progress.mjs'; @@ -704,6 +705,8 @@ function copyResponseHeaders(upstream) { 'retry-after', 'x-lloom-provider', 'x-lloom-provider-job-id', + 'x-lloom-provider-origin', + 'x-lloom-audio-format', 'x-lloom-upstream-model' ]) { const value = upstream.headers.get(name); @@ -3164,6 +3167,19 @@ export function createLloomServer( res }, async ({ signal, timing, progress, watchdog }) => { + if (resolved.backend.audioProvider) { + const upstream = await generateProviderAudio({ + backend: resolved.backend, + fetchFn: undiciFetch, + dispatcher: longRunningMediaDispatcher, + body: { ...body, model: resolved.model.upstreamModel }, + signal, + timeoutMs: resolved.backend.timeoutMs ?? 600000 + }); + // The provider stream is fully validated before this point, so the + // assembled artifact is written as a buffered binary response. + return proxyRawResponse(res, upstream, { signal, timing, corsConfig: config }); + } const upstream = await fetchUpstream({ headers: inferenceGatewayHeaders(req, resolved), backend: resolved.backend, diff --git a/test/audio-providers.test.mjs b/test/audio-providers.test.mjs new file mode 100644 index 0000000..74fa49a --- /dev/null +++ b/test/audio-providers.test.mjs @@ -0,0 +1,584 @@ +import assert from 'node:assert/strict'; +import test from 'node:test'; +import { generateProviderAudio } from '../src/audio-providers.mjs'; + +const WAV = Buffer.alloc(46); +WAV.write('RIFF'); +WAV.writeUInt32LE(38, 4); +WAV.write('WAVEfmt ', 8); +WAV.writeUInt32LE(16, 16); +WAV.writeUInt16LE(1, 20); +WAV.writeUInt16LE(1, 22); +WAV.writeUInt32LE(8000, 24); +WAV.writeUInt32LE(16000, 28); +WAV.writeUInt16LE(2, 32); +WAV.writeUInt16LE(16, 34); +WAV.write('data', 36); +WAV.writeUInt32LE(2, 40); +const MP3 = Buffer.alloc(417); +MP3.set([0xff, 0xfb, 0x90, 0x00]); + +/** Build SSE text where each part is delivered as its own bytes chunk. */ +function sse(...events) { + return events.map((event) => `data: ${JSON.stringify(event)}\n\n`).join(''); +} + +function audioEvent(base64, finish_reason = null) { + return { choices: [{ index: 0, delta: { audio: { data: base64 } }, finish_reason }] }; +} + +function bodyFrom(chunks) { + const encoder = new TextEncoder(); + let index = 0; + return { + cancel() {}, + async *[Symbol.asyncIterator]() { + while (index < chunks.length) yield encoder.encode(chunks[index++]); + } + }; +} + +function responseFor(chunks, init = {}) { + if (chunks && typeof chunks !== 'string' && !Array.isArray(chunks)) { + const response = new Response(null, { status: 200, headers: { 'content-type': 'text/event-stream' }, ...init }); + Object.defineProperty(response, 'body', { value: chunks }); + return response; + } + const response = new Response(null, { + status: 200, + headers: { 'content-type': 'text/event-stream' }, + ...init + }); + // A real Response would wrap and re-read the source; hand the adapter the + // raw iterator so per-chunk boundaries are deterministic under test. + Object.defineProperty(response, 'body', { value: bodyFrom(chunks) }); + return response; +} + +function collectFetch(chunks, init) { + const calls = []; + const fetchFn = async (url, options) => { + calls.push({ url, options }); + return responseFor(chunks, init); + }; + return { calls, fetchFn }; +} + +test('streams one provider event per chunk and returns wav bytes', async () => { + const { fetchFn } = collectFetch([ + sse({ choices: [{ index: 0, delta: { audio: { data: WAV.subarray(0, 3).toString('base64') } } }] }), + sse({ + choices: [{ index: 0, delta: { audio: { data: WAV.subarray(3).toString('base64') }, finish_reason: 'stop' } }] + }), + 'data: [DONE]\n\n' + ]); + const response = await generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKey: 'test-only' }, + body: { model: 'google/lyria-3-pro-preview', prompt: 'gentle piano' }, + fetchFn + }); + assert.equal(response.status, 200); + assert.equal(response.headers.get('content-type'), 'audio/wav'); + assert.equal(response.headers.get('x-lloom-provider'), 'openrouter'); + assert.equal(response.headers.get('x-lloom-upstream-model'), 'google/lyria-3-pro-preview'); + assert.deepEqual(Buffer.from(await response.arrayBuffer()), WAV); +}); + +test('honors mp3 format and maps the exact OpenRouter origin', async () => { + const { calls, fetchFn } = collectFetch([sse(audioEvent(MP3.toString('base64'), 'stop')), 'data: [DONE]\n\n']); + const response = await generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKey: 'test-only' }, + body: { model: 'google/lyria-3-pro-preview', instructions: 'lofi loop', format: 'mp3' }, + fetchFn + }); + assert.equal(response.headers.get('content-type'), 'audio/mpeg'); + assert.deepEqual(Buffer.from(await response.arrayBuffer()), MP3); + assert.equal(calls.length, 1); + assert.equal(calls[0].url, 'https://openrouter.ai/api/v1/chat/completions'); + assert.equal(calls[0].options.redirect, 'error'); + assert.equal(calls[0].options.headers.Authorization, 'Bearer test-only'); + const payload = JSON.parse(calls[0].options.body); + assert.equal(payload.model, 'google/lyria-3-pro-preview'); + assert.equal(payload.stream, true); + assert.deepEqual(payload.modalities, ['text', 'audio']); + assert.deepEqual(payload.audio, { format: 'mp3' }); + assert.equal(payload.messages[0].role, 'user'); + assert.equal(payload.messages[0].content, 'lofi loop'); +}); + +test('folds prompt, lyrics, input and duration into a single user message', async () => { + const { calls, fetchFn } = collectFetch([sse(audioEvent(WAV.toString('base64'), 'stop')), 'data: [DONE]\n\n']); + await generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKey: 'test-only' }, + body: { model: 'google/lyria-3-pro-preview', prompt: 'warm pads', lyrics: 'la la', input: 'BPM 90', duration: 30 }, + fetchFn + }); + const content = JSON.parse(calls[0].options.body).messages[0].content; + assert.equal(content, 'warm pads\nla la\nBPM 90\nDuration: 30 seconds.'); +}); + +test('uses apiKeyEnv when configured and rejects a missing credential', async () => { + process.env.LLOOM_TEST_AUDIO_KEY = 'env-key'; + try { + const { calls, fetchFn } = collectFetch([sse(audioEvent(WAV.toString('base64'), 'stop'))]); + await generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKeyEnv: 'LLOOM_TEST_AUDIO_KEY' }, + body: { model: 'google/lyria-3-pro-preview', prompt: 'p' }, + fetchFn + }); + assert.equal(calls[0].options.headers.Authorization, 'Bearer env-key'); + } finally { + delete process.env.LLOOM_TEST_AUDIO_KEY; + } + await assert.rejects( + generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKeyEnv: 'LLOOM_TEST_AUDIO_MISSING' }, + body: { model: 'google/lyria-3-pro-preview', prompt: 'p' }, + fetchFn: async () => { + throw new Error('must not call'); + } + }), + /credential is not configured/ + ); +}); + +test('rejects unsupported provider and unsupported local controls', async () => { + await assert.rejects( + generateProviderAudio({ + backend: { audioProvider: 'local' }, + body: { model: 'm', prompt: 'p' }, + fetchFn: async () => { + throw new Error('must not call'); + } + }), + /Unsupported audio provider/ + ); + for (const field of ['steps', 'cfg', 'seed', 'negative_prompt']) { + await assert.rejects( + generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKey: 'k' }, + body: { model: 'm', prompt: 'p', [field]: 1 }, + fetchFn: async () => { + throw new Error('must not call'); + } + }), + new RegExp(`does not support field "${field}"`) + ); + } +}); + +test('applies the configured OpenRouter provider policy', async () => { + const { calls, fetchFn } = collectFetch([sse(audioEvent(WAV.toString('base64'), 'stop'))]); + await generateProviderAudio({ + backend: { + audioProvider: 'openrouter', + apiKey: 'k', + baseUrl: 'https://stale.example/api/v1', + openrouterProvider: { only: ['google-vertex'], allow_fallbacks: false } + }, + body: { model: 'google/lyria-3-pro-preview', prompt: 'p' }, + fetchFn + }); + assert.deepEqual(JSON.parse(calls[0].options.body).provider, { only: ['google-vertex'], allow_fallbacks: false }); +}); + +test('changes accept both [DONE] and a finish_reason marker', async () => { + for (const chunks of [ + [sse(audioEvent(WAV.toString('base64'))), 'data: [DONE]\n\n'], + [sse(audioEvent(WAV.toString('base64'), 'stop'))] + ]) { + const { fetchFn } = collectFetch(chunks); + const response = await generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKey: 'k' }, + body: { model: 'm', prompt: 'p' }, + fetchFn + }); + assert.deepEqual(Buffer.from(await response.arrayBuffer()), WAV); + } +}); + +test('rejects provider errors, malformed events and non-SSE responses', async () => { + const base = { backend: { audioProvider: 'openrouter', apiKey: 'k' }, body: { model: 'm', prompt: 'p' } }; + await assert.rejects( + generateProviderAudio({ ...base, fetchFn: async () => responseFor([sse({ error: { message: 'boom' } })]) }), + /error event/ + ); + await assert.rejects( + generateProviderAudio({ ...base, fetchFn: async () => responseFor(['data: {oops}\n\n']) }), + /malformed event/ + ); + await assert.rejects( + generateProviderAudio({ + ...base, + fetchFn: async () => + new Response('{"error":"nope"}', { status: 200, headers: { 'content-type': 'application/json' } }) + }), + /non-SSE/ + ); +}); + +test('rejects upstream HTTP failures without leaking the provider body', async () => { + const { fetchFn } = collectFetch([], { status: 500 }); + const error = await generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKey: 'k' }, + body: { model: 'm', prompt: 'p' }, + fetchFn + }).then( + () => null, + (e) => e + ); + assert.match(error.message, /returned HTTP 500/); + assert.equal(error.statusCode, 500); +}); + +test('rejects truncated streams, empty audio and missing completion markers', async () => { + const base = { backend: { audioProvider: 'openrouter', apiKey: 'k' }, body: { model: 'm', prompt: 'p' } }; + await assert.rejects( + generateProviderAudio({ ...base, fetchFn: async () => responseFor([sse(audioEvent(WAV.toString('base64')))]) }), + /without a completion marker/ + ); + await assert.rejects( + generateProviderAudio({ + ...base, + fetchFn: async () => responseFor([sse(audioEvent(WAV.toString('base64'))), 'data: {"choices":[]}']) + }), + /mid-event|completion marker/ + ); + await assert.rejects( + generateProviderAudio({ + ...base, + fetchFn: async () => responseFor([sse({ choices: [{ delta: {} }] }), 'data: [DONE]\n\n']) + }), + /returned no audio/ + ); +}); + +test('enforces the decoded size cap and cancels the upstream body', async () => { + let cancelled = false; + const event = 'data: ' + JSON.stringify(audioEvent(WAV.toString('base64'))) + '\n\n'; + const encoder = new TextEncoder(); + let sent = false; + // One logical SSE event per chunk; the adapter must stop as soon as the + // decoded total crosses the cap and cancel the remaining stream. + const events = [event, event, event, event]; + const body = { + async *[Symbol.asyncIterator]() { + if (sent) return; + sent = true; + for (const chunk of events) yield encoder.encode(chunk); + }, + cancel() { + cancelled = true; + } + }; + await assert.rejects( + generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKey: 'k' }, + body: { model: 'm', prompt: 'p' }, + maxBytes: 4, + fetchFn: async () => responseFor(body) + }), + /size limit/ + ); + assert.equal(cancelled, true); +}); + +test('propagates an already-aborted signal and cancels the body', async () => { + let cancelled = false; + const controller = new AbortController(); + controller.abort(); + const encoder = new TextEncoder(); + const body = { + async *[Symbol.asyncIterator]() { + yield encoder.encode(sse(audioEvent(WAV.toString('base64')))); + }, + cancel() { + cancelled = true; + } + }; + await assert.rejects( + generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKey: 'k' }, + body: { model: 'm', prompt: 'p' }, + signal: controller.signal, + fetchFn: async () => responseFor(body) + }) + ); + assert.equal(cancelled, true); +}); + +test('tolerates CRLF, blank-line splits and multiline SSE blocks', async () => { + const half = WAV.subarray(0, 4).toString('base64'); + const rest = WAV.subarray(4).toString('base64'); + const text = + `event: message\r\ndata: ${JSON.stringify(audioEvent(half))}\r\n\r\n` + + `data: ${JSON.stringify(audioEvent(rest))}\r\n\r\n` + + 'data: [DONE]\r\n\r\n'; + // Split mid-way through an event boundary to exercise leftover buffering. + const parts = [text.slice(0, 17), text.slice(17, 60), text.slice(60)]; + const { fetchFn } = collectFetch(parts); + const response = await generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKey: 'k' }, + body: { model: 'm', prompt: 'p' }, + fetchFn + }); + assert.deepEqual(Buffer.from(await response.arrayBuffer()), WAV); +}); + +test('accepts one inline PNG reference image and rejects other inputs', async () => { + const image = `data:image/png;base64,${Buffer.from('png').toString('base64')}`; + const { calls, fetchFn } = collectFetch([sse(audioEvent(WAV.toString('base64'), 'stop'))]); + await generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKey: 'k' }, + body: { model: 'google/lyria-3-pro-preview', prompt: 'with art', image }, + fetchFn + }); + const content = JSON.parse(calls[0].options.body).messages[0].content; + assert.equal(Array.isArray(content), true); + assert.deepEqual(content[1], { type: 'image_url', image_url: { url: image } }); + + for (const bad of ['https://example.com/x.png', 'data:image/gif;base64,YQ==', '/tmp/x.png']) { + await assert.rejects( + generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKey: 'k' }, + body: { model: 'm', prompt: 'p', image: bad }, + fetchFn: async () => { + throw new Error('must not call'); + } + }), + /inline PNG or JPEG/ + ); + } +}); + +test('rejects invalid format and duration without calling the provider', async () => { + const base = { backend: { audioProvider: 'openrouter', apiKey: 'k' } }; + const noCall = { + fetchFn: async () => { + throw new Error('must not call'); + } + }; + await assert.rejects( + generateProviderAudio({ ...base, body: { model: 'm', prompt: 'p', format: 'flac' }, ...noCall }), + /wav or mp3/ + ); + await assert.rejects( + generateProviderAudio({ ...base, body: { model: 'm', prompt: 'p', duration: 0 }, ...noCall }), + /duration/i + ); + await assert.rejects( + generateProviderAudio({ ...base, body: { model: 'm', prompt: 'p', duration: 601 }, ...noCall }), + /duration/i + ); + await assert.rejects( + generateProviderAudio({ ...base, body: { model: 'm', prompt: 'p', duration: '30' }, ...noCall }), + /duration/i + ); +}); + +test('never returns audio before the stream completes', async () => { + const encoder = new TextEncoder(); + let release; + const gate = new Promise((resolve) => { + release = resolve; + }); + const body = { + async *[Symbol.asyncIterator]() { + yield encoder.encode(sse(audioEvent(WAV.toString('base64')))); + await gate; + yield encoder.encode('data: [DONE]\n\n'); + }, + cancel() {} + }; + let settled = false; + const pending = generateProviderAudio({ + backend: { audioProvider: 'openrouter', apiKey: 'k' }, + body: { model: 'm', prompt: 'p' }, + fetchFn: async () => responseFor(body) + }).then((value) => { + settled = true; + return value; + }); + await new Promise((resolve) => setTimeout(resolve, 25)); + assert.equal(settled, false); + release(); + assert.deepEqual(Buffer.from(await (await pending).arrayBuffer()), WAV); +}); + +const base = { backend: { audioProvider: 'openrouter', apiKey: 'test-only' }, body: { model: 'm', prompt: 'p' } }; +test('handles headerless Lyria streams, large events, combined stop and DONE, and every CRLF split', async () => { + const large = Buffer.concat([WAV, Buffer.alloc(1600000)]); + large.writeUInt32LE(large.length - 8, 4); + large.writeUInt32LE(large.length - 44, 40); + const response = await generateProviderAudio({ + ...base, + fetchFn: async () => + responseFor([sse(audioEvent(large.toString('base64'), 'stop')) + 'data: [DONE]\n\n'], { headers: {} }) + }); + assert.equal((await response.arrayBuffer()).byteLength, large.length); + const text = sse(audioEvent(WAV.toString('base64'), 'stop')).replaceAll('\n', '\r\n') + 'data: [DONE]\r\n\r\n'; + for (let at = 0; at < text.length; at++) { + const response = await generateProviderAudio({ + ...base, + fetchFn: async () => responseFor([text.slice(0, at), text.slice(at)]) + }); + assert.deepEqual(Buffer.from(await response.arrayBuffer()), WAV); + } +}); + +test('does not accept truncated, failed, foreign-choice or non-audio completions', async () => { + for (const stream of [ + sse(audioEvent(WAV.toString('base64'), 'length')) + 'data: [DONE]\n\n', + sse(audioEvent(WAV.toString('base64'), 'stop'), { error: { message: 'private' } }), + sse({ choices: [{ index: 1, delta: { audio: { data: WAV.toString('base64') } }, finish_reason: 'stop' }] }) + + 'data: [DONE]\n\n', + sse(audioEvent(Buffer.from('not audio').toString('base64'), 'stop')), + sse(audioEvent(WAV.toString('base64'), 'stop')) + 'data: {', + sse(audioEvent('YR==', 'stop')), + sse(audioEvent(Buffer.from('ID3abcdefghi').toString('base64'), 'stop')), + sse(audioEvent(WAV.subarray(0, 44).toString('base64'), 'stop')), + sse(audioEvent(MP3.subarray(0, 100).toString('base64'), 'stop')) + ]) + await assert.rejects(generateProviderAudio({ ...base, fetchFn: async () => responseFor([stream]) })); +}); + +test('validates callers before network IO and sanitizes network exceptions', async () => { + for (const body of [ + { ...base.body, seed: 1 }, + { ...base.body, format: 'mp3', response_format: 'wav' } + ]) { + await assert.rejects(generateProviderAudio({ ...base, body, fetchFn: () => assert.fail('network call') }), { + statusCode: 400 + }); + } + await assert.rejects( + generateProviderAudio({ + ...base, + fetchFn: () => { + throw new Error('secret'); + } + }), + (error) => !error.message.includes('secret') + ); +}); + +test('cancellation interrupts a stalled stream read', async () => { + const controller = new AbortController(); + let cancelled = false; + const body = new ReadableStream({ + start() {}, + cancel() { + cancelled = true; + } + }); + const response = new Response(body, { headers: { 'content-type': 'text/event-stream' } }); + const promise = generateProviderAudio({ ...base, signal: controller.signal, fetchFn: async () => response }); + setTimeout(() => controller.abort(), 15); + await assert.rejects(promise, /cancelled/); + assert.equal(cancelled, true); +}); + +test('ffmpeg normalizes provider format and final WAV length', async (t) => { + const { spawnSync } = await import('node:child_process'); + const encoded = spawnSync( + 'ffmpeg', + ['-hide_banner', '-loglevel', 'error', '-f', 'wav', '-i', 'pipe:0', '-f', 'mp3', 'pipe:1'], + { input: WAV } + ); + if (encoded.error?.code === 'ENOENT') { + t.skip('ffmpeg not installed'); + return; + } + assert.equal(encoded.status, 0); + const response = await generateProviderAudio({ + ...base, + body: { ...base.body, response_format: 'wav' }, + fetchFn: async () => responseFor([sse(audioEvent(encoded.stdout.toString('base64'), 'stop'))]) + }); + const output = Buffer.from(await response.arrayBuffer()); + assert.equal(response.headers.get('content-type'), 'audio/wav'); + assert.equal(output.toString('ascii', 8, 12), 'WAVE'); + assert.equal(output.readUInt32LE(4), output.length - 8); + await assert.rejects( + generateProviderAudio({ + ...base, + maxBytes: encoded.stdout.length, + fetchFn: async () => responseFor([sse(audioEvent(encoded.stdout.toString('base64'), 'stop'))]) + }), + /size limit/ + ); +}); + +test('gateway resolves the music alias and default through the adapter', async (t) => { + const { MockAgent, getGlobalDispatcher, setGlobalDispatcher } = await import('undici'); + const { once } = await import('node:events'); + const { createLloomServer } = await import('../src/server.mjs'); + const original = getGlobalDispatcher(); + const mock = new MockAgent(); + mock.disableNetConnect(); + mock.enableNetConnect(/^127\.0\.0\.1(?::\d+)?$/); + setGlobalDispatcher(mock); + const calls = []; + mock + .get('https://openrouter.ai') + .intercept({ path: '/api/v1/chat/completions', method: 'POST' }) + .reply(async (options) => { + const text = typeof options.body === 'string' ? options.body : await new Response(options.body).text(); + calls.push(JSON.parse(text)); + return { + statusCode: 200, + data: sse(audioEvent(WAV.toString('base64'), 'stop')) + 'data: [DONE]\n\n', + responseOptions: { headers: { 'content-type': 'text/event-stream' } } + }; + }) + .times(2); + const app = createLloomServer( + { + server: { host: '127.0.0.1', port: 0 }, + security: { allowMissingAuth: true, apiKeys: [] }, + logging: { metricsPersistence: false }, + telemetry: { performanceSampler: false }, + defaults: { audioGenerationModel: 'music' }, + aliases: { music: { members: ['lyria'] } }, + backends: { + cloud: { + type: 'openai', + baseUrl: 'https://openrouter.ai/api/v1', + audioProvider: 'openrouter', + apiKey: 'test-only' + } + }, + models: [{ id: 'lyria', backend: 'cloud', kind: 'audio_generation', upstreamModel: 'google/lyria-3-pro-preview' }] + }, + { logger: { error() {}, warn() {} }, upstreamDispatcher: mock } + ); + t.after(async () => { + app.server.closeAllConnections(); + await app.close({ stopRuntimes: false }); + setGlobalDispatcher(original); + await mock.close(); + }); + app.server.listen(0, '127.0.0.1'); + await once(app.server, 'listening'); + const url = `http://127.0.0.1:${app.server.address().port}/v1/audio/generations`; + for (const model of ['music', undefined]) { + const response = await fetch(url, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ model, prompt: 'p', response_format: 'wav' }) + }); + assert.equal(response.status, 200, await response.clone().text()); + assert.deepEqual(Buffer.from(await response.arrayBuffer()), WAV); + } + assert.deepEqual( + calls.map((c) => c.model), + ['google/lyria-3-pro-preview', 'google/lyria-3-pro-preview'] + ); + const bad = await fetch(url, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ prompt: 'p', seed: 1 }) + }); + assert.equal(bad.status, 400); + mock.assertNoPendingInterceptors(); +}); diff --git a/test/comfyui-media.test.mjs b/test/comfyui-media.test.mjs index d1dbcfa..b2596e0 100644 --- a/test/comfyui-media.test.mjs +++ b/test/comfyui-media.test.mjs @@ -45,10 +45,31 @@ try { /ennspark|spark03|enntitysparkadmin|\/Users\/|\/home\/|192\.168\.|100\.78\./ ); const standalone = await apply(empty, recipe); + const runtimeId = recipe.models[0].runtime; assert.equal(standalone.models.length, 1); assert.equal(Object.keys(standalone.runtimes).length, 1); - assert.equal(standalone.runtimes['comfyui-media'].bootstrap.image, image); - assert.match(standalone.runtimes['comfyui-media'].bootstrap.createArgs.join(' '), /127\.0\.0\.1:\d+:8000/); + assert.equal(standalone.runtimes[runtimeId].bootstrap.image, image); + assert.match(standalone.runtimes[runtimeId].bootstrap.createArgs.join(' '), /127\.0\.0\.1:\d+:8000/); + const args = standalone.runtimes[runtimeId].bootstrap.createArgs; + assert.ok(args.includes(`LLOOM_MEDIA_MODEL=${recipe.models[0].model}`)); + assert.ok(args.includes('LLOOM_MODELS_ROOT=/opt/ComfyUI/models')); + assert.ok(!args.some((arg) => arg.includes(':/opt/lloom-models'))); + const fileMounts = args.filter((arg) => arg.startsWith('type=bind,')); + const downloads = recipe.setup.steps.filter((step) => step.action === 'download-model'); + assert.equal( + fileMounts.length, + downloads.reduce((n, step) => n + step.include.length, 0) + ); + for (const step of downloads) + for (const file of step.include) { + assert.ok( + fileMounts.some( + (mount) => + mount.includes(`/${step.model.replaceAll('/', '--')}/${file},dst=/opt/ComfyUI/models/`) && + mount.endsWith(',readonly') + ) + ); + } const model = standalone.models[0]; assert.equal( model.kind, @@ -71,17 +92,20 @@ try { assert.deepEqual(plan.validationErrors, []); } for (const order of [recipes, recipes.toReversed()]) { - let config = await apply(empty, order[0]); - const runtime = structuredClone(config.runtimes['comfyui-media']); - const backend = structuredClone(config.backends['comfyui-media']); - for (const recipe of order.slice(1)) { + let config = structuredClone(empty); + for (const recipe of order) { + const previousRuntimes = structuredClone(config.runtimes); + const previousBackends = structuredClone(config.backends); config = await apply(config, recipe); - assert.deepEqual(config.runtimes['comfyui-media'], runtime); - assert.deepEqual(config.backends['comfyui-media'], backend); + for (const [id, value] of Object.entries(previousRuntimes)) assert.deepEqual(config.runtimes[id], value); + for (const [id, value] of Object.entries(previousBackends)) assert.deepEqual(config.backends[id], value); } assert.equal(config.models.length, 14); assert.equal(config.models.filter((m) => m.kind === 'audio_generation').length, 4); - assert.deepEqual(Object.keys(config.runtimes), ['comfyui-media']); + assert.equal(Object.keys(config.runtimes).length, 14); + assert.equal(Object.keys(config.backends).length, 14); + assert.equal(new Set(Object.values(config.runtimes).map((r) => r.port)).size, 14); + assert.equal(new Set(config.models.map((m) => m.runtime)).size, 14); } assert.throws( () => createModelImportPlan(empty, { modelRef: 'mlx-community/ACE-Step', backend: 'mlx-audio' }), @@ -89,27 +113,54 @@ try { ); // Explicit recipe selection refreshes legacy music classification while - // preserving a shared engine's established route and container settings. + // preserving the model's established route and container settings. const musicRecipe = recipes.find((recipe) => recipe.id.endsWith('ace-step-1-5-xl-turbo')); const legacy = await apply(empty, musicRecipe); const legacyModel = legacy.models[0]; + const musicRuntime = musicRecipe.models[0].runtime; legacyModel.kind = 'audio_speech'; legacyModel.capabilities = ['audio-speech', 'music-generation']; legacyModel.tts = { family: 'generic' }; legacyModel.upstreamModel = 'existing-upstream-name'; legacy.defaults.speechModel = legacyModel.id; - legacy.runtimes['comfyui-media'].recipe.id = 'existing-media-install'; - const preservedRuntime = structuredClone(legacy.runtimes['comfyui-media']); - const preservedBackend = structuredClone(legacy.backends['comfyui-media']); + legacy.runtimes[musicRuntime].recipe.id = 'existing-media-install'; + const preservedRuntime = structuredClone(legacy.runtimes[musicRuntime]); + const preservedBackend = structuredClone(legacy.backends[musicRuntime]); const migrated = await apply(legacy, musicRecipe); - assert.deepEqual(migrated.runtimes['comfyui-media'], preservedRuntime); - assert.deepEqual(migrated.backends['comfyui-media'], preservedBackend); + assert.deepEqual(migrated.runtimes[musicRuntime], preservedRuntime); + assert.deepEqual(migrated.backends[musicRuntime], preservedBackend); assert.equal(migrated.models[0].kind, 'audio_generation'); assert.equal(migrated.models[0].upstreamModel, 'existing-upstream-name'); assert.equal(migrated.models[0].tts, undefined); assert.equal(migrated.defaults.speechModel, undefined); assert.equal(migrated.defaults.audioGenerationModel, legacyModel.id); + // A catalog from the retired shared runtime moves each re-applied model to + // its own runtime; the shared one is dropped once no model references it. + const [imageRecipe, videoRecipe] = ['flux-2-klein-4b', 'minimax-h3'].map((suffix) => + recipes.find((recipe) => recipe.id.endsWith(suffix)) + ); + const shared = await apply(await apply(empty, imageRecipe), videoRecipe); + const sharedRuntime = structuredClone(shared.runtimes[imageRecipe.models[0].runtime]); + sharedRuntime.bootstrap.createArgs = sharedRuntime.bootstrap.createArgs.filter( + (arg) => !arg.startsWith('LLOOM_MEDIA_MODEL=') + ); + shared.runtimes = { 'comfyui-media': sharedRuntime }; + shared.backends = { 'comfyui-media': shared.backends[imageRecipe.models[0].backendConfig] }; + for (const model of shared.models) Object.assign(model, { runtime: 'comfyui-media', backend: 'comfyui-media' }); + const partlyMoved = await apply(shared, imageRecipe); + assert.equal(partlyMoved.models[0].runtime, imageRecipe.models[0].runtime); + assert.equal(partlyMoved.models[0].backend, imageRecipe.models[0].backendConfig); + assert.equal(partlyMoved.models[1].runtime, 'comfyui-media'); + assert.ok(partlyMoved.runtimes['comfyui-media']); + const fullyMoved = await apply(partlyMoved, videoRecipe); + assert.deepEqual( + fullyMoved.models.map((model) => model.runtime), + [imageRecipe, videoRecipe].map((recipe) => recipe.models[0].runtime) + ); + assert.equal(fullyMoved.runtimes['comfyui-media'], undefined); + assert.equal(fullyMoved.backends['comfyui-media'], undefined); + // Setup status must compose the actual download dependencies even when the // gateway model ID has no matching directory of its own. const dependencyRoot = path.join(dir, 'composed-models'); @@ -174,7 +225,7 @@ try { ); assert.ok(composedDestination.dependencies.every((dependency) => dependency.complete)); - console.log('ComfyUI recipes: all 14 standalone; shared runtime and backend unchanged in both application orders'); + console.log('ComfyUI recipes: all 14 have independent runtimes and file mounts in both application orders'); } finally { await fs.rm(dir, { recursive: true, force: true }); }