From 482d38f60f0d820a791d1fe565e04c177052c80f Mon Sep 17 00:00:00 2001 From: Jason McCartney Date: Sun, 27 Sep 2026 12:13:40 -0700 Subject: [PATCH 1/3] Add source-built Spark audio backend and per-model speech recipes Replace the host-local prebuilt media image with a locally built lloom/spark-audio:source- image and one public recipe per model: Qwen3-TTS 1.7B CustomVoice, Base (voice cloning) and VoiceDesign, and Whisper large-v3-turbo. Each recipe runs its checkpoint in its own managed container with a loopback port, a read-only mount of only that model's directory, and offline Hugging Face mode. Downloads are pinned to immutable revisions with per-file SHA-256. openai-whisper cannot load the Transformers-format repository, so the Whisper recipe fetches OpenAI's original large-v3-turbo.pt from a byte-identical Hugging Face copy whose hash matches openai-whisper v20250625. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 2 + backends/catalog.json | 55 ++++ backends/spark-audio/README.md | 2 + backends/spark-audio/install.py | 66 +++++ docs/backends.md | 2 + docs/recipes.md | 2 + docs/spark-audio.md | 55 ++++ package.json | 6 +- recipes/index.json | 112 ++++++++ ...vidia-spark-audio-qwen3-tts-1-7b-base.json | 255 +++++++++++++++++ ...park-audio-qwen3-tts-1-7b-customvoice.json | 256 ++++++++++++++++++ ...park-audio-qwen3-tts-1-7b-voicedesign.json | 254 +++++++++++++++++ ...ia-spark-audio-whisper-large-v3-turbo.json | 194 +++++++++++++ scripts/check-package.mjs | 2 +- test/spark-audio-recipes.test.mjs | 187 +++++++++++++ 15 files changed, 1446 insertions(+), 4 deletions(-) create mode 100644 backends/spark-audio/install.py create mode 100644 docs/spark-audio.md create mode 100644 recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-base.json create mode 100644 recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-customvoice.json create mode 100644 recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-voicedesign.json create mode 100644 recipes/linux-nvidia-spark-audio-whisper-large-v3-turbo.json create mode 100644 test/spark-audio-recipes.test.mjs diff --git a/CHANGELOG.md b/CHANGELOG.md index 78934a9..e86517f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,8 @@ All notable changes to LLooM will be documented in this file. The format follows - LLooM Hear CPU audio analysis, with structured estimates, optional dashboard images, and opt-in upstream interpretation. File and URL inputs require operator configuration. +- Four standalone NVIDIA Spark audio recipes (Qwen3-TTS 1.7B CustomVoice, Base voice cloning and VoiceDesign, and Whisper large-v3-turbo). Each runs one pinned, integrity-checked checkpoint in its own container from a locally built `spark-audio` image that replaces a host-local prebuilt image. + - Thirteen standalone NVIDIA ComfyUI media recipes, a public-source backend build, and shared runtime reuse for image, video and music generation. - Selective `include` paths or globs on recipe `download-model` steps, so a single-model lane fetches only the files its graph loads instead of every quantization in the model repository. Planned output reports the resolved `--include` command line. diff --git a/backends/catalog.json b/backends/catalog.json index 38e97f4..5293d38 100644 --- a/backends/catalog.json +++ b/backends/catalog.json @@ -1047,6 +1047,61 @@ "notes": "Per-request cancellation is checked between denoising steps. The backend retains its sole generation slot until the worker thread exits; encoding and decoding are not immediately interruptible." } }, + { + "id": "spark-audio", + "name": "Spark audio (NVIDIA Docker)", + "kind": "openai-compatible-server", + "description": "Source-built CUDA adapter serving one local Qwen3-TTS or Whisper checkpoint per container through the OpenAI speech and transcription APIs.", + "platforms": [ + "linux-arm64", + "linux-x64" + ], + "features": [ + "audio-speech", + "tts", + "tts-voice-clone", + "audio-transcription", + "stt", + "cuda", + "docker" + ], + "commands": [ + "docker", + "python3" + ], + "setup": [ + { + "id": "build-spark-audio", + "title": "Build or reuse the source-pinned Spark audio backend", + "action": "command", + "command": "python3", + "args": [ + "${repoRoot}/backends/spark-audio/install.py", + "--install-root", + "${installRoot}", + "--shim-dir", + "${shimDir}" + ], + "alwaysRun": true + } + ], + "server": { + "protocol": "openai", + "healthPath": "/health", + "speechPath": "/v1/audio/speech", + "transcriptionPath": "/v1/audio/transcriptions" + }, + "cancellation": { + "trigger": "client-disconnect", + "computeReclamation": "partial", + "modes": [ + "speech", + "transcription", + "streaming" + ], + "notes": "Buffered speech and transcription run to completion on the single GPU worker, which keeps the adapter's only slot until it exits. A streamed PCM voice clone on the optional faster engine stops at the next chunk and closes its generator." + } + }, { "id": "hear", "name": "LLooM Hear", diff --git a/backends/spark-audio/README.md b/backends/spark-audio/README.md index 9bda419..e110364 100644 --- a/backends/spark-audio/README.md +++ b/backends/spark-audio/README.md @@ -26,3 +26,5 @@ environment with FastAPI, httpx, numpy, soundfile, python-multipart, and pytest. These tests use fake models. Hardware acceptance must additionally prove PCM headers, named-profile routing, streaming onset, interruption/recovery, and transcription through the selected gateway on the actual host. + +Per-model recipes and setup commands are documented in [Spark audio](../../docs/spark-audio.md). diff --git a/backends/spark-audio/install.py b/backends/spark-audio/install.py new file mode 100644 index 0000000..7bf339a --- /dev/null +++ b/backends/spark-audio/install.py @@ -0,0 +1,66 @@ +#!/usr/bin/env python3 +"""Build/reuse the local image identified by its bundled source digest.""" +import argparse +import hashlib +import json +from pathlib import Path +import shutil +import subprocess +import sys + +ROOT = Path(__file__).resolve().parent + + +def source_digest(root=ROOT): + files = [root / name for name in ('Dockerfile', 'lloom_audio_cuda_server.py')] + digest = hashlib.sha256() + for file in sorted(files): + digest.update(file.relative_to(root).as_posix().encode() + b'\0') + digest.update(file.read_bytes() + b'\0') + return digest.hexdigest() + + +def image_name(): + return 'lloom/spark-audio:source-' + source_digest() + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--install-root', type=Path) + parser.add_argument('--shim-dir', type=Path) + parser.add_argument('--print-image', action='store_true') + args = parser.parse_args() + image = image_name() + if args.print_image: + print(image) + return + if args.install_root is None or args.shim_dir is None: + parser.error('--install-root and --shim-dir are required') + if sys.platform != 'linux': + parser.error('The CUDA backend requires Linux and NVIDIA Container Toolkit') + found = subprocess.run(['docker', 'image', 'inspect', image], capture_output=True, text=True) + expected = source_digest() + if found.returncode == 0: + labels = json.loads(found.stdout)[0].get('Config', {}).get('Labels', {}) or {} + if labels.get('dev.lloom.source-sha256') != expected: + raise SystemExit('Refusing an existing image with an unexpected source identity: ' + image) + print('Reusing ' + image) + else: + subprocess.run(['docker', 'build', '--label', 'dev.lloom.source-sha256=' + expected, + '--tag', image, str(ROOT)], check=True) + if not shutil.which('hf'): + venv = args.install_root / 'spark-audio' / 'huggingface' + if not (venv / 'bin/python').exists(): + subprocess.run([sys.executable, '-m', 'venv', str(venv)], check=True) + subprocess.run([str(venv / 'bin/python'), '-m', 'pip', 'install', 'huggingface-hub==1.7.1'], check=True) + args.shim_dir.mkdir(parents=True, exist_ok=True) + shim = args.shim_dir / 'hf' + if shim.exists() or shim.is_symlink(): + if shim.resolve() != (venv / 'bin/hf').resolve(): + raise SystemExit('Refusing to replace an unrelated hf shim') + else: + shim.symlink_to(venv / 'bin/hf') + + +if __name__ == '__main__': + main() diff --git a/docs/backends.md b/docs/backends.md index acc6478..27ee887 100644 --- a/docs/backends.md +++ b/docs/backends.md @@ -139,3 +139,5 @@ Authenticated hosted OpenAI-compatible providers use the same config-only unmana `lloom-host` publishes a lightweight backend index at `GET /v1/backends` and the full portable backend setup document at `GET /v1/backends/catalog`. Recipe authors and independent installers should use the catalog endpoint when they need setup actions, platform filters, server contracts, and idempotency guards rather than just the stable backend IDs. The `comfyui-media` backend builds a source-pinned NVIDIA Docker image for the [per-model media recipes](comfyui-media.md). Its idempotent build step uses `alwaysRun: true` so setup verifies the external image even when a previous installation was recorded as complete. + +The `spark-audio` backend builds the same kind of source-identified image for the [per-model speech and transcription recipes](spark-audio.md). Each recipe runs one checkpoint in its own container. diff --git a/docs/recipes.md b/docs/recipes.md index 292851a..fa8d7e8 100644 --- a/docs/recipes.md +++ b/docs/recipes.md @@ -341,3 +341,5 @@ Only the stable active file participates in planning and automatic recommendatio LLooM intentionally does not use stale model fallback aliases to make an index pass. Recipe `model` and `gatewayModel` values must be exact advertised IDs. Standalone image, video and music recipes with a shared ComfyUI backend are documented in [ComfyUI media](comfyui-media.md). + +Per-model NVIDIA speech and transcription recipes are documented in [Spark audio](spark-audio.md). diff --git a/docs/spark-audio.md b/docs/spark-audio.md new file mode 100644 index 0000000..714eb4d --- /dev/null +++ b/docs/spark-audio.md @@ -0,0 +1,55 @@ +# Spark audio speech and transcription recipes + +Each `linux-nvidia-spark-audio-*` recipe installs one checkpoint and runs it in +its own LLooM-managed container, port and runtime. All four recipes share only the +image, which is built locally from `backends/spark-audio`. + +| Recipe | Gateway model | Kind | +| ----------------------------------------------------- | -------------------------------------- | ------------------------- | +| `linux-nvidia-spark-audio-qwen3-tts-1-7b-customvoice` | `Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice` | speech, built-in speakers | +| `linux-nvidia-spark-audio-qwen3-tts-1-7b-base` | `Qwen/Qwen3-TTS-12Hz-1.7B-Base` | speech, voice cloning | +| `linux-nvidia-spark-audio-qwen3-tts-1-7b-voicedesign` | `Qwen/Qwen3-TTS-12Hz-1.7B-VoiceDesign` | speech, voice design | +| `linux-nvidia-spark-audio-whisper-large-v3-turbo` | `openai/whisper-large-v3-turbo` | transcription | + +## Install + +Use Linux on NVIDIA hardware with Docker, NVIDIA Container Toolkit and Python 3 +with venv support. Preview the complete plan, then apply and start: + +```sh +lloom setup --recipe linux-nvidia-spark-audio-qwen3-tts-1-7b-base --additive --json +lloom setup --recipe linux-nvidia-spark-audio-qwen3-tts-1-7b-base --additive --apply --yes --start +``` + +Repeat with another recipe ID to add a model. `--additive` preserves the existing +catalog and defaults. Speech models reserve 10 GiB and Whisper reserves 6 GiB in +LLooM's admission configuration. Each runtime accepts one request at a time and +is not kept warm. + +The backend installer builds `lloom/spark-audio:source-` from the digest-pinned +NGC PyTorch base, the Dockerfile and the adapter. The tag is the SHA-256 of those +build inputs; tests and documentation are excluded. Reapplying setup checks the +image's source label and reuses a matching image. Nothing is pulled from a private +registry. + +Downloads use immutable Hugging Face revisions with per-file SHA-256 checks. Each +container mounts only its own model directory, read-only, with Hugging Face +offline mode enabled. Backend ports stay bound to loopback; clients use the +authenticated LLooM gateway. + +openai-whisper cannot load the Transformers-format `openai/whisper-large-v3-turbo` +repository. The Whisper recipe therefore downloads OpenAI's original +`large-v3-turbo.pt` checkpoint from a byte-identical Hugging Face copy. Its SHA-256 +matches the value published in openai-whisper v20250625. + +Voice cloning uses the stock Qwen engine by default. Streamed PCM voice clones +and cancellation between chunks require `LLOOM_TTS_ENGINE=faster` in the +runtime's container environment; see the [adapter README](../backends/spark-audio/README.md). +Buffered synthesis and transcription run to completion after a client disconnects. + +Run CPU validation: + +```sh +python -m pytest -q backends/spark-audio/test_lloom_audio_cuda_server.py +node test/spark-audio-recipes.test.mjs +``` diff --git a/package.json b/package.json index 167b111..53796ae 100644 --- a/package.json +++ b/package.json @@ -64,8 +64,8 @@ "check": "npm run check:cluster && npm run check:js && npm run check:entity-stagger && npm run check:python && npm run check:shell && npm run check:spark-deploy && npm run check:workers", "check:cluster": "node --check src/cluster.mjs && node --check test/cluster.test.mjs", "check:entity-stagger": "node --check scripts/entity-stagger-benchmark.mjs && node --check test/entity-stagger-benchmark.test.mjs", - "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check src/runtime-residency.mjs && node --check src/dashboard-scene.mjs && node --check src/dashboard-presence.mjs && node --check src/dashboard-presence-client.mjs && node --check src/first-run.mjs && node --check src/first-run-page.mjs && node --check src/browser-setup.mjs && node --check src/runtime-preferences.mjs && node --check src/installation-jobs.mjs && node --check src/model-installation.mjs && node --check src/dashboard-installation.mjs && node --check src/runtime-memory-safety.mjs && node --check src/dashboard-memory.mjs && node --check src/runtime-memory-usage.mjs && node --check src/config-profiles.mjs && node --check test/config-profiles.test.mjs", - "check:python": "python3 -m py_compile backends/dspark-vllm/apply-patch-pack.py backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/vision_exp/*.py backends/qwen38-vllm/apply-ple-fp8-patch.py backends/qwen38-sglang/*.py backends/glm53-exl3/*.py backends/mlx-audio/lloom_audio_server.py backends/hear/lloom_hear_server.py backends/chatterbox/lloom_chatterbox_server.py patches/apply_mtplx_longctx_fix.py backends/comfyui-media/bridge/*.py backends/comfyui-media/graphs/*.py backends/comfyui-media/build/*.py backends/comfyui-media/install.py backends/qwen-image-diffusers/*.py backends/hear/test/test_hear_security.py backends/spark-audio/lloom_audio_cuda_server.py backends/spark-audio/test_lloom_audio_cuda_server.py src/darwin-memory-usage.py", + "check:js": "node --check bin/lloom.mjs && node --check bin/lloom-host.mjs && node --check src/backend-catalog.mjs && node --check src/benchmarks.mjs && node --check src/bootstrap.mjs && node --check src/client-integrations.mjs && node --check src/community-client.mjs && node --check src/community-site.mjs && node --check src/config.mjs && node --check src/dashboard.mjs && node --check src/doctor.mjs && node --check src/host-memory.mjs && node --check src/host-server.mjs && node --check src/init.mjs && node --check src/installer.mjs && node --check src/interchange.mjs && node --check src/machine-profile.mjs && node --check src/managed-environment.mjs && node --check src/model-acquisition.mjs && node --check src/model-intake.mjs && node --check src/model-removal.mjs && node --check src/onboarding.mjs && node --check src/recipe-index.mjs && node --check src/recipe-pack-export.mjs && node --check src/recipe-pack.mjs && node --check src/registry.mjs && node --check src/route-control.mjs && node --check src/recipes.mjs && node --check src/resource-fit.mjs && node --check src/runtime-manager.mjs && node --check src/runtime-policy.mjs && node --check src/server.mjs && node --check src/setup.mjs && node --check src/setup-status.mjs && node --check src/process-control.mjs && node --check src/security.mjs && node --check src/tts-catalog.mjs && node --check src/voice-profiles.mjs && node --check src/protocol/text.mjs && node --check src/protocol/reasoning-normalize.mjs && node --check src/protocol/reasoning-effort.mjs && node --check src/protocol/responses.mjs && node --check src/protocol/anthropic.mjs && node --check src/protocol/sse.mjs && node --check src/protocol/stream-anthropic.mjs && node --check src/protocol/stream-responses.mjs && node --check src/protocol/index.mjs && node --check scripts/apply-spark-route-catalog.mjs && node --check scripts/capture-live-metrics.mjs && node --check scripts/merge-live-metrics.mjs && node --check scripts/run-qualitative-chat-benchmark.mjs && node --check scripts/run-local-decision-benchmark.mjs && node --check scripts/check-community-deploy.mjs && node --check scripts/generate-client-configs.mjs && node --check scripts/check-interchange.mjs && node --check scripts/check-package.mjs && node --check scripts/release-lib.mjs && node --check scripts/build-release.mjs && node --check scripts/deploy-spark.mjs && node --check test/audio-generations.test.mjs && node --check test/community-host.test.mjs && node --check test/dashboard-status.test.mjs && node --check test/ds4fv-recipe.test.mjs && node --check test/dspark-hotfixes.test.mjs && node --check test/glm53-exl3-recipe.test.mjs && node --check test/host-memory.test.mjs && node --check test/metrics-persistence.test.mjs && node --check test/model-acquisition.test.mjs && node --check test/model-failover.test.mjs && node --check test/qwen38-vllm-recipe.test.mjs && node --check test/reasoning-normalize.test.mjs && node --check test/reasoning-effort.test.mjs && node --check test/resource-fit.test.mjs && node --check test/route-control.test.mjs && node --check test/spark-route-catalog.test.mjs && node --check test/runtime-watchdog.test.mjs && node --check test/tts-catalog.test.mjs && node --check test/voice-profiles.test.mjs && node --check test/chatterbox-recipe.test.mjs && node --check test/hear-recipe.test.mjs && node --check test/smoke.mjs && node --check src/runtime-supervisor.mjs && node --check test/runtime-supervisor.test.mjs && node --check test/comfyui-media.test.mjs && node --check test/spark-audio-recipes.test.mjs && node --check src/runtime-residency.mjs && node --check src/dashboard-scene.mjs && node --check src/dashboard-presence.mjs && node --check src/dashboard-presence-client.mjs && node --check src/first-run.mjs && node --check src/first-run-page.mjs && node --check src/browser-setup.mjs && node --check src/runtime-preferences.mjs && node --check src/installation-jobs.mjs && node --check src/model-installation.mjs && node --check src/dashboard-installation.mjs && node --check src/runtime-memory-safety.mjs && node --check src/dashboard-memory.mjs && node --check src/runtime-memory-usage.mjs && node --check src/config-profiles.mjs && node --check test/config-profiles.test.mjs", + "check:python": "python3 -m py_compile backends/dspark-vllm/apply-patch-pack.py backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.py backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/vision_exp/*.py backends/qwen38-vllm/apply-ple-fp8-patch.py backends/qwen38-sglang/*.py backends/glm53-exl3/*.py backends/mlx-audio/lloom_audio_server.py backends/hear/lloom_hear_server.py backends/chatterbox/lloom_chatterbox_server.py patches/apply_mtplx_longctx_fix.py backends/comfyui-media/bridge/*.py backends/comfyui-media/graphs/*.py backends/comfyui-media/build/*.py backends/comfyui-media/install.py backends/qwen-image-diffusers/*.py backends/hear/test/test_hear_security.py backends/spark-audio/install.py backends/spark-audio/lloom_audio_cuda_server.py backends/spark-audio/test_lloom_audio_cuda_server.py src/darwin-memory-usage.py", "check:shell": "bash -n backends/dspark-vllm/entrypoint.sh backends/dspark-vllm/runtime/build-hardened-image.sh backends/dspark-vllm/packs/miaai-dsv4flash-d1b76251-defaults/patches/*.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/apply-runtime.sh backends/dspark-vllm/packs/miaai-ds4fv-f5665e8/patches/*.sh backends/qwen38-vllm/entrypoint.sh backends/qwen38-sglang/entrypoint.sh backends/glm53-exl3/entrypoint.sh backends/mlx-audio/install.sh backends/hear/install.sh backends/chatterbox/install.sh clients/examples/lloom-claude clients/examples/lloom-codex clients/examples/lloom-hermes clients/examples/lloom-opencode clients/examples/lloom-zero scripts/remote-install-spark.sh scripts/remote-stage-spark-worker.sh", "check:spark-deploy": "node --check test/spark-deploy.test.mjs && node test/spark-deploy.test.mjs", "generate:clients": "node scripts/generate-client-configs.mjs", @@ -73,7 +73,7 @@ "community:check": "node scripts/check-community-deploy.mjs", "package:check": "node scripts/check-package.mjs", "test:adoption": "node test/resource-fit.test.mjs && node test/model-acquisition.test.mjs && node test/dashboard-status.test.mjs", - "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/openrouter-provider.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/config-profiles.test.mjs && node test/route-control.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/runtime-residency.test.mjs && npm run test:ux && npm run test:memory-safety && node --test test/dashboard-memory.test.mjs test/runtime-memory-usage.test.mjs && node --test test/speech-stream.test.mjs", + "test:unit": "node test/runtime-health-ownership.test.mjs && node test/long-prefill-transport.test.mjs && node test/sparkglm-managed.test.mjs && node test/protocol.test.mjs && node test/protocol-stream.test.mjs && node test/stream-metrics.test.mjs && node test/responses-codex.test.mjs && node test/security.test.mjs && node test/community-host.test.mjs && node test/openrouter-provider.test.mjs && node test/ds4fv-recipe.test.mjs && node test/dspark-hotfixes.test.mjs && node test/glm53-exl3-recipe.test.mjs && node test/host-memory.test.mjs && node test/config-profiles.test.mjs && node test/route-control.test.mjs && node test/runtime-policy.test.mjs && node --test test/rate-limit.test.mjs test/runtime-queue.test.mjs test/federated-priority.test.mjs test/runtime-supervisor.test.mjs && node test/interchange.test.mjs && node test/server-resilience.test.mjs && node test/metrics-persistence.test.mjs && node test/model-failover.test.mjs && node test/qwen38-vllm-recipe.test.mjs && node test/reasoning-normalize.test.mjs && node test/reasoning-effort.test.mjs && node test/route-control.test.mjs && node test/spark-route-catalog.test.mjs && node test/chat-lane-canary.test.mjs && node test/runtime-watchdog.test.mjs && node test/tts-catalog.test.mjs && node test/voice-profiles.test.mjs && node test/chatterbox-recipe.test.mjs && node test/hear-recipe.test.mjs && npm run test:adoption && node --test test/performance-sampler.test.mjs && node --test test/audio-generations.test.mjs && node test/video-providers.test.mjs && node --test test/web-functions.test.mjs && node --test test/config-mutation.test.mjs && node --test test/model-maintenance.test.mjs && node --test test/model-maintenance-control.test.mjs && node --test test/route-maintenance-concurrency.test.mjs && node --test test/model-maintenance-integration.test.mjs && npm run test:workers && node test/comfyui-media.test.mjs && node test/qwen-image-diffusers.test.mjs && node test/spark-audio-recipes.test.mjs && node test/runtime-residency.test.mjs && npm run test:ux && npm run test:memory-safety && node --test test/dashboard-memory.test.mjs test/runtime-memory-usage.test.mjs && node --test test/speech-stream.test.mjs", "test:cluster": "node test/cluster.test.mjs", "test": "npm run test:cluster && npm run test:unit && npm run test:entity-stagger && npm run smoke && npm run community:check", "test:entity-stagger": "node test/entity-stagger-benchmark.test.mjs", diff --git a/recipes/index.json b/recipes/index.json index b613189..f57e7c2 100644 --- a/recipes/index.json +++ b/recipes/index.json @@ -1121,6 +1121,118 @@ ], "name": "Qwen Image 2.1 generation and editing (Diffusers, NVIDIA)" }, + { + "id": "linux-nvidia-spark-audio-qwen3-tts-1-7b-base", + "path": "linux-nvidia-spark-audio-qwen3-tts-1-7b-base.json", + "currentVersion": 1, + "versions": [ + { + "version": 1, + "path": "linux-nvidia-spark-audio-qwen3-tts-1-7b-base.json", + "status": "current" + } + ], + "name": "Qwen3-TTS 1.7B Base (Spark audio, NVIDIA)", + "summary": "Qwen3-TTS 1.7B Base in its own source-built CUDA container: zero-shot voice cloning from a short inline reference clip through the OpenAI speech API.", + "tags": ["cuda", "nvidia", "dgx-spark", "gb10", "docker", "tts", "audio-speech", "qwen3-tts", "base", "bf16"], + "capabilities": ["audio-speech", "tts", "tts-voice-clone", "qwen3-tts"], + "source": { + "type": "local", + "url": "recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-base.json" + } + }, + { + "id": "linux-nvidia-spark-audio-qwen3-tts-1-7b-customvoice", + "path": "linux-nvidia-spark-audio-qwen3-tts-1-7b-customvoice.json", + "currentVersion": 1, + "versions": [ + { + "version": 1, + "path": "linux-nvidia-spark-audio-qwen3-tts-1-7b-customvoice.json", + "status": "current" + } + ], + "name": "Qwen3-TTS 1.7B CustomVoice (Spark audio, NVIDIA)", + "summary": "Qwen3-TTS 1.7B CustomVoice in its own source-built CUDA container: nine built-in speakers with optional style instructions through the OpenAI speech API.", + "tags": [ + "cuda", + "nvidia", + "dgx-spark", + "gb10", + "docker", + "tts", + "audio-speech", + "qwen3-tts", + "customvoice", + "bf16" + ], + "capabilities": ["audio-speech", "tts", "tts-custom-voice", "tts-style-instruct", "qwen3-tts"], + "source": { + "type": "local", + "url": "recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-customvoice.json" + } + }, + { + "id": "linux-nvidia-spark-audio-qwen3-tts-1-7b-voicedesign", + "path": "linux-nvidia-spark-audio-qwen3-tts-1-7b-voicedesign.json", + "currentVersion": 1, + "versions": [ + { + "version": 1, + "path": "linux-nvidia-spark-audio-qwen3-tts-1-7b-voicedesign.json", + "status": "current" + } + ], + "name": "Qwen3-TTS 1.7B VoiceDesign (Spark audio, NVIDIA)", + "summary": "Qwen3-TTS 1.7B VoiceDesign in its own source-built CUDA container: describe the voice in natural-language instructions through the OpenAI speech API.", + "tags": [ + "cuda", + "nvidia", + "dgx-spark", + "gb10", + "docker", + "tts", + "audio-speech", + "qwen3-tts", + "voicedesign", + "bf16" + ], + "capabilities": ["audio-speech", "tts", "tts-voice-design", "qwen3-tts"], + "source": { + "type": "local", + "url": "recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-voicedesign.json" + } + }, + { + "id": "linux-nvidia-spark-audio-whisper-large-v3-turbo", + "path": "linux-nvidia-spark-audio-whisper-large-v3-turbo.json", + "currentVersion": 1, + "versions": [ + { + "version": 1, + "path": "linux-nvidia-spark-audio-whisper-large-v3-turbo.json", + "status": "current" + } + ], + "name": "Whisper large-v3-turbo transcription (Spark audio, NVIDIA)", + "summary": "OpenAI Whisper large-v3-turbo in its own source-built CUDA container for OpenAI-compatible speech-to-text transcription.", + "tags": [ + "cuda", + "nvidia", + "dgx-spark", + "gb10", + "docker", + "stt", + "audio-transcription", + "whisper", + "whisper-large-v3-turbo" + ], + "capabilities": ["audio-transcription", "stt"], + "source": { + "type": "local", + "url": "recipes/linux-nvidia-spark-audio-whisper-large-v3-turbo.json" + } + }, { "id": "lloom-hear", "path": "lloom-hear.json", diff --git a/recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-base.json b/recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-base.json new file mode 100644 index 0000000..a89bc31 --- /dev/null +++ b/recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-base.json @@ -0,0 +1,255 @@ +{ + "$schema": "https://lloom.dev/schemas/recipe.v1.schema.json", + "schemaVersion": 1, + "profile": "https://lloom.dev/profiles/interchange/v1", + "id": "linux-nvidia-spark-audio-qwen3-tts-1-7b-base", + "name": "Qwen3-TTS 1.7B Base (Spark audio, NVIDIA)", + "version": 1, + "summary": "Qwen3-TTS 1.7B Base in its own source-built CUDA container: zero-shot voice cloning from a short inline reference clip through the OpenAI speech API.", + "license": { + "id": "MIT", + "name": "Recipe metadata only; Qwen3-TTS weights are under the Apache-2.0 license" + }, + "provenance": { + "createdBy": "lloom", + "generatedBy": "manual-curated", + "source": "Qwen/Qwen3-TTS-12Hz-1.7B-Base fd4b254389122332181a7c3db7f27e918eec64e3; qwen-tts 0.1.1 and Transformers 4.57.3 on a digest-pinned NGC PyTorch base; immutable revision and per-file SHA-256 below." + }, + "links": [ + { + "rel": "source", + "href": "https://huggingface.co/Qwen/Qwen3-TTS-12Hz-1.7B-Base/tree/fd4b254389122332181a7c3db7f27e918eec64e3" + }, + { + "rel": "runtime", + "href": "https://github.com/QwenLM/Qwen3-TTS" + } + ], + "keywords": [ + "cuda", + "nvidia", + "dgx-spark", + "gb10", + "docker", + "tts", + "audio-speech", + "qwen3-tts", + "base", + "bf16" + ], + "capabilities": [ + "audio-speech", + "tts", + "tts-voice-clone", + "qwen3-tts" + ], + "hardware": { + "family": "Linux NVIDIA CUDA", + "testedMachine": "DGX Spark / GB10 class NVIDIA system", + "notes": [ + "One synthesis at a time; a concurrent request to the adapter is refused with 429 while LLooM queues through maxActiveRequests 1.", + "BF16 weights on cuda:0 with SDPA attention. Cold model loading completes before the health endpoint answers.", + "Reference audio is accepted only inline (multipart upload or data: base64) and is bounded in size and duration; the adapter never opens caller-supplied paths or URLs. Only clone voices you own or have explicit permission to use.", + "Separate single-model runtime; each spark-audio recipe adds its own container and port and shares only the locally built image.", + "NVIDIA Container Toolkit with a driver compatible with the digest-pinned NGC PyTorch base image required." + ] + }, + "requirements": { + "platforms": [ + "linux-x64", + "linux-arm64" + ], + "memoryGb": 10, + "diskGb": 25, + "commands": [ + "docker", + "python3" + ], + "accelerators": [ + "cuda", + "nvidia-gpu" + ] + }, + "backend": { + "id": "spark-audio", + "name": "Spark audio", + "type": "openai-compatible-server", + "features": [ + "audio-speech", + "tts", + "tts-voice-clone", + "audio-transcription", + "stt", + "cuda", + "docker" + ] + }, + "setup": { + "steps": [ + { + "id": "check-docker", + "title": "Check Docker CLI", + "action": "check-command", + "command": "docker", + "args": [ + "--version" + ] + }, + { + "id": "download-qwen3-tts-1-7b-base", + "title": "Download Qwen/Qwen3-TTS-12Hz-1.7B-Base files", + "action": "download-model", + "provider": "huggingface", + "model": "Qwen/Qwen3-TTS-12Hz-1.7B-Base", + "revision": "fd4b254389122332181a7c3db7f27e918eec64e3", + "include": [ + "config.json", + "generation_config.json", + "merges.txt", + "model.safetensors", + "preprocessor_config.json", + "speech_tokenizer/config.json", + "speech_tokenizer/configuration.json", + "speech_tokenizer/model.safetensors", + "speech_tokenizer/preprocessor_config.json", + "tokenizer_config.json", + "vocab.json" + ], + "downloadSizeBytes": 4544170364, + "integrity": { + "files": [ + { + "path": "config.json", + "sizeBytes": 4494, + "sha256": "b4f01752d15a488abde3e1ab44723ae4f4b9e68a4037257b098b3737893cc1f9" + }, + { + "path": "generation_config.json", + "sizeBytes": 245, + "sha256": "f1b90b4513f3b34c62851049e2492d7b4c5940daf1276f89c82b8ef04127f3aa" + }, + { + "path": "merges.txt", + "sizeBytes": 1671839, + "sha256": "599bab54075088774b1733fde865d5bd747cbcc7a547c5bc12610e874e26f5e3" + }, + { + "path": "model.safetensors", + "sizeBytes": 3857413744, + "sha256": "38fc7fc51c5e776e840414b6fd443962e9411b9654888fd7913e4da643cb857c" + }, + { + "path": "preprocessor_config.json", + "sizeBytes": 127, + "sha256": "efdde1022ea9d76928bf7a9cd53139138f5ba2e466e837f08f6105ab1af1c119" + }, + { + "path": "speech_tokenizer/config.json", + "sizeBytes": 2336, + "sha256": "ee65bb901c876664ab8707c487157aa1a6ee57c65969b28fb5ec9dc211e68167" + }, + { + "path": "speech_tokenizer/configuration.json", + "sizeBytes": 76, + "sha256": "6bc26d64eb5024b4d1dab5a52371958b429256d6c9d59787f1f5294a54e0cebd" + }, + { + "path": "speech_tokenizer/model.safetensors", + "sizeBytes": 682293092, + "sha256": "836b7b357f5ea43e889936a3709af68dfe3751881acefe4ecf0dbd30ba571258" + }, + { + "path": "speech_tokenizer/preprocessor_config.json", + "sizeBytes": 234, + "sha256": "fcb3805e597e786d4067706e602f6688524640f8d3396790e2e09b5942fcbdfb" + }, + { + "path": "tokenizer_config.json", + "sizeBytes": 7344, + "sha256": "dc3c31c3bdaedd5016382bb3cbe07323026775ad51f5a4fb564505992ae4a670" + }, + { + "path": "vocab.json", + "sizeBytes": 2776833, + "sha256": "ca10d7e9fb3ed18575dd1e277a2579c16d108e32f27439684afa0e10b1440910" + } + ] + } + } + ] + }, + "models": [ + { + "role": "qwen3-tts-1-7b-base", + "model": "Qwen/Qwen3-TTS-12Hz-1.7B-Base", + "gatewayModel": "Qwen/Qwen3-TTS-12Hz-1.7B-Base", + "runtime": "spark-audio-qwen3-tts-1-7b-base", + "backendConfig": "spark-audio-qwen3-tts-1-7b-base", + "kind": "audio_speech", + "input": [ + "text" + ], + "output": [ + "audio" + ], + "capabilities": [ + "audio-speech", + "tts", + "tts-voice-clone", + "qwen3-tts" + ], + "tts": { + "family": "qwen3-tts", + "mode": "voice_clone" + }, + "settings": { + "memoryGb": 10, + "priority": 41, + "keepWarm": false, + "maxActiveRequests": 1, + "startupTimeoutMs": 900000, + "runtime": { + "warmup": false, + "adapter": "docker", + "management": "managed", + "containerName": "lloom-spark-audio-qwen3-tts-1-7b-base", + "healthUrl": "http://127.0.0.1:${port}/health", + "bootstrap": { + "adapter": "docker", + "image": "lloom/spark-audio:source-34b229132a097ab793a21c708a3b99be02ec3adcebd42d81c0305e87d5971c73", + "pull": false, + "createArgs": [ + "--restart", + "unless-stopped", + "--gpus", + "all", + "--ipc", + "host", + "-p", + "127.0.0.1:${port}:8000", + "-v", + "${modelPath}:/models/model:ro", + "-e", + "HF_HUB_OFFLINE=1", + "-e", + "TRANSFORMERS_OFFLINE=1" + ], + "command": [ + "--kind", + "tts", + "--model", + "Qwen/Qwen3-TTS-12Hz-1.7B-Base", + "--model-path", + "/models/model", + "--host", + "0.0.0.0", + "--port", + "8000" + ] + } + }, + "timeoutMs": 1800000 + } + } + ] +} diff --git a/recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-customvoice.json b/recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-customvoice.json new file mode 100644 index 0000000..0e3bded --- /dev/null +++ b/recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-customvoice.json @@ -0,0 +1,256 @@ +{ + "$schema": "https://lloom.dev/schemas/recipe.v1.schema.json", + "schemaVersion": 1, + "profile": "https://lloom.dev/profiles/interchange/v1", + "id": "linux-nvidia-spark-audio-qwen3-tts-1-7b-customvoice", + "name": "Qwen3-TTS 1.7B CustomVoice (Spark audio, NVIDIA)", + "version": 1, + "summary": "Qwen3-TTS 1.7B CustomVoice in its own source-built CUDA container: nine built-in speakers with optional style instructions through the OpenAI speech API.", + "license": { + "id": "MIT", + "name": "Recipe metadata only; Qwen3-TTS weights are under the Apache-2.0 license" + }, + "provenance": { + "createdBy": "lloom", + "generatedBy": "manual-curated", + "source": "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice 0c0e3051f131929182e2c023b9537f8b1c68adfe; qwen-tts 0.1.1 and Transformers 4.57.3 on a digest-pinned NGC PyTorch base; immutable revision and per-file SHA-256 below." + }, + "links": [ + { + "rel": "source", + "href": "https://huggingface.co/Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice/tree/0c0e3051f131929182e2c023b9537f8b1c68adfe" + }, + { + "rel": "runtime", + "href": "https://github.com/QwenLM/Qwen3-TTS" + } + ], + "keywords": [ + "cuda", + "nvidia", + "dgx-spark", + "gb10", + "docker", + "tts", + "audio-speech", + "qwen3-tts", + "customvoice", + "bf16" + ], + "capabilities": [ + "audio-speech", + "tts", + "tts-custom-voice", + "tts-style-instruct", + "qwen3-tts" + ], + "hardware": { + "family": "Linux NVIDIA CUDA", + "testedMachine": "DGX Spark / GB10 class NVIDIA system", + "notes": [ + "One synthesis at a time; a concurrent request to the adapter is refused with 429 while LLooM queues through maxActiveRequests 1.", + "BF16 weights on cuda:0 with SDPA attention. Cold model loading completes before the health endpoint answers.", + "Separate single-model runtime; each spark-audio recipe adds its own container and port and shares only the locally built image.", + "NVIDIA Container Toolkit with a driver compatible with the digest-pinned NGC PyTorch base image required." + ] + }, + "requirements": { + "platforms": [ + "linux-x64", + "linux-arm64" + ], + "memoryGb": 10, + "diskGb": 25, + "commands": [ + "docker", + "python3" + ], + "accelerators": [ + "cuda", + "nvidia-gpu" + ] + }, + "backend": { + "id": "spark-audio", + "name": "Spark audio", + "type": "openai-compatible-server", + "features": [ + "audio-speech", + "tts", + "tts-voice-clone", + "audio-transcription", + "stt", + "cuda", + "docker" + ] + }, + "setup": { + "steps": [ + { + "id": "check-docker", + "title": "Check Docker CLI", + "action": "check-command", + "command": "docker", + "args": [ + "--version" + ] + }, + { + "id": "download-qwen3-tts-1-7b-customvoice", + "title": "Download Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice files", + "action": "download-model", + "provider": "huggingface", + "model": "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice", + "revision": "0c0e3051f131929182e2c023b9537f8b1c68adfe", + "include": [ + "config.json", + "generation_config.json", + "merges.txt", + "model.safetensors", + "preprocessor_config.json", + "speech_tokenizer/config.json", + "speech_tokenizer/configuration.json", + "speech_tokenizer/model.safetensors", + "speech_tokenizer/preprocessor_config.json", + "tokenizer_config.json", + "vocab.json" + ], + "downloadSizeBytes": 4520159586, + "integrity": { + "files": [ + { + "path": "config.json", + "sizeBytes": 4908, + "sha256": "17a07f527a1c25ea30b4e023a184482a23d3e279d697b1dc81b1bde498d29cf9" + }, + { + "path": "generation_config.json", + "sizeBytes": 245, + "sha256": "f1b90b4513f3b34c62851049e2492d7b4c5940daf1276f89c82b8ef04127f3aa" + }, + { + "path": "merges.txt", + "sizeBytes": 1671839, + "sha256": "599bab54075088774b1733fde865d5bd747cbcc7a547c5bc12610e874e26f5e3" + }, + { + "path": "model.safetensors", + "sizeBytes": 3833402552, + "sha256": "38b1d5971bdbd982b561cccec982669a53b0537c3cf5e9bd4778ed07bb2f5137" + }, + { + "path": "preprocessor_config.json", + "sizeBytes": 127, + "sha256": "efdde1022ea9d76928bf7a9cd53139138f5ba2e466e837f08f6105ab1af1c119" + }, + { + "path": "speech_tokenizer/config.json", + "sizeBytes": 2336, + "sha256": "ee65bb901c876664ab8707c487157aa1a6ee57c65969b28fb5ec9dc211e68167" + }, + { + "path": "speech_tokenizer/configuration.json", + "sizeBytes": 76, + "sha256": "6bc26d64eb5024b4d1dab5a52371958b429256d6c9d59787f1f5294a54e0cebd" + }, + { + "path": "speech_tokenizer/model.safetensors", + "sizeBytes": 682293092, + "sha256": "836b7b357f5ea43e889936a3709af68dfe3751881acefe4ecf0dbd30ba571258" + }, + { + "path": "speech_tokenizer/preprocessor_config.json", + "sizeBytes": 234, + "sha256": "fcb3805e597e786d4067706e602f6688524640f8d3396790e2e09b5942fcbdfb" + }, + { + "path": "tokenizer_config.json", + "sizeBytes": 7344, + "sha256": "dc3c31c3bdaedd5016382bb3cbe07323026775ad51f5a4fb564505992ae4a670" + }, + { + "path": "vocab.json", + "sizeBytes": 2776833, + "sha256": "ca10d7e9fb3ed18575dd1e277a2579c16d108e32f27439684afa0e10b1440910" + } + ] + } + } + ] + }, + "models": [ + { + "role": "qwen3-tts-1-7b-customvoice", + "model": "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice", + "gatewayModel": "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice", + "runtime": "spark-audio-qwen3-tts-1-7b-customvoice", + "backendConfig": "spark-audio-qwen3-tts-1-7b-customvoice", + "kind": "audio_speech", + "input": [ + "text" + ], + "output": [ + "audio" + ], + "capabilities": [ + "audio-speech", + "tts", + "tts-custom-voice", + "tts-style-instruct", + "qwen3-tts" + ], + "tts": { + "family": "qwen3-tts", + "mode": "custom_voice" + }, + "settings": { + "memoryGb": 10, + "priority": 42, + "keepWarm": false, + "maxActiveRequests": 1, + "startupTimeoutMs": 900000, + "runtime": { + "warmup": false, + "adapter": "docker", + "management": "managed", + "containerName": "lloom-spark-audio-qwen3-tts-1-7b-customvoice", + "healthUrl": "http://127.0.0.1:${port}/health", + "bootstrap": { + "adapter": "docker", + "image": "lloom/spark-audio:source-34b229132a097ab793a21c708a3b99be02ec3adcebd42d81c0305e87d5971c73", + "pull": false, + "createArgs": [ + "--restart", + "unless-stopped", + "--gpus", + "all", + "--ipc", + "host", + "-p", + "127.0.0.1:${port}:8000", + "-v", + "${modelPath}:/models/model:ro", + "-e", + "HF_HUB_OFFLINE=1", + "-e", + "TRANSFORMERS_OFFLINE=1" + ], + "command": [ + "--kind", + "tts", + "--model", + "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice", + "--model-path", + "/models/model", + "--host", + "0.0.0.0", + "--port", + "8000" + ] + } + }, + "timeoutMs": 1800000 + } + } + ] +} diff --git a/recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-voicedesign.json b/recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-voicedesign.json new file mode 100644 index 0000000..67dbe9d --- /dev/null +++ b/recipes/linux-nvidia-spark-audio-qwen3-tts-1-7b-voicedesign.json @@ -0,0 +1,254 @@ +{ + "$schema": "https://lloom.dev/schemas/recipe.v1.schema.json", + "schemaVersion": 1, + "profile": "https://lloom.dev/profiles/interchange/v1", + "id": "linux-nvidia-spark-audio-qwen3-tts-1-7b-voicedesign", + "name": "Qwen3-TTS 1.7B VoiceDesign (Spark audio, NVIDIA)", + "version": 1, + "summary": "Qwen3-TTS 1.7B VoiceDesign in its own source-built CUDA container: describe the voice in natural-language instructions through the OpenAI speech API.", + "license": { + "id": "MIT", + "name": "Recipe metadata only; Qwen3-TTS weights are under the Apache-2.0 license" + }, + "provenance": { + "createdBy": "lloom", + "generatedBy": "manual-curated", + "source": "Qwen/Qwen3-TTS-12Hz-1.7B-VoiceDesign 5ecdb67327fd37bb2e042aab12ff7391903235d3; qwen-tts 0.1.1 and Transformers 4.57.3 on a digest-pinned NGC PyTorch base; immutable revision and per-file SHA-256 below." + }, + "links": [ + { + "rel": "source", + "href": "https://huggingface.co/Qwen/Qwen3-TTS-12Hz-1.7B-VoiceDesign/tree/5ecdb67327fd37bb2e042aab12ff7391903235d3" + }, + { + "rel": "runtime", + "href": "https://github.com/QwenLM/Qwen3-TTS" + } + ], + "keywords": [ + "cuda", + "nvidia", + "dgx-spark", + "gb10", + "docker", + "tts", + "audio-speech", + "qwen3-tts", + "voicedesign", + "bf16" + ], + "capabilities": [ + "audio-speech", + "tts", + "tts-voice-design", + "qwen3-tts" + ], + "hardware": { + "family": "Linux NVIDIA CUDA", + "testedMachine": "DGX Spark / GB10 class NVIDIA system", + "notes": [ + "One synthesis at a time; a concurrent request to the adapter is refused with 429 while LLooM queues through maxActiveRequests 1.", + "BF16 weights on cuda:0 with SDPA attention. Cold model loading completes before the health endpoint answers.", + "Separate single-model runtime; each spark-audio recipe adds its own container and port and shares only the locally built image.", + "NVIDIA Container Toolkit with a driver compatible with the digest-pinned NGC PyTorch base image required." + ] + }, + "requirements": { + "platforms": [ + "linux-x64", + "linux-arm64" + ], + "memoryGb": 10, + "diskGb": 25, + "commands": [ + "docker", + "python3" + ], + "accelerators": [ + "cuda", + "nvidia-gpu" + ] + }, + "backend": { + "id": "spark-audio", + "name": "Spark audio", + "type": "openai-compatible-server", + "features": [ + "audio-speech", + "tts", + "tts-voice-clone", + "audio-transcription", + "stt", + "cuda", + "docker" + ] + }, + "setup": { + "steps": [ + { + "id": "check-docker", + "title": "Check Docker CLI", + "action": "check-command", + "command": "docker", + "args": [ + "--version" + ] + }, + { + "id": "download-qwen3-tts-1-7b-voicedesign", + "title": "Download Qwen/Qwen3-TTS-12Hz-1.7B-VoiceDesign files", + "action": "download-model", + "provider": "huggingface", + "model": "Qwen/Qwen3-TTS-12Hz-1.7B-VoiceDesign", + "revision": "5ecdb67327fd37bb2e042aab12ff7391903235d3", + "include": [ + "config.json", + "generation_config.json", + "merges.txt", + "model.safetensors", + "preprocessor_config.json", + "speech_tokenizer/config.json", + "speech_tokenizer/configuration.json", + "speech_tokenizer/model.safetensors", + "speech_tokenizer/preprocessor_config.json", + "tokenizer_config.json", + "vocab.json" + ], + "downloadSizeBytes": 4520159099, + "integrity": { + "files": [ + { + "path": "config.json", + "sizeBytes": 4421, + "sha256": "aecd2cc4c1fe9edef1cb7ca7c401685a43879ad43f3f9e883f1c6760b61731e0" + }, + { + "path": "generation_config.json", + "sizeBytes": 245, + "sha256": "f1b90b4513f3b34c62851049e2492d7b4c5940daf1276f89c82b8ef04127f3aa" + }, + { + "path": "merges.txt", + "sizeBytes": 1671839, + "sha256": "599bab54075088774b1733fde865d5bd747cbcc7a547c5bc12610e874e26f5e3" + }, + { + "path": "model.safetensors", + "sizeBytes": 3833402552, + "sha256": "391e8db219f292c515297cdceeb43e4eae67cdde35fa57e79a6a8a532fca0522" + }, + { + "path": "preprocessor_config.json", + "sizeBytes": 127, + "sha256": "efdde1022ea9d76928bf7a9cd53139138f5ba2e466e837f08f6105ab1af1c119" + }, + { + "path": "speech_tokenizer/config.json", + "sizeBytes": 2336, + "sha256": "ee65bb901c876664ab8707c487157aa1a6ee57c65969b28fb5ec9dc211e68167" + }, + { + "path": "speech_tokenizer/configuration.json", + "sizeBytes": 76, + "sha256": "6bc26d64eb5024b4d1dab5a52371958b429256d6c9d59787f1f5294a54e0cebd" + }, + { + "path": "speech_tokenizer/model.safetensors", + "sizeBytes": 682293092, + "sha256": "836b7b357f5ea43e889936a3709af68dfe3751881acefe4ecf0dbd30ba571258" + }, + { + "path": "speech_tokenizer/preprocessor_config.json", + "sizeBytes": 234, + "sha256": "fcb3805e597e786d4067706e602f6688524640f8d3396790e2e09b5942fcbdfb" + }, + { + "path": "tokenizer_config.json", + "sizeBytes": 7344, + "sha256": "dc3c31c3bdaedd5016382bb3cbe07323026775ad51f5a4fb564505992ae4a670" + }, + { + "path": "vocab.json", + "sizeBytes": 2776833, + "sha256": "ca10d7e9fb3ed18575dd1e277a2579c16d108e32f27439684afa0e10b1440910" + } + ] + } + } + ] + }, + "models": [ + { + "role": "qwen3-tts-1-7b-voicedesign", + "model": "Qwen/Qwen3-TTS-12Hz-1.7B-VoiceDesign", + "gatewayModel": "Qwen/Qwen3-TTS-12Hz-1.7B-VoiceDesign", + "runtime": "spark-audio-qwen3-tts-1-7b-voicedesign", + "backendConfig": "spark-audio-qwen3-tts-1-7b-voicedesign", + "kind": "audio_speech", + "input": [ + "text" + ], + "output": [ + "audio" + ], + "capabilities": [ + "audio-speech", + "tts", + "tts-voice-design", + "qwen3-tts" + ], + "tts": { + "family": "qwen3-tts", + "mode": "voice_design" + }, + "settings": { + "memoryGb": 10, + "priority": 40, + "keepWarm": false, + "maxActiveRequests": 1, + "startupTimeoutMs": 900000, + "runtime": { + "warmup": false, + "adapter": "docker", + "management": "managed", + "containerName": "lloom-spark-audio-qwen3-tts-1-7b-voicedesign", + "healthUrl": "http://127.0.0.1:${port}/health", + "bootstrap": { + "adapter": "docker", + "image": "lloom/spark-audio:source-34b229132a097ab793a21c708a3b99be02ec3adcebd42d81c0305e87d5971c73", + "pull": false, + "createArgs": [ + "--restart", + "unless-stopped", + "--gpus", + "all", + "--ipc", + "host", + "-p", + "127.0.0.1:${port}:8000", + "-v", + "${modelPath}:/models/model:ro", + "-e", + "HF_HUB_OFFLINE=1", + "-e", + "TRANSFORMERS_OFFLINE=1" + ], + "command": [ + "--kind", + "tts", + "--model", + "Qwen/Qwen3-TTS-12Hz-1.7B-VoiceDesign", + "--model-path", + "/models/model", + "--host", + "0.0.0.0", + "--port", + "8000" + ] + } + }, + "timeoutMs": 1800000 + } + } + ] +} diff --git a/recipes/linux-nvidia-spark-audio-whisper-large-v3-turbo.json b/recipes/linux-nvidia-spark-audio-whisper-large-v3-turbo.json new file mode 100644 index 0000000..ff4b61c --- /dev/null +++ b/recipes/linux-nvidia-spark-audio-whisper-large-v3-turbo.json @@ -0,0 +1,194 @@ +{ + "$schema": "https://lloom.dev/schemas/recipe.v1.schema.json", + "schemaVersion": 1, + "profile": "https://lloom.dev/profiles/interchange/v1", + "id": "linux-nvidia-spark-audio-whisper-large-v3-turbo", + "name": "Whisper large-v3-turbo transcription (Spark audio, NVIDIA)", + "version": 1, + "summary": "OpenAI Whisper large-v3-turbo in its own source-built CUDA container for OpenAI-compatible speech-to-text transcription.", + "license": { + "id": "MIT", + "name": "Recipe metadata only; Whisper weights are under the MIT license" + }, + "provenance": { + "createdBy": "lloom", + "generatedBy": "manual-curated", + "source": "OpenAI large-v3-turbo.pt with the SHA-256 published in openai-whisper v20250625 (aff26ae408abcba5fbf8813c21e62b0941638c5f6eebfb145be0c9839262a19a), fetched from the byte-identical Hugging Face copy TheScenery/whisper-models 071f7d44e122e9964924a21c5bf8381d37595055; openai-whisper 20250625 on a digest-pinned NGC PyTorch base." + }, + "links": [ + { + "rel": "source", + "href": "https://huggingface.co/TheScenery/whisper-models/tree/071f7d44e122e9964924a21c5bf8381d37595055" + }, + { + "rel": "runtime", + "href": "https://github.com/openai/whisper/tree/v20250625" + }, + { + "rel": "model", + "href": "https://huggingface.co/openai/whisper-large-v3-turbo" + } + ], + "keywords": [ + "cuda", + "nvidia", + "dgx-spark", + "gb10", + "docker", + "stt", + "audio-transcription", + "whisper", + "whisper-large-v3-turbo" + ], + "capabilities": [ + "audio-transcription", + "stt" + ], + "hardware": { + "family": "Linux NVIDIA CUDA", + "testedMachine": "DGX Spark / GB10 class NVIDIA system", + "notes": [ + "One transcription at a time; a concurrent request to the adapter is refused with 429 while LLooM queues through maxActiveRequests 1.", + "Uploads are limited to 16 MiB and 10 minutes of decoded audio; json, text and verbose_json responses.", + "openai-whisper loads the original OpenAI checkpoint file; the Transformers-format openai/whisper-large-v3-turbo repository is not loadable by this adapter.", + "Separate single-model runtime; each spark-audio recipe adds its own container and port and shares only the locally built image.", + "NVIDIA Container Toolkit with a driver compatible with the digest-pinned NGC PyTorch base image required." + ] + }, + "requirements": { + "platforms": [ + "linux-x64", + "linux-arm64" + ], + "memoryGb": 6, + "diskGb": 22, + "commands": [ + "docker", + "python3" + ], + "accelerators": [ + "cuda", + "nvidia-gpu" + ] + }, + "backend": { + "id": "spark-audio", + "name": "Spark audio", + "type": "openai-compatible-server", + "features": [ + "audio-speech", + "tts", + "tts-voice-clone", + "audio-transcription", + "stt", + "cuda", + "docker" + ] + }, + "setup": { + "steps": [ + { + "id": "check-docker", + "title": "Check Docker CLI", + "action": "check-command", + "command": "docker", + "args": [ + "--version" + ] + }, + { + "id": "download-whisper-large-v3-turbo", + "title": "Download TheScenery/whisper-models files for openai/whisper-large-v3-turbo", + "action": "download-model", + "provider": "huggingface", + "model": "TheScenery/whisper-models", + "revision": "071f7d44e122e9964924a21c5bf8381d37595055", + "include": [ + "large-v3-turbo.pt" + ], + "downloadSizeBytes": 1617941637, + "integrity": { + "files": [ + { + "path": "large-v3-turbo.pt", + "sizeBytes": 1617941637, + "sha256": "aff26ae408abcba5fbf8813c21e62b0941638c5f6eebfb145be0c9839262a19a" + } + ] + } + } + ] + }, + "models": [ + { + "role": "whisper-large-v3-turbo", + "model": "TheScenery/whisper-models", + "gatewayModel": "openai/whisper-large-v3-turbo", + "upstreamModel": "openai/whisper-large-v3-turbo", + "runtime": "spark-audio-whisper-large-v3-turbo", + "backendConfig": "spark-audio-whisper-large-v3-turbo", + "kind": "audio_transcription", + "input": [ + "audio" + ], + "output": [ + "text" + ], + "capabilities": [ + "audio-transcription", + "stt" + ], + "stt": { + "family": "whisper" + }, + "settings": { + "memoryGb": 6, + "priority": 45, + "keepWarm": false, + "maxActiveRequests": 1, + "startupTimeoutMs": 900000, + "runtime": { + "warmup": false, + "adapter": "docker", + "management": "managed", + "containerName": "lloom-spark-audio-whisper-large-v3-turbo", + "healthUrl": "http://127.0.0.1:${port}/health", + "bootstrap": { + "adapter": "docker", + "image": "lloom/spark-audio:source-34b229132a097ab793a21c708a3b99be02ec3adcebd42d81c0305e87d5971c73", + "pull": false, + "createArgs": [ + "--restart", + "unless-stopped", + "--gpus", + "all", + "--ipc", + "host", + "-p", + "127.0.0.1:${port}:8000", + "-v", + "${modelPath}:/models/model:ro", + "-e", + "HF_HUB_OFFLINE=1", + "-e", + "TRANSFORMERS_OFFLINE=1" + ], + "command": [ + "--kind", + "stt", + "--model", + "openai/whisper-large-v3-turbo", + "--model-path", + "/models/model/large-v3-turbo.pt", + "--host", + "0.0.0.0", + "--port", + "8000" + ] + } + }, + "timeoutMs": 1800000 + } + } + ] +} diff --git a/scripts/check-package.mjs b/scripts/check-package.mjs index e2f16b5..390b848 100644 --- a/scripts/check-package.mjs +++ b/scripts/check-package.mjs @@ -381,7 +381,7 @@ try { LLOOM_HOME: path.join(homeRoot, '.lloom') }); const health = await waitForHostHealth(baseUrl); - if (health?.data?.recipeCount !== 41 || health?.data?.benchmarkCount !== 12) { + if (health?.data?.recipeCount !== 45 || health?.data?.benchmarkCount !== 12) { fail('installed lloom-host is not serving packaged seed community data', [JSON.stringify(health?.data ?? null)]); } diff --git a/test/spark-audio-recipes.test.mjs b/test/spark-audio-recipes.test.mjs new file mode 100644 index 0000000..8d00528 --- /dev/null +++ b/test/spark-audio-recipes.test.mjs @@ -0,0 +1,187 @@ +import assert from 'node:assert/strict'; +import fs from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; +import { execFileSync } from 'node:child_process'; +import { createInitPlan } from '../src/init.mjs'; +import { loadConfig } from '../src/config.mjs'; +import { getBackend, loadBackendCatalog } from '../src/backend-catalog.mjs'; +import { loadRecipes, planRecipe } from '../src/recipes.mjs'; +import { resolveSttDescriptor, resolveTtsDescriptor } from '../src/tts-catalog.mjs'; + +const expected = { + 'linux-nvidia-spark-audio-qwen3-tts-1-7b-customvoice': { + kind: 'audio_speech', + model: 'Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice', + mode: 'custom_voice', + modelPath: '/models/model' + }, + 'linux-nvidia-spark-audio-qwen3-tts-1-7b-base': { + kind: 'audio_speech', + model: 'Qwen/Qwen3-TTS-12Hz-1.7B-Base', + mode: 'voice_clone', + modelPath: '/models/model' + }, + 'linux-nvidia-spark-audio-qwen3-tts-1-7b-voicedesign': { + kind: 'audio_speech', + model: 'Qwen/Qwen3-TTS-12Hz-1.7B-VoiceDesign', + mode: 'voice_design', + modelPath: '/models/model' + }, + 'linux-nvidia-spark-audio-whisper-large-v3-turbo': { + kind: 'audio_transcription', + model: 'openai/whisper-large-v3-turbo', + modelPath: '/models/model/large-v3-turbo.pt' + } +}; + +const recipes = (await loadRecipes()).filter((r) => r.id.startsWith('linux-nvidia-spark-audio-')); +assert.deepEqual(recipes.map((r) => r.id).toSorted(), Object.keys(expected).toSorted()); +const image = execFileSync('python3', ['backends/spark-audio/install.py', '--print-image'], { + encoding: 'utf8' +}).trim(); +assert.match(image, /^lloom\/spark-audio:source-[a-f0-9]{64}$/); +const backend = getBackend(await loadBackendCatalog(), 'spark-audio'); +assert.ok(backend); +assert.ok(backend.setup.some((step) => step.args?.[0] === '${repoRoot}/backends/spark-audio/install.py')); + +const dir = await fs.mkdtemp(path.join(os.tmpdir(), 'lloom-spark-audio-recipes-')); +try { + // The image identity covers the build inputs only: tests and docs must not + // force a rebuild, while any change to the served adapter must. + const copy = path.join(dir, 'spark-audio'); + await fs.cp('backends/spark-audio', copy, { recursive: true }); + const copyImage = () => + execFileSync('python3', [path.join(copy, 'install.py'), '--print-image'], { encoding: 'utf8' }).trim(); + assert.equal(copyImage(), image); + await fs.appendFile(path.join(copy, 'test_lloom_audio_cuda_server.py'), '\n# changed\n'); + await fs.appendFile(path.join(copy, 'README.md'), '\nchanged\n'); + assert.equal(copyImage(), image); + await fs.appendFile(path.join(copy, 'lloom_audio_cuda_server.py'), '\n# changed\n'); + assert.notEqual(copyImage(), image); + + const modelRoot = path.join(dir, 'models'); + const file = path.join(dir, 'config.json'); + const empty = { + server: { host: '127.0.0.1', port: 8100 }, + models: [], + backends: {}, + runtimes: {}, + defaults: {}, + aliases: {} + }; + async function apply(config, recipe) { + await fs.writeFile(file, JSON.stringify(config)); + return ( + await createInitPlan(await loadConfig(file), { + recipeId: recipe.id, + additive: true, + modelRoot, + autoDetectModelRoot: false, + clientId: 'manifest' + }) + ).config; + } + function checkModel(config, recipe) { + const want = expected[recipe.id]; + const recipeModel = recipe.models[0]; + const model = config.models.find((m) => m.id === want.model); + assert.ok(model, `${recipe.id} did not materialize ${want.model}`); + assert.equal(model.kind, want.kind); + assert.equal(model.upstreamModel, want.model); + assert.equal(model.runtime, recipeModel.runtime); + if (want.kind === 'audio_speech') { + const descriptor = resolveTtsDescriptor(model); + assert.equal(descriptor.family, 'qwen3-tts'); + assert.equal(descriptor.mode, want.mode); + assert.equal(descriptor.capabilities.includes('tts-voice-clone'), want.mode === 'voice_clone'); + } else { + assert.equal(resolveSttDescriptor(model).family, 'whisper'); + } + const runtime = config.runtimes[recipeModel.runtime]; + assert.equal(runtime.warmup, undefined, 'audio runtime must not receive a chat warmup'); + assert.equal(runtime.maxConcurrency, 1); + assert.equal(runtime.containerName, `lloom-${recipeModel.runtime}`); + assert.equal(runtime.healthUrl, `http://127.0.0.1:${runtime.port}/health`); + assert.equal(config.backends[model.backend].baseUrl, `http://127.0.0.1:${runtime.port}/v1`); + assert.equal(runtime.bootstrap.image, image); + assert.equal(runtime.bootstrap.pull, false); + const createArgs = runtime.bootstrap.createArgs; + assert.ok(createArgs.join(' ').includes(`127.0.0.1:${runtime.port}:8000`)); + const download = recipe.setup.steps.find((s) => s.action === 'download-model'); + const mounts = createArgs.filter((arg, index) => ['-v', '--mount'].includes(createArgs[index - 1])); + assert.deepEqual(mounts, [`${path.join(modelRoot, download.model.replaceAll('/', '--'))}:/models/model:ro`]); + const command = runtime.bootstrap.command; + assert.deepEqual(command, [ + '--kind', + want.kind === 'audio_speech' ? 'tts' : 'stt', + '--model', + want.model, + '--model-path', + want.modelPath, + '--host', + '0.0.0.0', + '--port', + '8000' + ]); + if (want.modelPath !== '/models/model') { + assert.ok(download.include.includes(path.posix.relative('/models/model', want.modelPath))); + } + } + + const ports = new Set(); + for (const recipe of recipes) { + assert.equal(recipe.models.length, 1); + assert.equal(recipe.backend.id, 'spark-audio'); + assert.doesNotMatch( + JSON.stringify({ ...recipe, filePath: undefined }), + /ennspark|spark03|enntitysparkadmin|\/Users\/|\/home\/|192\.168\.|100\.78\./ + ); + for (const step of recipe.setup.steps.filter((s) => s.action === 'download-model')) { + assert.equal(step.provider, 'huggingface'); + assert.match(step.revision, /^[a-f0-9]{40}$/); + assert.deepEqual( + step.include, + step.integrity.files.map((f) => f.path) + ); + assert.equal( + step.downloadSizeBytes, + step.integrity.files.reduce((sum, f) => sum + f.sizeBytes, 0) + ); + for (const entry of step.integrity.files) assert.match(entry.sha256, /^[a-f0-9]{64}$/); + } + const standalone = await apply(empty, recipe); + assert.equal(standalone.models.length, 1); + assert.deepEqual(Object.keys(standalone.runtimes), [recipe.models[0].runtime]); + assert.deepEqual(standalone.defaults, {}); + checkModel(standalone, recipe); + const plan = planRecipe(recipe, standalone, { modelRoot, checkLocalReferences: false }); + assert.deepEqual(plan.validationErrors, []); + } + + for (const order of [recipes, recipes.toReversed()]) { + let config = await apply(empty, order[0]); + for (const recipe of order.slice(1)) { + const before = structuredClone(config); + config = await apply(config, recipe); + for (const [id, runtime] of Object.entries(before.runtimes)) assert.deepEqual(config.runtimes[id], runtime); + for (const [id, backendConfig] of Object.entries(before.backends)) + assert.deepEqual(config.backends[id], backendConfig); + assert.deepEqual(config.defaults, before.defaults); + } + assert.equal(config.models.length, 4); + assert.equal(Object.keys(config.runtimes).length, 4); + for (const recipe of recipes) checkModel(config, recipe); + const runtimePorts = Object.values(config.runtimes).map((runtime) => runtime.port); + assert.equal(new Set(runtimePorts).size, 4); + runtimePorts.forEach((port) => ports.add(port)); + assert.equal(new Set(Object.values(config.runtimes).map((runtime) => runtime.containerName)).size, 4); + assert.equal(config.models.filter((m) => m.kind === 'audio_speech').length, 3); + assert.equal(config.models.filter((m) => m.kind === 'audio_transcription').length, 1); + } + assert.ok(![...ports].includes(8100)); + + console.log('Spark audio recipes: 4 standalone and additive in both orders; independent runtimes, pinned image'); +} finally { + await fs.rm(dir, { recursive: true, force: true }); +} From f86857951b179c666369d75fa61c43513948cd20 Mon Sep 17 00:00:00 2001 From: Jason McCartney Date: Sun, 27 Sep 2026 13:16:27 -0700 Subject: [PATCH 2/3] Download Whisper large-v3-turbo from a mirror of OpenAI's checkpoint Replace the third-party Hugging Face copy with an unmodified mirror of OpenAI's large-v3-turbo.pt; the enforced SHA-256 is unchanged. Co-Authored-By: Claude Opus 5.5 --- docs/spark-audio.md | 6 ++++-- ...ux-nvidia-spark-audio-whisper-large-v3-turbo.json | 12 ++++++------ 2 files changed, 10 insertions(+), 8 deletions(-) diff --git a/docs/spark-audio.md b/docs/spark-audio.md index 714eb4d..e6f3b5a 100644 --- a/docs/spark-audio.md +++ b/docs/spark-audio.md @@ -39,8 +39,10 @@ authenticated LLooM gateway. openai-whisper cannot load the Transformers-format `openai/whisper-large-v3-turbo` repository. The Whisper recipe therefore downloads OpenAI's original -`large-v3-turbo.pt` checkpoint from a byte-identical Hugging Face copy. Its SHA-256 -matches the value published in openai-whisper v20250625. +`large-v3-turbo.pt` checkpoint from an unmodified mirror, +[`dataangel/whisper-large-v3-turbo-openai`](https://huggingface.co/dataangel/whisper-large-v3-turbo-openai), +copied from OpenAI's download URL. Its SHA-256 matches the value published in +openai-whisper v20250625. Voice cloning uses the stock Qwen engine by default. Streamed PCM voice clones and cancellation between chunks require `LLOOM_TTS_ENGINE=faster` in the diff --git a/recipes/linux-nvidia-spark-audio-whisper-large-v3-turbo.json b/recipes/linux-nvidia-spark-audio-whisper-large-v3-turbo.json index ff4b61c..74420d9 100644 --- a/recipes/linux-nvidia-spark-audio-whisper-large-v3-turbo.json +++ b/recipes/linux-nvidia-spark-audio-whisper-large-v3-turbo.json @@ -13,12 +13,12 @@ "provenance": { "createdBy": "lloom", "generatedBy": "manual-curated", - "source": "OpenAI large-v3-turbo.pt with the SHA-256 published in openai-whisper v20250625 (aff26ae408abcba5fbf8813c21e62b0941638c5f6eebfb145be0c9839262a19a), fetched from the byte-identical Hugging Face copy TheScenery/whisper-models 071f7d44e122e9964924a21c5bf8381d37595055; openai-whisper 20250625 on a digest-pinned NGC PyTorch base." + "source": "OpenAI large-v3-turbo.pt with the SHA-256 published in openai-whisper v20250625 (aff26ae408abcba5fbf8813c21e62b0941638c5f6eebfb145be0c9839262a19a), mirrored unmodified from OpenAI's download URL to dataangel/whisper-large-v3-turbo-openai b51fd912a062b3c3fdb993189a21ec2d93f790fe; openai-whisper 20250625 on a digest-pinned NGC PyTorch base." }, "links": [ { "rel": "source", - "href": "https://huggingface.co/TheScenery/whisper-models/tree/071f7d44e122e9964924a21c5bf8381d37595055" + "href": "https://huggingface.co/dataangel/whisper-large-v3-turbo-openai/tree/b51fd912a062b3c3fdb993189a21ec2d93f790fe" }, { "rel": "runtime", @@ -98,11 +98,11 @@ }, { "id": "download-whisper-large-v3-turbo", - "title": "Download TheScenery/whisper-models files for openai/whisper-large-v3-turbo", + "title": "Download dataangel/whisper-large-v3-turbo-openai files for openai/whisper-large-v3-turbo", "action": "download-model", "provider": "huggingface", - "model": "TheScenery/whisper-models", - "revision": "071f7d44e122e9964924a21c5bf8381d37595055", + "model": "dataangel/whisper-large-v3-turbo-openai", + "revision": "b51fd912a062b3c3fdb993189a21ec2d93f790fe", "include": [ "large-v3-turbo.pt" ], @@ -122,7 +122,7 @@ "models": [ { "role": "whisper-large-v3-turbo", - "model": "TheScenery/whisper-models", + "model": "dataangel/whisper-large-v3-turbo-openai", "gatewayModel": "openai/whisper-large-v3-turbo", "upstreamModel": "openai/whisper-large-v3-turbo", "runtime": "spark-audio-whisper-large-v3-turbo", From 67a17ec15f1d0792b1bf052700ffd2f5696e9064 Mon Sep 17 00:00:00 2001 From: Jason McCartney Date: Sun, 27 Sep 2026 13:24:00 -0700 Subject: [PATCH 3/3] List the Spark audio recipes in the smoke test's catalog checks Co-Authored-By: Claude Opus 5.5 --- test/smoke.mjs | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/test/smoke.mjs b/test/smoke.mjs index f213b99..0d75a8a 100644 --- a/test/smoke.mjs +++ b/test/smoke.mjs @@ -883,6 +883,10 @@ assert.deepEqual( 'linux-nvidia-gb10-thinkingcap-qwen36-27b-vllm', 'linux-nvidia-qwen-image-2-1-diffusers', 'linux-nvidia-qwen3-embedding-4b-vllm', + 'linux-nvidia-spark-audio-qwen3-tts-1-7b-base', + 'linux-nvidia-spark-audio-qwen3-tts-1-7b-customvoice', + 'linux-nvidia-spark-audio-qwen3-tts-1-7b-voicedesign', + 'linux-nvidia-spark-audio-whisper-large-v3-turbo', 'lloom-hear' ] ); @@ -1045,7 +1049,7 @@ const recipeIndexReport = await buildRecipeIndexReport(config, { }); assert.equal(recipeIndexReport.ok, true); assert.equal(recipeIndexReport.index.id, 'lloom-community-recipes'); -assert.equal(recipeIndexReport.recipes.length, 36); +assert.equal(recipeIndexReport.recipes.length, 40); const indexedSparkRecipe = recipeIndexReport.recipes.find( (candidate) => candidate.id === 'linux-nvidia-gb10-qwen36-unsloth-vllm' );