Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -83,7 +83,10 @@ transcript.](.github/images/workspace-light.png)
a folder of `.gguf` files and it serves them through its own managed llama.cpp
runtime -- no separate install. The `llama-server` binary is fetched once,
verified against a pinned SHA-256, and cached. GPU backend selection is
`auto` (try Vulkan, fall back to CPU), `vulkan`, or `cpu`.
`auto` (try Vulkan, fall back to CPU; Vulkan is skipped when the machine has
no Vulkan loader), `vulkan`, or `cpu`. A loaded model is released after 30
idle minutes (Settings > System; 0 keeps it loaded) or with "Unload model",
and its context window is held to what the model was trained for.
- **Bring your own GGUF.** Download a model into the local folder by direct URL
or Hugging Face repo, then select it from the same picker as everything else.
For a repository, Settings can list its `.gguf` files (with sizes, folders
Expand Down
2 changes: 2 additions & 0 deletions app_factory.py
Original file line number Diff line number Diff line change
Expand Up @@ -128,6 +128,8 @@ def gguf_directory() -> Path:
gpu_backend_setting=lambda: settings_repository.load().settings.llamacpp.gpu_backend,
models_directory=gguf_directory,
verify=ssl_context,
idle_unload_minutes=lambda: settings_repository.load().settings.llamacpp.idle_unload_minutes,
extra_args=lambda: settings_repository.load().settings.llamacpp.extra_args,
)
gguf_model_directory = GGUFModelDirectory(gguf_directory)
# ollama.Client.pull is overloaded on a Literal `stream`, one overload per
Expand Down
31 changes: 31 additions & 0 deletions backend/cortex_backend/api/routers/system.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,10 +17,12 @@
)
from cortex_backend.api.schemas import (
DiagnosticsResponse,
LlamaCppRuntimeStatus,
ShutdownResponse,
SystemResponse,
)
from cortex_backend.api.security import SessionPrincipal
from cortex_backend.llamacpp.errors import LlamaCppError, RuntimeBusyError
from fastapi import (
Depends,
HTTPException,
Expand Down Expand Up @@ -72,6 +74,35 @@ def system(
)


@router.post("/llamacpp/unload", response_model=LlamaCppRuntimeStatus)
def unload_llamacpp(
request: Request,
_: SessionPrincipal = Depends(require_session),
) -> LlamaCppRuntimeStatus:
"""Stop the loaded local model to free its memory; the next message loads it again.

Safe to repeat: with nothing loaded it changes nothing and reports the
status. Refused with 409 while a response is being generated or a model
is loading, because that would cut the answer off.
"""
manager = getattr(request.app.state, "llamacpp_manager", None)
unload = getattr(manager, "unload", None)
if not callable(unload):
raise HTTPException(status_code=409, detail="The local model runtime is unavailable in this preview.")
if request.app.state.jobs.active_snapshot(kind="generation") is not None:
raise HTTPException(
status_code=409,
detail="A response is being generated. Stop it or wait for it to finish, then unload the model.",
)
try:
unload()
except RuntimeBusyError as exc:
raise HTTPException(status_code=409, detail=exc.error) from exc
except LlamaCppError as exc:
raise HTTPException(status_code=500, detail=exc.error) from exc
return _llamacpp_status(request)


@router.post("/system/shutdown", response_model=ShutdownResponse)
def shutdown(
request: Request,
Expand Down
4 changes: 4 additions & 0 deletions backend/cortex_backend/api/routes.py
Original file line number Diff line number Diff line change
Expand Up @@ -1136,6 +1136,10 @@ def _llamacpp_status(request: Request) -> LlamaCppRuntimeStatus:
last_restart_reason=live.last_restart_reason,
loaded_context=live.loaded_context,
last_failure_code=live.last_failure_code,
gpu_layers_offloaded=live.gpu_layers_offloaded,
gpu_layers_total=live.gpu_layers_total,
backend_note=live.backend_note,
context_note=live.context_note,
)


Expand Down
13 changes: 13 additions & 0 deletions backend/cortex_backend/api/schemas.py
Original file line number Diff line number Diff line change
Expand Up @@ -112,6 +112,19 @@ class LlamaCppRuntimeStatus(APIModel):
# produced. Null when nothing failed, when the cause was not identified,
# and once a server is ready.
last_failure_code: LaunchFailureCode | None = None
# How many of the model's layers the running server put on the GPU, and how
# many it has, as the server reported them while loading. Null while
# nothing is ready and when the server said nothing Cortex recognises --
# unknown, not zero. ``active_backend`` says which build launched; these say
# whether the GPU is actually in use (0 offloaded means it is not).
gpu_layers_offloaded: int | None = None
gpu_layers_total: int | None = None
# Fixed text on why the GPU build was not used when it would have been the
# default (no Vulkan loader on this machine). Null otherwise.
backend_note: str | None = None
# Fixed text saying the context window was limited to what the model was
# trained for. Null when the window is as requested.
context_note: str | None = None


class SystemResponse(APIModel):
Expand Down
19 changes: 18 additions & 1 deletion backend/cortex_backend/core/settings.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,9 @@

from typing import Annotated, Literal

from pydantic import BaseModel, ConfigDict, Field, StringConstraints
from pydantic import BaseModel, ConfigDict, Field, StringConstraints, field_validator

from cortex_backend.llamacpp.extra_args import validate_extra_args


ModelTag = Annotated[
Expand Down Expand Up @@ -57,6 +59,21 @@ class LlamaCppSettings(_SettingsModel):
# "auto" tries Vulkan (broad GPU support, no extra toolkit) first and
# falls back to the CPU build if Vulkan can't launch on this machine.
gpu_backend: Literal["auto", "vulkan", "cpu"] = "auto"
# Minutes without a request after which the loaded model is released, so a
# 20 GB model does not keep its memory while the machine is used for
# something else. The next message loads it again. 0 keeps it loaded until
# Cortex exits or another model is chosen.
idle_unload_minutes: int = Field(default=30, ge=0, le=1440)
# Advanced runtime options appended to the launch (KV-cache types, flash
# attention, thread counts). Only an allow-list is accepted; the flags that
# define the launch -- model, context, address, key -- are refused. See
# cortex_backend.llamacpp.extra_args.
extra_args: tuple[str, ...] = ()

@field_validator("extra_args")
@classmethod
def _check_extra_args(cls, value: tuple[str, ...]) -> tuple[str, ...]:
return validate_extra_args(value)


class GenerationSettings(_SettingsModel):
Expand Down
105 changes: 105 additions & 0 deletions backend/cortex_backend/llamacpp/binary_fetcher.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
import hashlib
import logging
import os
import re
import shutil
import zipfile
from pathlib import Path
Expand Down Expand Up @@ -53,6 +54,16 @@
_MAX_BINARY_DOWNLOAD_BYTES = 2 * 1024 * 1024 * 1024
# Mirrors download.py's ``MIN_FREE_SPACE_BYTES`` safety reserve.
_MIN_FREE_SPACE_BYTES = 128 * 1024 * 1024
# The only directories pruning ever considers: a release build this fetcher
# extracts (``<tag>-<backend>``, tag being llama.cpp's ``b<build number>``) and
# the ``.prune-`` name a superseded one is renamed to on its way out. In-progress
# ``.download-`` and ``.extract-`` names deliberately match neither.
_RELEASE_DIR_RE = re.compile(r"^b(\d+)-(?:cpu|vulkan)$")
_PRUNING_DIR_RE = re.compile(r"^\.prune-[0-9a-f]{32}$")
# One pruning pass removes at most this many directories; a runtime folder holds
# a handful, so reaching it means something else is going on and the rest waits
# for the next launch.
_MAX_PRUNED_PER_PASS = 8


@runtime_checkable
Expand All @@ -67,6 +78,28 @@ def is_set(self) -> bool:
...


def _cancelled(cancellation_event: Cancellable | None) -> bool:
return cancellation_event is not None and cancellation_event.is_set()


def _directory_in_use(root: Path) -> bool:
"""Whether a program is running from ``root``, or holds a file in it open.

Opens each file for writing and closes it again without writing: a running
program's image and its loaded libraries refuse that, and so does a file
another process keeps open. Anything that cannot be listed or opened counts
as in use, because the caller's alternative is deleting it.
"""
try:
for path in root.rglob("*"):
if path.is_file():
with path.open("r+b"):
pass
except OSError:
return True
return False


def _raise_if_cancelled(cancellation_event: Cancellable | None) -> None:
if cancellation_event is not None and cancellation_event.is_set():
raise BinaryVerificationError("Local model runtime startup was cancelled.")
Expand Down Expand Up @@ -167,6 +200,7 @@ def __init__(self, runtime_dir: Path, *, http_client: httpx.Client | None = None
self._runtime_dir = runtime_dir
self._http = http_client
self._verification_cache: dict[Path, tuple[_TreeIdentity, bool]] = {}
self._pruned_for: str | None = None

def _verify_directory(
self,
Expand Down Expand Up @@ -248,6 +282,7 @@ def ensure_binary(
if self._verify_directory(
target_dir, asset, force_hash=True, cancellation_event=cancellation_event
):
self._prune_superseded_builds(release, cancellation_event)
return exe_path

self._runtime_dir.mkdir(parents=True, exist_ok=True)
Expand Down Expand Up @@ -279,8 +314,78 @@ def ensure_binary(
raise BinaryVerificationError(
f"Downloaded llama.cpp binary for '{backend}' failed verification."
)
self._prune_superseded_builds(release, cancellation_event)
return exe_path

def _prune_superseded_builds(
self, release: PinnedRelease, cancellation_event: Cancellable | None = None
) -> None:
"""Delete runtime builds older than the pinned release, once per process.

Every pin bump used to leave a 100-200 MB build behind for good. Run only
after ``release`` itself has been verified, so there is always a working
runtime to fall back on, and never allowed to fail the launch that
triggered it: whatever cannot be removed now is tried again next time.

Only a sibling named like one of this fetcher's own builds and with a
strictly lower build number is touched -- never the pinned release
(either backend), a newer one (a rolled-back Cortex sharing this data
folder), an in-progress download or extraction, or anything unrecognised.
A build a process is running from is left alone. Windows refuses to open
a running program's image, or a library it has loaded, for writing, so
every file is tried that way first (nothing is written); the directory
is then renamed, which is refused while any file in it is held open, and
only the renamed copy is deleted, so a half-deleted build is never left
under a name something could launch. Names are logged, never contents.

Whatever cannot be removed now is tried again the next time Cortex starts.
"""
if self._pruned_for == release.tag:
return
self._pruned_for = release.tag
current = _RELEASE_DIR_RE.match(f"{release.tag}-cpu")
if current is None:
return
current_build = int(current.group(1))
try:
root = self._runtime_dir.resolve()
candidates = sorted(self._runtime_dir.iterdir(), key=lambda entry: entry.name)
except OSError:
return
removed = 0
for entry in candidates:
if removed >= _MAX_PRUNED_PER_PASS or _cancelled(cancellation_event):
break
try:
leftover = _PRUNING_DIR_RE.match(entry.name) is not None
match = _RELEASE_DIR_RE.match(entry.name)
if not leftover and (match is None or int(match.group(1)) >= current_build):
continue
# Only a real directory directly inside the runtime folder: a
# link or junction could lead anywhere, and is not ours to delete.
if entry.is_symlink() or not entry.is_dir() or entry.resolve().parent != root:
continue
if leftover:
shutil.rmtree(entry, ignore_errors=True)
else:
if _directory_in_use(entry):
logger.info("Kept an older local runtime build that is in use (%s).", entry.name)
continue
claimed = self._runtime_dir / f".prune-{uuid4().hex}"
try:
os.replace(entry, claimed)
except OSError:
logger.info("Kept an older local runtime build that is in use (%s).", entry.name)
continue
self._verification_cache.pop(entry, None)
shutil.rmtree(claimed, ignore_errors=True)
logger.info("Removed a superseded local runtime build (%s).", entry.name)
removed += 1
except Exception as exc:
logger.warning(
"Could not remove a superseded local runtime build (%s).", type(exc).__name__
)

def _target_dir(self, release: PinnedRelease, backend: GpuBackend) -> Path:
return self._runtime_dir / f"{release.tag}-{backend}"

Expand Down
30 changes: 29 additions & 1 deletion backend/cortex_backend/llamacpp/chat_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@

from __future__ import annotations

import functools
import json
import logging
import ssl
Expand All @@ -19,7 +20,7 @@
from contextlib import AbstractContextManager, nullcontext
from pathlib import Path
from threading import Event, Lock
from typing import Any
from typing import Any, Concatenate, ParamSpec, TypeVar

import httpx

Expand All @@ -36,6 +37,31 @@
# quickly is not worth waiting for before the real one.
_TOKENIZE_TIMEOUT = httpx.Timeout(connect=2.0, read=10.0, write=10.0, pool=2.0)

_Params = ParamSpec("_Params")
_Result = TypeVar("_Result")


def _uses_the_server(
method: Callable[Concatenate[LlamaCppChatClient, _Params], _Result],
) -> Callable[Concatenate[LlamaCppChatClient, _Params], _Result]:
"""Tell the provider the server is in use for the whole call.

A generation can outlast the manager's idle period, and only this client
knows when its request begins and ends. Held from before the server is
made ready until the reply (or its failure) is over, so the server is
neither unloaded for being idle nor by a manual unload halfway through, and
the idle clock restarts when the call ends. A provider that does not track
use (a test double) is left alone.
"""

@functools.wraps(method)
def wrapper(self: LlamaCppChatClient, *args: _Params.args, **kwargs: _Params.kwargs) -> _Result:
scope = getattr(self._provider, "request_scope", None)
with scope() if callable(scope) else nullcontext():
return method(self, *args, **kwargs)

return wrapper


class LlamaCppChatClient:
"""``ChatClient`` implementation backed by a locally-managed llama-server."""
Expand Down Expand Up @@ -126,6 +152,7 @@ def set_status_callback(self, callback: Callable[[str], None] | None) -> None:
"""
self._status_callback = callback

@_uses_the_server
def chat(
self,
*,
Expand Down Expand Up @@ -192,6 +219,7 @@ def chat(
think=think,
)

@_uses_the_server
def tokenize(
self,
*,
Expand Down
8 changes: 8 additions & 0 deletions backend/cortex_backend/llamacpp/errors.py
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,14 @@ def user_message(self) -> str:
return self.error


class RuntimeBusyError(LlamaCppError):
"""Raised when the runtime cannot be unloaded because it is in use or loading.

Its message is already user-facing (written by the manager, never taken
from the child), and the API reports it as a conflict.
"""


class BinaryVerificationError(LlamaCppError):
"""Raised when a downloaded/cached llama-server binary fails verification."""

Expand Down
Loading
Loading