diff --git a/CHANGELOG.md b/CHANGELOG.md index aa16fd1..12f5245 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,38 @@ for 1.0.0 have something to be assembled from. ## Unreleased +## 1.9.0 + +The model menu says which model is loaded, and a new chat now matches the +model it is about to talk to. + +- **A dot on the model that is loaded.** Opening the model menu asks each + connection which of its models is in memory. llama-swap says so in its + ordinary model list, so the one it is holding gets a green dot, and one being + loaded gets a pulsing amber one. Picking a model without a dot means waiting + for it to load first. A hosted API such as DeepSeek never unloads anything and + does not report it, so its models show no dot, not a false "not loaded". Each + connection is asked once per menu opening, at most every five seconds. One + that reports nothing is asked again only after ten minutes, and one that is + slow or down just leaves the menu without dots. + +- **A new chat offers the model's own effort levels.** The new-chat screen + offered low, medium and high whatever the model took. On a model like Bonsai, + which takes low, medium and xhigh with xhigh as its default, the menu offered a + `high` it rejects. It had no xhigh, so it showed "off" while the chat it + created used xhigh. It now shows the same levels, and the same default, as the + chat will have. + +- **The message box is the same width for every model.** It was sized by its + widest content, so the "has no vision, so images will not be sent" line made + it wider for models without vision than for models with it. It is now always + the width of the conversation column. + +- **"Speak friend and enter."** The line under an empty chat (and on the "not + yours" error page) lost its commas. On the Doors of Durin it is a riddle: the + answer is to say *friend*, not to be greeted as one. Only the shipped wording + changed. An instance that has overridden the line keeps its own. + ## 1.8.5 - **No more grey slivers at the ends of the tab bars.** Tab bars fade at an edge diff --git a/src/lembas/__init__.py b/src/lembas/__init__.py index 85444f0..74b1c05 100644 --- a/src/lembas/__init__.py +++ b/src/lembas/__init__.py @@ -1,3 +1,3 @@ """LLeMbas - a Middle-earth themed web UI for OpenAI-compatible LLM endpoints.""" -__version__ = "1.8.5" +__version__ = "1.9.0" diff --git a/src/lembas/api/admin_audio.py b/src/lembas/api/admin_audio.py index 7123cce..da1a5c8 100644 --- a/src/lembas/api/admin_audio.py +++ b/src/lembas/api/admin_audio.py @@ -19,7 +19,7 @@ log = logging.getLogger(__name__) router = APIRouter(prefix="/admin/audio", tags=["admin-audio"]) # Read out by the speech test. Short, and the one line this project would pick. -TEST_PHRASE = "Speak, friend, and enter." +TEST_PHRASE = "Speak friend and enter." def _page_context(db: Db) -> dict: diff --git a/src/lembas/api/models.py b/src/lembas/api/models.py new file mode 100644 index 0000000..30de35d --- /dev/null +++ b/src/lembas/api/models.py @@ -0,0 +1,23 @@ +"""What the model menu asks for when it opens.""" + +from __future__ import annotations + +from fastapi import APIRouter + +from lembas.api.deps import Db, RequiredUser +from lembas.services import chat as chat_service +from lembas.services import model_state + +router = APIRouter(prefix="/api/models", tags=["models"]) + + +@router.get("/state") +async def model_states(db: Db, user: RequiredUser) -> dict: + """`{"states": {model_id: "loaded" | "loading" | "unloaded"}}`. + + Only models this reader may use, so the answer never names a model the + menu would not show. Only those whose endpoint reports a state, so a hosted + API's models are simply absent. See `services/model_state.py`. + """ + models = chat_service.available_models(db, user) + return {"states": await model_state.states_for(models)} diff --git a/src/lembas/api/pages.py b/src/lembas/api/pages.py index 61cec6f..100adde 100644 --- a/src/lembas/api/pages.py +++ b/src/lembas/api/pages.py @@ -736,6 +736,18 @@ async def chat_index( "bodies": {}, **context, "current_model": preselected, + # `_chat_context` reads the efforts off the *chat's* model, and there + # is no chat here -- so every new chat was offered the generic three + # whatever it was about to talk to. On Bonsai (low, medium, xhigh) + # the configured `xhigh` was not among them, and the picker fell + # through to "off". The chat created from this screen then got + # `xhigh` anyway, so the control said one thing and the first reply + # did another. + "efforts": ( + chat_service.efforts_for(preselected) + if preselected + else chat_service.DEFAULT_EFFORTS + ), "starting_temporary": temporary, "starting_kind": kind if kind in KINDS else KIND_CHAT, "starting_folder": starting_folder, diff --git a/src/lembas/main.py b/src/lembas/main.py index fe833d9..3f48949 100644 --- a/src/lembas/main.py +++ b/src/lembas/main.py @@ -38,6 +38,7 @@ from lembas.api import ( folders, library, messages, + models, pages, preferences, push, @@ -204,6 +205,7 @@ def create_app() -> FastAPI: app.include_router(folders.router) app.include_router(library.router) app.include_router(messages.router) + app.include_router(models.router) app.include_router(reports.router) app.include_router(schedules.router) app.include_router(agents.router) diff --git a/src/lembas/services/branding.py b/src/lembas/services/branding.py index b0d8cdf..c0d05f2 100644 --- a/src/lembas/services/branding.py +++ b/src/lembas/services/branding.py @@ -71,7 +71,11 @@ FLAVOUR: dict[str, tuple[str, str, str]] = { "chat_empty": ( "Empty chat", "Above the composer on a chat with nothing in it yet.", - "Speak, friend, and enter.", + # No commas, on purpose. It is the riddle on the Doors of Durin, and + # its answer is to *say* "friend" -- the password is the word itself. + # With commas it is an invitation to a friend, which is the misreading + # that kept the Fellowship outside the door. + "Speak friend and enter.", ), "offline_title": ( "Offline heading", @@ -87,7 +91,7 @@ FLAVOUR: dict[str, tuple[str, str, str]] = { "error_403": ( "403 — not yours", "Shown on a page somebody is not allowed to see.", - "Speak, friend, and enter. This door is not yours to open.", + "Speak friend and enter. This door is not yours to open.", ), "error_404": ( "404 — not found", diff --git a/src/lembas/services/model_state.py b/src/lembas/services/model_state.py new file mode 100644 index 0000000..e0d35fc --- /dev/null +++ b/src/lembas/services/model_state.py @@ -0,0 +1,114 @@ +"""Which models are loaded right now, where the endpoint is able to say. + +llama-swap holds one model at a time and reports which, inside the ordinary +`GET /v1/models` answer: every entry carries `"status": {"value": "loaded"}` +or `"unloaded"`. Choosing a model that is not loaded costs a load (seconds for +a small one, most of a minute for the 26B), so the model menu shows a dot on +the one that is ready. + +**Only what an endpoint states, and nothing inferred.** The OpenAI spec has +no such field. A hosted API such as DeepSeek leaves it out because nothing is +ever unloaded there, so its models get no state and no dot, rather than a +guess dressed up as a reading. The same shape covers the next runner that +reports it: `status` as an object with `value`, or as a bare string. + +**Cheap by construction**, because the menu asks every time it opens: + +- one `/v1/models` per *connection*, not per model, all at once; +- a short timeout, because a slow endpoint must never hold up a menu; +- five seconds of cache per connection, so opening the menu repeatedly costs + one request; +- and ten minutes for a connection that said nothing about state, so a hosted + API is not asked for its model list on every click only to answer nothing + again. + +Process-level, like the branding cache. With several workers each keeps its +own, which costs at most one extra request each and cannot be wrong for longer +than the TTL. +""" + +from __future__ import annotations + +import asyncio +import logging +import time +from typing import Any + +from lembas.services.llm.openai_client import Endpoint, list_models + +log = logging.getLogger(__name__) + +TIMEOUT = 3.0 +TTL = 5.0 +TTL_SILENT = 600.0 + +LOADED = "loaded" +LOADING = "loading" +UNLOADED = "unloaded" + +_LOADED_WORDS = frozenset({"loaded", "ready", "running"}) +_LOADING_WORDS = frozenset({"loading", "starting"}) + +# connection id -> (monotonic time read, TTL, {model_id: state}) +_CACHE: dict[str, tuple[float, float, dict[str, str]]] = {} + + +def state_of(entry: dict[str, Any]) -> str: + """One `/v1/models` entry's state, or "" when it states none.""" + status = entry.get("status") + value = status.get("value") if isinstance(status, dict) else status + if not isinstance(value, str) or not value.strip(): + return "" + word = value.strip().lower() + if word in _LOADED_WORDS: + return LOADED + if word in _LOADING_WORDS: + return LOADING + return UNLOADED + + +async def _read(connection) -> dict[str, str]: + now = time.monotonic() + cached = _CACHE.get(connection.id) + if cached and now - cached[0] < cached[1]: + return cached[2] + try: + entries = await asyncio.wait_for( + list_models(Endpoint.from_connection(connection)), TIMEOUT + ) + except Exception: # noqa: BLE001 - an unreachable endpoint has no state, not an error page + log.debug("model state unavailable for %s", connection.name, exc_info=True) + # Not cached: the next open asks again, which is right for an endpoint + # that is merely starting up. + return {} + states = {entry["id"]: state for entry in entries if (state := state_of(entry))} + _CACHE[connection.id] = (now, TTL if states else TTL_SILENT, states) + return states + + +async def states_for(models) -> dict[str, str]: + """`{model_id: state}` for the models whose endpoint reports one. + + Models without a stated state are absent, not `""`, so the page can treat + "no key" as "draw nothing". + """ + connections = {} + for model in models: + connection = getattr(model, "connection", None) + if connection is not None and connection.enabled: + connections[connection.id] = connection + if not connections: + return {} + results = await asyncio.gather(*(_read(c) for c in connections.values())) + by_connection = dict(zip(connections, results, strict=True)) + out: dict[str, str] = {} + for model in models: + state = by_connection.get(model.connection_id, {}).get(model.model_id) + if state: + out[model.model_id] = state + return out + + +def forget() -> None: + """Drop the cache. For tests.""" + _CACHE.clear() diff --git a/src/lembas/web/i18n/sk.py b/src/lembas/web/i18n/sk.py index 9757ab1..acadc4e 100644 --- a/src/lembas/web/i18n/sk.py +++ b/src/lembas/web/i18n/sk.py @@ -681,6 +681,8 @@ MESSAGES.update( "Model": "Model", "Context window": "Kontextové okno", "Sees images": "Vidí obrázky", + "Loaded": "Načítaný", + "Loading": "Načítava sa", "Groups": "Skupiny", "Members": "Členovia", "Account": "Účet", diff --git a/src/lembas/web/static/css/app.css b/src/lembas/web/static/css/app.css index 4dd6b18..68b7584 100644 --- a/src/lembas/web/static/css/app.css +++ b/src/lembas/web/static/css/app.css @@ -1775,6 +1775,33 @@ body.is-resizing .canvas__body { pointer-events: none; } gap: inherit; } .picker__list--models .picker__option .picker__avatar { margin-top: 0; } + +/* Whether a model is loaded, where its endpoint says so (llama-swap does; a + hosted API does not, and gets nothing). A dot on the avatar's corner, ringed + in the menu's own surface so it reads against any avatar colour. Nothing is + drawn until ui.js has an answer -- an empty `data-model-state` is "unknown", + which is not the same claim as "unloaded". */ +.model-slot { position: relative; display: flex; flex: none; } +.model-state { + position: absolute; + right: calc(var(--model-state-size) / -3); + bottom: calc(var(--model-state-size) / -3); + width: var(--model-state-size); + height: var(--model-state-size); + border-radius: var(--radius-full); + box-shadow: 0 0 0 var(--outline-w) var(--surface); + display: none; +} +.model-slot[data-model-state="loaded"] .model-state { display: block; background: var(--model-state-loaded); } +.model-slot[data-model-state="loading"] .model-state { + display: block; + background: var(--model-state-loading); + animation: model-state-pulse var(--dur-slow) var(--ease-in-out) infinite; +} +@keyframes model-state-pulse { 50% { opacity: 0.35; } } +@media (prefers-reduced-motion: reduce) { + .model-slot[data-model-state="loading"] .model-state { animation: none; } +} .picker__list--models .picker__option-name { min-width: 0; } .model-ctx { font-size: var(--text-xs); diff --git a/src/lembas/web/static/css/chat.css b/src/lembas/web/static/css/chat.css index 9aa7f07..4ab8e19 100644 --- a/src/lembas/web/static/css/chat.css +++ b/src/lembas/web/static/css/chat.css @@ -1003,9 +1003,17 @@ flex-direction: column; justify-content: flex-end; } -/* position: relative anchors the `@` and `/` menu to the box. */ +/* position: relative anchors the `@` and `/` menu to the box. + + `width: 100%` is the width; `max-width` only caps it. Without it the box was + as wide as its widest content: `.composer` is a flex column, and auto margins + on a flex item switch off the stretch it would otherwise get. So the hint + under it decided. "GPT-OSS has no vision, so images will not be sent" made + the box 768px, and the same screen with a model that sees images made it + 538px. */ .composer__inner { position: relative; + width: 100%; max-width: var(--thread-max-width); margin: 0 auto; } diff --git a/src/lembas/web/static/css/tokens.css b/src/lembas/web/static/css/tokens.css index e18f3e5..dba7099 100644 --- a/src/lembas/web/static/css/tokens.css +++ b/src/lembas/web/static/css/tokens.css @@ -50,6 +50,14 @@ --radius-xl: 18px; --radius-full: 999px; + /* The load-state dot on a model's avatar in the model menu. Its colours are + tokens of their own, defaulting to the theme's success and warning, so + an instance whose success colour is not green can still say "loaded" in + green -- that is what people read a dot beside a name as. */ + --model-state-size: 0.625rem; + --model-state-loaded: var(--success); + --model-state-loading: var(--warning); + /* --- Controls ---------------------------------------------------------- Every button, input and select resolves its height from these. That is the diff --git a/src/lembas/web/static/js/ui.js b/src/lembas/web/static/js/ui.js index cd42009..e8fadaa 100644 --- a/src/lembas/web/static/js/ui.js +++ b/src/lembas/web/static/js/ui.js @@ -334,6 +334,45 @@ // Keep the chosen model in view when the list is long. var current = menu.querySelector(".picker__option.is-selected"); if (current) current.scrollIntoView({ block: "nearest" }); + refreshStates(menu); + } + + /* Which models are loaded, asked for each time the model menu opens -- + llama-swap holds one at a time and it changes by the minute, so a value + rendered with the page would be stale by the time anybody looked. Only + models whose endpoint reports a state come back; everything else keeps an + empty `data-model-state`, which draws nothing. While one is loading the + menu asks again every two seconds, and stops when it closes. */ + function refreshStates(menu) { + var list = menu.querySelector(".picker__list--models"); + if (!list || !window.fetch) return; + clearTimeout(menu._stateTimer); + fetch("/api/models/state", { + credentials: "same-origin", + headers: { Accept: "application/json" } + }).then(function (response) { + return response.ok ? response.json() : null; + }).then(function (data) { + var states = (data && data.states) || {}; + var loading = false; + list.querySelectorAll(".picker__option[data-model-id]").forEach(function (option) { + var slot = option.querySelector("[data-model-state]"); + if (!slot) return; + var state = states[option.dataset.modelId] || ""; + slot.dataset.modelState = state; + if (state === "loading") loading = true; + var label = slot.querySelector("[data-model-state-label]"); + if (label) { + label.textContent = state === "loaded" ? list.dataset.labelLoaded + : state === "loading" ? list.dataset.labelLoading : ""; + } + }); + if (loading && !menu.hidden) { + menu._stateTimer = setTimeout(function () { + if (!menu.hidden) refreshStates(menu); + }, 2000); + } + }).catch(function () { /* No state is a menu without dots, not an error. */ }); } function applyFilter(menu, needle) { diff --git a/src/lembas/web/templates/chat/_model_picker.html b/src/lembas/web/templates/chat/_model_picker.html index 2a159f5..2ad6fab 100644 --- a/src/lembas/web/templates/chat/_model_picker.html +++ b/src/lembas/web/templates/chat/_model_picker.html @@ -38,14 +38,28 @@ {% endif %} -
+
{% for model in models %}