From cb8a223fa4e88caaa91cbd666bc63d2be983f503 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jaroslav=20Bene=C5=A1?= Date: Mon, 28 Sep 2026 08:48:58 +0000 Subject: [PATCH] A dot on the loaded model, and a new chat that matches its model The model menu asks each connection's /v1/models for the load state llama-swap reports there and marks the loaded model; endpoints that state nothing (a hosted API) get no dot. The new-chat screen offered the generic three efforts whatever the model took, so Bonsai's xhigh default showed as off. The composer was as wide as its widest hint. And the Doors of Durin are a riddle: Speak friend and enter. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 32 ++++ src/lembas/__init__.py | 2 +- src/lembas/api/admin_audio.py | 2 +- src/lembas/api/models.py | 23 +++ src/lembas/api/pages.py | 12 ++ src/lembas/main.py | 2 + src/lembas/services/branding.py | 8 +- src/lembas/services/model_state.py | 114 +++++++++++++ src/lembas/web/i18n/sk.py | 2 + src/lembas/web/static/css/app.css | 27 ++++ src/lembas/web/static/css/chat.css | 10 +- src/lembas/web/static/css/tokens.css | 8 + src/lembas/web/static/js/ui.js | 39 +++++ .../web/templates/chat/_model_picker.html | 20 ++- tests/test_branding.py | 13 ++ tests/test_effort.py | 22 +++ tests/test_model_state.py | 152 ++++++++++++++++++ tests/test_ui_js.py | 13 ++ 18 files changed, 493 insertions(+), 8 deletions(-) create mode 100644 src/lembas/api/models.py create mode 100644 src/lembas/services/model_state.py create mode 100644 tests/test_model_state.py diff --git a/CHANGELOG.md b/CHANGELOG.md index aa16fd1..12f5245 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,38 @@ for 1.0.0 have something to be assembled from. ## Unreleased +## 1.9.0 + +The model menu says which model is loaded, and a new chat now matches the +model it is about to talk to. + +- **A dot on the model that is loaded.** Opening the model menu asks each + connection which of its models is in memory. llama-swap says so in its + ordinary model list, so the one it is holding gets a green dot, and one being + loaded gets a pulsing amber one. Picking a model without a dot means waiting + for it to load first. A hosted API such as DeepSeek never unloads anything and + does not report it, so its models show no dot, not a false "not loaded". Each + connection is asked once per menu opening, at most every five seconds. One + that reports nothing is asked again only after ten minutes, and one that is + slow or down just leaves the menu without dots. + +- **A new chat offers the model's own effort levels.** The new-chat screen + offered low, medium and high whatever the model took. On a model like Bonsai, + which takes low, medium and xhigh with xhigh as its default, the menu offered a + `high` it rejects. It had no xhigh, so it showed "off" while the chat it + created used xhigh. It now shows the same levels, and the same default, as the + chat will have. + +- **The message box is the same width for every model.** It was sized by its + widest content, so the "has no vision, so images will not be sent" line made + it wider for models without vision than for models with it. It is now always + the width of the conversation column. + +- **"Speak friend and enter."** The line under an empty chat (and on the "not + yours" error page) lost its commas. On the Doors of Durin it is a riddle: the + answer is to say *friend*, not to be greeted as one. Only the shipped wording + changed. An instance that has overridden the line keeps its own. + ## 1.8.5 - **No more grey slivers at the ends of the tab bars.** Tab bars fade at an edge diff --git a/src/lembas/__init__.py b/src/lembas/__init__.py index 85444f0..74b1c05 100644 --- a/src/lembas/__init__.py +++ b/src/lembas/__init__.py @@ -1,3 +1,3 @@ """LLeMbas - a Middle-earth themed web UI for OpenAI-compatible LLM endpoints.""" -__version__ = "1.8.5" +__version__ = "1.9.0" diff --git a/src/lembas/api/admin_audio.py b/src/lembas/api/admin_audio.py index 7123cce..da1a5c8 100644 --- a/src/lembas/api/admin_audio.py +++ b/src/lembas/api/admin_audio.py @@ -19,7 +19,7 @@ log = logging.getLogger(__name__) router = APIRouter(prefix="/admin/audio", tags=["admin-audio"]) # Read out by the speech test. Short, and the one line this project would pick. -TEST_PHRASE = "Speak, friend, and enter." +TEST_PHRASE = "Speak friend and enter." def _page_context(db: Db) -> dict: diff --git a/src/lembas/api/models.py b/src/lembas/api/models.py new file mode 100644 index 0000000..30de35d --- /dev/null +++ b/src/lembas/api/models.py @@ -0,0 +1,23 @@ +"""What the model menu asks for when it opens.""" + +from __future__ import annotations + +from fastapi import APIRouter + +from lembas.api.deps import Db, RequiredUser +from lembas.services import chat as chat_service +from lembas.services import model_state + +router = APIRouter(prefix="/api/models", tags=["models"]) + + +@router.get("/state") +async def model_states(db: Db, user: RequiredUser) -> dict: + """`{"states": {model_id: "loaded" | "loading" | "unloaded"}}`. + + Only models this reader may use, so the answer never names a model the + menu would not show. Only those whose endpoint reports a state, so a hosted + API's models are simply absent. See `services/model_state.py`. + """ + models = chat_service.available_models(db, user) + return {"states": await model_state.states_for(models)} diff --git a/src/lembas/api/pages.py b/src/lembas/api/pages.py index 61cec6f..100adde 100644 --- a/src/lembas/api/pages.py +++ b/src/lembas/api/pages.py @@ -736,6 +736,18 @@ async def chat_index( "bodies": {}, **context, "current_model": preselected, + # `_chat_context` reads the efforts off the *chat's* model, and there + # is no chat here -- so every new chat was offered the generic three + # whatever it was about to talk to. On Bonsai (low, medium, xhigh) + # the configured `xhigh` was not among them, and the picker fell + # through to "off". The chat created from this screen then got + # `xhigh` anyway, so the control said one thing and the first reply + # did another. + "efforts": ( + chat_service.efforts_for(preselected) + if preselected + else chat_service.DEFAULT_EFFORTS + ), "starting_temporary": temporary, "starting_kind": kind if kind in KINDS else KIND_CHAT, "starting_folder": starting_folder, diff --git a/src/lembas/main.py b/src/lembas/main.py index fe833d9..3f48949 100644 --- a/src/lembas/main.py +++ b/src/lembas/main.py @@ -38,6 +38,7 @@ from lembas.api import ( folders, library, messages, + models, pages, preferences, push, @@ -204,6 +205,7 @@ def create_app() -> FastAPI: app.include_router(folders.router) app.include_router(library.router) app.include_router(messages.router) + app.include_router(models.router) app.include_router(reports.router) app.include_router(schedules.router) app.include_router(agents.router) diff --git a/src/lembas/services/branding.py b/src/lembas/services/branding.py index b0d8cdf..c0d05f2 100644 --- a/src/lembas/services/branding.py +++ b/src/lembas/services/branding.py @@ -71,7 +71,11 @@ FLAVOUR: dict[str, tuple[str, str, str]] = { "chat_empty": ( "Empty chat", "Above the composer on a chat with nothing in it yet.", - "Speak, friend, and enter.", + # No commas, on purpose. It is the riddle on the Doors of Durin, and + # its answer is to *say* "friend" -- the password is the word itself. + # With commas it is an invitation to a friend, which is the misreading + # that kept the Fellowship outside the door. + "Speak friend and enter.", ), "offline_title": ( "Offline heading", @@ -87,7 +91,7 @@ FLAVOUR: dict[str, tuple[str, str, str]] = { "error_403": ( "403 — not yours", "Shown on a page somebody is not allowed to see.", - "Speak, friend, and enter. This door is not yours to open.", + "Speak friend and enter. This door is not yours to open.", ), "error_404": ( "404 — not found", diff --git a/src/lembas/services/model_state.py b/src/lembas/services/model_state.py new file mode 100644 index 0000000..e0d35fc --- /dev/null +++ b/src/lembas/services/model_state.py @@ -0,0 +1,114 @@ +"""Which models are loaded right now, where the endpoint is able to say. + +llama-swap holds one model at a time and reports which, inside the ordinary +`GET /v1/models` answer: every entry carries `"status": {"value": "loaded"}` +or `"unloaded"`. Choosing a model that is not loaded costs a load (seconds for +a small one, most of a minute for the 26B), so the model menu shows a dot on +the one that is ready. + +**Only what an endpoint states, and nothing inferred.** The OpenAI spec has +no such field. A hosted API such as DeepSeek leaves it out because nothing is +ever unloaded there, so its models get no state and no dot, rather than a +guess dressed up as a reading. The same shape covers the next runner that +reports it: `status` as an object with `value`, or as a bare string. + +**Cheap by construction**, because the menu asks every time it opens: + +- one `/v1/models` per *connection*, not per model, all at once; +- a short timeout, because a slow endpoint must never hold up a menu; +- five seconds of cache per connection, so opening the menu repeatedly costs + one request; +- and ten minutes for a connection that said nothing about state, so a hosted + API is not asked for its model list on every click only to answer nothing + again. + +Process-level, like the branding cache. With several workers each keeps its +own, which costs at most one extra request each and cannot be wrong for longer +than the TTL. +""" + +from __future__ import annotations + +import asyncio +import logging +import time +from typing import Any + +from lembas.services.llm.openai_client import Endpoint, list_models + +log = logging.getLogger(__name__) + +TIMEOUT = 3.0 +TTL = 5.0 +TTL_SILENT = 600.0 + +LOADED = "loaded" +LOADING = "loading" +UNLOADED = "unloaded" + +_LOADED_WORDS = frozenset({"loaded", "ready", "running"}) +_LOADING_WORDS = frozenset({"loading", "starting"}) + +# connection id -> (monotonic time read, TTL, {model_id: state}) +_CACHE: dict[str, tuple[float, float, dict[str, str]]] = {} + + +def state_of(entry: dict[str, Any]) -> str: + """One `/v1/models` entry's state, or "" when it states none.""" + status = entry.get("status") + value = status.get("value") if isinstance(status, dict) else status + if not isinstance(value, str) or not value.strip(): + return "" + word = value.strip().lower() + if word in _LOADED_WORDS: + return LOADED + if word in _LOADING_WORDS: + return LOADING + return UNLOADED + + +async def _read(connection) -> dict[str, str]: + now = time.monotonic() + cached = _CACHE.get(connection.id) + if cached and now - cached[0] < cached[1]: + return cached[2] + try: + entries = await asyncio.wait_for( + list_models(Endpoint.from_connection(connection)), TIMEOUT + ) + except Exception: # noqa: BLE001 - an unreachable endpoint has no state, not an error page + log.debug("model state unavailable for %s", connection.name, exc_info=True) + # Not cached: the next open asks again, which is right for an endpoint + # that is merely starting up. + return {} + states = {entry["id"]: state for entry in entries if (state := state_of(entry))} + _CACHE[connection.id] = (now, TTL if states else TTL_SILENT, states) + return states + + +async def states_for(models) -> dict[str, str]: + """`{model_id: state}` for the models whose endpoint reports one. + + Models without a stated state are absent, not `""`, so the page can treat + "no key" as "draw nothing". + """ + connections = {} + for model in models: + connection = getattr(model, "connection", None) + if connection is not None and connection.enabled: + connections[connection.id] = connection + if not connections: + return {} + results = await asyncio.gather(*(_read(c) for c in connections.values())) + by_connection = dict(zip(connections, results, strict=True)) + out: dict[str, str] = {} + for model in models: + state = by_connection.get(model.connection_id, {}).get(model.model_id) + if state: + out[model.model_id] = state + return out + + +def forget() -> None: + """Drop the cache. For tests.""" + _CACHE.clear() diff --git a/src/lembas/web/i18n/sk.py b/src/lembas/web/i18n/sk.py index 9757ab1..acadc4e 100644 --- a/src/lembas/web/i18n/sk.py +++ b/src/lembas/web/i18n/sk.py @@ -681,6 +681,8 @@ MESSAGES.update( "Model": "Model", "Context window": "Kontextové okno", "Sees images": "Vidí obrázky", + "Loaded": "Načítaný", + "Loading": "Načítava sa", "Groups": "Skupiny", "Members": "Členovia", "Account": "Účet", diff --git a/src/lembas/web/static/css/app.css b/src/lembas/web/static/css/app.css index 4dd6b18..68b7584 100644 --- a/src/lembas/web/static/css/app.css +++ b/src/lembas/web/static/css/app.css @@ -1775,6 +1775,33 @@ body.is-resizing .canvas__body { pointer-events: none; } gap: inherit; } .picker__list--models .picker__option .picker__avatar { margin-top: 0; } + +/* Whether a model is loaded, where its endpoint says so (llama-swap does; a + hosted API does not, and gets nothing). A dot on the avatar's corner, ringed + in the menu's own surface so it reads against any avatar colour. Nothing is + drawn until ui.js has an answer -- an empty `data-model-state` is "unknown", + which is not the same claim as "unloaded". */ +.model-slot { position: relative; display: flex; flex: none; } +.model-state { + position: absolute; + right: calc(var(--model-state-size) / -3); + bottom: calc(var(--model-state-size) / -3); + width: var(--model-state-size); + height: var(--model-state-size); + border-radius: var(--radius-full); + box-shadow: 0 0 0 var(--outline-w) var(--surface); + display: none; +} +.model-slot[data-model-state="loaded"] .model-state { display: block; background: var(--model-state-loaded); } +.model-slot[data-model-state="loading"] .model-state { + display: block; + background: var(--model-state-loading); + animation: model-state-pulse var(--dur-slow) var(--ease-in-out) infinite; +} +@keyframes model-state-pulse { 50% { opacity: 0.35; } } +@media (prefers-reduced-motion: reduce) { + .model-slot[data-model-state="loading"] .model-state { animation: none; } +} .picker__list--models .picker__option-name { min-width: 0; } .model-ctx { font-size: var(--text-xs); diff --git a/src/lembas/web/static/css/chat.css b/src/lembas/web/static/css/chat.css index 9aa7f07..4ab8e19 100644 --- a/src/lembas/web/static/css/chat.css +++ b/src/lembas/web/static/css/chat.css @@ -1003,9 +1003,17 @@ flex-direction: column; justify-content: flex-end; } -/* position: relative anchors the `@` and `/` menu to the box. */ +/* position: relative anchors the `@` and `/` menu to the box. + + `width: 100%` is the width; `max-width` only caps it. Without it the box was + as wide as its widest content: `.composer` is a flex column, and auto margins + on a flex item switch off the stretch it would otherwise get. So the hint + under it decided. "GPT-OSS has no vision, so images will not be sent" made + the box 768px, and the same screen with a model that sees images made it + 538px. */ .composer__inner { position: relative; + width: 100%; max-width: var(--thread-max-width); margin: 0 auto; } diff --git a/src/lembas/web/static/css/tokens.css b/src/lembas/web/static/css/tokens.css index e18f3e5..dba7099 100644 --- a/src/lembas/web/static/css/tokens.css +++ b/src/lembas/web/static/css/tokens.css @@ -50,6 +50,14 @@ --radius-xl: 18px; --radius-full: 999px; + /* The load-state dot on a model's avatar in the model menu. Its colours are + tokens of their own, defaulting to the theme's success and warning, so + an instance whose success colour is not green can still say "loaded" in + green -- that is what people read a dot beside a name as. */ + --model-state-size: 0.625rem; + --model-state-loaded: var(--success); + --model-state-loading: var(--warning); + /* --- Controls ---------------------------------------------------------- Every button, input and select resolves its height from these. That is the diff --git a/src/lembas/web/static/js/ui.js b/src/lembas/web/static/js/ui.js index cd42009..e8fadaa 100644 --- a/src/lembas/web/static/js/ui.js +++ b/src/lembas/web/static/js/ui.js @@ -334,6 +334,45 @@ // Keep the chosen model in view when the list is long. var current = menu.querySelector(".picker__option.is-selected"); if (current) current.scrollIntoView({ block: "nearest" }); + refreshStates(menu); + } + + /* Which models are loaded, asked for each time the model menu opens -- + llama-swap holds one at a time and it changes by the minute, so a value + rendered with the page would be stale by the time anybody looked. Only + models whose endpoint reports a state come back; everything else keeps an + empty `data-model-state`, which draws nothing. While one is loading the + menu asks again every two seconds, and stops when it closes. */ + function refreshStates(menu) { + var list = menu.querySelector(".picker__list--models"); + if (!list || !window.fetch) return; + clearTimeout(menu._stateTimer); + fetch("/api/models/state", { + credentials: "same-origin", + headers: { Accept: "application/json" } + }).then(function (response) { + return response.ok ? response.json() : null; + }).then(function (data) { + var states = (data && data.states) || {}; + var loading = false; + list.querySelectorAll(".picker__option[data-model-id]").forEach(function (option) { + var slot = option.querySelector("[data-model-state]"); + if (!slot) return; + var state = states[option.dataset.modelId] || ""; + slot.dataset.modelState = state; + if (state === "loading") loading = true; + var label = slot.querySelector("[data-model-state-label]"); + if (label) { + label.textContent = state === "loaded" ? list.dataset.labelLoaded + : state === "loading" ? list.dataset.labelLoading : ""; + } + }); + if (loading && !menu.hidden) { + menu._stateTimer = setTimeout(function () { + if (!menu.hidden) refreshStates(menu); + }, 2000); + } + }).catch(function () { /* No state is a menu without dots, not an error. */ }); } function applyFilter(menu, needle) { diff --git a/src/lembas/web/templates/chat/_model_picker.html b/src/lembas/web/templates/chat/_model_picker.html index 2a159f5..2ad6fab 100644 --- a/src/lembas/web/templates/chat/_model_picker.html +++ b/src/lembas/web/templates/chat/_model_picker.html @@ -38,14 +38,28 @@ {% endif %} -
+
{% for model in models %}