Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e65ea90fe6
|
||
|
|
cb8a223fa4
|
@@ -16,6 +16,47 @@ for 1.0.0 have something to be assembled from.
|
|||||||
|
|
||||||
## Unreleased
|
## Unreleased
|
||||||
|
|
||||||
|
## 1.9.1
|
||||||
|
|
||||||
|
- **A temporary chat can be started on any model.** On the new-chat screen,
|
||||||
|
turning on Temporary switched the model back to the default, and choosing a
|
||||||
|
model switched Temporary off, so a temporary chat could only ever be started
|
||||||
|
on the default model. The Temporary button, the model menu and `/temp` now
|
||||||
|
keep each other's choice, and also keep the folder a chat was started in
|
||||||
|
("New chat here") and whether it is an agent chat.
|
||||||
|
|
||||||
|
## 1.9.0
|
||||||
|
|
||||||
|
The model menu says which model is loaded, and a new chat now matches the
|
||||||
|
model it is about to talk to.
|
||||||
|
|
||||||
|
- **A dot on the model that is loaded.** Opening the model menu asks each
|
||||||
|
connection which of its models is in memory. llama-swap says so in its
|
||||||
|
ordinary model list, so the one it is holding gets a green dot, and one being
|
||||||
|
loaded gets a pulsing amber one. Picking a model without a dot means waiting
|
||||||
|
for it to load first. A hosted API such as DeepSeek never unloads anything and
|
||||||
|
does not report it, so its models show no dot, not a false "not loaded". Each
|
||||||
|
connection is asked once per menu opening, at most every five seconds. One
|
||||||
|
that reports nothing is asked again only after ten minutes, and one that is
|
||||||
|
slow or down just leaves the menu without dots.
|
||||||
|
|
||||||
|
- **A new chat offers the model's own effort levels.** The new-chat screen
|
||||||
|
offered low, medium and high whatever the model took. On a model like Bonsai,
|
||||||
|
which takes low, medium and xhigh with xhigh as its default, the menu offered a
|
||||||
|
`high` it rejects. It had no xhigh, so it showed "off" while the chat it
|
||||||
|
created used xhigh. It now shows the same levels, and the same default, as the
|
||||||
|
chat will have.
|
||||||
|
|
||||||
|
- **The message box is the same width for every model.** It was sized by its
|
||||||
|
widest content, so the "has no vision, so images will not be sent" line made
|
||||||
|
it wider for models without vision than for models with it. It is now always
|
||||||
|
the width of the conversation column.
|
||||||
|
|
||||||
|
- **"Speak friend and enter."** The line under an empty chat (and on the "not
|
||||||
|
yours" error page) lost its commas. On the Doors of Durin it is a riddle: the
|
||||||
|
answer is to say *friend*, not to be greeted as one. Only the shipped wording
|
||||||
|
changed. An instance that has overridden the line keeps its own.
|
||||||
|
|
||||||
## 1.8.5
|
## 1.8.5
|
||||||
|
|
||||||
- **No more grey slivers at the ends of the tab bars.** Tab bars fade at an edge
|
- **No more grey slivers at the ends of the tab bars.** Tab bars fade at an edge
|
||||||
|
|||||||
@@ -1,3 +1,3 @@
|
|||||||
"""LLeMbas - a Middle-earth themed web UI for OpenAI-compatible LLM endpoints."""
|
"""LLeMbas - a Middle-earth themed web UI for OpenAI-compatible LLM endpoints."""
|
||||||
|
|
||||||
__version__ = "1.8.5"
|
__version__ = "1.9.1"
|
||||||
|
|||||||
@@ -19,7 +19,7 @@ log = logging.getLogger(__name__)
|
|||||||
router = APIRouter(prefix="/admin/audio", tags=["admin-audio"])
|
router = APIRouter(prefix="/admin/audio", tags=["admin-audio"])
|
||||||
|
|
||||||
# Read out by the speech test. Short, and the one line this project would pick.
|
# Read out by the speech test. Short, and the one line this project would pick.
|
||||||
TEST_PHRASE = "Speak, friend, and enter."
|
TEST_PHRASE = "Speak friend and enter."
|
||||||
|
|
||||||
|
|
||||||
def _page_context(db: Db) -> dict:
|
def _page_context(db: Db) -> dict:
|
||||||
|
|||||||
@@ -0,0 +1,23 @@
|
|||||||
|
"""What the model menu asks for when it opens."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from fastapi import APIRouter
|
||||||
|
|
||||||
|
from lembas.api.deps import Db, RequiredUser
|
||||||
|
from lembas.services import chat as chat_service
|
||||||
|
from lembas.services import model_state
|
||||||
|
|
||||||
|
router = APIRouter(prefix="/api/models", tags=["models"])
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/state")
|
||||||
|
async def model_states(db: Db, user: RequiredUser) -> dict:
|
||||||
|
"""`{"states": {model_id: "loaded" | "loading" | "unloaded"}}`.
|
||||||
|
|
||||||
|
Only models this reader may use, so the answer never names a model the
|
||||||
|
menu would not show. Only those whose endpoint reports a state, so a hosted
|
||||||
|
API's models are simply absent. See `services/model_state.py`.
|
||||||
|
"""
|
||||||
|
models = chat_service.available_models(db, user)
|
||||||
|
return {"states": await model_state.states_for(models)}
|
||||||
@@ -2,6 +2,7 @@
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from urllib.parse import urlencode
|
||||||
from zoneinfo import available_timezones
|
from zoneinfo import available_timezones
|
||||||
|
|
||||||
from fastapi import APIRouter, HTTPException, Request, Response, status
|
from fastapi import APIRouter, HTTPException, Request, Response, status
|
||||||
@@ -727,6 +728,24 @@ async def chat_index(
|
|||||||
if preselected is None and context["models"]:
|
if preselected is None and context["models"]:
|
||||||
preselected = context["models"][0]
|
preselected = context["models"][0]
|
||||||
|
|
||||||
|
# Every preselection lives in the URL, so every link that changes one of
|
||||||
|
# them has to carry the rest. The temporary toggle used to link to a bare
|
||||||
|
# `/chat?temporary=1` and the model picker to a bare `/chat?model=`, so
|
||||||
|
# each undid the other: temporary chats could only ever be started on the
|
||||||
|
# default model. The model goes last in the picker's URL because ui.js
|
||||||
|
# appends the chosen id to it.
|
||||||
|
carried = {
|
||||||
|
"model": model if model and preselected and preselected.model_id == model else "",
|
||||||
|
"temporary": "1" if temporary else "",
|
||||||
|
"kind": kind if kind in KINDS and kind != KIND_CHAT else "",
|
||||||
|
"folder": starting_folder.id if starting_folder is not None else "",
|
||||||
|
}
|
||||||
|
|
||||||
|
def new_chat_url(**changes: str) -> str:
|
||||||
|
query = urlencode({k: v for k, v in {**carried, **changes}.items() if v})
|
||||||
|
return f"/chat?{query}" if query else "/chat"
|
||||||
|
|
||||||
|
without_model = new_chat_url(model="")
|
||||||
return render(
|
return render(
|
||||||
request,
|
request,
|
||||||
"chat/index.html",
|
"chat/index.html",
|
||||||
@@ -736,7 +755,21 @@ async def chat_index(
|
|||||||
"bodies": {},
|
"bodies": {},
|
||||||
**context,
|
**context,
|
||||||
"current_model": preselected,
|
"current_model": preselected,
|
||||||
|
# `_chat_context` reads the efforts off the *chat's* model, and there
|
||||||
|
# is no chat here -- so every new chat was offered the generic three
|
||||||
|
# whatever it was about to talk to. On Bonsai (low, medium, xhigh)
|
||||||
|
# the configured `xhigh` was not among them, and the picker fell
|
||||||
|
# through to "off". The chat created from this screen then got
|
||||||
|
# `xhigh` anyway, so the control said one thing and the first reply
|
||||||
|
# did another.
|
||||||
|
"efforts": (
|
||||||
|
chat_service.efforts_for(preselected)
|
||||||
|
if preselected
|
||||||
|
else chat_service.DEFAULT_EFFORTS
|
||||||
|
),
|
||||||
"starting_temporary": temporary,
|
"starting_temporary": temporary,
|
||||||
|
"temporary_toggle_url": new_chat_url(temporary="" if temporary else "1"),
|
||||||
|
"model_navigate_url": without_model + ("&" if "?" in without_model else "?") + "model=",
|
||||||
"starting_kind": kind if kind in KINDS else KIND_CHAT,
|
"starting_kind": kind if kind in KINDS else KIND_CHAT,
|
||||||
"starting_folder": starting_folder,
|
"starting_folder": starting_folder,
|
||||||
"suggestions": suggestions_service.visible(db),
|
"suggestions": suggestions_service.visible(db),
|
||||||
|
|||||||
@@ -38,6 +38,7 @@ from lembas.api import (
|
|||||||
folders,
|
folders,
|
||||||
library,
|
library,
|
||||||
messages,
|
messages,
|
||||||
|
models,
|
||||||
pages,
|
pages,
|
||||||
preferences,
|
preferences,
|
||||||
push,
|
push,
|
||||||
@@ -204,6 +205,7 @@ def create_app() -> FastAPI:
|
|||||||
app.include_router(folders.router)
|
app.include_router(folders.router)
|
||||||
app.include_router(library.router)
|
app.include_router(library.router)
|
||||||
app.include_router(messages.router)
|
app.include_router(messages.router)
|
||||||
|
app.include_router(models.router)
|
||||||
app.include_router(reports.router)
|
app.include_router(reports.router)
|
||||||
app.include_router(schedules.router)
|
app.include_router(schedules.router)
|
||||||
app.include_router(agents.router)
|
app.include_router(agents.router)
|
||||||
|
|||||||
@@ -71,7 +71,11 @@ FLAVOUR: dict[str, tuple[str, str, str]] = {
|
|||||||
"chat_empty": (
|
"chat_empty": (
|
||||||
"Empty chat",
|
"Empty chat",
|
||||||
"Above the composer on a chat with nothing in it yet.",
|
"Above the composer on a chat with nothing in it yet.",
|
||||||
"Speak, friend, and enter.",
|
# No commas, on purpose. It is the riddle on the Doors of Durin, and
|
||||||
|
# its answer is to *say* "friend" -- the password is the word itself.
|
||||||
|
# With commas it is an invitation to a friend, which is the misreading
|
||||||
|
# that kept the Fellowship outside the door.
|
||||||
|
"Speak friend and enter.",
|
||||||
),
|
),
|
||||||
"offline_title": (
|
"offline_title": (
|
||||||
"Offline heading",
|
"Offline heading",
|
||||||
@@ -87,7 +91,7 @@ FLAVOUR: dict[str, tuple[str, str, str]] = {
|
|||||||
"error_403": (
|
"error_403": (
|
||||||
"403 — not yours",
|
"403 — not yours",
|
||||||
"Shown on a page somebody is not allowed to see.",
|
"Shown on a page somebody is not allowed to see.",
|
||||||
"Speak, friend, and enter. This door is not yours to open.",
|
"Speak friend and enter. This door is not yours to open.",
|
||||||
),
|
),
|
||||||
"error_404": (
|
"error_404": (
|
||||||
"404 — not found",
|
"404 — not found",
|
||||||
|
|||||||
@@ -0,0 +1,114 @@
|
|||||||
|
"""Which models are loaded right now, where the endpoint is able to say.
|
||||||
|
|
||||||
|
llama-swap holds one model at a time and reports which, inside the ordinary
|
||||||
|
`GET /v1/models` answer: every entry carries `"status": {"value": "loaded"}`
|
||||||
|
or `"unloaded"`. Choosing a model that is not loaded costs a load (seconds for
|
||||||
|
a small one, most of a minute for the 26B), so the model menu shows a dot on
|
||||||
|
the one that is ready.
|
||||||
|
|
||||||
|
**Only what an endpoint states, and nothing inferred.** The OpenAI spec has
|
||||||
|
no such field. A hosted API such as DeepSeek leaves it out because nothing is
|
||||||
|
ever unloaded there, so its models get no state and no dot, rather than a
|
||||||
|
guess dressed up as a reading. The same shape covers the next runner that
|
||||||
|
reports it: `status` as an object with `value`, or as a bare string.
|
||||||
|
|
||||||
|
**Cheap by construction**, because the menu asks every time it opens:
|
||||||
|
|
||||||
|
- one `/v1/models` per *connection*, not per model, all at once;
|
||||||
|
- a short timeout, because a slow endpoint must never hold up a menu;
|
||||||
|
- five seconds of cache per connection, so opening the menu repeatedly costs
|
||||||
|
one request;
|
||||||
|
- and ten minutes for a connection that said nothing about state, so a hosted
|
||||||
|
API is not asked for its model list on every click only to answer nothing
|
||||||
|
again.
|
||||||
|
|
||||||
|
Process-level, like the branding cache. With several workers each keeps its
|
||||||
|
own, which costs at most one extra request each and cannot be wrong for longer
|
||||||
|
than the TTL.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
import time
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from lembas.services.llm.openai_client import Endpoint, list_models
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
TIMEOUT = 3.0
|
||||||
|
TTL = 5.0
|
||||||
|
TTL_SILENT = 600.0
|
||||||
|
|
||||||
|
LOADED = "loaded"
|
||||||
|
LOADING = "loading"
|
||||||
|
UNLOADED = "unloaded"
|
||||||
|
|
||||||
|
_LOADED_WORDS = frozenset({"loaded", "ready", "running"})
|
||||||
|
_LOADING_WORDS = frozenset({"loading", "starting"})
|
||||||
|
|
||||||
|
# connection id -> (monotonic time read, TTL, {model_id: state})
|
||||||
|
_CACHE: dict[str, tuple[float, float, dict[str, str]]] = {}
|
||||||
|
|
||||||
|
|
||||||
|
def state_of(entry: dict[str, Any]) -> str:
|
||||||
|
"""One `/v1/models` entry's state, or "" when it states none."""
|
||||||
|
status = entry.get("status")
|
||||||
|
value = status.get("value") if isinstance(status, dict) else status
|
||||||
|
if not isinstance(value, str) or not value.strip():
|
||||||
|
return ""
|
||||||
|
word = value.strip().lower()
|
||||||
|
if word in _LOADED_WORDS:
|
||||||
|
return LOADED
|
||||||
|
if word in _LOADING_WORDS:
|
||||||
|
return LOADING
|
||||||
|
return UNLOADED
|
||||||
|
|
||||||
|
|
||||||
|
async def _read(connection) -> dict[str, str]:
|
||||||
|
now = time.monotonic()
|
||||||
|
cached = _CACHE.get(connection.id)
|
||||||
|
if cached and now - cached[0] < cached[1]:
|
||||||
|
return cached[2]
|
||||||
|
try:
|
||||||
|
entries = await asyncio.wait_for(
|
||||||
|
list_models(Endpoint.from_connection(connection)), TIMEOUT
|
||||||
|
)
|
||||||
|
except Exception: # noqa: BLE001 - an unreachable endpoint has no state, not an error page
|
||||||
|
log.debug("model state unavailable for %s", connection.name, exc_info=True)
|
||||||
|
# Not cached: the next open asks again, which is right for an endpoint
|
||||||
|
# that is merely starting up.
|
||||||
|
return {}
|
||||||
|
states = {entry["id"]: state for entry in entries if (state := state_of(entry))}
|
||||||
|
_CACHE[connection.id] = (now, TTL if states else TTL_SILENT, states)
|
||||||
|
return states
|
||||||
|
|
||||||
|
|
||||||
|
async def states_for(models) -> dict[str, str]:
|
||||||
|
"""`{model_id: state}` for the models whose endpoint reports one.
|
||||||
|
|
||||||
|
Models without a stated state are absent, not `""`, so the page can treat
|
||||||
|
"no key" as "draw nothing".
|
||||||
|
"""
|
||||||
|
connections = {}
|
||||||
|
for model in models:
|
||||||
|
connection = getattr(model, "connection", None)
|
||||||
|
if connection is not None and connection.enabled:
|
||||||
|
connections[connection.id] = connection
|
||||||
|
if not connections:
|
||||||
|
return {}
|
||||||
|
results = await asyncio.gather(*(_read(c) for c in connections.values()))
|
||||||
|
by_connection = dict(zip(connections, results, strict=True))
|
||||||
|
out: dict[str, str] = {}
|
||||||
|
for model in models:
|
||||||
|
state = by_connection.get(model.connection_id, {}).get(model.model_id)
|
||||||
|
if state:
|
||||||
|
out[model.model_id] = state
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def forget() -> None:
|
||||||
|
"""Drop the cache. For tests."""
|
||||||
|
_CACHE.clear()
|
||||||
@@ -681,6 +681,8 @@ MESSAGES.update(
|
|||||||
"Model": "Model",
|
"Model": "Model",
|
||||||
"Context window": "Kontextové okno",
|
"Context window": "Kontextové okno",
|
||||||
"Sees images": "Vidí obrázky",
|
"Sees images": "Vidí obrázky",
|
||||||
|
"Loaded": "Načítaný",
|
||||||
|
"Loading": "Načítava sa",
|
||||||
"Groups": "Skupiny",
|
"Groups": "Skupiny",
|
||||||
"Members": "Členovia",
|
"Members": "Členovia",
|
||||||
"Account": "Účet",
|
"Account": "Účet",
|
||||||
|
|||||||
@@ -1775,6 +1775,33 @@ body.is-resizing .canvas__body { pointer-events: none; }
|
|||||||
gap: inherit;
|
gap: inherit;
|
||||||
}
|
}
|
||||||
.picker__list--models .picker__option .picker__avatar { margin-top: 0; }
|
.picker__list--models .picker__option .picker__avatar { margin-top: 0; }
|
||||||
|
|
||||||
|
/* Whether a model is loaded, where its endpoint says so (llama-swap does; a
|
||||||
|
hosted API does not, and gets nothing). A dot on the avatar's corner, ringed
|
||||||
|
in the menu's own surface so it reads against any avatar colour. Nothing is
|
||||||
|
drawn until ui.js has an answer -- an empty `data-model-state` is "unknown",
|
||||||
|
which is not the same claim as "unloaded". */
|
||||||
|
.model-slot { position: relative; display: flex; flex: none; }
|
||||||
|
.model-state {
|
||||||
|
position: absolute;
|
||||||
|
right: calc(var(--model-state-size) / -3);
|
||||||
|
bottom: calc(var(--model-state-size) / -3);
|
||||||
|
width: var(--model-state-size);
|
||||||
|
height: var(--model-state-size);
|
||||||
|
border-radius: var(--radius-full);
|
||||||
|
box-shadow: 0 0 0 var(--outline-w) var(--surface);
|
||||||
|
display: none;
|
||||||
|
}
|
||||||
|
.model-slot[data-model-state="loaded"] .model-state { display: block; background: var(--model-state-loaded); }
|
||||||
|
.model-slot[data-model-state="loading"] .model-state {
|
||||||
|
display: block;
|
||||||
|
background: var(--model-state-loading);
|
||||||
|
animation: model-state-pulse var(--dur-slow) var(--ease-in-out) infinite;
|
||||||
|
}
|
||||||
|
@keyframes model-state-pulse { 50% { opacity: 0.35; } }
|
||||||
|
@media (prefers-reduced-motion: reduce) {
|
||||||
|
.model-slot[data-model-state="loading"] .model-state { animation: none; }
|
||||||
|
}
|
||||||
.picker__list--models .picker__option-name { min-width: 0; }
|
.picker__list--models .picker__option-name { min-width: 0; }
|
||||||
.model-ctx {
|
.model-ctx {
|
||||||
font-size: var(--text-xs);
|
font-size: var(--text-xs);
|
||||||
|
|||||||
@@ -1003,9 +1003,17 @@
|
|||||||
flex-direction: column;
|
flex-direction: column;
|
||||||
justify-content: flex-end;
|
justify-content: flex-end;
|
||||||
}
|
}
|
||||||
/* position: relative anchors the `@` and `/` menu to the box. */
|
/* position: relative anchors the `@` and `/` menu to the box.
|
||||||
|
|
||||||
|
`width: 100%` is the width; `max-width` only caps it. Without it the box was
|
||||||
|
as wide as its widest content: `.composer` is a flex column, and auto margins
|
||||||
|
on a flex item switch off the stretch it would otherwise get. So the hint
|
||||||
|
under it decided. "GPT-OSS has no vision, so images will not be sent" made
|
||||||
|
the box 768px, and the same screen with a model that sees images made it
|
||||||
|
538px. */
|
||||||
.composer__inner {
|
.composer__inner {
|
||||||
position: relative;
|
position: relative;
|
||||||
|
width: 100%;
|
||||||
max-width: var(--thread-max-width);
|
max-width: var(--thread-max-width);
|
||||||
margin: 0 auto;
|
margin: 0 auto;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -50,6 +50,14 @@
|
|||||||
--radius-xl: 18px;
|
--radius-xl: 18px;
|
||||||
--radius-full: 999px;
|
--radius-full: 999px;
|
||||||
|
|
||||||
|
/* The load-state dot on a model's avatar in the model menu. Its colours are
|
||||||
|
tokens of their own, defaulting to the theme's success and warning, so
|
||||||
|
an instance whose success colour is not green can still say "loaded" in
|
||||||
|
green -- that is what people read a dot beside a name as. */
|
||||||
|
--model-state-size: 0.625rem;
|
||||||
|
--model-state-loaded: var(--success);
|
||||||
|
--model-state-loading: var(--warning);
|
||||||
|
|
||||||
/*
|
/*
|
||||||
--- Controls ----------------------------------------------------------
|
--- Controls ----------------------------------------------------------
|
||||||
Every button, input and select resolves its height from these. That is the
|
Every button, input and select resolves its height from these. That is the
|
||||||
|
|||||||
@@ -222,7 +222,16 @@
|
|||||||
{
|
{
|
||||||
name: "temp",
|
name: "temp",
|
||||||
summary: "Start a temporary chat, gone after a day",
|
summary: "Start a temporary chat, gone after a day",
|
||||||
run: function () { window.location = "/chat?temporary=1"; }
|
run: function () {
|
||||||
|
/* On the new-chat screen the model, folder and kind already chosen are
|
||||||
|
in the query string. Add the flag to them rather than starting over,
|
||||||
|
or the chat is made on the default model. */
|
||||||
|
var query = new URLSearchParams(
|
||||||
|
window.location.pathname === "/chat" ? window.location.search : ""
|
||||||
|
);
|
||||||
|
query.set("temporary", "1");
|
||||||
|
window.location = "/chat?" + query.toString();
|
||||||
|
}
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
name: "stop",
|
name: "stop",
|
||||||
|
|||||||
@@ -334,6 +334,45 @@
|
|||||||
// Keep the chosen model in view when the list is long.
|
// Keep the chosen model in view when the list is long.
|
||||||
var current = menu.querySelector(".picker__option.is-selected");
|
var current = menu.querySelector(".picker__option.is-selected");
|
||||||
if (current) current.scrollIntoView({ block: "nearest" });
|
if (current) current.scrollIntoView({ block: "nearest" });
|
||||||
|
refreshStates(menu);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Which models are loaded, asked for each time the model menu opens --
|
||||||
|
llama-swap holds one at a time and it changes by the minute, so a value
|
||||||
|
rendered with the page would be stale by the time anybody looked. Only
|
||||||
|
models whose endpoint reports a state come back; everything else keeps an
|
||||||
|
empty `data-model-state`, which draws nothing. While one is loading the
|
||||||
|
menu asks again every two seconds, and stops when it closes. */
|
||||||
|
function refreshStates(menu) {
|
||||||
|
var list = menu.querySelector(".picker__list--models");
|
||||||
|
if (!list || !window.fetch) return;
|
||||||
|
clearTimeout(menu._stateTimer);
|
||||||
|
fetch("/api/models/state", {
|
||||||
|
credentials: "same-origin",
|
||||||
|
headers: { Accept: "application/json" }
|
||||||
|
}).then(function (response) {
|
||||||
|
return response.ok ? response.json() : null;
|
||||||
|
}).then(function (data) {
|
||||||
|
var states = (data && data.states) || {};
|
||||||
|
var loading = false;
|
||||||
|
list.querySelectorAll(".picker__option[data-model-id]").forEach(function (option) {
|
||||||
|
var slot = option.querySelector("[data-model-state]");
|
||||||
|
if (!slot) return;
|
||||||
|
var state = states[option.dataset.modelId] || "";
|
||||||
|
slot.dataset.modelState = state;
|
||||||
|
if (state === "loading") loading = true;
|
||||||
|
var label = slot.querySelector("[data-model-state-label]");
|
||||||
|
if (label) {
|
||||||
|
label.textContent = state === "loaded" ? list.dataset.labelLoaded
|
||||||
|
: state === "loading" ? list.dataset.labelLoading : "";
|
||||||
|
}
|
||||||
|
});
|
||||||
|
if (loading && !menu.hidden) {
|
||||||
|
menu._stateTimer = setTimeout(function () {
|
||||||
|
if (!menu.hidden) refreshStates(menu);
|
||||||
|
}, 2000);
|
||||||
|
}
|
||||||
|
}).catch(function () { /* No state is a menu without dots, not an error. */ });
|
||||||
}
|
}
|
||||||
|
|
||||||
function applyFilter(menu, needle) {
|
function applyFilter(menu, needle) {
|
||||||
|
|||||||
@@ -38,14 +38,28 @@
|
|||||||
</div>
|
</div>
|
||||||
{% endif %}
|
{% endif %}
|
||||||
|
|
||||||
<div class="picker__list picker__list--models">
|
<div class="picker__list picker__list--models"
|
||||||
|
data-label-loaded="{{ t('Loaded') }}" data-label-loading="{{ t('Loading') }}">
|
||||||
{% for model in models %}
|
{% for model in models %}
|
||||||
<button class="picker__option {{ 'is-selected' if current_model and model.model_id == current_model.model_id }}"
|
<button class="picker__option {{ 'is-selected' if current_model and model.model_id == current_model.model_id }}"
|
||||||
type="button" role="option"
|
type="button" role="option"
|
||||||
aria-selected="{{ 'true' if current_model and model.model_id == current_model.model_id else 'false' }}"
|
aria-selected="{{ 'true' if current_model and model.model_id == current_model.model_id else 'false' }}"
|
||||||
data-picker-value="{{ model.model_id }}"
|
data-picker-value="{{ model.model_id }}"
|
||||||
data-picker-search="{{ model.label|lower }} {{ model.model_id|lower }}">
|
data-picker-search="{{ model.label|lower }} {{ model.model_id|lower }}"
|
||||||
{{ model_avatar(model, cls="picker__avatar") }}
|
data-model-id="{{ model.model_id }}">
|
||||||
|
{#
|
||||||
|
The avatar in a slot of its own size, carrying the load-state dot on
|
||||||
|
its corner. A dot there takes no track, so the columns the context
|
||||||
|
sizes and the eyes line up on are the ones they had. The state is
|
||||||
|
fetched when the menu opens (ui.js, /api/models/state) rather than
|
||||||
|
rendered here: it changes by the minute, and asking every endpoint on
|
||||||
|
every page render would put a network call in front of each page.
|
||||||
|
#}
|
||||||
|
<span class="model-slot" data-model-state="">
|
||||||
|
{{ model_avatar(model, cls="picker__avatar") }}
|
||||||
|
<span class="model-state" aria-hidden="true"></span>
|
||||||
|
<span class="visually-hidden" data-model-state-label></span>
|
||||||
|
</span>
|
||||||
<span class="picker__option-name">
|
<span class="picker__option-name">
|
||||||
<span class="truncate">{{ model.label }}</span>
|
<span class="truncate">{{ model.label }}</span>
|
||||||
{% if model.pinned %}{{ icon("pin", "icon--sm picker__pin") }}{% endif %}
|
{% if model.pinned %}{{ icon("pin", "icon--sm picker__pin") }}{% endif %}
|
||||||
@@ -77,7 +91,9 @@
|
|||||||
</form>
|
</form>
|
||||||
{% else %}
|
{% else %}
|
||||||
{# No chat yet: selecting navigates so the whole composer re-renders with the
|
{# No chat yet: selecting navigates so the whole composer re-renders with the
|
||||||
right vision warning and the right hidden model_id. #}
|
right vision warning and the right hidden model_id. The URL carries the
|
||||||
<span hidden data-picker-navigate="/chat?model="></span>
|
other preselections -- temporary, folder, kind -- and ends in `model=`,
|
||||||
|
which ui.js completes. #}
|
||||||
|
<span hidden data-picker-navigate="{{ model_navigate_url }}"></span>
|
||||||
{% endif %}
|
{% endif %}
|
||||||
</div>
|
</div>
|
||||||
|
|||||||
@@ -65,11 +65,12 @@
|
|||||||
<div class="topbar__actions">
|
<div class="topbar__actions">
|
||||||
{#
|
{#
|
||||||
A link, not a script: the flag lives in the URL, so it survives a
|
A link, not a script: the flag lives in the URL, so it survives a
|
||||||
reload and can be bookmarked.
|
reload and can be bookmarked. The URL is built in `chat_index` so it
|
||||||
|
keeps the chosen model, folder and kind.
|
||||||
#}
|
#}
|
||||||
{% if not chat and can.get("chat.create") %}
|
{% if not chat and can.get("chat.create") %}
|
||||||
<a class="btn btn--icon {{ 'is-active' if starting_temporary }}"
|
<a class="btn btn--icon {{ 'is-active' if starting_temporary }}"
|
||||||
href="{{ '/chat' if starting_temporary else '/chat?temporary=1' }}"
|
href="{{ temporary_toggle_url }}"
|
||||||
aria-label="{{ t('Temporary chat') }}"
|
aria-label="{{ t('Temporary chat') }}"
|
||||||
title="{% if starting_temporary %}Starting a temporary chat. Click to go back to a normal one.{% else %}Start a temporary chat: not listed in the sidebar, and removed after a day.{% endif %}">
|
title="{% if starting_temporary %}Starting a temporary chat. Click to go back to a normal one.{% else %}Start a temporary chat: not listed in the sidebar, and removed after a day.{% endif %}">
|
||||||
{{ icon("clock") }}
|
{{ icon("clock") }}
|
||||||
|
|||||||
@@ -448,3 +448,16 @@ def test_only_an_administrator_may_customise(db, client, registered):
|
|||||||
|
|
||||||
for path in ("identity", "flavour", "css", "themes"):
|
for path in ("identity", "flavour", "css", "themes"):
|
||||||
assert client.post(f"/admin/customization/{path}", data={}).status_code == 403
|
assert client.post(f"/admin/customization/{path}", data={}).status_code == 403
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_doors_of_durin_are_a_riddle_not_a_greeting():
|
||||||
|
""""Speak, friend, and enter" invites a friend to speak. The inscription is a
|
||||||
|
riddle whose answer is to say the word *friend*, so the shipped line has no
|
||||||
|
commas. The owner caught it, and the commas must not come back in a
|
||||||
|
tidy-up."""
|
||||||
|
from lembas.services.branding import FLAVOUR
|
||||||
|
|
||||||
|
for key in ("chat_empty", "error_403"):
|
||||||
|
line = FLAVOUR[key][2]
|
||||||
|
assert line.startswith("Speak friend and enter.")
|
||||||
|
assert "Speak, friend" not in line
|
||||||
|
|||||||
@@ -58,3 +58,12 @@ def test_send_from_anywhere_never_means_stop():
|
|||||||
stops."""
|
stops."""
|
||||||
window = SOURCE[SOURCE.index('event.code === "Enter"') :][:600]
|
window = SOURCE[SOURCE.index('event.code === "Enter"') :][:600]
|
||||||
assert 'composerAction === "send"' in window
|
assert 'composerAction === "send"' in window
|
||||||
|
|
||||||
|
|
||||||
|
def test_temp_keeps_what_the_new_chat_screen_already_chose():
|
||||||
|
"""`/temp` went to a bare `/chat?temporary=1`, so on a new-chat screen with
|
||||||
|
a model picked it quietly swapped back to the default model."""
|
||||||
|
window = SOURCE[SOURCE.index('name: "temp"') :][:600]
|
||||||
|
assert 'window.location = "/chat?temporary=1"' not in window
|
||||||
|
assert "window.location.search" in window
|
||||||
|
assert 'query.set("temporary", "1")' in window
|
||||||
|
|||||||
@@ -582,3 +582,25 @@ def test_an_endpoint_with_no_props_leaves_the_list_alone(client, db, registered,
|
|||||||
|
|
||||||
db.expire_all()
|
db.expire_all()
|
||||||
assert db.get(Model, model.id).reasoning_efforts == ["low", "high"]
|
assert db.get(Model, model.id).reasoning_efforts == ["low", "high"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_new_chat_offers_the_models_own_efforts(client: TestClient, db, registered):
|
||||||
|
"""`/chat?model=` offered the generic three whatever the model took.
|
||||||
|
|
||||||
|
On Bonsai (low, medium, xhigh, default xhigh) that drew `high`, which it
|
||||||
|
rejects, and no `xhigh`, so the configured default was not an option and
|
||||||
|
the picker fell through to "off". The chat created from that screen got
|
||||||
|
`xhigh` anyway, so the control said one thing and the first reply did
|
||||||
|
another. Reported from the live instance.
|
||||||
|
"""
|
||||||
|
model = _model(db)
|
||||||
|
model.model_id = "bonsai"
|
||||||
|
model.reasoning_efforts = ["low", "medium", "xhigh"]
|
||||||
|
model.params_json = {"reasoning_effort": "xhigh"}
|
||||||
|
db.commit()
|
||||||
|
|
||||||
|
html = client.get("/chat?model=bonsai").text.replace("\n", "").replace(" ", "")
|
||||||
|
|
||||||
|
assert '<option value="xhigh" selected>' in html
|
||||||
|
assert '<option value="high"' not in html
|
||||||
|
assert '<option value="off" selected' not in html
|
||||||
|
|||||||
@@ -0,0 +1,152 @@
|
|||||||
|
"""The load-state dot in the model menu: only what an endpoint states.
|
||||||
|
|
||||||
|
llama-swap reports `"status": {"value": "loaded" | "unloaded"}` on every entry
|
||||||
|
of `GET /v1/models`, verified against the live one on 2026-09-28. A hosted API
|
||||||
|
such as DeepSeek has no such field, so its models must get no state at all --
|
||||||
|
not "unloaded", which would be a claim nobody made.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from fastapi.testclient import TestClient
|
||||||
|
|
||||||
|
from lembas.db.models import Connection, Model
|
||||||
|
from lembas.services import model_state
|
||||||
|
|
||||||
|
LLAMA_SWAP = [
|
||||||
|
{"id": "bonsai", "status": {"value": "loaded"}},
|
||||||
|
{"id": "gpt-oss", "status": {"value": "unloaded"}},
|
||||||
|
{"id": "qwen36", "status": {"value": "starting"}},
|
||||||
|
]
|
||||||
|
HOSTED = [{"id": "deepseek-flash", "object": "model"}]
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(autouse=True)
|
||||||
|
def _fresh_cache():
|
||||||
|
model_state.forget()
|
||||||
|
yield
|
||||||
|
model_state.forget()
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("entry", "state"),
|
||||||
|
[
|
||||||
|
({"status": {"value": "loaded"}}, "loaded"),
|
||||||
|
({"status": {"value": "ready"}}, "loaded"),
|
||||||
|
({"status": "loaded"}, "loaded"),
|
||||||
|
({"status": {"value": "starting"}}, "loading"),
|
||||||
|
({"status": {"value": "unloaded"}}, "unloaded"),
|
||||||
|
({"status": {"value": "stopped"}}, "unloaded"),
|
||||||
|
({}, ""),
|
||||||
|
({"status": {}}, ""),
|
||||||
|
({"status": 3}, ""),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_state_is_read_from_the_entry_or_not_at_all(entry, state):
|
||||||
|
assert model_state.state_of({"id": "x", **entry}) == state
|
||||||
|
|
||||||
|
|
||||||
|
def _two_connections(db):
|
||||||
|
local = Connection(name="llama", base_url="http://llama.test/v1", api_key_encrypted="")
|
||||||
|
hosted = Connection(name="deepseek", base_url="http://hosted.test/v1", api_key_encrypted="")
|
||||||
|
db.add_all([local, hosted])
|
||||||
|
db.commit()
|
||||||
|
served = ((local, ("bonsai", "gpt-oss", "qwen36")), (hosted, ("deepseek-flash",)))
|
||||||
|
for connection, ids in served:
|
||||||
|
for model_id in ids:
|
||||||
|
db.add(Model(connection_id=connection.id, model_id=model_id))
|
||||||
|
db.commit()
|
||||||
|
return local, hosted
|
||||||
|
|
||||||
|
|
||||||
|
def _fake_endpoints(monkeypatch, calls):
|
||||||
|
async def fake(endpoint):
|
||||||
|
calls.append(endpoint.base_url)
|
||||||
|
return LLAMA_SWAP if "llama" in endpoint.base_url else HOSTED
|
||||||
|
|
||||||
|
monkeypatch.setattr(model_state, "list_models", fake)
|
||||||
|
|
||||||
|
|
||||||
|
def test_only_models_whose_endpoint_states_one_get_a_state(
|
||||||
|
client: TestClient, db, registered, monkeypatch
|
||||||
|
):
|
||||||
|
_two_connections(db)
|
||||||
|
calls: list[str] = []
|
||||||
|
_fake_endpoints(monkeypatch, calls)
|
||||||
|
|
||||||
|
states = client.get("/api/models/state").json()["states"]
|
||||||
|
|
||||||
|
assert states == {"bonsai": "loaded", "gpt-oss": "unloaded", "qwen36": "loading"}
|
||||||
|
assert "deepseek-flash" not in states
|
||||||
|
# One request per connection, not per model.
|
||||||
|
assert sorted(calls) == ["http://hosted.test/v1", "http://llama.test/v1"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_silent_endpoint_is_not_asked_again_on_every_open(
|
||||||
|
client: TestClient, db, registered, monkeypatch
|
||||||
|
):
|
||||||
|
"""A hosted API answers with no state every time. Asking it on each click
|
||||||
|
only to hear nothing again is a request to a third party for no reason."""
|
||||||
|
_two_connections(db)
|
||||||
|
calls: list[str] = []
|
||||||
|
_fake_endpoints(monkeypatch, calls)
|
||||||
|
|
||||||
|
client.get("/api/models/state")
|
||||||
|
client.get("/api/models/state")
|
||||||
|
|
||||||
|
assert calls.count("http://hosted.test/v1") == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_an_unreachable_endpoint_is_a_menu_without_dots(
|
||||||
|
client: TestClient, db, registered, monkeypatch
|
||||||
|
):
|
||||||
|
_two_connections(db)
|
||||||
|
|
||||||
|
async def broken(endpoint):
|
||||||
|
raise OSError("connection refused")
|
||||||
|
|
||||||
|
monkeypatch.setattr(model_state, "list_models", broken)
|
||||||
|
response = client.get("/api/models/state")
|
||||||
|
|
||||||
|
assert response.status_code == 200
|
||||||
|
assert response.json() == {"states": {}}
|
||||||
|
|
||||||
|
|
||||||
|
def test_a_slow_endpoint_cannot_hold_the_menu(db, monkeypatch):
|
||||||
|
connection = Connection(name="slow", base_url="http://slow.test/v1", api_key_encrypted="")
|
||||||
|
db.add(connection)
|
||||||
|
db.commit()
|
||||||
|
db.add(Model(connection_id=connection.id, model_id="m"))
|
||||||
|
db.commit()
|
||||||
|
|
||||||
|
async def slow(endpoint):
|
||||||
|
await asyncio.sleep(10)
|
||||||
|
return LLAMA_SWAP
|
||||||
|
|
||||||
|
monkeypatch.setattr(model_state, "list_models", slow)
|
||||||
|
monkeypatch.setattr(model_state, "TIMEOUT", 0.05)
|
||||||
|
models = db.query(Model).all()
|
||||||
|
|
||||||
|
assert asyncio.run(model_state.states_for(models)) == {}
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_menu_has_a_slot_for_every_model(client: TestClient, db, registered):
|
||||||
|
"""Every option emits the slot, whatever its endpoint says: the dot is
|
||||||
|
placed by ui.js after the menu opens, so a model with no slot could never
|
||||||
|
show one."""
|
||||||
|
_two_connections(db)
|
||||||
|
html = client.get("/chat").text
|
||||||
|
|
||||||
|
for model_id in ("bonsai", "gpt-oss", "qwen36", "deepseek-flash"):
|
||||||
|
start = html.index(f'data-model-id="{model_id}"')
|
||||||
|
option = html[start : html.index("</button>", start)]
|
||||||
|
assert 'data-model-state=""' in option
|
||||||
|
assert "data-label-loaded=" in html
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_state_needs_a_signed_in_reader(client: TestClient):
|
||||||
|
response = client.get("/api/models/state", follow_redirects=False)
|
||||||
|
assert response.status_code in (401, 303, 307)
|
||||||
@@ -2,6 +2,8 @@
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import html
|
||||||
|
import re
|
||||||
from datetime import UTC, datetime, timedelta
|
from datetime import UTC, datetime, timedelta
|
||||||
|
|
||||||
from fastapi.testclient import TestClient
|
from fastapi.testclient import TestClient
|
||||||
@@ -100,6 +102,59 @@ def test_the_new_chat_screen_carries_the_flag(client: TestClient, db, registered
|
|||||||
assert 'name="temporary"' in client.get("/chat?temporary=1").text
|
assert 'name="temporary"' in client.get("/chat?temporary=1").text
|
||||||
|
|
||||||
|
|
||||||
|
def _temporary_link(page: str) -> str:
|
||||||
|
match = re.search(r'href="([^"]*)"\s+aria-label="Temporary chat"', page)
|
||||||
|
assert match, "no temporary toggle on the page"
|
||||||
|
return html.unescape(match.group(1))
|
||||||
|
|
||||||
|
|
||||||
|
def _picker_url(page: str) -> str:
|
||||||
|
match = re.search(r'data-picker-navigate="([^"]*)"', page)
|
||||||
|
assert match, "no navigating model picker on the page"
|
||||||
|
return html.unescape(match.group(1))
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_temporary_toggle_keeps_the_chosen_model(client: TestClient, db, registered):
|
||||||
|
"""Temporary chats could only be started on the default model: the toggle
|
||||||
|
linked to a bare `/chat?temporary=1`, so the model picked a moment before
|
||||||
|
was dropped."""
|
||||||
|
connection = _connection(db)
|
||||||
|
db.add(Model(connection_id=connection.id, model_id="other-model"))
|
||||||
|
db.commit()
|
||||||
|
|
||||||
|
on = _temporary_link(client.get("/chat?model=other-model").text)
|
||||||
|
assert "model=other-model" in on and "temporary=1" in on
|
||||||
|
assert 'name="model_id" value="other-model"' in client.get(on).text
|
||||||
|
|
||||||
|
off = _temporary_link(client.get(on).text)
|
||||||
|
assert "model=other-model" in off and "temporary" not in off
|
||||||
|
|
||||||
|
|
||||||
|
def test_choosing_a_model_keeps_the_temporary_flag(client: TestClient, db, registered):
|
||||||
|
"""The other half: the picker navigated to a bare `/chat?model=`, so
|
||||||
|
picking a model after the toggle quietly made the chat an ordinary one."""
|
||||||
|
connection = _connection(db)
|
||||||
|
db.add(Model(connection_id=connection.id, model_id="other-model"))
|
||||||
|
db.commit()
|
||||||
|
|
||||||
|
url = _picker_url(client.get("/chat?temporary=1").text)
|
||||||
|
assert url.endswith("model=")
|
||||||
|
page = client.get(url + "other-model").text
|
||||||
|
assert 'name="temporary"' in page
|
||||||
|
assert 'name="model_id" value="other-model"' in page
|
||||||
|
assert _picker_url(client.get("/chat").text) == "/chat?model="
|
||||||
|
|
||||||
|
|
||||||
|
def test_both_links_keep_the_folder(client: TestClient, db, registered):
|
||||||
|
_connection(db)
|
||||||
|
client.post("/api/folders", data={"name": "Quests"})
|
||||||
|
folder = db.scalar(select(Folder))
|
||||||
|
|
||||||
|
page = client.get(f"/chat?folder={folder.id}").text
|
||||||
|
assert f"folder={folder.id}" in _temporary_link(page)
|
||||||
|
assert f"folder={folder.id}" in _picker_url(page)
|
||||||
|
|
||||||
|
|
||||||
def test_starting_a_temporary_chat_sets_the_flag(client: TestClient, db, registered):
|
def test_starting_a_temporary_chat_sets_the_flag(client: TestClient, db, registered):
|
||||||
_connection(db)
|
_connection(db)
|
||||||
client.post("/api/chats/start", data={"content": "hello", "temporary": "true"})
|
client.post("/api/chats/start", data={"content": "hello", "temporary": "true"})
|
||||||
|
|||||||
@@ -219,6 +219,19 @@ def test_the_tab_bar_edge_fade_is_covered_when_nothing_overflows():
|
|||||||
assert solid >= float(shadow.group(1)) == float(sizes.group(2))
|
assert solid >= float(shadow.group(1)) == float(sizes.group(2))
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_composer_is_as_wide_as_its_column_not_its_hint():
|
||||||
|
"""With `max-width` and auto margins alone, the box was as wide as its
|
||||||
|
widest content inside a flex column, so the vision hint under it decided:
|
||||||
|
768px for GPT-OSS ("has no vision, so images will not be sent") and 538px
|
||||||
|
for a model that sees images. Reported from the live instance with four
|
||||||
|
screenshots."""
|
||||||
|
chat = (ROOT / "web/static/css/chat.css").read_text(encoding="utf-8")
|
||||||
|
start = chat.index(".composer__inner {")
|
||||||
|
rule = chat[start : chat.index("}", start)]
|
||||||
|
assert "width: 100%" in rule
|
||||||
|
assert "max-width: var(--thread-max-width)" in rule
|
||||||
|
|
||||||
|
|
||||||
def test_the_two_ends_of_the_shell_stay_level():
|
def test_the_two_ends_of_the_shell_stay_level():
|
||||||
"""The sidebar footer and the composer sit either side of the same vertical
|
"""The sidebar footer and the composer sit either side of the same vertical
|
||||||
edge and are both content-sized, so without a common floor they end at
|
edge and are both content-sized, so without a common floor they end at
|
||||||
|
|||||||
Reference in New Issue
Block a user