A dot on the loaded model, and a new chat that matches its model
The model menu asks each connection's /v1/models for the load state llama-swap reports there and marks the loaded model; endpoints that state nothing (a hosted API) get no dot. The new-chat screen offered the generic three efforts whatever the model took, so Bonsai's xhigh default showed as off. The composer was as wide as its widest hint. And the Doors of Durin are a riddle: Speak friend and enter. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -448,3 +448,16 @@ def test_only_an_administrator_may_customise(db, client, registered):
|
||||
|
||||
for path in ("identity", "flavour", "css", "themes"):
|
||||
assert client.post(f"/admin/customization/{path}", data={}).status_code == 403
|
||||
|
||||
|
||||
def test_the_doors_of_durin_are_a_riddle_not_a_greeting():
|
||||
""""Speak, friend, and enter" invites a friend to speak. The inscription is a
|
||||
riddle whose answer is to say the word *friend*, so the shipped line has no
|
||||
commas. The owner caught it, and the commas must not come back in a
|
||||
tidy-up."""
|
||||
from lembas.services.branding import FLAVOUR
|
||||
|
||||
for key in ("chat_empty", "error_403"):
|
||||
line = FLAVOUR[key][2]
|
||||
assert line.startswith("Speak friend and enter.")
|
||||
assert "Speak, friend" not in line
|
||||
|
||||
@@ -582,3 +582,25 @@ def test_an_endpoint_with_no_props_leaves_the_list_alone(client, db, registered,
|
||||
|
||||
db.expire_all()
|
||||
assert db.get(Model, model.id).reasoning_efforts == ["low", "high"]
|
||||
|
||||
|
||||
def test_a_new_chat_offers_the_models_own_efforts(client: TestClient, db, registered):
|
||||
"""`/chat?model=` offered the generic three whatever the model took.
|
||||
|
||||
On Bonsai (low, medium, xhigh, default xhigh) that drew `high`, which it
|
||||
rejects, and no `xhigh`, so the configured default was not an option and
|
||||
the picker fell through to "off". The chat created from that screen got
|
||||
`xhigh` anyway, so the control said one thing and the first reply did
|
||||
another. Reported from the live instance.
|
||||
"""
|
||||
model = _model(db)
|
||||
model.model_id = "bonsai"
|
||||
model.reasoning_efforts = ["low", "medium", "xhigh"]
|
||||
model.params_json = {"reasoning_effort": "xhigh"}
|
||||
db.commit()
|
||||
|
||||
html = client.get("/chat?model=bonsai").text.replace("\n", "").replace(" ", "")
|
||||
|
||||
assert '<option value="xhigh" selected>' in html
|
||||
assert '<option value="high"' not in html
|
||||
assert '<option value="off" selected' not in html
|
||||
|
||||
@@ -0,0 +1,152 @@
|
||||
"""The load-state dot in the model menu: only what an endpoint states.
|
||||
|
||||
llama-swap reports `"status": {"value": "loaded" | "unloaded"}` on every entry
|
||||
of `GET /v1/models`, verified against the live one on 2026-09-28. A hosted API
|
||||
such as DeepSeek has no such field, so its models must get no state at all --
|
||||
not "unloaded", which would be a claim nobody made.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from lembas.db.models import Connection, Model
|
||||
from lembas.services import model_state
|
||||
|
||||
LLAMA_SWAP = [
|
||||
{"id": "bonsai", "status": {"value": "loaded"}},
|
||||
{"id": "gpt-oss", "status": {"value": "unloaded"}},
|
||||
{"id": "qwen36", "status": {"value": "starting"}},
|
||||
]
|
||||
HOSTED = [{"id": "deepseek-flash", "object": "model"}]
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _fresh_cache():
|
||||
model_state.forget()
|
||||
yield
|
||||
model_state.forget()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("entry", "state"),
|
||||
[
|
||||
({"status": {"value": "loaded"}}, "loaded"),
|
||||
({"status": {"value": "ready"}}, "loaded"),
|
||||
({"status": "loaded"}, "loaded"),
|
||||
({"status": {"value": "starting"}}, "loading"),
|
||||
({"status": {"value": "unloaded"}}, "unloaded"),
|
||||
({"status": {"value": "stopped"}}, "unloaded"),
|
||||
({}, ""),
|
||||
({"status": {}}, ""),
|
||||
({"status": 3}, ""),
|
||||
],
|
||||
)
|
||||
def test_state_is_read_from_the_entry_or_not_at_all(entry, state):
|
||||
assert model_state.state_of({"id": "x", **entry}) == state
|
||||
|
||||
|
||||
def _two_connections(db):
|
||||
local = Connection(name="llama", base_url="http://llama.test/v1", api_key_encrypted="")
|
||||
hosted = Connection(name="deepseek", base_url="http://hosted.test/v1", api_key_encrypted="")
|
||||
db.add_all([local, hosted])
|
||||
db.commit()
|
||||
served = ((local, ("bonsai", "gpt-oss", "qwen36")), (hosted, ("deepseek-flash",)))
|
||||
for connection, ids in served:
|
||||
for model_id in ids:
|
||||
db.add(Model(connection_id=connection.id, model_id=model_id))
|
||||
db.commit()
|
||||
return local, hosted
|
||||
|
||||
|
||||
def _fake_endpoints(monkeypatch, calls):
|
||||
async def fake(endpoint):
|
||||
calls.append(endpoint.base_url)
|
||||
return LLAMA_SWAP if "llama" in endpoint.base_url else HOSTED
|
||||
|
||||
monkeypatch.setattr(model_state, "list_models", fake)
|
||||
|
||||
|
||||
def test_only_models_whose_endpoint_states_one_get_a_state(
|
||||
client: TestClient, db, registered, monkeypatch
|
||||
):
|
||||
_two_connections(db)
|
||||
calls: list[str] = []
|
||||
_fake_endpoints(monkeypatch, calls)
|
||||
|
||||
states = client.get("/api/models/state").json()["states"]
|
||||
|
||||
assert states == {"bonsai": "loaded", "gpt-oss": "unloaded", "qwen36": "loading"}
|
||||
assert "deepseek-flash" not in states
|
||||
# One request per connection, not per model.
|
||||
assert sorted(calls) == ["http://hosted.test/v1", "http://llama.test/v1"]
|
||||
|
||||
|
||||
def test_a_silent_endpoint_is_not_asked_again_on_every_open(
|
||||
client: TestClient, db, registered, monkeypatch
|
||||
):
|
||||
"""A hosted API answers with no state every time. Asking it on each click
|
||||
only to hear nothing again is a request to a third party for no reason."""
|
||||
_two_connections(db)
|
||||
calls: list[str] = []
|
||||
_fake_endpoints(monkeypatch, calls)
|
||||
|
||||
client.get("/api/models/state")
|
||||
client.get("/api/models/state")
|
||||
|
||||
assert calls.count("http://hosted.test/v1") == 1
|
||||
|
||||
|
||||
def test_an_unreachable_endpoint_is_a_menu_without_dots(
|
||||
client: TestClient, db, registered, monkeypatch
|
||||
):
|
||||
_two_connections(db)
|
||||
|
||||
async def broken(endpoint):
|
||||
raise OSError("connection refused")
|
||||
|
||||
monkeypatch.setattr(model_state, "list_models", broken)
|
||||
response = client.get("/api/models/state")
|
||||
|
||||
assert response.status_code == 200
|
||||
assert response.json() == {"states": {}}
|
||||
|
||||
|
||||
def test_a_slow_endpoint_cannot_hold_the_menu(db, monkeypatch):
|
||||
connection = Connection(name="slow", base_url="http://slow.test/v1", api_key_encrypted="")
|
||||
db.add(connection)
|
||||
db.commit()
|
||||
db.add(Model(connection_id=connection.id, model_id="m"))
|
||||
db.commit()
|
||||
|
||||
async def slow(endpoint):
|
||||
await asyncio.sleep(10)
|
||||
return LLAMA_SWAP
|
||||
|
||||
monkeypatch.setattr(model_state, "list_models", slow)
|
||||
monkeypatch.setattr(model_state, "TIMEOUT", 0.05)
|
||||
models = db.query(Model).all()
|
||||
|
||||
assert asyncio.run(model_state.states_for(models)) == {}
|
||||
|
||||
|
||||
def test_the_menu_has_a_slot_for_every_model(client: TestClient, db, registered):
|
||||
"""Every option emits the slot, whatever its endpoint says: the dot is
|
||||
placed by ui.js after the menu opens, so a model with no slot could never
|
||||
show one."""
|
||||
_two_connections(db)
|
||||
html = client.get("/chat").text
|
||||
|
||||
for model_id in ("bonsai", "gpt-oss", "qwen36", "deepseek-flash"):
|
||||
start = html.index(f'data-model-id="{model_id}"')
|
||||
option = html[start : html.index("</button>", start)]
|
||||
assert 'data-model-state=""' in option
|
||||
assert "data-label-loaded=" in html
|
||||
|
||||
|
||||
def test_the_state_needs_a_signed_in_reader(client: TestClient):
|
||||
response = client.get("/api/models/state", follow_redirects=False)
|
||||
assert response.status_code in (401, 303, 307)
|
||||
@@ -219,6 +219,19 @@ def test_the_tab_bar_edge_fade_is_covered_when_nothing_overflows():
|
||||
assert solid >= float(shadow.group(1)) == float(sizes.group(2))
|
||||
|
||||
|
||||
def test_the_composer_is_as_wide_as_its_column_not_its_hint():
|
||||
"""With `max-width` and auto margins alone, the box was as wide as its
|
||||
widest content inside a flex column, so the vision hint under it decided:
|
||||
768px for GPT-OSS ("has no vision, so images will not be sent") and 538px
|
||||
for a model that sees images. Reported from the live instance with four
|
||||
screenshots."""
|
||||
chat = (ROOT / "web/static/css/chat.css").read_text(encoding="utf-8")
|
||||
start = chat.index(".composer__inner {")
|
||||
rule = chat[start : chat.index("}", start)]
|
||||
assert "width: 100%" in rule
|
||||
assert "max-width: var(--thread-max-width)" in rule
|
||||
|
||||
|
||||
def test_the_two_ends_of_the_shell_stay_level():
|
||||
"""The sidebar footer and the composer sit either side of the same vertical
|
||||
edge and are both content-sized, so without a common floor they end at
|
||||
|
||||
Reference in New Issue
Block a user