From 32e2326d416e32fa27c3b9d16e4734077ebea82f Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?Jaroslav=20Bene=C5=A1?=
Date: Fri, 25 Sep 2026 21:20:06 +0000
Subject: [PATCH] An effort the model had never heard of
Reported from a live instance, on Bonsai:
Jinja Exception: Unexpected reasoning effort high. Supported types are
xhigh (default), medium, and low.
Effort goes out two ways because no single field works, and the second --
chat_template_kwargs -- is not a parameter the server interprets. It is
rendered into the model's own chat template, which does not ignore a value it
does not know: it calls raise_exception, and the request dies before a token.
So a perfectly ordinary option, drawn by this application in its own menu, took
the whole reply with it.
The vocabulary is per model and nobody agrees. gpt-oss takes low/medium/high.
Bonsai takes low/medium/xhigh and refuses high. OpenAI has added minimal, xhigh
and max at different points, and which of them a given model accepts varies
again. One global tuple was going to be wrong for somebody whatever it held.
A model carries its own list now, and the picker, the slash command and the
request builder all read it. A column rather than a key in capabilities_json,
for the reason context_length is one: that dict is rebuilt wholesale from the
submitted checkboxes on every save.
And it corrects itself. A refusal retries the reply once without the effort
rather than losing it -- safe only because the template renders before any
token, so nothing has been emitted, and there is a guard that keeps it that way
-- then narrows the model's list. Bonsai's error states what it does take, so
that is what gets stored.
Note the parser bug, because it is a good one: "high" is a substring of
"xhigh", so reading the advertised list by substring learned `high` from a
sentence explaining that `high` is the problem. Whole words now, with a test
named after it.
/effort reads its levels off the picker instead of a second copy of the list
kept in the browser.
Co-Authored-By: Claude Opus 5
---
CHANGELOG.md | 24 ++++
src/lembas/__init__.py | 2 +-
src/lembas/api/admin_models.py | 17 ++-
src/lembas/api/pages.py | 10 +-
src/lembas/db/models/connection.py | 16 ++-
src/lembas/services/chat.py | 63 +++++++++-
src/lembas/services/generation.py | 119 +++++++++++++++++-
src/lembas/web/static/js/commands.js | 32 +++--
.../web/templates/admin/model_detail.html | 30 ++++-
tests/test_effort.py | 90 +++++++++++++
10 files changed, 382 insertions(+), 21 deletions(-)
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 64092df..5f94f2d 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -16,6 +16,30 @@ for 1.0.0 have something to be assembled from.
## Unreleased
+## 1.2.0
+
+- Fixed: **choosing a reasoning effort could kill the reply outright**, with a
+ Jinja traceback where the answer should have been. Reasoning effort is sent
+ two ways, and the second — `chat_template_kwargs` — is rendered into the
+ model's own chat template, which does not ignore a value it has never heard
+ of: it raises, and the whole request fails. The catch is that the vocabulary
+ is **not the same for every model**. gpt-oss takes `low/medium/high`; Bonsai
+ takes `low/medium/xhigh` and refuses `high`; OpenAI has added `minimal`,
+ `xhigh` and `max` at various points. This application offered the same three
+ to everything, so on some models the top setting was one the model would
+ throw for.
+- **A model now has its own list of the efforts it accepts**, on its page under
+ Models, and the composer's picker and `/effort` offer only those. Tick none
+ and the familiar three are used, which is right for nearly everything.
+- **And it corrects itself.** If an endpoint refuses an effort anyway — a model
+ swapped underneath a name, a runtime upgraded — that reply is retried once
+ without it instead of being lost, and the model's list is narrowed so the
+ menu stops offering something that does not work. Where the endpoint says
+ what it *does* take, that is what gets stored.
+- `/effort` now reads the levels from the picker rather than from a second copy
+ of the list kept in the browser, so the two can no longer disagree about what
+ a valid effort is.
+
## 1.1.2
Two things a phone found that 1.1.0's phone pass had not.
diff --git a/src/lembas/__init__.py b/src/lembas/__init__.py
index f7b6d4c..62184fc 100644
--- a/src/lembas/__init__.py
+++ b/src/lembas/__init__.py
@@ -1,3 +1,3 @@
"""LLeMbas - a Middle-earth themed web UI for OpenAI-compatible LLM endpoints."""
-__version__ = "1.1.2"
+__version__ = "1.2.0"
diff --git a/src/lembas/api/admin_models.py b/src/lembas/api/admin_models.py
index 1a6af95..9343975 100644
--- a/src/lembas/api/admin_models.py
+++ b/src/lembas/api/admin_models.py
@@ -177,7 +177,11 @@ async def model_detail(
"groups": list(db.scalars(select(Group).order_by(Group.name))),
"capabilities": PROTOCOL_CAPABILITIES,
"tool_capabilities": TOOL_CAPABILITIES,
+ # Every effort this application understands, so an administrator
+ # can tick the ones their model actually takes -- and the model's
+ # current answer, which is the common three until somebody says.
"efforts": chat_service.EFFORTS,
+ "model_efforts": chat_service.efforts_for(model),
# Rows predating the split have no tool_* keys at all. Showing them
# unticked would be a lie: tools.enabled_tools treats absent as on
# when `tools` is on, so that an upgrade does not silently take web
@@ -238,6 +242,7 @@ async def update_model(
position: str = Form(""),
context_length: str = Form(""),
default_effort: str = Form(""),
+ reasoning_efforts: list[str] = Form(default=[]),
group_ids: list[str] = Form(default=[]),
capability: list[str] = Form(default=[]),
) -> Response:
@@ -260,9 +265,19 @@ async def update_model(
# Merged rather than rebuilt, unlike the capabilities below: params_json
# holds whatever sampling defaults an administrator has set and this form
# only carries one of them.
+ # Which efforts this model takes at all. Submitted as a list of ticked
+ # values; empty means "nobody has said", and `chat.efforts_for` answers with
+ # the common three. Stored in the order `EFFORTS` declares rather than the
+ # order a browser happened to send.
+ chosen = [value for value in chat_service.EFFORTS if value in (reasoning_efforts or [])]
+ model.reasoning_efforts = chosen
+
params = dict(model.params_json or {})
wanted = default_effort.strip().lower()
- if wanted in chat_service.EFFORTS:
+ # Checked against what this model takes, not against everything this
+ # application has heard of -- a default of `high` on a model whose template
+ # refuses it is a chat that fails on its first turn.
+ if wanted in chat_service.efforts_for(model):
params["reasoning_effort"] = wanted
else:
params.pop("reasoning_effort", None)
diff --git a/src/lembas/api/pages.py b/src/lembas/api/pages.py
index efbdce4..3877b45 100644
--- a/src/lembas/api/pages.py
+++ b/src/lembas/api/pages.py
@@ -82,10 +82,12 @@ def _chat_context(db: DBSession, user: User, chat: Chat | None) -> dict:
else []
),
"attached_base_ids": [base.id for base in chat.knowledge_bases] if chat else [],
- # The three a reasoning model understands. From the service so the
- # command, the control and the request builder cannot disagree about
- # what is a valid effort.
- "efforts": chat_service.EFFORTS,
+ # What *this* model takes, not the three every model used to be assumed
+ # to take. The vocabulary is per model -- gpt-oss has no `xhigh` and
+ # Bonsai has no `high`, and sending the wrong one does not degrade, it
+ # raises inside the chat template and fails the reply. From the service
+ # so the command, the control and the request builder cannot disagree.
+ "efforts": chat_service.efforts_for(current) if current else chat_service.DEFAULT_EFFORTS,
# What the picker shows, and what `build_request` will send. One
# resolver so the two cannot disagree.
"resolved_effort": chat_service.resolved_effort(chat) if chat else "",
diff --git a/src/lembas/db/models/connection.py b/src/lembas/db/models/connection.py
index c91c8f9..27b4667 100644
--- a/src/lembas/db/models/connection.py
+++ b/src/lembas/db/models/connection.py
@@ -19,7 +19,7 @@ from sqlalchemy import (
from sqlalchemy.orm import Mapped, mapped_column, relationship
from lembas.db.base import Base, Timestamps, UUIDPrimaryKey
-from lembas.db.types import JSONDict
+from lembas.db.types import JSONDict, JSONList
if TYPE_CHECKING:
# Import only for the annotation; at runtime SQLAlchemy resolves the
@@ -137,6 +137,20 @@ class Model(UUIDPrimaryKey, Timestamps, Base):
# ticked anything.
context_length: Mapped[int] = mapped_column(Integer, default=0, nullable=False)
+ # Which reasoning efforts this model actually accepts. Empty means "nobody
+ # has said", and `services/chat.efforts_for` answers with the common set.
+ #
+ # It has to be per model, because the vocabulary is: gpt-oss takes
+ # low/medium/high, Bonsai takes low/medium/xhigh and *raises* on high, and
+ # OpenAI's own list has grown minimal, xhigh and max at different times. A
+ # single global tuple is a guess that is wrong for somebody.
+ #
+ # ⚠ A column and not a key in `capabilities_json`, for exactly the reason
+ # `context_length` is one: that dict is rebuilt wholesale from the submitted
+ # checkboxes on every save, so anything in it that is not a checkbox is
+ # destroyed the next time an administrator ticks anything.
+ reasoning_efforts: Mapped[list[str]] = mapped_column(JSONList, default=list)
+
connection: Mapped[Connection] = relationship(back_populates="models")
groups: Mapped[list[Group]] = relationship(
"Group", secondary=model_groups, back_populates="models"
diff --git a/src/lembas/services/chat.py b/src/lembas/services/chat.py
index 5ea1cae..eb82617 100644
--- a/src/lembas/services/chat.py
+++ b/src/lembas/services/chat.py
@@ -387,7 +387,15 @@ def build_request(
):
body["tool_choice"] = {"type": "function", "function": {"name": force_tool}}
- apply_effort(body, (chat.params_json or {}).get("reasoning_effort"))
+ # The model's own vocabulary, looked up here rather than passed in: every
+ # caller of `build_request` would otherwise have to remember, which is the
+ # trap `audio_service.template_flags` fell into.
+ chat_model = model_for(db, chat)
+ apply_effort(
+ body,
+ (chat.params_json or {}).get("reasoning_effort"),
+ efforts_for(chat_model) if chat_model is not None else None,
+ )
return body
@@ -405,7 +413,42 @@ def build_request(
# an effort on sends neither field and is byte-for-byte what it was. An endpoint
# strict about unknown parameters will refuse the extra one -- but on a chat
# somebody deliberately set an effort on, not on every chat in the instance.
-EFFORTS = ("low", "medium", "high")
+# Every reasoning effort this application understands, and the subset a model
+# gets when nobody has said otherwise.
+#
+# 🚨 These are two different questions and conflating them is what broke a
+# chat on Bonsai: `EFFORTS` was `("low", "medium", "high")` and was used both to
+# validate what somebody chose *and* to decide what to offer, so a model whose
+# vocabulary is low/medium/**xhigh** could not be given its own top setting,
+# and the one it was given -- `high` -- made its chat template call
+# `raise_exception` and took the whole reply with it.
+#
+# The known list is the union across providers, which have not agreed: OpenAI
+# has added `minimal`, `xhigh` and `max` at different points; gpt-oss takes
+# low/medium/high; Bonsai takes low/medium/xhigh and refuses high. `none` is
+# deliberately absent -- this application already spells that `off`, and two
+# spellings of off is the failure this codebase keeps cataloguing.
+EFFORTS = ("minimal", "low", "medium", "high", "xhigh", "max")
+
+# What a model is offered when its own list is empty. The three every reasoning
+# model since the first one has understood.
+DEFAULT_EFFORTS = ("low", "medium", "high")
+
+
+def efforts_for(model) -> tuple[str, ...]:
+ """The efforts this model accepts, in the order they should be offered.
+
+ A model's own list when an administrator has set one or the endpoint has
+ taught us one (see `generation._narrow_efforts`), and the common three
+ otherwise. Filtered against `EFFORTS` on the way out, so a value stored by
+ an older release -- or learned from an endpoint that advertised something
+ this application has never heard of -- cannot reach a request body.
+ """
+ stored = list(getattr(model, "reasoning_efforts", None) or [])
+ chosen = [value for value in stored if value in EFFORTS]
+ if not chosen:
+ return DEFAULT_EFFORTS
+ return tuple(value for value in EFFORTS if value in chosen)
def resolved_effort(chat) -> str:
@@ -427,9 +470,19 @@ def resolved_effort(chat) -> str:
return value if value in EFFORTS else ""
-def apply_effort(body: dict[str, Any], effort: str | None) -> None:
- """Put a chosen reasoning effort into a request body, in both forms."""
- if not effort or effort not in EFFORTS:
+def apply_effort(
+ body: dict[str, Any], effort: str | None, supported: tuple[str, ...] | None = None
+) -> None:
+ """Put a chosen reasoning effort into a request body, in both forms.
+
+ `supported` is the model's own vocabulary. An effort outside it is dropped
+ rather than sent, because the second form below is not advisory: it reaches
+ the model's Jinja chat template, and a template that does not know the value
+ raises rather than ignoring it -- which fails the whole request, not the
+ parameter.
+ """
+ allowed = supported or DEFAULT_EFFORTS
+ if not effort or effort not in allowed:
return
body["reasoning_effort"] = effort
kwargs = dict(body.get("chat_template_kwargs") or {})
diff --git a/src/lembas/services/generation.py b/src/lembas/services/generation.py
index 27a115e..8d79c41 100644
--- a/src/lembas/services/generation.py
+++ b/src/lembas/services/generation.py
@@ -19,6 +19,7 @@ import asyncio
import contextlib
import json
import logging
+import re
import time
import uuid
from dataclasses import dataclass, field, replace
@@ -454,6 +455,122 @@ def _narrower(instance: float, quota: int) -> float:
return float(min(instance, quota))
+# --- A reasoning effort the model will not take ------------------------------
+#
+# `chat_template_kwargs.reasoning_effort` is not advisory. It reaches the
+# model's Jinja chat template, and a template that does not know the value does
+# not ignore it -- gpt-oss and Bonsai both call `raise_exception`, which fails
+# the whole request. The reader sees their reply die with a Jinja traceback in
+# it, having chosen a perfectly ordinary-looking option from a menu this
+# application drew.
+#
+# So the value is checked against the model's own vocabulary before it is sent
+# (`chat.apply_effort`), and this is the second line: when it is refused anyway
+# -- an endpoint upgraded underneath us, a model whose list nobody has set --
+# the reply is retried once without it rather than lost, and the model's list is
+# narrowed so the menu stops offering something that does not work.
+
+
+def _effort_was_refused(message: str) -> bool:
+ """Whether this error is the chat template refusing the effort we sent.
+
+ Deliberately narrow. Anything that merely mentions reasoning would also
+ match a model politely declining to think, and retrying *that* silently
+ would hide a real failure behind a second request.
+ """
+ lowered = message.lower()
+ return "effort" in lowered and ("unexpected" in lowered or "supported" in lowered)
+
+
+def _advertised_efforts(message: str) -> list[str]:
+ """The efforts an error message says it will take, if it says.
+
+ Bonsai's is "Unexpected reasoning effort high. Supported types are xhigh
+ (default), medium, and low." -- which is the answer, written out, in the
+ failure. Read only from the part after "supported", so the *rejected* value
+ named in the first sentence is not collected as a supported one.
+
+ Best-effort by design: it only ever narrows what is offered, an
+ administrator can set the list by hand, and anything unrecognised is
+ dropped by `efforts_for` on the way out.
+ """
+ lowered = message.lower()
+ if "supported" not in lowered:
+ return []
+ tail = lowered.split("supported", 1)[1]
+ # Whole words. `"high" in "xhigh"` is true, so a substring test reads
+ # Bonsai's "Supported types are xhigh (default), medium, and low" as
+ # advertising `high` -- the very value it has just refused -- and the list
+ # would learn the opposite of what the endpoint said.
+ words = set(re.findall(r"[a-z]+", tail))
+ return [effort for effort in chat_service.EFFORTS if effort in words]
+
+
+def _learn_refused_effort(model_id: str, refused: str, message: str) -> None:
+ """Write what the endpoint just taught us onto the model.
+
+ Its own session: this runs from inside a generation, which outlives the
+ request's session, and the whole point is that it survives to the next turn.
+ """
+ from lembas.db.models import Model
+
+ if not model_id:
+ return
+ try:
+ with session_scope() as db:
+ models = list(db.scalars(select(Model).where(Model.model_id == model_id)))
+ for model in models:
+ advertised = _advertised_efforts(message)
+ current = list(model.reasoning_efforts or chat_service.DEFAULT_EFFORTS)
+ # What the endpoint advertised, when it did; otherwise simply
+ # the list it had, minus the one it has just refused.
+ wanted = advertised or [e for e in current if e != refused]
+ wanted = [e for e in wanted if e in chat_service.EFFORTS and e != refused]
+ if wanted and wanted != list(model.reasoning_efforts or []):
+ model.reasoning_efforts = wanted
+ log.info(
+ "model %s refused reasoning effort %r; efforts narrowed to %s",
+ model_id, refused, wanted,
+ )
+ except Exception: # noqa: BLE001 - never let bookkeeping fail a reply
+ log.exception("could not record the refused effort for model %s", model_id)
+
+
+async def _stream_once(endpoint, payload, generation, model_id: str):
+ """`stream_chat`, retried once without the reasoning effort if that is what
+ the endpoint objected to.
+
+ ⚠ The retry is only safe because the template is rendered *before* any token
+ is produced, so a refusal arrives with nothing yet emitted. `sent` is the
+ guard that keeps it that way: once a single chunk has reached the caller,
+ the reply is under way and a second request would duplicate it.
+ """
+ sent = False
+ try:
+ async for chunk in stream_chat(endpoint, payload):
+ sent = True
+ yield chunk
+ return
+ except LLMError as exc:
+ refused = str((payload.get("chat_template_kwargs") or {}).get("reasoning_effort") or "")
+ if sent or not refused or not _effort_was_refused(exc.message):
+ raise
+ log.info("retrying without reasoning effort %r: %s", refused, exc.message)
+ _learn_refused_effort(model_id, refused, exc.message)
+
+ retry = dict(payload)
+ retry.pop("reasoning_effort", None)
+ kwargs = dict(retry.get("chat_template_kwargs") or {})
+ kwargs.pop("reasoning_effort", None)
+ if kwargs:
+ retry["chat_template_kwargs"] = kwargs
+ else:
+ retry.pop("chat_template_kwargs", None)
+
+ async for chunk in stream_chat(endpoint, retry):
+ yield chunk
+
+
async def _run(generation: Generation) -> None:
"""Produce one reply, then persist it. Never raises into the task.
@@ -643,7 +760,7 @@ async def _run(generation: Generation) -> None:
# round thinks at all -- plenty of rounds do not.
round_thinking: tuple[float, float] | None = None
- async for chunk in stream_chat(endpoint, payload):
+ async for chunk in _stream_once(endpoint, payload, generation, model_id):
counts = chunk_usage(chunk)
if counts is not None:
generation.reported_usage = True
diff --git a/src/lembas/web/static/js/commands.js b/src/lembas/web/static/js/commands.js
index c63acf7..41aee98 100644
--- a/src/lembas/web/static/js/commands.js
+++ b/src/lembas/web/static/js/commands.js
@@ -271,8 +271,21 @@
/* --- Reasoning effort ---------------------------------------------------
The command drives the same select the composer shows, so there is one
- piece of state and the control updates itself when the command is used. */
- var EFFORTS = ["low", "medium", "high"];
+ piece of state and the control updates itself when the command is used.
+
+ Which efforts exist is read off that select's own options rather than
+ kept here. It used to be a second copy of `["low","medium","high"]`, which
+ was wrong the moment the vocabulary became per model: a Bonsai takes
+ `xhigh` and no `high`, so the list the server rendered and the list this
+ file believed in disagreed -- and the one that decides what `/effort xhigh`
+ does was this one. The select is the table; nothing else should hold it. */
+ function efforts() {
+ var select = el("[data-effort]");
+ if (!select) return [];
+ return Array.prototype.map
+ .call(select.options, function (option) { return option.value; })
+ .filter(function (value) { return value !== "off"; });
+ }
function setEffort(rest) {
var select = el("[data-effort]");
@@ -283,12 +296,14 @@
"error"
);
}
+ var available = efforts();
+ var listed = available.join(", ");
var wanted = (rest || "").trim().toLowerCase();
if (!wanted) {
return note(
- EFFORTS.indexOf(select.value) === -1
- ? "No effort is being sent. Try low, medium or high."
- : "Effort is " + select.value + ". /effort low, medium, high, or off."
+ available.indexOf(select.value) === -1
+ ? "No effort is being sent. Try " + listed + "."
+ : "Effort is " + select.value + ". /effort " + listed + ", or off."
);
}
/* "off" is the option's real value, not an empty string: the new-chat form
@@ -296,8 +311,11 @@
sentinel and this has to match it. "default" and "none" still work,
because somebody's fingers will type them. */
if (wanted === "default" || wanted === "none") wanted = "off";
- else if (wanted !== "off" && EFFORTS.indexOf(wanted) === -1) {
- return note("“" + wanted + "” is not an effort. Try low, medium, high or off.", "error");
+ else if (wanted !== "off" && available.indexOf(wanted) === -1) {
+ return note(
+ "“" + wanted + "” is not an effort this model takes. Try " + listed + " or off.",
+ "error"
+ );
}
select.value = wanted;
select.dispatchEvent(new Event("change", { bubbles: true }));
diff --git a/src/lembas/web/templates/admin/model_detail.html b/src/lembas/web/templates/admin/model_detail.html
index b4b4abc..76e1a04 100644
--- a/src/lembas/web/templates/admin/model_detail.html
+++ b/src/lembas/web/templates/admin/model_detail.html
@@ -104,11 +104,39 @@
+
+ Reasoning efforts this model accepts
+
+ {% for value in efforts %}
+
+ {% endfor %}
+
+
+ The vocabulary is not the same for every model, and
+ sending one a model does not know is not ignored — it is rendered into
+ the model's chat template, which raises and fails the whole reply.
+ gpt-oss takes low/medium/high; Bonsai takes
+ low/medium/xhigh and refuses
+ high; OpenAI has added
+ minimal, xhigh and
+ max at various points.
+
+ Tick none and the common three are offered, which is right for almost
+ everything. If an endpoint ever refuses one anyway, that reply is
+ retried without it and this list corrects itself — so this is worth
+ setting by hand only to save that one round trip.
+