An effort the model had never heard of
Reported from a live instance, on Bonsai: Jinja Exception: Unexpected reasoning effort high. Supported types are xhigh (default), medium, and low. Effort goes out two ways because no single field works, and the second -- chat_template_kwargs -- is not a parameter the server interprets. It is rendered into the model's own chat template, which does not ignore a value it does not know: it calls raise_exception, and the request dies before a token. So a perfectly ordinary option, drawn by this application in its own menu, took the whole reply with it. The vocabulary is per model and nobody agrees. gpt-oss takes low/medium/high. Bonsai takes low/medium/xhigh and refuses high. OpenAI has added minimal, xhigh and max at different points, and which of them a given model accepts varies again. One global tuple was going to be wrong for somebody whatever it held. A model carries its own list now, and the picker, the slash command and the request builder all read it. A column rather than a key in capabilities_json, for the reason context_length is one: that dict is rebuilt wholesale from the submitted checkboxes on every save. And it corrects itself. A refusal retries the reply once without the effort rather than losing it -- safe only because the template renders before any token, so nothing has been emitted, and there is a guard that keeps it that way -- then narrows the model's list. Bonsai's error states what it does take, so that is what gets stored. Note the parser bug, because it is a good one: "high" is a substring of "xhigh", so reading the advertised list by substring learned `high` from a sentence explaining that `high` is the problem. Whole words now, with a test named after it. /effort reads its levels off the picker instead of a second copy of the list kept in the browser. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -366,3 +366,93 @@ def test_the_picker_never_says_default(client: TestClient, db, registered):
|
||||
assert "Effort: default" not in html
|
||||
assert "Effort: off" in html
|
||||
assert '<option value="medium" selected>' in html.replace("\n", "").replace(" ", "")
|
||||
|
||||
|
||||
# --- A vocabulary that is not the same for every model -----------------------
|
||||
#
|
||||
# Reported from a real instance, on a model called Bonsai:
|
||||
#
|
||||
# Jinja Exception: Unexpected reasoning effort high. Supported types are
|
||||
# xhigh (default), medium, and low.
|
||||
#
|
||||
# `chat_template_kwargs.reasoning_effort` is rendered into the model's own chat
|
||||
# template, and a template that does not know the value calls `raise_exception`
|
||||
# rather than ignoring it -- so the whole reply died, from an option this
|
||||
# application had drawn in a menu.
|
||||
BONSAI_ERROR = (
|
||||
"Jinja Exception: Unexpected reasoning effort high. "
|
||||
"Supported types are xhigh (default), medium, and low."
|
||||
)
|
||||
|
||||
|
||||
class _FakeModel:
|
||||
def __init__(self, efforts=None):
|
||||
self.reasoning_efforts = efforts or []
|
||||
|
||||
|
||||
def test_a_model_that_has_said_nothing_gets_the_common_three():
|
||||
from lembas.services import chat as chat_service
|
||||
|
||||
assert chat_service.efforts_for(_FakeModel()) == ("low", "medium", "high")
|
||||
|
||||
|
||||
def test_a_model_can_take_xhigh_and_not_high():
|
||||
from lembas.services import chat as chat_service
|
||||
|
||||
bonsai = _FakeModel(["xhigh", "medium", "low"])
|
||||
assert chat_service.efforts_for(bonsai) == ("low", "medium", "xhigh")
|
||||
assert "high" not in chat_service.efforts_for(bonsai)
|
||||
|
||||
|
||||
def test_an_effort_the_model_refuses_is_never_sent():
|
||||
"""The check that stops the crash happening at all."""
|
||||
from lembas.services import chat as chat_service
|
||||
|
||||
supported = chat_service.efforts_for(_FakeModel(["xhigh", "medium", "low"]))
|
||||
body: dict = {}
|
||||
chat_service.apply_effort(body, "high", supported)
|
||||
assert body == {}
|
||||
|
||||
chat_service.apply_effort(body, "xhigh", supported)
|
||||
assert body["reasoning_effort"] == "xhigh"
|
||||
assert body["chat_template_kwargs"]["reasoning_effort"] == "xhigh"
|
||||
|
||||
|
||||
def test_a_value_this_application_never_heard_of_cannot_reach_a_request():
|
||||
from lembas.services import chat as chat_service
|
||||
|
||||
assert chat_service.efforts_for(_FakeModel(["ludicrous"])) == ("low", "medium", "high")
|
||||
|
||||
|
||||
def test_the_refusal_is_recognised_and_the_supported_list_read_out_of_it():
|
||||
from lembas.services import generation
|
||||
|
||||
assert generation._effort_was_refused(BONSAI_ERROR)
|
||||
assert generation._advertised_efforts(BONSAI_ERROR) == ["low", "medium", "xhigh"]
|
||||
|
||||
|
||||
def test_the_rejected_value_is_not_collected_as_a_supported_one():
|
||||
"""The message names the refused effort first and the supported ones after,
|
||||
so anything reading the whole string would learn `high` from a sentence
|
||||
saying `high` is the problem."""
|
||||
from lembas.services import generation
|
||||
|
||||
assert "high" not in generation._advertised_efforts(BONSAI_ERROR)
|
||||
|
||||
|
||||
def test_an_ordinary_failure_is_not_retried_as_an_effort_problem():
|
||||
"""Retrying a genuine failure would hide it behind a second request."""
|
||||
from lembas.services import generation
|
||||
|
||||
for message in (
|
||||
"Connection refused.",
|
||||
"The model is still loading.",
|
||||
"context length exceeded",
|
||||
):
|
||||
assert not generation._effort_was_refused(message)
|
||||
|
||||
|
||||
def test_a_model_with_no_advertisement_simply_loses_the_refused_value():
|
||||
from lembas.services import generation
|
||||
|
||||
assert generation._advertised_efforts("Unexpected reasoning effort high.") == []
|
||||
|
||||
Reference in New Issue
Block a user