Files
LLeMbas/src/lembas/services/settings_store.py
T
Jaroslav Beneš 7977d4ef25 One round for a chat, as many as it takes for an agent
Two different jobs were sharing one number. A plain conversation asking a
question is one round of looking things up and then an answer; the rounds after
that were a small model that had decided searching was the answer searching
until the context ran out, at a full request each. MAX_ROUNDS is 1 now. Several
tools can still be called within that round, which is the thing worth telling
the model.

The trade is real and worth naming: a plain chat can no longer search and then
read one of the results, because reading is a second round. That is what an
agent chat is for.

An agent chat is sized by Limits instead, where steps is now a runaway backstop
and not a working budget. It was 40 and it was reached -- a step count low
enough to be the thing that ends a reply is a count that ends it halfway. What
bounds one now is the wall clock and a new completion-token ceiling, with zero
meaning no ceiling, the same convention index_chars already uses.

That ceiling would have been decorative. generation.completion_tokens is only
populated when the endpoint sends a usage block, and llama.cpp, Ollama and
friends never do; the fallback estimate is computed once, in _run's finally,
long after the loop that needs it. So _written takes the larger of reported and
estimated, and there is a test that runs the whole thing against a stream
reporting no usage at all. A limit that works on OpenAI and silently does
nothing everywhere else is the worst kind: one that looks configured.

core.rounds could not stay one fragment. "You get at most N rounds" is not the
same sentence with a different number in it -- a model told it has a budget
rations it and stops early to report progress, which is exactly the behaviour
that strands a long piece of work. So it splits: core.rounds keeps the
one-round case and gates on a new round_budget variable that _agent_values
blanks, and core.keep_working says the other thing to an agent chat.

A queued message during a one-round reply is now never taken mid-reply -- there
is no work under way to steer -- and falls through to _drain, which gives it a
reply of its own. No code change went with that; it falls out of the guard, and
there is a test so that "it happens to work" and "it is meant to work" stop
looking the same.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-03 11:11:05 +02:00

282 lines
11 KiB
Python

"""Instance-wide settings that administrators can change at runtime.
Distinct from ``lembas.config``, which holds deployment configuration read from
the environment at startup. Anything here is editable from the admin UI and
lives in the ``settings`` table.
Environment variables act as the *initial* value only. Once an administrator
sets something in the UI, the stored value wins -- otherwise a toggle in the
interface would silently revert on the next restart, which is worse than not
offering the toggle at all.
"""
from __future__ import annotations
from typing import Any
from sqlalchemy.orm import Session as DBSession
from lembas.config import settings as env_settings
from lembas.db.models import Setting
GENERAL = "general"
AUDIO = "audio"
SEARCH = "search"
PROMPTS = "prompts"
AGENTS = "agents"
def _general_defaults() -> dict[str, Any]:
return {
"allow_signup": env_settings.allow_signup,
# When on, new accounts land in the `pending` role and cannot sign in
# until an administrator approves them. Reserved for the users pass.
"require_approval": False,
"instance_name": "LLeMbas",
# Applied to every chat that has no model or chat prompt of its
# own. See services.chat.effective_system_prompt.
"system_prompt": "",
# Percentage of a model's context length at which the earlier turns are
# summarised automatically. 0 turns it off; the Compact button still
# works, because a person asking for it does not need a threshold.
# Never fires for a model whose context_length is 0, since that is
# "unknown" rather than "small". See services/compaction.py.
"compact_threshold": 95,
}
def _agents_defaults() -> dict[str, Any]:
"""Agentic execution: running commands on a machine reached over SSH.
Deliberately never on the machine LLeMbas runs on. Executing here would put
the blast radius on the host holding the database and the encryption key,
and buying it back needs a sandbox, a bind list, a second unix account and
an argument about every one of them. Over SSH, isolation is somebody's
considered choice of host -- a throwaway container with one project mounted
into it, or a VM -- made with tools far better at it than anything that
could be built here.
"""
return {
# Off until an administrator turns it on. Not caution for its own sake:
# a model reads web pages, files and command output, all of them
# untrusted, so a shell is a capability somebody chooses on purpose.
"enabled": False,
# Per command.
"default_timeout": 60,
"max_timeout": 600,
"max_output_bytes": 64 * 1024,
# Per reply. See services/agent/policy.py:Limits.
#
# `max_steps` is a runaway backstop rather than a working budget: an
# agent reply is meant to run until the task is done, and a step count
# low enough to be the thing that stops it is a count that stops it
# halfway. What actually bounds a long reply is the wall clock and
# `max_completion_tokens`.
"max_steps": 200,
"max_wall_seconds": 900,
"max_total_output_bytes": 1024 * 1024,
# How much the model may *write* in one reply, across every round.
# Zero means no ceiling, which is a thing somebody may want and has no
# other way of being said -- the same convention as `index_chars`.
"max_completion_tokens": 200_000,
# How long a reply waits for someone to answer. Clamped on read: a zero
# here would park a background task forever.
"approval_timeout": 900,
"allow_default": ["file_read", "file_list", "ls *", "pwd", "git status"],
"deny_default": ["shutdown *", "reboot *", "mkfs*"],
"ask_free_text": True,
# The terminal panel: a person's own shell on their own connection.
# Separate from `enabled` because the two are different capabilities --
# one lets a model run commands, the other lets a human do what they
# could already do with an ssh client. Neither implies the other.
"terminal_enabled": True,
# Seconds with nobody watching *and* nothing typed before the session is
# closed. A build running with the panel shut is not idle. Clamped on
# read: zero would leave a shell open until the next restart.
"terminal_idle_timeout": 1800,
# Open shells across the instance, and per person. Each is a PTY and an
# SSH connection held open, so this is a real resource, not a scruple.
"terminal_max_sessions": 20,
"terminal_max_per_user": 3,
# Whether the panel's shell is given hooks that mark where one
# command ends and the next begins. Off means the Copy and Send
# buttons fall back to scraping the screen, and Auto is
# unavailable -- there is nothing to key it on.
"terminal_integration": True,
# A listing of the project directory, put in front of the model so the
# first rounds of a reply are not spent discovering what is there. It
# costs its budget on *every* request in an agent chat, forever, which
# is why it is a switch and a number rather than a constant.
"index_enabled": True,
# Characters. Clamped on read: a huge value here would quietly spend
# somebody's whole context window on filenames.
"index_chars": 2000,
}
def _audio_defaults() -> dict[str, Any]:
"""Speech-to-text and text-to-speech endpoints.
Two separate endpoints rather than one, because they usually are: a local
install runs whisper.cpp for one and Kokoro for the other. Both speak the
OpenAI audio API, so the shape below is the same on each side.
"""
return {
"stt_enabled": False,
"stt_base_url": "",
"stt_api_key_encrypted": "",
"stt_model": "whisper-1",
# Empty means "let the server detect it", which is what whisper does
# best. A forced language is an override, not a default.
"stt_language": "",
"tts_enabled": False,
"tts_base_url": "",
"tts_api_key_encrypted": "",
"tts_model": "tts-1",
"tts_voice": "",
"tts_format": "mp3",
"tts_speed": 1.0,
# The instance-wide starting point for the per-user toggle, not a
# setting that forces anything on anyone.
"tts_autoplay": False,
}
def _search_defaults() -> dict[str, Any]:
return {
"enabled": False,
"provider": "ddgs",
"max_results": 5,
"region": "wt-wt",
"safesearch": "moderate",
"searxng_base_url": "",
"firecrawl_base_url": "https://api.firecrawl.dev",
"firecrawl_api_key_encrypted": "",
"timeout": 20.0,
# Whether saving a link may reach addresses on this machine or this
# network. Off, because a server that fetches any URL it is handed can
# be pointed at a router's admin page or at LLeMbas itself, and the URL
# can come from a model. See services/fetch.py.
"allow_private_fetch": False,
}
def _prompts_defaults() -> dict[str, Any]:
"""Deliberately carries no prompt text.
The default wording of every fragment lives in ``services/prompts.py``, and
only an administrator's *override* is stored here. That is what lets a later
release improve a default and have the improvement reach every instance that
never touched that fragment -- copying the defaults in here at first save
would freeze them forever.
"""
return {
# 0 means "use services.harness.MAX_HARNESS_CHARS".
"max_harness_chars": 0,
}
_DEFAULTS: dict[str, Any] = {
GENERAL: _general_defaults,
AUDIO: _audio_defaults,
SEARCH: _search_defaults,
PROMPTS: _prompts_defaults,
AGENTS: _agents_defaults,
}
def defaults(key: str = GENERAL) -> dict[str, Any]:
"""The built-in values for a settings group, with nothing stored applied."""
factory = _DEFAULTS.get(key)
return factory() if factory else {}
def get_group(db: DBSession, key: str = GENERAL) -> dict[str, Any]:
"""Stored settings for a group, with defaults filled in for absent keys."""
values = defaults(key)
row = db.get(Setting, key)
if row is not None and isinstance(row.value, dict):
values.update(row.value)
return values
def get(db: DBSession, name: str, *, key: str = GENERAL) -> Any:
return get_group(db, key).get(name)
def update(db: DBSession, changes: dict[str, Any], *, key: str = GENERAL) -> dict[str, Any]:
"""Merge changes into a settings group and persist them."""
row = db.get(Setting, key)
if row is None:
row = Setting(key=key, value={})
db.add(row)
# Reassigned rather than mutated: SQLAlchemy only reliably detects a change
# to a JSON column when the whole value is replaced.
row.value = {**(row.value or {}), **changes}
db.commit()
return get_group(db, key)
def replace(db: DBSession, values: dict[str, Any], *, key: str = GENERAL) -> dict[str, Any]:
"""Set a settings group to exactly these values, dropping anything absent.
`update` merges, which is right for a form that posts a fixed set of fields
and wrong for one whose fields come and go -- the prompt editor stores only
the fragments an administrator has actually changed, so "no longer present"
has to mean "no longer stored". There is no other way to delete a key.
"""
row = db.get(Setting, key)
if row is None:
row = Setting(key=key, value={})
db.add(row)
row.value = dict(values)
db.commit()
return get_group(db, key)
def signup_allowed(db: DBSession) -> bool:
return bool(get(db, "allow_signup"))
def audio(db: DBSession) -> dict[str, Any]:
return get_group(db, AUDIO)
def search(db: DBSession) -> dict[str, Any]:
return get_group(db, SEARCH)
def agents(db: DBSession) -> dict[str, Any]:
"""Agent settings, with the numbers that must not be zero clamped.
`approval_timeout` of 0 would park a background task on a question nobody
is going to answer, and nothing else prunes a generation that is not
finished. `terminal_idle_timeout` of 0 would keep a PTY and an SSH
connection open until the next restart. Clamped on read rather than on save,
so a value already stored by an earlier version cannot bite either.
"""
values = get_group(db, AGENTS)
values["approval_timeout"] = min(max(int(values.get("approval_timeout") or 0), 60), 3600)
values["max_timeout"] = min(max(int(values.get("max_timeout") or 0), 1), 3600)
values["terminal_idle_timeout"] = min(
max(int(values.get("terminal_idle_timeout") or 0), 60), 86400
)
values["terminal_max_sessions"] = min(
max(int(values.get("terminal_max_sessions") or 0), 1), 500
)
values["terminal_max_per_user"] = min(max(int(values.get("terminal_max_per_user") or 0), 1), 50)
# Zero is meaningful here and is not clamped away: it means "index the
# directory for the file picker, but put none of it in the prompt", which
# is a reasonable thing to want and has no other way of being said.
values["index_chars"] = min(max(int(values.get("index_chars") or 0), 0), 20_000)
values["instructions_chars"] = min(
max(int(values.get("instructions_chars") or 0), 0), 20_000
)
# Zero is meaningful here too: no ceiling on what one reply may write.
values["max_completion_tokens"] = min(
max(int(values.get("max_completion_tokens") or 0), 0), 5_000_000
)
return values