Files
LLeMbas/src/lembas/api/admin.py
T
Jaroslav Beneš 17f3fa1946 Compaction: a button, and automatically when the window fills
A long conversation eventually just stops working. Compaction summarises the
earlier turns and sends the summary in their place.

The messages are kept. They stay in the transcript behind a collapsed
divider and simply stop being part of the request, which is what makes the
button safe to press and automatic compaction safe to have at all: a summary
that came out badly is a bad turn, not a lost conversation.

Stored on the Chat, not as a synthetic Message. A synthetic row needs a
role -- `system` breaks the one-system-message rule the moment build_messages
emits it beside the harness, and user/assistant makes it a turn people can
edit, regenerate from and copy, indistinguishable from a real one in all
four places a bubble is rendered. Worse, "editing rewinds, it does not
branch" would silently delete it and leave no marker that compaction had
happened at all.

The summary goes out as a user turn and an assistant turn, not one. A
leading assistant breaks templates requiring the first non-system message to
be user; a lone leading user produces user, user whenever the kept history
starts on a user turn -- which it always does, because the cutoff lands on a
finished reply.

compacted_through_id is a plain id rather than a foreign key: migrations.py
compiles only the column type, so a REFERENCES clause would exist on a fresh
database and not on an upgraded one, and a constraint half the fleet has is
worse than none. cutoff_message validates it on every read instead, and a
rewind past the boundary clears it.

Compacting again summarises only the delta, with the previous summary
supplied to be subsumed. Re-summarising the whole chat each time grows
quadratically and eventually exceeds the window it is protecting.

Automatically at the top of _run, not in post_message: that route's contract
is to return immediately and leave the slow part to a resumable connection,
and it also means build_request is called once, after compaction, with no
second assembly path. The trigger is the last reply's recorded usage plus an
estimate of the new turn -- retrospective because true prompt_tokens are only
knowable after a response, plus the delta because otherwise fifty thousand
characters pasted into the composer overflow a window that read 90% last
turn. It never fires when the context length is unknown. It does fire on
estimated counts, which is safe here precisely because nothing is lost.

_maybe_compact never raises: a failure logs and sends the uncompacted
request. A `status` event says "Summarising earlier messages…" in the
meantime, because a silent multi-second pause before the first token is what
a hang looks like.

The wording is three fragments under Admin - Prompts. Clearing task.compact
turns compaction off entirely.

Also adds compaction.moment(): SQLite does not store the offset, so a row
loaded from disk is naive while one in the session's identity map keeps its
tzinfo, and comparing the two raises. Every comparison here is between
exactly those.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-01 01:02:02 +02:00

247 lines
8.5 KiB
Python

"""Administration: OpenAI-compatible connections and their models."""
from __future__ import annotations
import logging
from datetime import UTC, datetime
from fastapi import APIRouter, Form, HTTPException, Request, Response, status
from fastapi.responses import RedirectResponse
from sqlalchemy import func, select
from sqlalchemy.orm import Session as DBSession
from lembas.api.deps import AdminUser, Db
from lembas.db.models import Connection, Model, User
from lembas.services import settings_store
from lembas.services.crypto import UNCHANGED_SENTINEL, decrypt, encrypt, mask
from lembas.services.llm.openai_client import Endpoint, LLMError, context_from, list_models
from lembas.web.templating import render
log = logging.getLogger(__name__)
router = APIRouter(prefix="/admin", tags=["admin"])
def _connection(db: DBSession, connection_id: str) -> Connection:
connection = db.get(Connection, connection_id)
if connection is None:
raise HTTPException(status.HTTP_404_NOT_FOUND, "That connection no longer exists.")
return connection
def _connections(db: DBSession) -> list[Connection]:
return list(db.scalars(select(Connection).order_by(Connection.position, Connection.name)))
@router.get("")
async def admin_home(user: AdminUser):
return RedirectResponse("/admin/general", status_code=status.HTTP_303_SEE_OTHER)
@router.get("/general")
async def general_page(request: Request, db: Db, user: AdminUser, saved: bool = False):
return render(
request,
"admin/general.html",
{
"values": settings_store.get_group(db),
"saved": saved,
"user_count": db.scalar(select(func.count()).select_from(User)),
},
)
@router.post("/general")
async def save_general(
db: Db,
user: AdminUser,
instance_name: str = Form("LLeMbas"),
allow_signup: bool = Form(False),
system_prompt: str = Form(""),
compact_threshold: int = Form(95),
) -> Response:
"""Save instance settings.
Unchecked checkboxes are simply absent from a form post, which is why
allow_signup defaults to False here -- that absence *is* the "off" signal.
"""
settings_store.update(
db,
{
"instance_name": instance_name.strip()[:120] or "LLeMbas",
"allow_signup": allow_signup,
"system_prompt": system_prompt.strip()[:8000],
# 0 is "never"; anything else is clamped into a band where it can
# do some good. 100 is useless -- you cannot compact after
# overflowing -- and below 50 it fires while there is plenty left.
"compact_threshold": (
0 if compact_threshold <= 0 else min(max(compact_threshold, 50), 99)
),
},
)
log.info("registration %s by %s", "opened" if allow_signup else "closed", user.email)
return RedirectResponse("/admin/general?saved=1", status_code=status.HTTP_303_SEE_OTHER)
@router.get("/connections")
async def connections_page(request: Request, db: Db, user: AdminUser, message: str = ""):
connections = _connections(db)
return render(
request,
"admin/connections.html",
{
"connections": connections,
"masked": {c.id: mask(decrypt(c.api_key_encrypted)) for c in connections},
"model_counts": {
c.id: sum(1 for m in c.models if m.enabled) for c in connections
},
"message": message,
"unchanged": UNCHANGED_SENTINEL,
},
)
@router.post("/connections")
async def create_connection(
db: Db,
user: AdminUser,
name: str = Form(...),
base_url: str = Form(...),
api_key: str = Form(""),
) -> Response:
base_url = base_url.strip().rstrip("/")
if not base_url.startswith(("http://", "https://")):
raise HTTPException(
status.HTTP_400_BAD_REQUEST,
"The base URL must start with http:// or https://",
)
position = db.scalar(select(func.coalesce(func.max(Connection.position), -1))) + 1
connection = Connection(
name=name.strip()[:120] or "Connection",
base_url=base_url,
api_key_encrypted=encrypt(api_key.strip()),
position=position,
)
db.add(connection)
db.commit()
# Discover models immediately: a connection that lists nothing is
# indistinguishable from a broken one, and finding out now is the point.
await _refresh_models(db, connection)
return RedirectResponse("/admin/connections", status_code=status.HTTP_303_SEE_OTHER)
@router.post("/connections/{connection_id}")
async def update_connection(
db: Db,
user: AdminUser,
connection_id: str,
name: str = Form(...),
base_url: str = Form(...),
api_key: str = Form(""),
enabled: bool = Form(False),
) -> Response:
connection = _connection(db, connection_id)
connection.name = name.strip()[:120] or connection.name
connection.base_url = base_url.strip().rstrip("/")
connection.enabled = enabled
submitted = api_key.strip()
if submitted and submitted != UNCHANGED_SENTINEL:
connection.api_key_encrypted = encrypt(submitted)
elif not submitted:
# An explicitly emptied field means "this endpoint needs no key".
connection.api_key_encrypted = ""
db.commit()
return RedirectResponse("/admin/connections", status_code=status.HTTP_303_SEE_OTHER)
@router.post("/connections/{connection_id}/test")
async def test_connection(
request: Request, db: Db, user: AdminUser, connection_id: str
) -> Response:
"""Contact the endpoint and refresh its model list."""
connection = _connection(db, connection_id)
count, error = await _refresh_models(db, connection)
message = (
f"{connection.name}: {error}"
if error
else f"{connection.name}: found {count} model{'s' if count != 1 else ''}."
)
return render(
request,
"admin/_connection_row.html",
{
"connection": connection,
"masked": mask(decrypt(connection.api_key_encrypted)),
"message": message,
"message_kind": "error" if error else "success",
"unchanged": UNCHANGED_SENTINEL,
},
)
async def _refresh_models(db: DBSession, connection: Connection) -> tuple[int, str]:
"""Sync the cached model list. Returns (count, error message)."""
try:
discovered = await list_models(Endpoint.from_connection(connection))
except LLMError as exc:
connection.last_error = exc.message
connection.last_checked_at = datetime.now(UTC)
db.commit()
return 0, exc.message
existing = {model.model_id: model for model in connection.models}
seen: set[str] = set()
# New models land after everything already ordered, rather than all at
# position 0 where they would sort by id and shuffle the existing list.
# No `or -1` after the coalesce: position 0 is falsy, so that idiom sent the
# second discovered model back to 0 on top of the first.
highest = db.scalar(select(func.coalesce(func.max(Model.position), -1)))
next_position = int(highest if highest is not None else -1) + 1
for entry in discovered:
model_id = str(entry["id"])[:300]
seen.add(model_id)
if model_id in existing:
# A context length is filled in only when nobody has one yet. A
# refresh must never overwrite a number an administrator typed --
# they are usually correcting the endpoint.
model = existing[model_id]
if not model.context_length:
model.context_length = context_from(entry)
continue
db.add(
Model(
connection_id=connection.id,
model_id=model_id,
position=next_position,
context_length=context_from(entry),
)
)
next_position += 1
# Models that vanished upstream are dropped, so the picker never offers
# something the endpoint will reject.
for model_id, model in existing.items():
if model_id not in seen:
db.delete(model)
connection.last_error = ""
connection.last_checked_at = datetime.now(UTC)
db.commit()
log.info("connection %s: %d models", connection.name, len(seen))
return len(seen), ""
@router.post("/connections/{connection_id}/delete")
async def delete_connection(db: Db, user: AdminUser, connection_id: str) -> Response:
connection = _connection(db, connection_id)
db.delete(connection)
db.commit()
return RedirectResponse("/admin/connections", status_code=status.HTTP_303_SEE_OTHER)