59739cc7fd
The second audit pass. Four things, and the first two were reported. The Prompts page put a screen of variables and a screen of preview above the editor, so the tabs began two screens down and switching one had to drag the whole page to be any use -- and on a short tab it could not drag far enough, leaving the panel stranded above a screenful of nothing. Editor first, reference after, bar sticky. Custom themes were three fixed slots: fifty-seven empty colour boxes on a fresh instance and no way to make a fourth theme. One block per theme plus a blank one, colours behind a disclosure. Both measured rather than argued about -- rendered through TestClient and driven under headless Chromium, where the tab bar moved 385->642px before and does not move now, and the themes page went from 5495px to 2820px. Asking where generated images go found the other two. Deleting a chat cascades to the attachment rows and leaves every file on disk; the helper written for exactly that was called from one place, and it was not the delete button, a schedule's chat, a helper's chat or deleting an account. Underneath it, `claim` bound message_id and never chat_id, so anything picked before a chat existed kept an empty chat_id forever -- which six readers filter on, so those files were also unnamed in the prompt, unopenable in the canvas, and invisible to the one caller the cleanup had. And folders nest now. The route has handled parent_id since folders existed, with a cycle guard and a depth cap the move path never applied; the sidebar has always drawn a tree. Nothing could ask for one. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
713 lines
28 KiB
Python
713 lines
28 KiB
Python
"""Chat orchestration: building requests, streaming replies, naming chats."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from datetime import UTC, datetime, timedelta
|
|
from typing import Any
|
|
|
|
from sqlalchemy import func, select
|
|
from sqlalchemy.orm import Session as DBSession
|
|
|
|
from lembas.db.models import (
|
|
KIND_MESSAGES,
|
|
ROLE_ASSISTANT,
|
|
ROLE_SYSTEM,
|
|
ROLE_USER,
|
|
Chat,
|
|
Connection,
|
|
Message,
|
|
Model,
|
|
)
|
|
from lembas.services import files as files_service
|
|
from lembas.services.llm.openai_client import Endpoint, LLMError, complete
|
|
|
|
log = logging.getLogger(__name__)
|
|
|
|
# Sampling keys forwarded upstream. Anything else a user puts in params_json is
|
|
# ignored rather than passed through, so a typo cannot produce a 400 from the
|
|
# provider that looks like a LLeMbas bug.
|
|
FORWARDED_PARAMS = frozenset(
|
|
{"temperature", "top_p", "max_tokens", "presence_penalty", "frequency_penalty",
|
|
"seed", "stop"}
|
|
)
|
|
|
|
MAX_TITLE_LENGTH = 60
|
|
|
|
# What one title call may spend. A title is a handful of words; the rest of this
|
|
# is headroom for a model that thinks before it answers, which is most of the
|
|
# interesting local ones. Too small is not a shorter title -- it is no title at
|
|
# all, because the thinking consumes the budget and the content field comes back
|
|
# empty or holding an unclosed `<think>`.
|
|
TITLE_MAX_TOKENS = 512
|
|
|
|
# How long a temporary chat survives after the last thing said in it.
|
|
TEMPORARY_LIFETIME = timedelta(hours=24)
|
|
|
|
|
|
def resolve_endpoint(db: DBSession, chat: Chat) -> tuple[Endpoint, str]:
|
|
"""Find the connection and model a chat should use.
|
|
|
|
Chats store the model id as text rather than a foreign key so history
|
|
survives an admin deleting a connection, which means the mapping back to a
|
|
live connection has to be resolved at send time and can legitimately fail.
|
|
"""
|
|
if not chat.model_id:
|
|
raise LLMError("This chat has no model selected.")
|
|
|
|
connection: Connection | None = None
|
|
if chat.connection_id:
|
|
connection = db.get(Connection, chat.connection_id)
|
|
|
|
if connection is None or not connection.enabled:
|
|
# The original connection is gone or disabled. Any enabled connection
|
|
# still offering this model id will do.
|
|
model = db.scalar(
|
|
select(Model)
|
|
.join(Connection)
|
|
.where(
|
|
Model.model_id == chat.model_id,
|
|
Model.enabled.is_(True),
|
|
Connection.enabled.is_(True),
|
|
)
|
|
.order_by(Connection.position)
|
|
)
|
|
if model is None:
|
|
raise LLMError(
|
|
f"No enabled connection currently offers the model "
|
|
f"'{chat.model_id}'. Pick another model for this chat."
|
|
)
|
|
connection = model.connection
|
|
chat.connection_id = connection.id
|
|
db.commit()
|
|
|
|
return Endpoint.from_connection(connection), chat.model_id
|
|
|
|
|
|
def document_context(message: Message) -> str:
|
|
"""Extracted text from a message's non-image attachments.
|
|
|
|
Wrapped in named tags so the model can tell one document from another, and
|
|
tell all of them from what the user actually typed. Truncation is stated
|
|
inline rather than silently, so a model asked about page 400 of a 300-page
|
|
extract can say it did not see it.
|
|
"""
|
|
blocks: list[str] = []
|
|
for attachment in message.documents:
|
|
if not attachment.extracted_text.strip():
|
|
continue
|
|
note = " (truncated)" if attachment.truncated else ""
|
|
# Where it came from, when there is a where. A model handed `main.py`
|
|
# cannot tell which of four it is looking at, and cannot name the file
|
|
# back when asked to change something -- so a file read off a machine
|
|
# says which machine and which path. Quotes are stripped rather than
|
|
# escaped: these are attribute values in a tag the model reads, and a
|
|
# path containing one would otherwise close it early.
|
|
where = ""
|
|
if attachment.source_path:
|
|
where += f' path="{_attr(attachment.source_path)}"'
|
|
if attachment.source_label:
|
|
where += f' from="{_attr(attachment.source_label)}"'
|
|
blocks.append(
|
|
f'<document name="{_attr(attachment.filename)}"{where}{note}>\n'
|
|
f"{attachment.extracted_text.strip()}\n"
|
|
f"</document>"
|
|
)
|
|
return "\n\n".join(blocks)
|
|
|
|
|
|
def _attr(value: str) -> str:
|
|
"""A value safe to sit inside the double quotes of a tag we are writing."""
|
|
return value.replace('"', "").replace("<", "").replace(">", "").replace("\n", " ")
|
|
|
|
|
|
def message_payload(message: Message, *, vision: bool) -> dict[str, Any]:
|
|
"""One history entry in the shape the endpoint expects.
|
|
|
|
Plain text stays a plain string: sending the multimodal list form to an
|
|
endpoint that does not implement it is a reliable way to get a 400, and
|
|
most local runners do not.
|
|
"""
|
|
text = message.content.strip()
|
|
|
|
documents = document_context(message)
|
|
if documents:
|
|
# Documents lead so the question that follows has its material already
|
|
# in view, which is how these models are trained to read a prompt.
|
|
text = f"{documents}\n\n{text}" if text else documents
|
|
|
|
# Images ride on a *user* turn and nowhere else. Until image generation
|
|
# existed no assistant message had ever carried one, so this was never a
|
|
# distinction worth drawing -- and the moment one does, the multimodal list
|
|
# form on an `assistant` turn is rejected outright by OpenAI and by most
|
|
# local runners, which would break not that turn but every later one in the
|
|
# chat. What follows from it, and is worth knowing rather than discovering:
|
|
# a model cannot see the picture it made on a *subsequent* turn (tool
|
|
# results are not replayed either), so "make it bluer" regenerates rather
|
|
# than edits. Honest for a text-to-image workflow with no img2img path.
|
|
images = message.images if (vision and message.role == ROLE_USER) else []
|
|
if not images:
|
|
return {"role": message.role, "content": text}
|
|
|
|
parts: list[dict[str, Any]] = []
|
|
if text:
|
|
parts.append({"type": "text", "text": text})
|
|
for attachment in images:
|
|
uri = files_service.data_uri(attachment)
|
|
if uri is None:
|
|
# The row survived but the file did not. Better to say so than to
|
|
# send a turn that silently lost its picture.
|
|
log.warning("attachment %s has no file on disk", attachment.id)
|
|
continue
|
|
parts.append({"type": "image_url", "image_url": {"url": uri}})
|
|
|
|
if not parts:
|
|
return {"role": message.role, "content": text}
|
|
return {"role": message.role, "content": parts}
|
|
|
|
|
|
def folder_system_prompt(db: DBSession, chat: Chat) -> str:
|
|
"""The nearest prompt on the chat's folder, or on a folder above it.
|
|
|
|
Walks up rather than reading one level, because folders nest and a project's
|
|
prompt belongs on the project rather than on each sub-folder of it. The
|
|
nearest one wins, which is the same rule the ladder as a whole follows.
|
|
|
|
Bounded and cycle-safe the way `api/folders.py:_depth_of` is. Reparenting
|
|
already refuses to build a cycle, but this runs on the request path for
|
|
every reply and a row written by something else must not be able to hang it.
|
|
"""
|
|
from lembas.db.models import Folder
|
|
|
|
folder = chat.folder
|
|
seen: set[str] = set()
|
|
while folder is not None and folder.id not in seen:
|
|
seen.add(folder.id)
|
|
if (folder.system_prompt or "").strip():
|
|
return folder.system_prompt.strip()
|
|
folder = db.get(Folder, folder.parent_id) if folder.parent_id else None
|
|
return ""
|
|
|
|
|
|
def effective_system_prompt(db: DBSession, chat: Chat) -> str:
|
|
"""The system prompt a chat actually runs with.
|
|
|
|
Four layers, most specific wins outright:
|
|
|
|
chat > folder > model > instance
|
|
|
|
Precedence rather than concatenation. Stacking them reads well in a
|
|
settings screen and badly in practice: the moment two layers disagree the
|
|
model gets contradictory instructions and nobody can tell which one is
|
|
losing. With precedence, "why is it behaving like this" has one answer.
|
|
|
|
The folder sits above the model because it is the more specific statement:
|
|
a model's prompt describes the model wherever it is used, and a folder's
|
|
describes this piece of work whichever model is pointed at it.
|
|
"""
|
|
from lembas.services import settings_store
|
|
|
|
if chat.system_prompt.strip():
|
|
return chat.system_prompt.strip()
|
|
|
|
if inherited := folder_system_prompt(db, chat):
|
|
return inherited
|
|
|
|
model = db.scalar(
|
|
select(Model).where(Model.model_id == chat.model_id).order_by(Model.position)
|
|
)
|
|
if model is not None and (model.system_prompt or "").strip():
|
|
return model.system_prompt.strip()
|
|
|
|
return (settings_store.get(db, "system_prompt") or "").strip()
|
|
|
|
|
|
def build_messages(
|
|
db: DBSession,
|
|
chat: Chat,
|
|
*,
|
|
upto: Message | None = None,
|
|
vision: bool = False,
|
|
system_prompt: str | None = None,
|
|
) -> list[dict]:
|
|
"""Assemble the message list to send upstream.
|
|
|
|
`upto` excludes the placeholder assistant row being generated into, and
|
|
everything after it. `system_prompt` overrides what would otherwise be
|
|
resolved, which is how the harness gets in front of the authored prompt
|
|
without this function knowing anything about tools.
|
|
"""
|
|
from lembas.services import compaction as compaction_service
|
|
from lembas.services import prompts as prompts_service
|
|
|
|
payload: list[dict[str, Any]] = []
|
|
system = effective_system_prompt(db, chat) if system_prompt is None else system_prompt
|
|
if system:
|
|
payload.append({"role": ROLE_SYSTEM, "content": system})
|
|
|
|
# Compacted turns are replaced by a summary carried in two turns rather than
|
|
# one. A leading `assistant` breaks templates that require the first
|
|
# non-system message to be `user`; a lone leading `user` produces user, user
|
|
# whenever the kept history starts on a user turn -- which it always does,
|
|
# because the cutoff lands on a finished reply. The pair alternates
|
|
# correctly in both directions and keeps exactly one system message.
|
|
cutoff = compaction_service.cutoff_message(db, chat)
|
|
if cutoff is not None:
|
|
lead = prompts_service.resolve(db, "task.compact_lead").strip()
|
|
ack = prompts_service.resolve(db, "task.compact_ack").strip()
|
|
summary = chat.compact_summary.strip()
|
|
payload.append(
|
|
{"role": ROLE_USER, "content": f"{lead}\n\n{summary}" if lead else summary}
|
|
)
|
|
if ack:
|
|
payload.append({"role": ROLE_ASSISTANT, "content": ack})
|
|
|
|
history = db.scalars(
|
|
select(Message).where(Message.chat_id == chat.id).order_by(Message.created_at)
|
|
).all()
|
|
|
|
# The Messages conversation never ends, so it cannot all be sent. Only the
|
|
# most recent turns go; everything before them stays on screen and out of
|
|
# the request. One branch, and the bound is applied before the loop rather
|
|
# than inside it so the filters below still see a contiguous tail.
|
|
#
|
|
# Not compaction: that summarises with a model call and a threshold, on a
|
|
# conversation somebody decided to shorten. This is mechanical, lossless and
|
|
# permanent, which is why `compaction.should_compact` refuses this kind --
|
|
# two mechanisms fighting over one transcript is how you get a summary of a
|
|
# summary.
|
|
if chat.kind == KIND_MESSAGES:
|
|
from lembas.services import messages as messages_service
|
|
|
|
history = history[-messages_service.LIVE_CHUNK :]
|
|
|
|
for message in history:
|
|
if upto is not None and message.id == upto.id:
|
|
break
|
|
if cutoff is not None and compaction_service.moment(
|
|
message
|
|
) <= compaction_service.moment(cutoff):
|
|
continue
|
|
# Typed while the previous reply was still being written, and not yet
|
|
# handed to a model. It is in the transcript and it is not in the
|
|
# request; delivery is what moves it from one to the other.
|
|
if message.queued:
|
|
continue
|
|
# Skip turns that failed or produced nothing -- but a message carrying
|
|
# only an attachment has no text and must still be sent.
|
|
if message.error:
|
|
continue
|
|
if not message.content.strip() and not message.attachments:
|
|
continue
|
|
payload.append(message_payload(message, vision=vision))
|
|
|
|
return payload
|
|
|
|
|
|
def model_for(db: DBSession, chat: Chat) -> Model | None:
|
|
"""The Model row a chat is using, or None if it has gone.
|
|
|
|
Looked up by id rather than held as a foreign key, for the same reason
|
|
resolve_endpoint does: chats store the model as text so history survives an
|
|
administrator deleting a connection.
|
|
"""
|
|
return db.scalar(
|
|
select(Model).where(Model.model_id == chat.model_id).order_by(Model.position)
|
|
)
|
|
|
|
|
|
def model_supports(db: DBSession, chat: Chat, capability: str) -> bool:
|
|
"""Whether the chat's current model is marked as having a capability."""
|
|
model = model_for(db, chat)
|
|
return bool(model and (model.capabilities_json or {}).get(capability))
|
|
|
|
|
|
def build_request(
|
|
db: DBSession,
|
|
chat: Chat,
|
|
*,
|
|
upto: Message | None = None,
|
|
tools: list[dict[str, Any]] | None = None,
|
|
user=None,
|
|
force_tool: str = "",
|
|
) -> dict[str, Any]:
|
|
"""The whole request body, tools and harness included.
|
|
|
|
Composed here rather than in the generation loop so that "what gets sent"
|
|
has one answer, and so the harness cannot be forgotten by a future caller
|
|
that offers tools.
|
|
"""
|
|
from lembas.services import harness as harness_service
|
|
from lembas.services import prompts as prompts_service
|
|
|
|
params = {
|
|
key: value
|
|
for key, value in (chat.params_json or {}).items()
|
|
if key in FORWARDED_PARAMS and value not in (None, "")
|
|
}
|
|
# Images are only sent to a model an administrator has marked as having
|
|
# vision. Sending them to one that has not is not a graceful degradation:
|
|
# most endpoints reject the whole request.
|
|
vision = model_supports(db, chat, "vision")
|
|
|
|
if user is None:
|
|
from lembas.db.models import User
|
|
|
|
user = db.get(User, chat.user_id)
|
|
|
|
# The harness describes the tools; the authored prompt describes the
|
|
# behaviour. See services/harness.py for why these are joined rather than
|
|
# being two competing layers.
|
|
system = harness_service.join(
|
|
harness_service.compose(db, user, tools, chat),
|
|
effective_system_prompt(db, chat),
|
|
lead=prompts_service.render(db, "seam.authored_lead", {}),
|
|
)
|
|
|
|
body: dict[str, Any] = {
|
|
"model": chat.model_id,
|
|
"messages": build_messages(
|
|
db, chat, upto=upto, vision=vision, system_prompt=system
|
|
),
|
|
**params,
|
|
}
|
|
if tools:
|
|
body["tools"] = tools
|
|
# Making the model call one particular tool, for `/image` -- the whole
|
|
# of what that command is. Only ever sent alongside a tools array and
|
|
# only when something asked for it, so a provider strict about unknown
|
|
# parameters sees exactly the request it always did until somebody types
|
|
# a slash command.
|
|
#
|
|
# An endpoint that ignores `tool_choice` is not a failure here: the turn
|
|
# still carries the instruction in words, so the model is being steered
|
|
# twice and the weaker half is the one that can be dropped.
|
|
if force_tool and any(
|
|
(tool.get("function") or {}).get("name") == force_tool for tool in tools
|
|
):
|
|
body["tool_choice"] = {"type": "function", "function": {"name": force_tool}}
|
|
|
|
apply_effort(body, (chat.params_json or {}).get("reasoning_effort"))
|
|
return body
|
|
|
|
|
|
# Reasoning effort, and why it goes out twice.
|
|
#
|
|
# There is no one field that works. OpenAI and vLLM read a plain
|
|
# `reasoning_effort`. llama.cpp reads it too and, per its own documentation,
|
|
# "other values (e.g. 'low', 'max') have no effect" -- its maintainer is blunter
|
|
# still: "llama-server cannot support reasoning_effort at all", and the field
|
|
# "simply gets dropped without error or logging". What *does* reach a gpt-oss
|
|
# behind llama.cpp is `chat_template_kwargs`, which it accepts per request.
|
|
#
|
|
# So both are sent, and only when an effort has actually been chosen. That
|
|
# second half is what keeps this from being a regression: a chat nobody has set
|
|
# an effort on sends neither field and is byte-for-byte what it was. An endpoint
|
|
# strict about unknown parameters will refuse the extra one -- but on a chat
|
|
# somebody deliberately set an effort on, not on every chat in the instance.
|
|
EFFORTS = ("low", "medium", "high")
|
|
|
|
|
|
def resolved_effort(chat) -> str:
|
|
"""The effort this chat will actually send, or "" for none.
|
|
|
|
Its own value, and nothing else. The model's default is a **seed** applied
|
|
when the chat is created (`api/chats.py:_new_chat`) and on a model change,
|
|
and is deliberately not consulted here for two reasons. A chat's request
|
|
should be a function of the chat row alone -- the same rule that has PDF
|
|
text extracted once at upload and knowledge attachments copied. And a
|
|
fallback would break "off": `update_chat` stores `None` for a cleared
|
|
effort, a fallback would resurrect the model's default underneath it, and
|
|
the off option would silently do nothing.
|
|
|
|
The picker shows exactly this, which is the whole point of it existing:
|
|
"Effort: default" named no level and was true of nothing in particular.
|
|
"""
|
|
value = (getattr(chat, "params_json", None) or {}).get("reasoning_effort")
|
|
return value if value in EFFORTS else ""
|
|
|
|
|
|
def apply_effort(body: dict[str, Any], effort: str | None) -> None:
|
|
"""Put a chosen reasoning effort into a request body, in both forms."""
|
|
if not effort or effort not in EFFORTS:
|
|
return
|
|
body["reasoning_effort"] = effort
|
|
kwargs = dict(body.get("chat_template_kwargs") or {})
|
|
kwargs["reasoning_effort"] = effort
|
|
body["chat_template_kwargs"] = kwargs
|
|
|
|
|
|
def default_model(db: DBSession, user=None) -> tuple[str, str] | None:
|
|
"""The model a new chat should start with, as (model_id, connection_id).
|
|
|
|
Preference order: the user's own choice, then the instance default, then
|
|
whatever is first in the admin's ordering. Each is checked against what the
|
|
user may actually reach, so a default they have lost access to falls
|
|
through rather than producing a chat they cannot use.
|
|
"""
|
|
from lembas.security import permissions
|
|
from lembas.services import settings_store
|
|
|
|
reachable = permissions.models_visible_to(db, user)
|
|
if not reachable:
|
|
return None
|
|
|
|
by_id = {model.model_id: model for model in reachable}
|
|
|
|
preferred = (user.settings_json or {}).get("default_model") if user is not None else None
|
|
if preferred and preferred in by_id:
|
|
return preferred, by_id[preferred].connection_id
|
|
|
|
instance_default = settings_store.get(db, "default_model")
|
|
if instance_default and instance_default in by_id:
|
|
return instance_default, by_id[instance_default].connection_id
|
|
|
|
# First in the administrator's ordering. Pinning is a sidebar shortcut, not
|
|
# a reordering, so it deliberately does not influence this.
|
|
chosen = sorted(reachable, key=lambda m: (m.position, m.model_id))[0]
|
|
return chosen.model_id, chosen.connection_id
|
|
|
|
|
|
def available_models(db: DBSession, user=None) -> list[Model]:
|
|
"""Models this user may start a chat with, in the administrator's order.
|
|
|
|
Pinning does NOT hoist a model up this list: pinned models get their own
|
|
shortcuts in the sidebar, and a picker whose order silently differs from
|
|
the one configured in the admin screen is just confusing.
|
|
"""
|
|
from lembas.security import permissions
|
|
|
|
reachable = permissions.models_visible_to(db, user)
|
|
return sorted(reachable, key=lambda m: (m.position, m.model_id))
|
|
|
|
|
|
def fallback_title(text: str) -> str:
|
|
"""Derive a chat title from the opening message, without calling a model."""
|
|
cleaned = " ".join(text.split())
|
|
if not cleaned:
|
|
return "New chat"
|
|
if len(cleaned) <= MAX_TITLE_LENGTH:
|
|
return cleaned
|
|
# Prefer a word boundary, but only if it does not cut the title in half.
|
|
clipped = cleaned[:MAX_TITLE_LENGTH]
|
|
space = clipped.rfind(" ")
|
|
if space > MAX_TITLE_LENGTH * 0.6:
|
|
clipped = clipped[:space]
|
|
return clipped.rstrip(" ,.;:-") + "…"
|
|
|
|
|
|
async def generate_title(
|
|
endpoint: Endpoint, model_id: str, question: str, answer: str, *, template: str
|
|
) -> str:
|
|
"""Ask the model for a short chat title.
|
|
|
|
Best-effort by design: any failure falls back to trimming the first
|
|
message. Naming a chat is never worth surfacing an error for.
|
|
|
|
`template` is passed in rather than read here because this runs after the
|
|
generation's session has closed -- see `generation._run`. An empty one means
|
|
an administrator cleared the fragment, which is how auto-titling is turned
|
|
off: no request is made at all.
|
|
"""
|
|
from lembas.services import prompts as prompts_service
|
|
|
|
if not template.strip():
|
|
return fallback_title(question)
|
|
|
|
from lembas.services.reasoning import strip_reasoning
|
|
|
|
prompt = prompts_service.substitute(
|
|
template, {"question": question[:500], "answer": answer[:500]}
|
|
)
|
|
body = {
|
|
"model": model_id,
|
|
"messages": [{"role": ROLE_USER, "content": prompt}],
|
|
# Enough that a model which thinks before answering can do both. It was
|
|
# 24, which is ample for six words and nowhere near enough for a
|
|
# reasoning model: the whole budget went on thinking and the reply came
|
|
# back either empty or as an unclosed `<think>`, so every chat on such a
|
|
# model silently fell back to its first prompt and looked as though
|
|
# titling had never run.
|
|
"max_tokens": TITLE_MAX_TOKENS,
|
|
"temperature": 0.2,
|
|
}
|
|
# Deliberately *not* `apply_effort(body, "low")`, tempting as it is: naming
|
|
# a chat does not reward deliberation and a low effort would make this call
|
|
# much cheaper. But `reasoning_effort` and `chat_template_kwargs` appear
|
|
# only when somebody has opted in, precisely so a provider strict about
|
|
# unknown parameters sees exactly the request it always did — and sending
|
|
# them here would put them on every instance's title call, where a 400 is
|
|
# caught and turned into a fallback title. That is titling silently
|
|
# switching itself off, which is the failure this whole change is fixing.
|
|
# The token budget above is what makes room for the thinking instead.
|
|
try:
|
|
raw = await complete(endpoint, body)
|
|
except LLMError as exc:
|
|
log.debug("auto-title failed, using fallback: %s", exc)
|
|
return fallback_title(question)
|
|
|
|
# `complete` hands back `message.content` as it arrived. A model that emits
|
|
# `<think>` tags inline puts them in exactly that field, so without this the
|
|
# title was "<think>Okay, the user wants a short title for". Reasoning sent
|
|
# in a separate `reasoning_content` field is ignored by `complete` already.
|
|
answered, _thinking = strip_reasoning(raw)
|
|
|
|
title = " ".join(answered.split()).strip().strip('"“”\'')
|
|
# Small models sometimes ignore the instruction and answer the question
|
|
# instead; an over-long reply is a better signal of that than anything else.
|
|
if not title or len(title) > MAX_TITLE_LENGTH * 1.5:
|
|
return fallback_title(question)
|
|
return title[:MAX_TITLE_LENGTH]
|
|
|
|
|
|
def create_message(
|
|
db: DBSession,
|
|
chat: Chat,
|
|
role: str,
|
|
content: str = "",
|
|
*,
|
|
complete_: bool = True,
|
|
model_id: str = "",
|
|
queued: bool = False,
|
|
machine: bool = False,
|
|
) -> Message:
|
|
message = Message(
|
|
chat_id=chat.id,
|
|
role=role,
|
|
content=content,
|
|
complete=complete_,
|
|
model_id=model_id,
|
|
queued=queued,
|
|
machine=machine,
|
|
)
|
|
db.add(message)
|
|
db.commit()
|
|
return message
|
|
|
|
|
|
async def summarise_for_compaction(
|
|
endpoint: Endpoint,
|
|
model_id: str,
|
|
*,
|
|
transcript: str,
|
|
previous_summary: str,
|
|
template: str,
|
|
) -> str:
|
|
"""Ask the model to summarise the earlier turns.
|
|
|
|
`template` is passed in for the same reason `generate_title`'s is: this runs
|
|
after the generation's session has closed, and opening another one there is
|
|
how you get a session that outlives its scope. An empty template means an
|
|
administrator cleared the fragment, and nothing is asked of anyone.
|
|
"""
|
|
from lembas.services import prompts as prompts_service
|
|
|
|
if not template.strip() or not transcript.strip():
|
|
return ""
|
|
|
|
prompt = prompts_service.substitute(
|
|
template, {"transcript": transcript, "previous_summary": previous_summary}
|
|
)
|
|
raw = await complete(
|
|
endpoint,
|
|
{
|
|
"model": model_id,
|
|
"messages": [{"role": ROLE_USER, "content": prompt}],
|
|
"max_tokens": 1200,
|
|
# Low, but not zero: this is recall, not invention.
|
|
"temperature": 0.3,
|
|
},
|
|
)
|
|
return raw.strip()
|
|
|
|
|
|
def delete_chats(db: DBSession, chats) -> int:
|
|
"""Delete chats, and the files their attachments point at.
|
|
|
|
**The one way to delete a chat.** `db.delete(chat)` cascades to its messages
|
|
and to its attachment *rows*, and leaves every file on disk -- a generated
|
|
image, an uploaded PDF, a photo -- with nothing that will ever look at them
|
|
again: `sweep_orphans` only considers uploads that were never attached.
|
|
|
|
`files_service.remove_files_for_chats` was written for exactly this and was
|
|
called from one place, the temporary sweep. The delete button, a schedule's
|
|
task chat, a helper's hidden chat and deleting an account all went straight
|
|
to `db.delete`, so four of the five ways a chat can end leaked its files.
|
|
That is `sharing.forget_principal` again: a helper that exists, is correct,
|
|
and is not called on the path that needs it.
|
|
|
|
The order matters and is why this is a function rather than a note. The
|
|
files have to be unlinked **while the rows still say which they are**, so it
|
|
happens before the delete and in the same session.
|
|
|
|
Does not commit -- the caller decides, because some of them are deleting
|
|
other things in the same transaction.
|
|
"""
|
|
live = [chat for chat in chats if chat is not None]
|
|
if not live:
|
|
return 0
|
|
files_service.remove_files_for_chats(db, [chat.id for chat in live])
|
|
for chat in live:
|
|
db.delete(chat)
|
|
return len(live)
|
|
|
|
|
|
def sweep_temporary(db: DBSession, older_than: timedelta = TEMPORARY_LIFETIME) -> int:
|
|
"""Delete temporary chats nobody has touched for a day.
|
|
|
|
Age is measured from the newest message rather than from the chat row's own
|
|
timestamps. `created_at` would destroy a conversation still in use at hour
|
|
23, and `updated_at` does not move when a message is inserted -- `onupdate`
|
|
fires on an UPDATE of the chat, and adding a message is not one.
|
|
|
|
Startup only, like files.sweep_orphans beside it. A server that runs for a
|
|
month sweeps once; that is the trade the existing sweep already makes, and a
|
|
scheduler is a whole new concern for a single-worker application.
|
|
"""
|
|
cutoff = datetime.now(UTC) - older_than
|
|
newest = (
|
|
select(Message.chat_id, func.max(Message.created_at).label("last"))
|
|
.group_by(Message.chat_id)
|
|
.subquery()
|
|
)
|
|
stale = list(
|
|
db.scalars(
|
|
select(Chat)
|
|
.outerjoin(newest, newest.c.chat_id == Chat.id)
|
|
.where(
|
|
Chat.temporary.is_(True),
|
|
func.coalesce(newest.c.last, Chat.created_at) < cutoff,
|
|
)
|
|
)
|
|
)
|
|
if not stale:
|
|
return 0
|
|
|
|
delete_chats(db, stale)
|
|
db.commit()
|
|
log.info("swept %d temporary chat(s)", len(stale))
|
|
return len(stale)
|
|
|
|
|
|
def user_chats(db: DBSession, user_id: str, *, folder_id: str | None = None) -> list[Chat]:
|
|
query = select(Chat).where(
|
|
Chat.user_id == user_id, Chat.archived.is_(False), Chat.temporary.is_(False)
|
|
)
|
|
if folder_id is not None:
|
|
query = query.where(Chat.folder_id == folder_id)
|
|
return list(db.scalars(query.order_by(Chat.pinned.desc(), Chat.updated_at.desc())))
|
|
|
|
|
|
__all__ = [
|
|
"ROLE_ASSISTANT",
|
|
"ROLE_USER",
|
|
"available_models",
|
|
"build_request",
|
|
"create_message",
|
|
"default_model",
|
|
"fallback_title",
|
|
"generate_title",
|
|
"resolve_endpoint",
|
|
"user_chats",
|
|
]
|