2576755f79
Reported: chat names never regenerate after the first reply. They were regenerating; the request was being made and the answer thrown away. `complete()` returns `message.content` verbatim, and a model that emits `<think>` inline puts its thinking in exactly the field the title is read from. So the title came back as "<think>Okay, the user wants a short title for" -- or, once the too-long guard caught that, as the first prompt trimmed, which is indistinguishable from titling never having run. That is what was being seen. Underneath it, `max_tokens: 24`. Ample for six words, and nowhere near enough for a model that reasons first: the budget goes on thinking and the content field comes back empty or holding an unclosed tag. Too small is not a shorter title, it is no title at all. Both fixed: the reply goes through `reasoning.strip_reasoning`, and the budget is `TITLE_MAX_TOKENS` with room to think. Reproduced first against the four shapes an endpoint actually answers with -- three of them were broken -- and the tests are written from those. What I did *not* do is ask for a low reasoning effort on the call, which would make it much cheaper and was the obvious move. `reasoning_effort` and `chat_template_kwargs` appear only where somebody has opted in, so that a provider strict about unknown parameters sees exactly the request it always did. An LLMError here is caught and turned into a fallback title -- so a 400 would be titling silently switching itself off, which is the failure this commit exists to fix. The token budget makes the room instead. The shipped prompt now asks for a leading emoji, as requested. Asked for rather than assumed: a model that ignores it gives a title without one, and an administrator who does not want them clears the word. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
643 lines
24 KiB
Python
643 lines
24 KiB
Python
"""Chat orchestration: building requests, streaming replies, naming chats."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from datetime import UTC, datetime, timedelta
|
|
from typing import Any
|
|
|
|
from sqlalchemy import func, select
|
|
from sqlalchemy.orm import Session as DBSession
|
|
|
|
from lembas.db.models import (
|
|
ROLE_ASSISTANT,
|
|
ROLE_SYSTEM,
|
|
ROLE_USER,
|
|
Chat,
|
|
Connection,
|
|
Message,
|
|
Model,
|
|
)
|
|
from lembas.services import files as files_service
|
|
from lembas.services.llm.openai_client import Endpoint, LLMError, complete
|
|
|
|
log = logging.getLogger(__name__)
|
|
|
|
# Sampling keys forwarded upstream. Anything else a user puts in params_json is
|
|
# ignored rather than passed through, so a typo cannot produce a 400 from the
|
|
# provider that looks like a LLeMbas bug.
|
|
FORWARDED_PARAMS = frozenset(
|
|
{"temperature", "top_p", "max_tokens", "presence_penalty", "frequency_penalty",
|
|
"seed", "stop"}
|
|
)
|
|
|
|
MAX_TITLE_LENGTH = 60
|
|
|
|
# What one title call may spend. A title is a handful of words; the rest of this
|
|
# is headroom for a model that thinks before it answers, which is most of the
|
|
# interesting local ones. Too small is not a shorter title -- it is no title at
|
|
# all, because the thinking consumes the budget and the content field comes back
|
|
# empty or holding an unclosed `<think>`.
|
|
TITLE_MAX_TOKENS = 512
|
|
|
|
# How long a temporary chat survives after the last thing said in it.
|
|
TEMPORARY_LIFETIME = timedelta(hours=24)
|
|
|
|
|
|
def resolve_endpoint(db: DBSession, chat: Chat) -> tuple[Endpoint, str]:
|
|
"""Find the connection and model a chat should use.
|
|
|
|
Chats store the model id as text rather than a foreign key so history
|
|
survives an admin deleting a connection, which means the mapping back to a
|
|
live connection has to be resolved at send time and can legitimately fail.
|
|
"""
|
|
if not chat.model_id:
|
|
raise LLMError("This chat has no model selected.")
|
|
|
|
connection: Connection | None = None
|
|
if chat.connection_id:
|
|
connection = db.get(Connection, chat.connection_id)
|
|
|
|
if connection is None or not connection.enabled:
|
|
# The original connection is gone or disabled. Any enabled connection
|
|
# still offering this model id will do.
|
|
model = db.scalar(
|
|
select(Model)
|
|
.join(Connection)
|
|
.where(
|
|
Model.model_id == chat.model_id,
|
|
Model.enabled.is_(True),
|
|
Connection.enabled.is_(True),
|
|
)
|
|
.order_by(Connection.position)
|
|
)
|
|
if model is None:
|
|
raise LLMError(
|
|
f"No enabled connection currently offers the model "
|
|
f"'{chat.model_id}'. Pick another model for this chat."
|
|
)
|
|
connection = model.connection
|
|
chat.connection_id = connection.id
|
|
db.commit()
|
|
|
|
return Endpoint.from_connection(connection), chat.model_id
|
|
|
|
|
|
def document_context(message: Message) -> str:
|
|
"""Extracted text from a message's non-image attachments.
|
|
|
|
Wrapped in named tags so the model can tell one document from another, and
|
|
tell all of them from what the user actually typed. Truncation is stated
|
|
inline rather than silently, so a model asked about page 400 of a 300-page
|
|
extract can say it did not see it.
|
|
"""
|
|
blocks: list[str] = []
|
|
for attachment in message.documents:
|
|
if not attachment.extracted_text.strip():
|
|
continue
|
|
note = " (truncated)" if attachment.truncated else ""
|
|
# Where it came from, when there is a where. A model handed `main.py`
|
|
# cannot tell which of four it is looking at, and cannot name the file
|
|
# back when asked to change something -- so a file read off a machine
|
|
# says which machine and which path. Quotes are stripped rather than
|
|
# escaped: these are attribute values in a tag the model reads, and a
|
|
# path containing one would otherwise close it early.
|
|
where = ""
|
|
if attachment.source_path:
|
|
where += f' path="{_attr(attachment.source_path)}"'
|
|
if attachment.source_label:
|
|
where += f' from="{_attr(attachment.source_label)}"'
|
|
blocks.append(
|
|
f'<document name="{_attr(attachment.filename)}"{where}{note}>\n'
|
|
f"{attachment.extracted_text.strip()}\n"
|
|
f"</document>"
|
|
)
|
|
return "\n\n".join(blocks)
|
|
|
|
|
|
def _attr(value: str) -> str:
|
|
"""A value safe to sit inside the double quotes of a tag we are writing."""
|
|
return value.replace('"', "").replace("<", "").replace(">", "").replace("\n", " ")
|
|
|
|
|
|
def message_payload(message: Message, *, vision: bool) -> dict[str, Any]:
|
|
"""One history entry in the shape the endpoint expects.
|
|
|
|
Plain text stays a plain string: sending the multimodal list form to an
|
|
endpoint that does not implement it is a reliable way to get a 400, and
|
|
most local runners do not.
|
|
"""
|
|
text = message.content.strip()
|
|
|
|
documents = document_context(message)
|
|
if documents:
|
|
# Documents lead so the question that follows has its material already
|
|
# in view, which is how these models are trained to read a prompt.
|
|
text = f"{documents}\n\n{text}" if text else documents
|
|
|
|
images = message.images if vision else []
|
|
if not images:
|
|
return {"role": message.role, "content": text}
|
|
|
|
parts: list[dict[str, Any]] = []
|
|
if text:
|
|
parts.append({"type": "text", "text": text})
|
|
for attachment in images:
|
|
uri = files_service.data_uri(attachment)
|
|
if uri is None:
|
|
# The row survived but the file did not. Better to say so than to
|
|
# send a turn that silently lost its picture.
|
|
log.warning("attachment %s has no file on disk", attachment.id)
|
|
continue
|
|
parts.append({"type": "image_url", "image_url": {"url": uri}})
|
|
|
|
if not parts:
|
|
return {"role": message.role, "content": text}
|
|
return {"role": message.role, "content": parts}
|
|
|
|
|
|
def folder_system_prompt(db: DBSession, chat: Chat) -> str:
|
|
"""The nearest prompt on the chat's folder, or on a folder above it.
|
|
|
|
Walks up rather than reading one level, because folders nest and a project's
|
|
prompt belongs on the project rather than on each sub-folder of it. The
|
|
nearest one wins, which is the same rule the ladder as a whole follows.
|
|
|
|
Bounded and cycle-safe the way `api/folders.py:_depth_of` is. Reparenting
|
|
already refuses to build a cycle, but this runs on the request path for
|
|
every reply and a row written by something else must not be able to hang it.
|
|
"""
|
|
from lembas.db.models import Folder
|
|
|
|
folder = chat.folder
|
|
seen: set[str] = set()
|
|
while folder is not None and folder.id not in seen:
|
|
seen.add(folder.id)
|
|
if (folder.system_prompt or "").strip():
|
|
return folder.system_prompt.strip()
|
|
folder = db.get(Folder, folder.parent_id) if folder.parent_id else None
|
|
return ""
|
|
|
|
|
|
def effective_system_prompt(db: DBSession, chat: Chat) -> str:
|
|
"""The system prompt a chat actually runs with.
|
|
|
|
Four layers, most specific wins outright:
|
|
|
|
chat > folder > model > instance
|
|
|
|
Precedence rather than concatenation. Stacking them reads well in a
|
|
settings screen and badly in practice: the moment two layers disagree the
|
|
model gets contradictory instructions and nobody can tell which one is
|
|
losing. With precedence, "why is it behaving like this" has one answer.
|
|
|
|
The folder sits above the model because it is the more specific statement:
|
|
a model's prompt describes the model wherever it is used, and a folder's
|
|
describes this piece of work whichever model is pointed at it.
|
|
"""
|
|
from lembas.services import settings_store
|
|
|
|
if chat.system_prompt.strip():
|
|
return chat.system_prompt.strip()
|
|
|
|
if inherited := folder_system_prompt(db, chat):
|
|
return inherited
|
|
|
|
model = db.scalar(
|
|
select(Model).where(Model.model_id == chat.model_id).order_by(Model.position)
|
|
)
|
|
if model is not None and (model.system_prompt or "").strip():
|
|
return model.system_prompt.strip()
|
|
|
|
return (settings_store.get(db, "system_prompt") or "").strip()
|
|
|
|
|
|
def build_messages(
|
|
db: DBSession,
|
|
chat: Chat,
|
|
*,
|
|
upto: Message | None = None,
|
|
vision: bool = False,
|
|
system_prompt: str | None = None,
|
|
) -> list[dict]:
|
|
"""Assemble the message list to send upstream.
|
|
|
|
`upto` excludes the placeholder assistant row being generated into, and
|
|
everything after it. `system_prompt` overrides what would otherwise be
|
|
resolved, which is how the harness gets in front of the authored prompt
|
|
without this function knowing anything about tools.
|
|
"""
|
|
from lembas.services import compaction as compaction_service
|
|
from lembas.services import prompts as prompts_service
|
|
|
|
payload: list[dict[str, Any]] = []
|
|
system = effective_system_prompt(db, chat) if system_prompt is None else system_prompt
|
|
if system:
|
|
payload.append({"role": ROLE_SYSTEM, "content": system})
|
|
|
|
# Compacted turns are replaced by a summary carried in two turns rather than
|
|
# one. A leading `assistant` breaks templates that require the first
|
|
# non-system message to be `user`; a lone leading `user` produces user, user
|
|
# whenever the kept history starts on a user turn -- which it always does,
|
|
# because the cutoff lands on a finished reply. The pair alternates
|
|
# correctly in both directions and keeps exactly one system message.
|
|
cutoff = compaction_service.cutoff_message(db, chat)
|
|
if cutoff is not None:
|
|
lead = prompts_service.resolve(db, "task.compact_lead").strip()
|
|
ack = prompts_service.resolve(db, "task.compact_ack").strip()
|
|
summary = chat.compact_summary.strip()
|
|
payload.append(
|
|
{"role": ROLE_USER, "content": f"{lead}\n\n{summary}" if lead else summary}
|
|
)
|
|
if ack:
|
|
payload.append({"role": ROLE_ASSISTANT, "content": ack})
|
|
|
|
history = db.scalars(
|
|
select(Message).where(Message.chat_id == chat.id).order_by(Message.created_at)
|
|
).all()
|
|
|
|
for message in history:
|
|
if upto is not None and message.id == upto.id:
|
|
break
|
|
if cutoff is not None and compaction_service.moment(
|
|
message
|
|
) <= compaction_service.moment(cutoff):
|
|
continue
|
|
# Typed while the previous reply was still being written, and not yet
|
|
# handed to a model. It is in the transcript and it is not in the
|
|
# request; delivery is what moves it from one to the other.
|
|
if message.queued:
|
|
continue
|
|
# Skip turns that failed or produced nothing -- but a message carrying
|
|
# only an attachment has no text and must still be sent.
|
|
if message.error:
|
|
continue
|
|
if not message.content.strip() and not message.attachments:
|
|
continue
|
|
payload.append(message_payload(message, vision=vision))
|
|
|
|
return payload
|
|
|
|
|
|
def model_for(db: DBSession, chat: Chat) -> Model | None:
|
|
"""The Model row a chat is using, or None if it has gone.
|
|
|
|
Looked up by id rather than held as a foreign key, for the same reason
|
|
resolve_endpoint does: chats store the model as text so history survives an
|
|
administrator deleting a connection.
|
|
"""
|
|
return db.scalar(
|
|
select(Model).where(Model.model_id == chat.model_id).order_by(Model.position)
|
|
)
|
|
|
|
|
|
def model_supports(db: DBSession, chat: Chat, capability: str) -> bool:
|
|
"""Whether the chat's current model is marked as having a capability."""
|
|
model = model_for(db, chat)
|
|
return bool(model and (model.capabilities_json or {}).get(capability))
|
|
|
|
|
|
def build_request(
|
|
db: DBSession,
|
|
chat: Chat,
|
|
*,
|
|
upto: Message | None = None,
|
|
tools: list[dict[str, Any]] | None = None,
|
|
user=None,
|
|
) -> dict[str, Any]:
|
|
"""The whole request body, tools and harness included.
|
|
|
|
Composed here rather than in the generation loop so that "what gets sent"
|
|
has one answer, and so the harness cannot be forgotten by a future caller
|
|
that offers tools.
|
|
"""
|
|
from lembas.services import harness as harness_service
|
|
from lembas.services import prompts as prompts_service
|
|
|
|
params = {
|
|
key: value
|
|
for key, value in (chat.params_json or {}).items()
|
|
if key in FORWARDED_PARAMS and value not in (None, "")
|
|
}
|
|
# Images are only sent to a model an administrator has marked as having
|
|
# vision. Sending them to one that has not is not a graceful degradation:
|
|
# most endpoints reject the whole request.
|
|
vision = model_supports(db, chat, "vision")
|
|
|
|
if user is None:
|
|
from lembas.db.models import User
|
|
|
|
user = db.get(User, chat.user_id)
|
|
|
|
# The harness describes the tools; the authored prompt describes the
|
|
# behaviour. See services/harness.py for why these are joined rather than
|
|
# being two competing layers.
|
|
system = harness_service.join(
|
|
harness_service.compose(db, user, tools, chat),
|
|
effective_system_prompt(db, chat),
|
|
lead=prompts_service.render(db, "seam.authored_lead", {}),
|
|
)
|
|
|
|
body: dict[str, Any] = {
|
|
"model": chat.model_id,
|
|
"messages": build_messages(
|
|
db, chat, upto=upto, vision=vision, system_prompt=system
|
|
),
|
|
**params,
|
|
}
|
|
if tools:
|
|
body["tools"] = tools
|
|
|
|
apply_effort(body, (chat.params_json or {}).get("reasoning_effort"))
|
|
return body
|
|
|
|
|
|
# Reasoning effort, and why it goes out twice.
|
|
#
|
|
# There is no one field that works. OpenAI and vLLM read a plain
|
|
# `reasoning_effort`. llama.cpp reads it too and, per its own documentation,
|
|
# "other values (e.g. 'low', 'max') have no effect" -- its maintainer is blunter
|
|
# still: "llama-server cannot support reasoning_effort at all", and the field
|
|
# "simply gets dropped without error or logging". What *does* reach a gpt-oss
|
|
# behind llama.cpp is `chat_template_kwargs`, which it accepts per request.
|
|
#
|
|
# So both are sent, and only when an effort has actually been chosen. That
|
|
# second half is what keeps this from being a regression: a chat nobody has set
|
|
# an effort on sends neither field and is byte-for-byte what it was. An endpoint
|
|
# strict about unknown parameters will refuse the extra one -- but on a chat
|
|
# somebody deliberately set an effort on, not on every chat in the instance.
|
|
EFFORTS = ("low", "medium", "high")
|
|
|
|
|
|
def resolved_effort(chat) -> str:
|
|
"""The effort this chat will actually send, or "" for none.
|
|
|
|
Its own value, and nothing else. The model's default is a **seed** applied
|
|
when the chat is created (`api/chats.py:_new_chat`) and on a model change,
|
|
and is deliberately not consulted here for two reasons. A chat's request
|
|
should be a function of the chat row alone -- the same rule that has PDF
|
|
text extracted once at upload and knowledge attachments copied. And a
|
|
fallback would break "off": `update_chat` stores `None` for a cleared
|
|
effort, a fallback would resurrect the model's default underneath it, and
|
|
the off option would silently do nothing.
|
|
|
|
The picker shows exactly this, which is the whole point of it existing:
|
|
"Effort: default" named no level and was true of nothing in particular.
|
|
"""
|
|
value = (getattr(chat, "params_json", None) or {}).get("reasoning_effort")
|
|
return value if value in EFFORTS else ""
|
|
|
|
|
|
def apply_effort(body: dict[str, Any], effort: str | None) -> None:
|
|
"""Put a chosen reasoning effort into a request body, in both forms."""
|
|
if not effort or effort not in EFFORTS:
|
|
return
|
|
body["reasoning_effort"] = effort
|
|
kwargs = dict(body.get("chat_template_kwargs") or {})
|
|
kwargs["reasoning_effort"] = effort
|
|
body["chat_template_kwargs"] = kwargs
|
|
|
|
|
|
def default_model(db: DBSession, user=None) -> tuple[str, str] | None:
|
|
"""The model a new chat should start with, as (model_id, connection_id).
|
|
|
|
Preference order: the user's own choice, then the instance default, then
|
|
whatever is first in the admin's ordering. Each is checked against what the
|
|
user may actually reach, so a default they have lost access to falls
|
|
through rather than producing a chat they cannot use.
|
|
"""
|
|
from lembas.security import permissions
|
|
from lembas.services import settings_store
|
|
|
|
reachable = permissions.models_visible_to(db, user)
|
|
if not reachable:
|
|
return None
|
|
|
|
by_id = {model.model_id: model for model in reachable}
|
|
|
|
preferred = (user.settings_json or {}).get("default_model") if user is not None else None
|
|
if preferred and preferred in by_id:
|
|
return preferred, by_id[preferred].connection_id
|
|
|
|
instance_default = settings_store.get(db, "default_model")
|
|
if instance_default and instance_default in by_id:
|
|
return instance_default, by_id[instance_default].connection_id
|
|
|
|
# First in the administrator's ordering. Pinning is a sidebar shortcut, not
|
|
# a reordering, so it deliberately does not influence this.
|
|
chosen = sorted(reachable, key=lambda m: (m.position, m.model_id))[0]
|
|
return chosen.model_id, chosen.connection_id
|
|
|
|
|
|
def available_models(db: DBSession, user=None) -> list[Model]:
|
|
"""Models this user may start a chat with, in the administrator's order.
|
|
|
|
Pinning does NOT hoist a model up this list: pinned models get their own
|
|
shortcuts in the sidebar, and a picker whose order silently differs from
|
|
the one configured in the admin screen is just confusing.
|
|
"""
|
|
from lembas.security import permissions
|
|
|
|
reachable = permissions.models_visible_to(db, user)
|
|
return sorted(reachable, key=lambda m: (m.position, m.model_id))
|
|
|
|
|
|
def fallback_title(text: str) -> str:
|
|
"""Derive a chat title from the opening message, without calling a model."""
|
|
cleaned = " ".join(text.split())
|
|
if not cleaned:
|
|
return "New chat"
|
|
if len(cleaned) <= MAX_TITLE_LENGTH:
|
|
return cleaned
|
|
# Prefer a word boundary, but only if it does not cut the title in half.
|
|
clipped = cleaned[:MAX_TITLE_LENGTH]
|
|
space = clipped.rfind(" ")
|
|
if space > MAX_TITLE_LENGTH * 0.6:
|
|
clipped = clipped[:space]
|
|
return clipped.rstrip(" ,.;:-") + "…"
|
|
|
|
|
|
async def generate_title(
|
|
endpoint: Endpoint, model_id: str, question: str, answer: str, *, template: str
|
|
) -> str:
|
|
"""Ask the model for a short chat title.
|
|
|
|
Best-effort by design: any failure falls back to trimming the first
|
|
message. Naming a chat is never worth surfacing an error for.
|
|
|
|
`template` is passed in rather than read here because this runs after the
|
|
generation's session has closed -- see `generation._run`. An empty one means
|
|
an administrator cleared the fragment, which is how auto-titling is turned
|
|
off: no request is made at all.
|
|
"""
|
|
from lembas.services import prompts as prompts_service
|
|
|
|
if not template.strip():
|
|
return fallback_title(question)
|
|
|
|
from lembas.services.reasoning import strip_reasoning
|
|
|
|
prompt = prompts_service.substitute(
|
|
template, {"question": question[:500], "answer": answer[:500]}
|
|
)
|
|
body = {
|
|
"model": model_id,
|
|
"messages": [{"role": ROLE_USER, "content": prompt}],
|
|
# Enough that a model which thinks before answering can do both. It was
|
|
# 24, which is ample for six words and nowhere near enough for a
|
|
# reasoning model: the whole budget went on thinking and the reply came
|
|
# back either empty or as an unclosed `<think>`, so every chat on such a
|
|
# model silently fell back to its first prompt and looked as though
|
|
# titling had never run.
|
|
"max_tokens": TITLE_MAX_TOKENS,
|
|
"temperature": 0.2,
|
|
}
|
|
# Deliberately *not* `apply_effort(body, "low")`, tempting as it is: naming
|
|
# a chat does not reward deliberation and a low effort would make this call
|
|
# much cheaper. But `reasoning_effort` and `chat_template_kwargs` appear
|
|
# only when somebody has opted in, precisely so a provider strict about
|
|
# unknown parameters sees exactly the request it always did — and sending
|
|
# them here would put them on every instance's title call, where a 400 is
|
|
# caught and turned into a fallback title. That is titling silently
|
|
# switching itself off, which is the failure this whole change is fixing.
|
|
# The token budget above is what makes room for the thinking instead.
|
|
try:
|
|
raw = await complete(endpoint, body)
|
|
except LLMError as exc:
|
|
log.debug("auto-title failed, using fallback: %s", exc)
|
|
return fallback_title(question)
|
|
|
|
# `complete` hands back `message.content` as it arrived. A model that emits
|
|
# `<think>` tags inline puts them in exactly that field, so without this the
|
|
# title was "<think>Okay, the user wants a short title for". Reasoning sent
|
|
# in a separate `reasoning_content` field is ignored by `complete` already.
|
|
answered, _thinking = strip_reasoning(raw)
|
|
|
|
title = " ".join(answered.split()).strip().strip('"“”\'')
|
|
# Small models sometimes ignore the instruction and answer the question
|
|
# instead; an over-long reply is a better signal of that than anything else.
|
|
if not title or len(title) > MAX_TITLE_LENGTH * 1.5:
|
|
return fallback_title(question)
|
|
return title[:MAX_TITLE_LENGTH]
|
|
|
|
|
|
def create_message(
|
|
db: DBSession,
|
|
chat: Chat,
|
|
role: str,
|
|
content: str = "",
|
|
*,
|
|
complete_: bool = True,
|
|
model_id: str = "",
|
|
queued: bool = False,
|
|
) -> Message:
|
|
message = Message(
|
|
chat_id=chat.id,
|
|
role=role,
|
|
content=content,
|
|
complete=complete_,
|
|
model_id=model_id,
|
|
queued=queued,
|
|
)
|
|
db.add(message)
|
|
db.commit()
|
|
return message
|
|
|
|
|
|
async def summarise_for_compaction(
|
|
endpoint: Endpoint,
|
|
model_id: str,
|
|
*,
|
|
transcript: str,
|
|
previous_summary: str,
|
|
template: str,
|
|
) -> str:
|
|
"""Ask the model to summarise the earlier turns.
|
|
|
|
`template` is passed in for the same reason `generate_title`'s is: this runs
|
|
after the generation's session has closed, and opening another one there is
|
|
how you get a session that outlives its scope. An empty template means an
|
|
administrator cleared the fragment, and nothing is asked of anyone.
|
|
"""
|
|
from lembas.services import prompts as prompts_service
|
|
|
|
if not template.strip() or not transcript.strip():
|
|
return ""
|
|
|
|
prompt = prompts_service.substitute(
|
|
template, {"transcript": transcript, "previous_summary": previous_summary}
|
|
)
|
|
raw = await complete(
|
|
endpoint,
|
|
{
|
|
"model": model_id,
|
|
"messages": [{"role": ROLE_USER, "content": prompt}],
|
|
"max_tokens": 1200,
|
|
# Low, but not zero: this is recall, not invention.
|
|
"temperature": 0.3,
|
|
},
|
|
)
|
|
return raw.strip()
|
|
|
|
|
|
def sweep_temporary(db: DBSession, older_than: timedelta = TEMPORARY_LIFETIME) -> int:
|
|
"""Delete temporary chats nobody has touched for a day.
|
|
|
|
Age is measured from the newest message rather than from the chat row's own
|
|
timestamps. `created_at` would destroy a conversation still in use at hour
|
|
23, and `updated_at` does not move when a message is inserted -- `onupdate`
|
|
fires on an UPDATE of the chat, and adding a message is not one.
|
|
|
|
Startup only, like files.sweep_orphans beside it. A server that runs for a
|
|
month sweeps once; that is the trade the existing sweep already makes, and a
|
|
scheduler is a whole new concern for a single-worker application.
|
|
"""
|
|
cutoff = datetime.now(UTC) - older_than
|
|
newest = (
|
|
select(Message.chat_id, func.max(Message.created_at).label("last"))
|
|
.group_by(Message.chat_id)
|
|
.subquery()
|
|
)
|
|
stale = list(
|
|
db.scalars(
|
|
select(Chat)
|
|
.outerjoin(newest, newest.c.chat_id == Chat.id)
|
|
.where(
|
|
Chat.temporary.is_(True),
|
|
func.coalesce(newest.c.last, Chat.created_at) < cutoff,
|
|
)
|
|
)
|
|
)
|
|
if not stale:
|
|
return 0
|
|
|
|
files_service.remove_files_for_chats(db, [chat.id for chat in stale])
|
|
for chat in stale:
|
|
db.delete(chat)
|
|
db.commit()
|
|
log.info("swept %d temporary chat(s)", len(stale))
|
|
return len(stale)
|
|
|
|
|
|
def user_chats(db: DBSession, user_id: str, *, folder_id: str | None = None) -> list[Chat]:
|
|
query = select(Chat).where(
|
|
Chat.user_id == user_id, Chat.archived.is_(False), Chat.temporary.is_(False)
|
|
)
|
|
if folder_id is not None:
|
|
query = query.where(Chat.folder_id == folder_id)
|
|
return list(db.scalars(query.order_by(Chat.pinned.desc(), Chat.updated_at.desc())))
|
|
|
|
|
|
__all__ = [
|
|
"ROLE_ASSISTANT",
|
|
"ROLE_USER",
|
|
"available_models",
|
|
"build_request",
|
|
"create_message",
|
|
"default_model",
|
|
"fallback_title",
|
|
"generate_title",
|
|
"resolve_endpoint",
|
|
"user_chats",
|
|
]
|