Files
LLeMbas/src/lembas/services/personas.py
T
HomerandClaude Opus 5.5 9970bb43c6 Data groups: a provider's models read only their own group's data
Every connection is in a data group. Its models are handed, and can find,
only that group's memories, notes, skills, knowledge, reports and
personality -- by search and by id. A chat stays in the group it was
started in: switching its model, the endpoint fallback, the crowd, friends,
bases and the @ menu all stay inside it, and a chat whose model has moved
is refused rather than sent. A group may name its own embedder and image
reviewer. data.manage lets a person make personal groups, remap
connections for themselves and move their own records.

Also: a search no longer mixes two embedders of the same width.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-29 16:07:55 +00:00

358 lines
12 KiB
Python

"""A model's personality with one person, and what it makes of them.
Both are per (model, person) -- see `db/models/persona.py` for the shape and for
why they are two tables. The administrator's default persona (`owner_id IS NULL`)
is a **starting point**, resolved by `effective` and never stacked on top of
somebody's own.
Three rules, and each is here rather than in the column so a write that breaks
one can be trimmed with an explanation instead of failing somebody's turn -- the
rule `memories.py` already follows:
* **Capped.** Both texts are in front of the model on every single request, so
a personality that grows without limit is a context window that shrinks
without anybody noticing.
* **A personality is snapshotted before every change.** A model may rewrite its
own, so what stops a bad rewrite being permanent is a record and a way back.
Not a gate: the roadmap states the same limit for model-written skills. An
impression is not snapshotted, for the reason its own docstring gives.
* **Both belong to the person they concern.** Keyed on their id, read only for
them, and shown to them in their own settings. A model-written note about
somebody that they cannot see is not something this application should hold.
"""
from __future__ import annotations
import logging
from sqlalchemy import select
from sqlalchemy.orm import Session as DBSession
from lembas.db.models import (
AUTHOR_MODEL,
AUTHOR_USER,
DEFAULT_GROUP,
Impression,
Persona,
PersonaRevision,
User,
)
log = logging.getLogger(__name__)
# Who a model is. Room for a real character -- a voice, what it cares about, how
# it argues -- and not room for a second system prompt. An administrator who
# wants more than this wants `Model.system_prompt`, which is the layer meant for
# instructions and is not rewritten by the model.
MAX_PERSONA_CHARS = 1200
# What one model has made of one person. Shorter on purpose: it is a standing
# impression, not a file. Anything that needs more than this is either a memory
# (a fact) or a note (a document).
MAX_VIEW_CHARS = 800
# How many "before" states are kept. Enough to undo a bad afternoon, bounded so
# a model editing itself every turn cannot grow the table without limit.
MAX_REVISIONS = 20
# Between a model id and a data group in a person's key. Two characters, because
# one `@` is a character a model id could plausibly contain and this must never
# split one.
KEY_SEPARATOR = "@@"
def key_for(model_id: str, group: str | None) -> str:
"""The key a person's personality and impression are stored under.
**Namespaced by data group, and the reason is a constraint.** Both tables
are `UNIQUE(model_key, owner_id)`, SQLite cannot alter a constraint, and
this project's schema changes are additive only -- so a `data_group_id`
column could not let one person hold a personality for the same model id in
two groups, which is exactly what one model id served by two providers in
different groups needs.
The default group keeps the bare model id, which is what every row written
before groups existed already holds, so an upgrade moves nothing. The
administrator's default (`owner_id NULL`) is always bare: it is their text,
not a person's data, and every group falls back to it.
"""
if not model_id or not group or group == DEFAULT_GROUP:
return model_id
return f"{model_id}{KEY_SEPARATOR}{group}"
def split_key(model_key: str) -> tuple[str, str]:
"""(model id, data group) out of a stored key."""
model_id, separator, group = (model_key or "").rpartition(KEY_SEPARATOR)
if not separator:
return model_key or "", DEFAULT_GROUP
return model_id, group or DEFAULT_GROUP
def get(db: DBSession, model_key: str, owner: User | None) -> Persona | None:
"""One personality row, exactly as asked for and with no fallback.
`owner=None` asks for the administrator's default. Use `effective` to ask the
question the prompt asks -- "who is this model with this person" -- which is
where the fallback belongs.
"""
if not model_key:
return None
return db.scalars(
select(Persona).where(
Persona.model_key == model_key,
Persona.owner_id == (owner.id if owner is not None else None),
)
).first()
def effective(db: DBSession, model_key: str, owner: User | None) -> Persona | None:
"""This person's personality for this model, or the default if they have none.
The fallback is what makes an administrator's default mean anything: until
the model has written something of its own with somebody, that is who it is.
Once it has, the default stops applying to them -- it is a starting point and
not a layer, because two personalities stacked would contradict each other and
nobody could tell which was losing.
"""
own = get(db, model_key, owner)
if own is not None:
return own
# The default is keyed on the bare model id whatever group asked.
return get(db, split_key(model_key)[0], None) if owner is not None else None
def personas_of(db: DBSession, owner: User | None) -> list[Persona]:
"""Every personality this person has, for their own settings page."""
if owner is None:
return []
return list(
db.scalars(
select(Persona)
.where(Persona.owner_id == owner.id)
.order_by(Persona.model_key)
)
)
def impression(db: DBSession, model_key: str, owner: User | None) -> Impression | None:
if not model_key or owner is None:
return None
return db.scalars(
select(Impression).where(
Impression.model_key == model_key, Impression.owner_id == owner.id
)
).first()
def impressions_for(db: DBSession, owner: User | None) -> list[Impression]:
"""Every model's read of one person, for that person's own settings page."""
if owner is None:
return []
return list(
db.scalars(
select(Impression)
.where(Impression.owner_id == owner.id)
.order_by(Impression.model_key)
)
)
def write_impression(
db: DBSession,
*,
model_key: str,
owner: User,
content: str,
author: str = AUTHOR_MODEL,
) -> Impression:
"""Set what a model makes of somebody. Replaces; no history kept.
Deliberately without the snapshotting `write` does. An impression is meant to
change as the model learns, so a history of it would be a log of somebody
being reassessed -- and the control that matters is that they can read it and
delete it, which they can.
"""
if not model_key:
raise ValueError("There is no model to write an impression for.")
text = (content or "").strip()[:MAX_VIEW_CHARS]
row = impression(db, model_key, owner)
if row is None:
row = Impression(
model_key=model_key,
owner_id=owner.id,
content=text,
author=author if author in (AUTHOR_USER, AUTHOR_MODEL) else AUTHOR_MODEL,
)
db.add(row)
else:
row.content = text
row.author = author if author in (AUTHOR_USER, AUTHOR_MODEL) else AUTHOR_MODEL
db.commit()
return row
def clear_impression(db: DBSession, row: Impression) -> None:
db.delete(row)
db.commit()
def personas_for(db: DBSession, model_keys: list[str]) -> dict[str, Persona]:
"""Every model's own persona, keyed by model id. For the admin screens."""
if not model_keys:
return {}
rows = db.scalars(
select(Persona).where(
Persona.model_key.in_(model_keys), Persona.owner_id.is_(None)
)
)
return {row.model_key: row for row in rows}
def write(
db: DBSession,
*,
model_key: str,
owner: User | None,
content: str,
author: str = AUTHOR_MODEL,
note: str = "",
) -> Persona:
"""Set a persona or a reflection, keeping what was there.
Returns the row. Raises `ValueError` only for a write with no model to
attach to -- an over-long text is trimmed rather than refused, because the
alternative is a model losing a turn to a length it could not have known.
"""
if not model_key:
raise ValueError("There is no model to write a personality for.")
text = (content or "").strip()[:MAX_PERSONA_CHARS]
row = get(db, model_key, owner)
if row is None:
row = Persona(
model_key=model_key,
owner_id=owner.id if owner is not None else None,
content=text,
author=author if author in (AUTHOR_USER, AUTHOR_MODEL) else AUTHOR_MODEL,
)
db.add(row)
db.commit()
return row
if row.content == text:
# Nothing changed, so nothing is snapshotted. Otherwise a model that
# rewrites itself with the same words every turn fills the history with
# identical revisions and pushes the real "before" out of it.
return row
db.add(
PersonaRevision(
persona_id=row.id,
content=row.content,
author=row.author,
note=(note or "").strip()[:200],
)
)
row.content = text
row.author = author if author in (AUTHOR_USER, AUTHOR_MODEL) else AUTHOR_MODEL
db.commit()
_prune(db, row)
return row
def _prune(db: DBSession, row: Persona) -> None:
"""Drop the oldest revisions past the ceiling.
Queried rather than read off `row.revisions`, and ordered with the id as a
tiebreak. Both matter. The session is built with `expire_on_commit=False`, so
the loaded collection can be a version of the list from before the write that
prompted this -- which is how the first draft of this deleted a row that was
already gone and left one that should have been. And revisions written in the
same microsecond order arbitrarily under `created_at` alone, so which ones
"the oldest" names would not be stable.
"""
extra = list(
db.scalars(
select(PersonaRevision)
.where(PersonaRevision.persona_id == row.id)
.order_by(PersonaRevision.created_at.desc(), PersonaRevision.id.desc())
.offset(MAX_REVISIONS)
)
)
if not extra:
return
for revision in extra:
db.delete(revision)
db.commit()
# Or the caller's next read of `row.revisions` is the list that still has
# them in it.
db.expire(row, ["revisions"])
def revert(db: DBSession, row: Persona, revision: PersonaRevision) -> Persona:
"""Put a previous text back, as the person doing the reverting.
Goes through `write`, so the text being replaced is itself snapshotted: an
undo that cannot be undone is a second way to lose the same work.
"""
owner = db.get(User, row.owner_id) if row.owner_id else None
return write(
db,
model_key=row.model_key,
owner=owner,
content=revision.content,
author=AUTHOR_USER,
note="reverted",
)
def clear(db: DBSession, row: Persona) -> None:
db.delete(row)
db.commit()
def block(db: DBSession, model_key: str, owner: User | None) -> str:
"""The personality as the prompt carries it, or "" when there is none.
Empty and disabled are the same answer on purpose: the fragments that read
this are gated on it with `requires`, so both make the whole section vanish
rather than leaving a heading above nothing.
"""
row = effective(db, model_key, owner)
if row is None or not row.enabled:
return ""
return (row.content or "").strip()
def view_block(db: DBSession, model_key: str, owner: User | None) -> str:
"""What the model makes of this person, as the prompt carries it."""
row = impression(db, model_key, owner)
if row is None or not row.enabled:
return ""
return (row.content or "").strip()
__all__ = [
"KEY_SEPARATOR",
"MAX_PERSONA_CHARS",
"MAX_REVISIONS",
"MAX_VIEW_CHARS",
"block",
"clear",
"clear_impression",
"effective",
"get",
"impression",
"impressions_for",
"key_for",
"personas_for",
"personas_of",
"split_key",
"view_block",
"revert",
"write",
"write_impression",
]