Data groups: a provider's models read only their own group's data

Every connection is in a data group. Its models are handed, and can find,
only that group's memories, notes, skills, knowledge, reports and
personality -- by search and by id. A chat stays in the group it was
started in: switching its model, the endpoint fallback, the crowd, friends,
bases and the @ menu all stay inside it, and a chat whose model has moved
is refused rather than sent. A group may name its own embedder and image
reviewer. data.manage lets a person make personal groups, remap
connections for themselves and move their own records.

Also: a search no longer mixes two embedders of the same width.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-29 16:07:55 +00:00
co-authored by Claude Opus 5.5
parent e65ea90fe6
commit 9970bb43c6
61 changed files with 3595 additions and 200 deletions
+53 -7
View File
@@ -55,6 +55,7 @@ from lembas.db.models import (
Chunk,
Connection,
Document,
KnowledgeBase,
Model,
Note,
Report,
@@ -92,19 +93,40 @@ class Embedder:
batch: int = 16
def embedder(db: DBSession) -> Embedder | None:
"""The configured embedding model, or None.
def embedder(db: DBSession, group: str | None = None) -> Embedder | None:
"""The embedding model for one data group, or the instance's, or None.
None is the answer to every "no" -- none chosen, the model row deleted, its
connection disabled -- and every caller reads it the same way: do nothing,
and let the keyword search stand. That is deliberately not an error. An
instance that never configured this is the common case, not a broken one.
**A group may name its own.** The embedder is sent the full text of every
record it indexes, so a group that keeps its data away from a provider has
to be able to keep it away from that provider's embedder too. A group that
names none uses the instance's, which is what every group starts with --
and the Data groups page says so, per group, beside the providers.
Mixing is impossible by construction rather than by care: a `Chunk` carries
the model that made its vector and `retrieval.semantic_ids` skips any other,
so a query embedded by one group's model never meets another's vectors.
"""
values = settings_store.extraction(db)
batch = int(values.get("embed_batch") or 16)
if group:
from lembas.services import data_groups
row = data_groups.get(db, group)
if row is not None and row.embedding_model_id:
return _resolve(db, row.embedding_model_id, row.embedding_connection_id, batch)
wanted = str(values.get("embedding_model_id") or "").strip()
if not wanted:
return None
model = db.scalar(
return _resolve(db, wanted, "", batch)
def _resolve(db: DBSession, wanted: str, connection_id: str, batch: int) -> Embedder | None:
query = (
select(Model)
.join(Connection)
.where(
@@ -112,8 +134,9 @@ def embedder(db: DBSession) -> Embedder | None:
Model.enabled.is_(True),
Connection.enabled.is_(True),
)
.order_by(Connection.position)
.order_by(Connection.id != (connection_id or ""), Connection.position)
)
model = db.scalar(query)
if model is None:
log.info("embedding model %r is configured but not available", wanted)
return None
@@ -123,7 +146,30 @@ def embedder(db: DBSession) -> Embedder | None:
return Embedder(
endpoint=Endpoint.from_connection(connection),
model_id=model.model_id,
batch=int(values.get("embed_batch") or 16),
batch=batch,
)
def group_of_row(db: DBSession, row) -> str:
"""The data group a record is in. A document is in its base's."""
from lembas.services import data_groups
if isinstance(row, Document):
base = db.get(KnowledgeBase, row.base_id) if row.base_id else None
return data_groups.group_of(base) if base is not None else data_groups.DEFAULT_GROUP
return data_groups.group_of(row)
def any_configured(db: DBSession) -> bool:
"""Whether any group -- or the instance -- has an embedder to index with."""
from lembas.services import data_groups
if embedder(db) is not None:
return True
return any(
embedder(db, group.id) is not None
for group in data_groups.all_groups(db)
if group.embedding_model_id
)
@@ -199,7 +245,7 @@ async def index_resource(kind: str, resource_id: str, *, force: bool = False) ->
if row is None:
forget_resource(db, kind, resource_id)
return 0
worker = embedder(db)
worker = embedder(db, group_of_row(db, row))
if worker is None:
return 0
body = text_of(row)
@@ -438,7 +484,7 @@ async def rebuild_all(*, force: bool = True) -> None:
_PROGRESS = Progress(running=True)
try:
with session_scope() as db:
if embedder(db) is None:
if not any_configured(db):
_PROGRESS.error = "No embedding model is configured."
return
work: list[tuple[str, str]] = []