Data groups: a provider's models read only their own group's data
Every connection is in a data group. Its models are handed, and can find, only that group's memories, notes, skills, knowledge, reports and personality -- by search and by id. A chat stays in the group it was started in: switching its model, the endpoint fallback, the crowd, friends, bases and the @ menu all stay inside it, and a chat whose model has moved is refused rather than sent. A group may name its own embedder and image reviewer. data.manage lets a person make personal groups, remap connections for themselves and move their own records. Also: a search no longer mixes two embedders of the same width. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -55,6 +55,7 @@ from lembas.db.models import (
|
||||
Chunk,
|
||||
Connection,
|
||||
Document,
|
||||
KnowledgeBase,
|
||||
Model,
|
||||
Note,
|
||||
Report,
|
||||
@@ -92,19 +93,40 @@ class Embedder:
|
||||
batch: int = 16
|
||||
|
||||
|
||||
def embedder(db: DBSession) -> Embedder | None:
|
||||
"""The configured embedding model, or None.
|
||||
def embedder(db: DBSession, group: str | None = None) -> Embedder | None:
|
||||
"""The embedding model for one data group, or the instance's, or None.
|
||||
|
||||
None is the answer to every "no" -- none chosen, the model row deleted, its
|
||||
connection disabled -- and every caller reads it the same way: do nothing,
|
||||
and let the keyword search stand. That is deliberately not an error. An
|
||||
instance that never configured this is the common case, not a broken one.
|
||||
|
||||
**A group may name its own.** The embedder is sent the full text of every
|
||||
record it indexes, so a group that keeps its data away from a provider has
|
||||
to be able to keep it away from that provider's embedder too. A group that
|
||||
names none uses the instance's, which is what every group starts with --
|
||||
and the Data groups page says so, per group, beside the providers.
|
||||
|
||||
Mixing is impossible by construction rather than by care: a `Chunk` carries
|
||||
the model that made its vector and `retrieval.semantic_ids` skips any other,
|
||||
so a query embedded by one group's model never meets another's vectors.
|
||||
"""
|
||||
values = settings_store.extraction(db)
|
||||
batch = int(values.get("embed_batch") or 16)
|
||||
if group:
|
||||
from lembas.services import data_groups
|
||||
|
||||
row = data_groups.get(db, group)
|
||||
if row is not None and row.embedding_model_id:
|
||||
return _resolve(db, row.embedding_model_id, row.embedding_connection_id, batch)
|
||||
wanted = str(values.get("embedding_model_id") or "").strip()
|
||||
if not wanted:
|
||||
return None
|
||||
model = db.scalar(
|
||||
return _resolve(db, wanted, "", batch)
|
||||
|
||||
|
||||
def _resolve(db: DBSession, wanted: str, connection_id: str, batch: int) -> Embedder | None:
|
||||
query = (
|
||||
select(Model)
|
||||
.join(Connection)
|
||||
.where(
|
||||
@@ -112,8 +134,9 @@ def embedder(db: DBSession) -> Embedder | None:
|
||||
Model.enabled.is_(True),
|
||||
Connection.enabled.is_(True),
|
||||
)
|
||||
.order_by(Connection.position)
|
||||
.order_by(Connection.id != (connection_id or ""), Connection.position)
|
||||
)
|
||||
model = db.scalar(query)
|
||||
if model is None:
|
||||
log.info("embedding model %r is configured but not available", wanted)
|
||||
return None
|
||||
@@ -123,7 +146,30 @@ def embedder(db: DBSession) -> Embedder | None:
|
||||
return Embedder(
|
||||
endpoint=Endpoint.from_connection(connection),
|
||||
model_id=model.model_id,
|
||||
batch=int(values.get("embed_batch") or 16),
|
||||
batch=batch,
|
||||
)
|
||||
|
||||
|
||||
def group_of_row(db: DBSession, row) -> str:
|
||||
"""The data group a record is in. A document is in its base's."""
|
||||
from lembas.services import data_groups
|
||||
|
||||
if isinstance(row, Document):
|
||||
base = db.get(KnowledgeBase, row.base_id) if row.base_id else None
|
||||
return data_groups.group_of(base) if base is not None else data_groups.DEFAULT_GROUP
|
||||
return data_groups.group_of(row)
|
||||
|
||||
|
||||
def any_configured(db: DBSession) -> bool:
|
||||
"""Whether any group -- or the instance -- has an embedder to index with."""
|
||||
from lembas.services import data_groups
|
||||
|
||||
if embedder(db) is not None:
|
||||
return True
|
||||
return any(
|
||||
embedder(db, group.id) is not None
|
||||
for group in data_groups.all_groups(db)
|
||||
if group.embedding_model_id
|
||||
)
|
||||
|
||||
|
||||
@@ -199,7 +245,7 @@ async def index_resource(kind: str, resource_id: str, *, force: bool = False) ->
|
||||
if row is None:
|
||||
forget_resource(db, kind, resource_id)
|
||||
return 0
|
||||
worker = embedder(db)
|
||||
worker = embedder(db, group_of_row(db, row))
|
||||
if worker is None:
|
||||
return 0
|
||||
body = text_of(row)
|
||||
@@ -438,7 +484,7 @@ async def rebuild_all(*, force: bool = True) -> None:
|
||||
_PROGRESS = Progress(running=True)
|
||||
try:
|
||||
with session_scope() as db:
|
||||
if embedder(db) is None:
|
||||
if not any_configured(db):
|
||||
_PROGRESS.error = "No embedding model is configured."
|
||||
return
|
||||
work: list[tuple[str, str]] = []
|
||||
|
||||
Reference in New Issue
Block a user