diff --git a/CHANGELOG.md b/CHANGELOG.md index 3d0ed83..23afd05 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,43 @@ for 1.0.0 have something to be assembled from. ## Unreleased +## 1.12.0 + +Helpers on another model. This is the last of the three releases. + +- **A helper can run on a different model.** A helper is still the chat's own + model by default. On each model's page, **Helpers** lists other models this + one may send helpers to. Each is either *offered to the model*, so it can + choose that model itself, or *by hand only*, so it is used once somebody adds + it to a chat. When there is more than one choice, `subagent_run` gains a + `model` argument limited to exactly those models, and the model is told what + each is for. A helper on another model runs with that model's own default + reasoning effort, never the parent's, which it might refuse. It reads its own + data group's memories and notes. +- **The composer's Helpers button** lists the models designated for the chat's + model. Tick one to add it to this chat, on the new-chat screen as well. +- **Capacity is decided by logic, not by the model.** Two new switches, both off + by default, so nothing changes until they are set: + - **Serves one request at a time**, on a model. It cannot be its own helper, + because the helper would wait behind the reply that is waiting for it. + - **Holds one model at a time**, on a connection (llama-swap in front of one + GPU). Its models may still be their own helpers, but never send one to + another model on the same connection, because loading it would unload the + model whose reply is waiting. + + A model with no helper it can use is no longer offered `subagent_run` at all, + instead of being offered a tool that refuses every call. Asking a friend and + the crowd are not affected: both are sequential, and the wait for a model to + load is accepted there. +- **The model rules apply to helpers too.** A designation the model chooses + itself must be allowed by the rules. One added to a chat by hand needs what a + crowd member added by hand needs. A model in another data group needs a rule + that allows it. +- **Your own helper models.** With the new *Choose their own helper models* + permission (off by default), Settings → Models lets a person designate helpers + for their models. These are added to the instance's, and a row of theirs for + the same pair replaces the instance's. + ## 1.11.0 Rules for which model may talk to which. This is the second of three releases; diff --git a/scripts/shoot.py b/scripts/shoot.py index 21b5304..a321fb1 100644 --- a/scripts/shoot.py +++ b/scripts/shoot.py @@ -324,6 +324,12 @@ def build_client(): owner = db.query(User).first() db.add(Note(owner_id=owner.id, title="A note at home", body="x", data_group_id="default")) db.add(Note(owner_id=owner.id, title="A hosted note", body="x", data_group_id="hosted")) + # A rule and a designation, so the rules page, the matrix and the + # model page's Helpers card render with something in them. + from lembas.db.models import HelperDesignation, TalkRule + + db.add(TalkRule(from_model="gemma4-moe", to_model="qwen3-coder", effect="deny")) + db.add(HelperDesignation(main_model="gemma4-moe", helper_model="qwen3-coder")) local = db.query(Connection).filter_by(name="local").first() db.add( Chat( diff --git a/src/lembas/__init__.py b/src/lembas/__init__.py index 84cdf83..8dea925 100644 --- a/src/lembas/__init__.py +++ b/src/lembas/__init__.py @@ -1,3 +1,3 @@ """LLeMbas - a Middle-earth themed web UI for OpenAI-compatible LLM endpoints.""" -__version__ = "1.11.0" +__version__ = "1.12.0" diff --git a/src/lembas/api/admin.py b/src/lembas/api/admin.py index a8d6c5e..875b093 100644 --- a/src/lembas/api/admin.py +++ b/src/lembas/api/admin.py @@ -177,8 +177,10 @@ async def update_connection( unload_method: str = Form("POST"), extra_headers: str = Form(""), data_group_id: str = Form(""), + one_model_at_a_time: bool = Form(False), ) -> Response: connection = _connection(db, connection_id) + connection.one_model_at_a_time = one_model_at_a_time # Which of the instance's data groups this provider reads. Empty is "not # submitted" -- an older page -- and leaves it alone; a personal group is # somebody else's arrangement and cannot be chosen here. diff --git a/src/lembas/api/admin_models.py b/src/lembas/api/admin_models.py index 2142851..1499988 100644 --- a/src/lembas/api/admin_models.py +++ b/src/lembas/api/admin_models.py @@ -14,6 +14,7 @@ from sqlalchemy.orm import Session as DBSession from lembas.api.deps import AdminUser, Db, RequiredUser from lembas.db.models import AUTHOR_USER, Connection, Group, Model, PersonaRevision from lembas.services import chat as chat_service +from lembas.services import helpers as helpers_service from lembas.services import personas as personas_service from lembas.services import settings_store, uploads from lembas.services.llm.openai_client import MAX_CONTEXT @@ -209,6 +210,16 @@ async def model_detail( # and undo what a model wrote *before* they switched it off, which is # exactly when they would come looking. "persona": personas_service.get(db, model.model_id, None), + # Helpers designated for this model by the instance, and every other + # model id that could be one. + "designations": [ + d for d in helpers_service.own_designations(db, None) + if d.main_model == model.model_id + ], + "helper_choices": sorted( + {m.model_id for m in chat_service.available_models(db, user)} + - {model.model_id} + ), "persona_limit": personas_service.MAX_PERSONA_CHARS, "position_of": index + 1, "total": len(ordered), @@ -261,6 +272,7 @@ async def update_model( enabled: bool = Form(False), pinned: bool = Form(False), public: bool = Form(False), + single_session: bool = Form(False), position: str = Form(""), context_length: str = Form(""), default_effort: str = Form(""), @@ -284,6 +296,7 @@ async def update_model( model.enabled = enabled model.pinned = pinned model.public = public + model.single_session = single_session # Merged rather than rebuilt, unlike the capabilities below: params_json # holds whatever sampling defaults an administrator has set and this form @@ -338,6 +351,35 @@ async def update_model( ) +@router.post("/admin/models/{model_id}/helpers") +async def add_helper( + db: Db, + user: AdminUser, + model_id: str, + helper_model: str = Form(""), + offer: bool = Form(False), +) -> Response: + """Designate a model this one may send helpers to, for everybody. + + Its own form, for the persona's reason: a list edited row by row, not a field + carried by the big save. + """ + model = _model(db, model_id) + helpers_service.set_designation(db, None, model.model_id, helper_model, offer=offer) + return RedirectResponse( + f"/admin/models/{model.id}/edit?saved=Helpers+updated.", status_code=303 + ) + + +@router.post("/admin/models/{model_id}/helpers/{designation_id}/delete") +async def remove_helper(db: Db, user: AdminUser, model_id: str, designation_id: str) -> Response: + model = _model(db, model_id) + helpers_service.delete_designation(db, None, designation_id) + return RedirectResponse( + f"/admin/models/{model.id}/edit?saved=Helpers+updated.", status_code=303 + ) + + @router.post("/admin/models/{model_id}/persona") async def update_persona( db: Db, diff --git a/src/lembas/api/chats.py b/src/lembas/api/chats.py index c132916..e58dedb 100644 --- a/src/lembas/api/chats.py +++ b/src/lembas/api/chats.py @@ -37,6 +37,7 @@ from lembas.services import compaction as compaction_service from lembas.services import data_groups, interaction, settings_store, sse from lembas.services import files as files_service from lembas.services import generation as generation_service +from lembas.services import helpers as helpers_service from lembas.services import metrics as metrics_service from lembas.services import prompts as prompts_service from lembas.services import reports as reports_service @@ -309,6 +310,7 @@ async def start_chat( # -- the same mechanism the scope switches above use, and the reason the control # lives inside the composer's form rather than in the topbar. crowd_model_ids: list[str] = Form(default=[]), + helper_model_ids: list[str] = Form(default=[]), ) -> Response: """Create a chat from its first message. @@ -343,6 +345,7 @@ async def start_chat( ) _apply_crowd(db, chat, user, crowd_model_ids) + helpers_service.apply_chat_helpers(db, chat, user, helper_model_ids) _adopt_draft(db, user, draft_id, chat) @@ -2200,6 +2203,10 @@ async def update_chat(request: Request, db: Db, user: RequiredUser, chat_id: str # every box clears the crowd. _apply_crowd(db, chat, user, form.getlist("crowd_model_ids")) + if "helper_model_ids" in form: + # The helpers picker, the same shape again. + helpers_service.apply_chat_helpers(db, chat, user, form.getlist("helper_model_ids")) + submitted_params = {name: form[name] for name in _PARAM_RANGES if name in form} if submitted_params: if not allowed.get("chat.params"): diff --git a/src/lembas/api/pages.py b/src/lembas/api/pages.py index 7177c89..e3ed24a 100644 --- a/src/lembas/api/pages.py +++ b/src/lembas/api/pages.py @@ -100,6 +100,7 @@ def _chat_context(db: DBSession, user: User, chat: Chat | None) -> dict: **_crowd_context( db, user, chat, in_group, current.model_id if current is not None else "" ), + **_helper_context(db, user, chat, current.model_id if current is not None else ""), # What *this* model takes, not the three every model used to be assumed # to take. The vocabulary is per model -- gpt-oss has no `xhigh` and # Bonsai has no `high`, and sending the wrong one does not degrade, it @@ -231,6 +232,22 @@ def _talk_settings(db: DBSession, user: User) -> dict: "allow": EFFECT_ALLOW, "deny": EFFECT_DENY, "describe_verdict": describe_verdict, + **_helper_settings(db, user), + } + + +def _helper_settings(db: DBSession, user: User) -> dict: + """The person's own helper designations, and whether they may add any.""" + from lembas.services import helpers as helpers_service + + may = helpers_service.may_designate(db, user) + shown = bool(settings_store.subagents(db).get("enabled")) and permissions.has( + db, user, "tools.subagent" + ) + return { + "helper_settings_shown": shown, + "may_designate": may, + "own_designations": helpers_service.own_designations(db, user), } @@ -298,6 +315,42 @@ def describe_verdict(verdict) -> str: }.get(verdict.why, "") +def _helper_context(db: DBSession, user: User, chat: Chat | None, own: str) -> dict: + """The helpers picker: models designated for this chat's model, to add by hand. + + Only when helpers are on and this person may send one, and only when + something is designated -- with nothing designated the model is its own + helper and there is nothing to choose. On the new-chat screen the choice + rides along with the first message, exactly as the crowd's does. + """ + from lembas.services import helpers as helpers_service + + empty = {"helper_choices": [], "helper_member_ids": []} + if not settings_store.subagents(db).get("enabled"): + return empty + if not permissions.has(db, user, "tools.subagent"): + return empty + main = chat if chat is not None else own + if not main: + return empty + choices = helpers_service.picker(db, main, user) + return { + "helper_choices": choices, + "helper_reasons": { + choice.model.model_id: ( + # One literal, not two adjacent ones: the catalogue extractor + # reads the first string of a `t()` call only, so a sentence split + # across two would be looked up under a key that exists nowhere. + i18n.t("It cannot run beside this model's reply: one of them serves one request at a time, or their connection holds one model at a time.") # noqa: E501 + if choice.capacity + else (describe_verdict(choice.verdict) if not choice.addable else "") + ) + for choice in choices + }, + "helper_member_ids": [row.model_id for row in chat.helpers] if chat else [], + } + + def _crowd_context( db: DBSession, user: User, chat: Chat | None, models: list, default_model_id: str = "" ) -> dict: @@ -902,6 +955,9 @@ async def chat_index( context["models"], preselected.model_id if preselected is not None else "", ), + **_helper_context( + db, user, None, preselected.model_id if preselected is not None else "" + ), "starting_temporary": temporary, "temporary_toggle_url": new_chat_url(temporary="" if temporary else "1"), "model_navigate_url": without_model + ("&" if "?" in without_model else "?") + "model=", diff --git a/src/lembas/api/preferences.py b/src/lembas/api/preferences.py index df38859..27f6739 100644 --- a/src/lembas/api/preferences.py +++ b/src/lembas/api/preferences.py @@ -412,3 +412,52 @@ async def delete_talk_rule(db: Db, user: RequiredUser, rule_id: str) -> Response talk.delete_rule(db, user, rule_id) return RedirectResponse("/settings?saved=Model+rules+updated.", status_code=303) + + +# --- Helper models ------------------------------------------------------------- +# A person's own designations: for each of their models, others it may send +# helpers to. Added to the instance's, for them alone. Needs `helpers.designate`. +def _refuse_without_designate(db, user) -> Response | None: + from lembas.services import helpers + + if helpers.may_designate(db, user): + return None + return RedirectResponse( + "/settings?error=You+may+not+choose+your+own+helper+models.", status_code=303 + ) + + +@router.post("/helpers") +async def add_own_helper( + db: Db, + user: RequiredUser, + main_model: str = Form(""), + helper_model: str = Form(""), + offer: bool = Form(False), +) -> Response: + from lembas.security import permissions + from lembas.services import helpers + + refused = _refuse_without_designate(db, user) + if refused is not None: + return refused + # Only models this person can reach, on both sides: a designation naming one + # they cannot use would never be offered and would sit there unexplained. + if main_model == helper_model or not ( + permissions.can_use_model(db, user, main_model) + and permissions.can_use_model(db, user, helper_model) + ): + return RedirectResponse("/settings?error=Choose+two+different+models.", status_code=303) + helpers.set_designation(db, user, main_model, helper_model, offer=offer) + return RedirectResponse("/settings?saved=Helper+models+updated.", status_code=303) + + +@router.post("/helpers/{designation_id}/delete") +async def delete_own_helper(db: Db, user: RequiredUser, designation_id: str) -> Response: + from lembas.services import helpers + + refused = _refuse_without_designate(db, user) + if refused is not None: + return refused + helpers.delete_designation(db, user, designation_id) + return RedirectResponse("/settings?saved=Helper+models+updated.", status_code=303) diff --git a/src/lembas/db/models/__init__.py b/src/lembas/db/models/__init__.py index e20143e..3461498 100644 --- a/src/lembas/db/models/__init__.py +++ b/src/lembas/db/models/__init__.py @@ -31,12 +31,14 @@ from lembas.db.models.chat import ( ROLE_TOOL, ROLE_USER, Chat, + ChatHelper, CrowdMember, Folder, Message, ) from lembas.db.models.connection import Connection, Model, model_groups from lembas.db.models.data_group import DEFAULT_GROUP, DataGroup, InDataGroup +from lembas.db.models.helper import HelperDesignation from lembas.db.models.image import ImageWorkflow from lembas.db.models.library import ( AUTHOR_MODEL, @@ -166,7 +168,9 @@ __all__ = [ "Report", "Schedule", "Chat", + "ChatHelper", "CrowdMember", + "HelperDesignation", "Job", "Connection", "DEFAULT_GROUP", diff --git a/src/lembas/db/models/chat.py b/src/lembas/db/models/chat.py index 88256c8..80adf78 100644 --- a/src/lembas/db/models/chat.py +++ b/src/lembas/db/models/chat.py @@ -325,6 +325,12 @@ class Chat(UUIDPrimaryKey, Timestamps, InDataGroup, Base): cascade="all, delete-orphan", order_by="CrowdMember.position", ) + # Helper models somebody added to this chat by hand. See `ChatHelper`. + helpers: Mapped[list[ChatHelper]] = relationship( + back_populates="chat", + cascade="all, delete-orphan", + order_by="ChatHelper.position", + ) def __repr__(self) -> str: return f"" @@ -372,6 +378,30 @@ class CrowdMember(UUIDPrimaryKey, Timestamps, Base): return f"" +class ChatHelper(UUIDPrimaryKey, Timestamps, Base): + """A model added to this chat by hand, for its model to send helpers to. + + `CrowdMember`'s shape and `CrowdMember`'s reasoning: the model is text with + no foreign key, so a "Test & refresh" cannot silently empty the list. A + helper that no longer resolves is simply not offered. + """ + + __tablename__ = "chat_helpers" + __table_args__ = (UniqueConstraint("chat_id", "model_id"),) + + chat_id: Mapped[str] = mapped_column( + String(32), ForeignKey("chats.id", ondelete="CASCADE"), nullable=False, index=True + ) + model_id: Mapped[str] = mapped_column(String(300), nullable=False) + connection_id: Mapped[str | None] = mapped_column(String(32), nullable=True) + position: Mapped[int] = mapped_column(Integer, default=0, nullable=False) + + chat: Mapped[Chat] = relationship(back_populates="helpers") + + def __repr__(self) -> str: + return f"" + + class Message(UUIDPrimaryKey, Timestamps, Base): __tablename__ = "messages" diff --git a/src/lembas/db/models/connection.py b/src/lembas/db/models/connection.py index dbed933..23eb919 100644 --- a/src/lembas/db/models/connection.py +++ b/src/lembas/db/models/connection.py @@ -71,6 +71,14 @@ class Connection(UUIDPrimaryKey, Timestamps, InDataGroup, Base): unload_url: Mapped[str] = mapped_column(String(500), default="") unload_method: Mapped[str] = mapped_column(String(8), default="POST") + # Whether this endpoint holds one model at a time -- llama-swap in front of + # one GPU, which swaps the model out to serve another. A model here may then + # be its own helper, never another model from this connection: the helper + # would evict the model whose reply is waiting on it. Named for the + # *restrictive* state on purpose: `sync_schema` backfills a NOT NULL boolean + # with False, so False has to mean "as before" (any number of models at once). + one_model_at_a_time: Mapped[bool] = mapped_column(Boolean, default=False, nullable=False) + # Result of the most recent "Test & refresh", surfaced in the admin list. last_checked_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True)) last_error: Mapped[str] = mapped_column(Text, default="") @@ -119,6 +127,12 @@ class Model(UUIDPrimaryKey, Timestamps, Base): position: Mapped[int] = mapped_column(Integer, default=0, nullable=False) # Pinned models are offered first, before the full list. pinned: Mapped[bool] = mapped_column(Boolean, default=False, nullable=False) + # Whether this model serves one request at a time, so it cannot be its own + # helper: the helper's request would queue behind the reply that is waiting + # for it. The restrictive state, for the backfill reason on + # `Connection.one_model_at_a_time`. A column and not a `capabilities_json` + # key, because that dict is rebuilt from the checkboxes on every save. + single_session: Mapped[bool] = mapped_column(Boolean, default=False, nullable=False) # Public models are usable by anyone; otherwise access comes from `groups`. public: Mapped[bool] = mapped_column(Boolean, default=True, nullable=False) diff --git a/src/lembas/db/models/helper.py b/src/lembas/db/models/helper.py new file mode 100644 index 0000000..3d2b170 --- /dev/null +++ b/src/lembas/db/models/helper.py @@ -0,0 +1,37 @@ +"""Which models a main model may send helpers to, besides itself. + +Designated **per main model**, on the owner's word: gpt-oss may use qwen35, +bonsai may use deepseek-flash, and neither says anything about the other. The +instance designates (`owner_id` NULL); a person holding `helpers.designate` may +add their own, and theirs are added to the instance's -- a row of theirs for the +same pair wins, which is how they change whether it is offered. + +`offer` says whether the main model may choose the helper itself. Off, it can be +added to a chat only by hand, and is then the chat's. + +Text model ids and no foreign key, for the reason every such reference here has: +"Test & refresh" recreates `Model` rows. +""" + +from __future__ import annotations + +from sqlalchemy import Boolean, ForeignKey, String, UniqueConstraint +from sqlalchemy.orm import Mapped, mapped_column + +from lembas.db.base import Base, Timestamps, UUIDPrimaryKey + + +class HelperDesignation(UUIDPrimaryKey, Timestamps, Base): + __tablename__ = "helper_designations" + __table_args__ = (UniqueConstraint("owner_id", "main_model", "helper_model"),) + + owner_id: Mapped[str | None] = mapped_column( + String(32), ForeignKey("users.id", ondelete="CASCADE"), nullable=True, index=True + ) + main_model: Mapped[str] = mapped_column(String(300), nullable=False) + helper_model: Mapped[str] = mapped_column(String(300), nullable=False) + helper_connection_id: Mapped[str] = mapped_column(String(32), default="") + offer: Mapped[bool] = mapped_column(Boolean, default=True, nullable=False) + + def __repr__(self) -> str: + return f" {self.helper_model}>" diff --git a/src/lembas/security/permissions.py b/src/lembas/security/permissions.py index ad118a0..5be9d58 100644 --- a/src/lembas/security/permissions.py +++ b/src/lembas/security/permissions.py @@ -347,6 +347,18 @@ PERMISSION_DEFS: tuple[PermissionDef, ...] = ( False, "Chat", ), + # Helpers on another model. The instance designates which models each main + # model may send helpers to; this lets a person add their own designations + # for themselves. Off by default: a designation decides where a person's + # tasks -- and whatever context a model writes into them -- are sent. + PermissionDef( + "helpers.designate", + "Choose their own helper models", + "Let this person designate, for each model, other models it may send " + "helpers to -- added to the instance's designations, for them alone.", + False, + "Chat", + ), PermissionDef( "data.manage", "Manage their own data groups", diff --git a/src/lembas/services/harness.py b/src/lembas/services/harness.py index d3e77e3..0f8b29a 100644 --- a/src/lembas/services/harness.py +++ b/src/lembas/services/harness.py @@ -292,6 +292,9 @@ def context_variables( # opinion. One fragment covering all three would say nothing useful to # any of them. "friend": "", + # Where a helper may run, when that is more than the answering model. + # Filled below, from the same candidates the tool's enum was built from. + "helper_models": "", # Who else is here. Filled below, where the chat's own model is known -- # a model does not need telling that it exists. "model_roster": "", @@ -364,6 +367,9 @@ def context_variables( db, user, exclude=speaking.model_id, group=group ) + if "subagent" in families and speaking.model_id == chat.model_id: + values["helper_models"] = _helper_models(db, chat, user) + if "persona" in families: # This person's own personality for this model, falling back to the # administrator's default until the model has written one with them; @@ -376,6 +382,37 @@ def context_variables( return values +def _helper_models(db: DBSession, chat, user) -> str: + """One line per model a helper may run on, or "" when it is only this one. + + The same candidates `tools._subagent_defs` built the `model` enum from, so + the list the model reads and the values the call accepts cannot disagree. + Roster-shaped and capped the same way. + """ + from lembas.services import chat as chat_service + from lembas.services import helpers as helpers_service + + found = helpers_service.candidates(db, chat, user) + if len(found) < 2: + return "" + lines: list[str] = [] + budget = chat_service.MAX_ROSTER_CHARS + for candidate in found[: chat_service.MAX_ROSTER_MODELS]: + model = candidate.model + parts = ((model.description or "").strip(), (model.notes or "").strip()) + about = " ".join(" ".join(p for p in parts if p).split())[: chat_service.MAX_ROSTER_ENTRY] + line = f"- {model.label} ({model.model_id})" + if candidate.source == helpers_service.SOURCE_SELF: + line += " — you" + elif about: + line += f" — {about}" + if len(line) > budget: + break + budget -= len(line) + lines.append(line) + return "\n".join(lines) + + def _schedule_values(db: DBSession, chat, user) -> dict[str, str]: """What a scheduled task's chat is for, and how often it comes round. diff --git a/src/lembas/services/helpers.py b/src/lembas/services/helpers.py new file mode 100644 index 0000000..b9eb082 --- /dev/null +++ b/src/lembas/services/helpers.py @@ -0,0 +1,363 @@ +"""Which model a helper runs on: the main model itself, or one designated for it. + +`subagent_run` used to have one answer -- the parent's own model. It has three +sources now, in this order, and they are decided here by logic, never by the +model: + +1. **The main model itself**, unless it cannot run two requests at once. +2. **Helpers somebody added to this chat by hand** (`ChatHelper`). +3. **Helpers designated for the main model with `offer` on** -- by the instance, + or by a person holding `helpers.designate` -- which the model may choose. + +Each candidate passes two checks before it is offered: + +* **Capacity.** A model marked *serves one request at a time* cannot be its own + helper: the helper's request would queue behind the very reply waiting for it. + A connection marked *holds one model at a time* (llama-swap in front of one + GPU) cannot serve a helper on another of its models: loading it evicts the + model whose reply is waiting. The main model may still help itself there. + Both flags default off, so an instance that never sets them behaves exactly as + before -- the main model is its own helper. +* **The talk rules**, read from the main model: a chat's hand-added helpers need + to be `addable`, a designation the model chooses itself needs to be `offered`. + A different data group is the deny a rule has to open, exactly as for a crowd. + +`ask_friend` and the crowd do not take the capacity check, on the owner's word: +both are sequential, and waiting for a model to load is accepted there. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +from sqlalchemy import select +from sqlalchemy.orm import Session as DBSession + +from lembas.db.models import Chat, ChatHelper, HelperDesignation, Model, User + +PERMISSION = "helpers.designate" + +SOURCE_SELF = "self" +SOURCE_CHAT = "chat" +SOURCE_OFFERED = "offered" + + +@dataclass(frozen=True) +class Candidate: + """A model a helper may run on, and why it is on the list.""" + + model: Model + source: str + + +@dataclass(frozen=True) +class Choice: + """One designation as the helpers picker shows it. + + `reason` is in English, for a model; `capacity` and `verdict` are what a + screen needs to say the same thing in the reader's language. + """ + + model: Model + offer: bool + addable: bool + reason: str + capacity: bool = False + verdict: object = None + + +def may_designate(db: DBSession, user: User | None) -> bool: + from lembas.security import permissions + + return user is not None and permissions.has(db, user, PERMISSION) + + +def main_row(db: DBSession, chat: Chat) -> Model | None: + """The chat's own model as a row, on the connection it is reached through.""" + from lembas.services import chat as chat_service + + return chat_service.model_row(db, chat_service.speaker_for(db, chat)) + + +def capacity_refusal(main: Model | None, helper: Model) -> str: + """Why this helper cannot run beside the main model's waiting reply, or "".""" + if main is None: + return "" + if helper.model_id == main.model_id and helper.connection_id == main.connection_id: + if helper.single_session: + return ( + f"{helper.label} serves one request at a time, so it cannot be its own " + f"helper: the helper would wait behind the reply that is waiting for it." + ) + return "" + if helper.connection_id == main.connection_id: + connection = helper.connection + if connection is not None and connection.one_model_at_a_time: + return ( + f"{connection.name} holds one model at a time, so a helper on " + f"{helper.label} would unload {main.label} while its reply waits." + ) + return "" + + +def designations(db: DBSession, user: User | None, main_model: str) -> list[HelperDesignation]: + """The helpers designated for one main model: the instance's, then the person's. + + A person's own row for the same helper replaces the instance's -- that is + how they change whether it is offered -- and theirs count only while they + hold `helpers.designate`. + """ + rows = list( + db.scalars( + select(HelperDesignation) + .where( + HelperDesignation.owner_id.is_(None), + HelperDesignation.main_model == main_model, + ) + .order_by(HelperDesignation.helper_model) + ) + ) + if user is not None and may_designate(db, user): + own = list( + db.scalars( + select(HelperDesignation) + .where( + HelperDesignation.owner_id == user.id, + HelperDesignation.main_model == main_model, + ) + .order_by(HelperDesignation.helper_model) + ) + ) + replaced = {row.helper_model for row in own} + rows = [row for row in rows if row.helper_model not in replaced] + own + return rows + + +def _reachable(db: DBSession, user: User | None, model_id: str, connection_id: str | None): + """The row this person can reach for a model id, preferring a named connection.""" + from lembas.services import chat as chat_service + + rows = [m for m in chat_service.available_models(db, user) if m.model_id == model_id] + rows.sort(key=lambda m: m.connection_id != (connection_id or "")) + return rows[0] if rows else None + + +def candidates(db: DBSession, chat: Chat, user: User | None) -> list[Candidate]: + """Every model a helper of this chat may run on, in the order it is chosen.""" + from lembas.services import data_groups, talk + + main = main_row(db, chat) + group = data_groups.for_chat(db, chat) + judge = talk.Judge(db, user) + found: list[Candidate] = [] + seen: set[str] = set() + + if main is not None and not capacity_refusal(main, main): + found.append(Candidate(main, SOURCE_SELF)) + seen.add(main.model_id) + + for row in chat.helpers: + model = _reachable(db, user, row.model_id, row.connection_id) + if model is None or model.model_id in seen: + continue + if not judge.verdict(chat.model_id, group, model).addable: + continue + if capacity_refusal(main, model): + continue + found.append(Candidate(model, SOURCE_CHAT)) + seen.add(model.model_id) + + for row in designations(db, user, chat.model_id): + if not row.offer or row.helper_model in seen: + continue + model = _reachable(db, user, row.helper_model, row.helper_connection_id) + if model is None: + continue + if not judge.verdict(chat.model_id, group, model).offered: + continue + if capacity_refusal(main, model): + continue + found.append(Candidate(model, SOURCE_OFFERED)) + seen.add(model.model_id) + return found + + +def choose( + db: DBSession, chat: Chat, user: User | None, wanted: str +) -> tuple[Model | None, str]: + """The model a helper runs on, or a refusal saying what could have been named. + + `wanted` comes from a tool call, so it is matched against the candidates and + nothing else -- by model id, then by label. Empty means the default: the main + model itself, or failing that the first helper added to this chat by hand. + """ + found = candidates(db, chat, user) + names = ", ".join(c.model.model_id for c in found) + wanted = (wanted or "").strip() + if not wanted: + for source in (SOURCE_SELF, SOURCE_CHAT): + for candidate in found: + if candidate.source == source: + return candidate.model, "" + if found: + return None, f"Name the model to send the helper to. You may use: {names}." + return None, _why_none(db, chat) + + lowered = wanted.lower() + for candidate in found: + if lowered in (candidate.model.model_id.lower(), candidate.model.label.lower()): + return candidate.model, "" + + main = main_row(db, chat) + target = _reachable(db, user, wanted, None) + if target is not None and capacity_refusal(main, target): + reason = capacity_refusal(main, target) + else: + reason = f"{wanted!r} is not a helper you may use." + return None, f"{reason} You may use: {names}." if names else f"{reason} {_why_none(db, chat)}" + + +def _why_none(db: DBSession, chat: Chat) -> str: + main = main_row(db, chat) + if main is not None and capacity_refusal(main, main): + return capacity_refusal(main, main) + " No other helper is set up for it." + return "No helper model is available here. Do this part yourself." + + +def picker(db: DBSession, chat_or_main: Chat | str, user: User | None) -> list[Choice]: + """The designations for a main model, as the composer's helpers picker lists them. + + Every designation, offered or not, with whether this person may add it to + the chat by hand and, where not, why. On the new-chat screen there is no + chat yet, so a main model id is passed instead. + """ + from lembas.services import data_groups, talk + + if isinstance(chat_or_main, Chat): + main_model = chat_or_main.model_id + main = main_row(db, chat_or_main) + group = data_groups.for_chat(db, chat_or_main) + else: + main_model = chat_or_main + main = _reachable(db, user, main_model, None) + group = data_groups.for_pair(db, user, main_model) + judge = talk.Judge(db, user) + shown: list[Choice] = [] + for row in designations(db, user, main_model): + model = _reachable(db, user, row.helper_model, row.helper_connection_id) + if model is None: + continue + verdict = judge.verdict(main_model, group, model) + capacity = capacity_refusal(main, model) + reason = capacity or ("" if verdict.addable else verdict.reason) + shown.append( + Choice( + model, + row.offer, + bool(verdict.addable and not capacity), + reason, + capacity=bool(capacity), + verdict=verdict, + ) + ) + return shown + + +def set_designation( + db: DBSession, + owner: User | None, + main_model: str, + helper_model: str, + *, + offer: bool, + connection_id: str = "", +) -> None: + """Designate a helper for a main model, or change whether it is offered.""" + main_model = (main_model or "").strip()[:300] + helper_model = (helper_model or "").strip()[:300] + if not main_model or not helper_model: + return + existing = db.scalar( + select(HelperDesignation).where( + ( + HelperDesignation.owner_id.is_(None) + if owner is None + else HelperDesignation.owner_id == owner.id + ), + HelperDesignation.main_model == main_model, + HelperDesignation.helper_model == helper_model, + ) + ) + if existing is not None: + existing.offer = offer + existing.helper_connection_id = connection_id or existing.helper_connection_id + else: + db.add( + HelperDesignation( + owner_id=owner.id if owner is not None else None, + main_model=main_model, + helper_model=helper_model, + helper_connection_id=connection_id or "", + offer=offer, + ) + ) + db.commit() + + +def delete_designation(db: DBSession, owner: User | None, designation_id: str) -> bool: + row = db.get(HelperDesignation, designation_id) + if row is None or row.owner_id != (owner.id if owner is not None else None): + return False + db.delete(row) + db.commit() + return True + + +def own_designations(db: DBSession, owner: User | None) -> list[HelperDesignation]: + return list( + db.scalars( + select(HelperDesignation) + .where( + HelperDesignation.owner_id.is_(None) + if owner is None + else HelperDesignation.owner_id == owner.id + ) + .order_by(HelperDesignation.main_model, HelperDesignation.helper_model) + ) + ) + + +def apply_chat_helpers(db: DBSession, chat: Chat, user: User | None, values: list[str]) -> None: + """Replace a chat's hand-added helpers with the models named. + + Only designations for the chat's model this person may add -- the talk rules + and the capacity check, the same as the picker shows -- and never the chat's + own model, which is its own helper already. + """ + allowed = {c.model.model_id: c.model for c in picker(db, chat, user) if c.addable} + wanted: list[str] = [] + for value in values: + value = str(value).strip() + if value and value in allowed and value != chat.model_id and value not in wanted: + wanted.append(value) + chat.helpers = [ + ChatHelper(model_id=model_id, connection_id=allowed[model_id].connection_id, position=i) + for i, model_id in enumerate(wanted) + ] + + +__all__ = [ + "PERMISSION", + "Candidate", + "Choice", + "apply_chat_helpers", + "candidates", + "capacity_refusal", + "choose", + "delete_designation", + "designations", + "may_designate", + "own_designations", + "picker", + "set_designation", +] diff --git a/src/lembas/services/prompts.py b/src/lembas/services/prompts.py index f6b54ce..274138c 100644 --- a/src/lembas/services/prompts.py +++ b/src/lembas/services/prompts.py @@ -163,6 +163,14 @@ VARIABLES: tuple[Variable, ...] = ( "page, bounded, and empty unless this model may ask one of them a " "question — a list of peers it cannot reach is context spent on nothing.", ), + Variable( + "helper_models", + "The models a helper may run on", + "One line per model subagent_run may send a helper to, the answering model " + "first when it may be its own helper: its name, the id to pass as model, " + "and what it is for. Empty when the only choice is the model itself, so the " + "section vanishes and the tool reads as it always did.", + ), Variable( "persona", "Its personality with this person", @@ -1366,6 +1374,28 @@ BUILTIN: tuple[Fragment, ...] = ( "because a helper made it." ), ), + Fragment( + key="tool.subagent_models", + label="Which model a helper runs on", + group=GROUP_TOOLS, + order=252, + families=("subagent",), + variables=("helper_models",), + requires=("helper_models",), + hint="Appears only when a helper may run on a model other than the one " + "answering -- the chat's own helpers added by hand, or ones designated for " + "this model and offered to it. The list is built after the capacity check " + "and the model rules, so every line is one the call will accept.", + default=( + "### Where a helper can run\n" + "\n" + "subagent_run takes a model. Leave it out for the first one below; name " + "another, by the id in brackets, when its strengths suit the piece of work " + "better. Only these are accepted:\n" + "\n" + "{{helper_models}}" + ), + ), Fragment( key="tool.subagent_agent", label="Helpers on a machine", diff --git a/src/lembas/services/subagent.py b/src/lembas/services/subagent.py index fb7dbb5..41d2086 100644 --- a/src/lembas/services/subagent.py +++ b/src/lembas/services/subagent.py @@ -196,6 +196,7 @@ def _create_child( title: str, write: bool, friend: Model | None = None, + helper: Model | None = None, ) -> Chat: """The hidden chat one helper or one friend runs in. @@ -225,6 +226,10 @@ def _create_child( peer = friend is not None owner = db.get(User, parent.user_id) if parent.user_id else None + # A helper on another model runs on that model -- as a pair, for the reason + # the friend is -- and reads that model's data group, exactly as a friend + # does. It is still a helper: the parent's kind, machine and scope. + other = friend or helper child = Chat( user_id=parent.user_id, # An ordinary chat for a friend even when the asking one is an agent @@ -232,14 +237,14 @@ def _create_child( # and a peer being asked a question is not working on one. kind=KIND_CHAT if peer else parent.kind, title=title[:200] or ("Question" if peer else "Helper"), - model_id=friend.model_id if peer else parent.model_id, - connection_id=friend.connection_id if peer else parent.connection_id, + model_id=other.model_id if other is not None else parent.model_id, + connection_id=other.connection_id if other is not None else parent.connection_id, # The helper is in its parent's group, doing its parent's work. A friend # is in its own model's, so it reads its own group's data and never the # asker's. data_group_id=( - data_groups.for_pair(db, owner, friend.model_id, friend.connection_id) - if peer + data_groups.for_pair(db, owner, other.model_id, other.connection_id) + if other is not None else data_groups.for_chat(db, parent) ), # Never in a listing, and swept a day later even if it is kept. @@ -254,8 +259,11 @@ def _create_child( child.agent_mode = MODE_WRITING if write else MODE_READING if peer: child.scope_json = {**(child.scope_json or {}), "role": ROLE_FRIEND} - effort = str((friend.params_json or {}).get("reasoning_effort") or "") - if effort not in chat_service.efforts_for(friend): + if other is not None: + # The other model's own default, never the parent's: the vocabularies + # differ, and `high` handed to a Bonsai raises inside its chat template. + effort = str((other.params_json or {}).get("reasoning_effort") or "") + if effort not in chat_service.efforts_for(other): effort = "" else: effort = chat_service.resolved_effort(parent) @@ -455,6 +463,7 @@ async def _run_subagent(context: ToolContext, args: dict[str, Any]) -> ToolOutco title = str(args.get("title") or "").strip() or task[:60] briefing = str(args.get("context") or "") want_write = bool(args.get("write")) + wanted_model = str(args.get("model") or "") if not task: return _error( @@ -493,6 +502,15 @@ async def _run_subagent(context: ToolContext, args: dict[str, Any]) -> ToolOutco if owner is None: # pragma: no cover - a chat outliving its owner return _error("That account no longer exists.", task=task) + # Which model the helper runs on -- decided by logic, never taken on the + # model's word: the name is matched against the candidates the tool was + # built from, which already passed the capacity check and the talk rules. + from lembas.services import helpers as helpers_service + + helper, refusal = helpers_service.choose(db, parent, owner, wanted_model) + if helper is None: + return _error(refusal, task=task) + # After the refusals above and before anything is created. The order is # the design: a call that could never have worked should be told *why* # rather than told it has run out of helpers, and the counter should @@ -508,7 +526,13 @@ async def _run_subagent(context: ToolContext, args: dict[str, Any]) -> ToolOutco if refusal: return _error(refusal, task=task) - child = _create_child(db, parent, title=title, write=write) + own = ( + helper.model_id == parent.model_id + and (not parent.connection_id or helper.connection_id == parent.connection_id) + ) + child = _create_child( + db, parent, title=title, write=write, helper=None if own else helper + ) child_id = child.id _LIVE.add(child_id) @@ -812,10 +836,27 @@ def friend_tool_defs() -> list[ToolDef]: ] -def tool_defs() -> list[ToolDef]: - """The one tool, built here so `services/tools.py` need not know the wording.""" +def tool_defs(models: list[str] | None = None) -> list[ToolDef]: + """The one tool, built here so `services/tools.py` need not know the wording. + + `models` is who a helper may run on, from `helpers.candidates`, the main + model first when it is its own helper. One of them, or none given, is the + tool as it always was; more adds a `model` argument whose enum is exactly + that list, so a small model cannot type a name that was never offered. + """ from lembas.services.tools import FAMILY_SUBAGENT, RISK_READ, ToolDef + properties = _subagent_properties() + if models and len(models) > 1: + properties["model"] = { + "type": "string", + "enum": list(models), + "description": ( + "Which model the helper runs on. Leave it out for the default " + f"({models[0]}). Choose another when its strengths suit the task " + "better -- the list of helpers you were given says what each is." + ), + } return [ ToolDef( name="subagent_run", @@ -836,39 +877,7 @@ def tool_defs() -> list[ToolDef]: ), parameters={ "type": "object", - "properties": { - "task": { - "type": "string", - "description": ( - "What the helper is to do, written out in full and as " - "an instruction. Say what a good answer contains and " - "how long it should be. It is read on its own, with " - "none of this conversation around it." - ), - }, - "title": { - "type": "string", - "description": "A few words naming this piece of work.", - }, - "context": { - "type": "string", - "description": ( - "Facts the helper needs that it cannot look up — what " - "the reader asked for, decisions already made, names " - "and paths. Not a summary of the conversation." - ), - }, - "write": { - "type": "boolean", - "description": ( - "True if the helper must change something: write a " - "file, keep a note, file a report. Leave it out for " - "anything that only reads, which is nearly always. A " - "writing helper is refused where you would have been " - "stopped for approval yourself." - ), - }, - }, + "properties": properties, "required": ["task"], }, run=_run_subagent, @@ -894,3 +903,40 @@ __all__ = [ "live_count", "tool_defs", ] + + +def _subagent_properties() -> dict[str, Any]: + """The arguments every `subagent_run` has; `tool_defs` may add `model`.""" + return { + "task": { + "type": "string", + "description": ( + "What the helper is to do, written out in full and as " + "an instruction. Say what a good answer contains and " + "how long it should be. It is read on its own, with " + "none of this conversation around it." + ), + }, + "title": { + "type": "string", + "description": "A few words naming this piece of work.", + }, + "context": { + "type": "string", + "description": ( + "Facts the helper needs that it cannot look up — what " + "the reader asked for, decisions already made, names " + "and paths. Not a summary of the conversation." + ), + }, + "write": { + "type": "boolean", + "description": ( + "True if the helper must change something: write a " + "file, keep a note, file a report. Leave it out for " + "anything that only reads, which is nearly always. A " + "writing helper is refused where you would have been " + "stopped for approval yourself." + ), + }, + } diff --git a/src/lembas/services/tools.py b/src/lembas/services/tools.py index 43875bd..268dd58 100644 --- a/src/lembas/services/tools.py +++ b/src/lembas/services/tools.py @@ -1750,13 +1750,31 @@ def _schedule_defs() -> list[ToolDef]: return schedule_tool.tool_defs() -def _subagent_defs() -> list[ToolDef]: - """The subagent tool. Imported inside the call for the reason above.""" +def _subagent_registry_defs() -> list[ToolDef]: from lembas.services import subagent as subagent_service return subagent_service.tool_defs() +def _subagent_defs(db: DBSession, chat: Chat, user: User | None) -> list[ToolDef]: + """The subagent tool, shaped by which models a helper may run on. + + Withdrawn when there are none -- a main model that cannot help itself and + has no other helper set up would be offered a tool every call of which is + refused, and a model finds that out one wasted round at a time. With only + itself it is the tool it always was; with more, it gains a `model` + argument listing exactly those. Imported inside the call for the reason + above. + """ + from lembas.services import helpers as helpers_service + from lembas.services import subagent as subagent_service + + found = helpers_service.candidates(db, chat, user) + if not found: + return [] + return subagent_service.tool_defs([c.model.model_id for c in found]) + + def _friend_defs() -> list[ToolDef]: """The ask-a-friend tool. Same module, same reason for the late import.""" from lembas.services import subagent as subagent_service @@ -1824,7 +1842,9 @@ def registry(db: DBSession) -> dict[str, ToolDef]: # reaches the model. That omission has cost two features their # instructions already. *_schedule_defs(), - *_subagent_defs(), + # The registry wants the tool's name and family, not this chat's + # candidates -- so the bare definition, which needs no chat. + *_subagent_registry_defs(), *_friend_defs(), *_crowd_defs(), ] @@ -1886,7 +1906,7 @@ def resolve_tools( *_agent_defs(db, chat, user), *(_image_defs(db, image_values) if images_ready else []), *(_schedule_defs() if schedules_on else []), - *(_subagent_defs() if subagents_on else []), + *(_subagent_defs(db, chat, user) if subagents_on else []), *(_friend_defs() if subagents_on else []), # Only on the closing turn, and only with a round left. Not gated on a # capability or a permission: a tool that exists on exactly one turn of diff --git a/src/lembas/web/i18n/sk.py b/src/lembas/web/i18n/sk.py index fe4a516..17eec29 100644 --- a/src/lembas/web/i18n/sk.py +++ b/src/lembas/web/i18n/sk.py @@ -1906,3 +1906,28 @@ MESSAGES.update( 'Your setting lets no model talk to another.': 'Vaše nastavenie nedovoľuje žiadnemu modelu hovoriť s iným.', } ) + +# --- Helpers on another model (1.12.0) ----------------------------------------- +MESSAGES.update( + { + 'Holds one model at a time': 'Drží naraz jeden model', + 'Tick this for llama-swap in front of one GPU, or anything else that unloads one model to serve another. A model here may then be its own helper, but never send a helper to another model on this connection: loading it would unload the model whose reply is waiting.': 'Označte pri llama-swap pred jednou GPU alebo čomkoľvek, čo jeden model uvoľní, aby obslúžilo iný. Model tu potom môže byť vlastným pomocníkom, ale nikdy nepošle pomocníka inému modelu na tomto spojení: jeho načítanie by uvoľnilo model, ktorého odpoveď čaká.', + 'Serves one request at a time': 'Obsluhuje naraz jednu požiadavku', + 'Tick this for a model with a single slot. It then cannot be its own helper, because the helper would wait behind the reply that is waiting for it. Other models can still be its helpers.': 'Označte pri modeli s jedným slotom. Potom nemôže byť vlastným pomocníkom, lebo pomocník by čakal za odpoveďou, ktorá čaká naňho. Iné modely môžu byť jeho pomocníkmi aj tak.', + "Other models this one may send helpers to, for everybody. Offered ones it may choose itself; the rest can only be added to a chat by hand. It is always its own helper too, unless it serves one request at a time. The model rules and the connection's capacity are checked when a helper is sent.": 'Iné modely, ku ktorým môže tento posielať pomocníkov, pre všetkých. Ponúkané si môže vybrať sám; ostatné sa dajú do konverzácie pridať len ručne. Vždy je aj vlastným pomocníkom, pokiaľ neobsluhuje naraz jednu požiadavku. Pravidlá modelov a kapacita spojenia sa kontrolujú pri odoslaní pomocníka.', + 'offered to the model': 'ponúkané modelu', + 'by hand only': 'len ručne', + 'Remove this helper': 'Odstrániť tohto pomocníka', + 'Helper model': 'Model pomocníka', + 'Offer it to the model': 'Ponúknuť ho modelu', + 'Add helper': 'Pridať pomocníka', + 'Models this one may send helpers to. Tick one to add it to this chat; one offered to the model it may choose anyway.': 'Modely, ku ktorým môže tento posielať pomocníkov. Označením ho pridáte do tejto konverzácie; ponúkaný si model môže vybrať aj tak.', + 'Offered to the model.': 'Ponúkaný modelu.', + 'Only when added here.': 'Len keď ho pridáte tu.', + 'Your helper models': 'Vaše modely pomocníkov', + "Models each of your models may send helpers to, for you alone, on top of the instance's. Your row for the same pair replaces the instance's.": 'Modely, ku ktorým môže každý z vašich modelov posielať pomocníkov, len pre vás, navyše k tým od inštancie. Váš riadok pre tú istú dvojicu nahradí riadok inštancie.', + "Set by the administrator on each model's page. Adding your own needs the permission to choose your own helper models.": 'Nastavuje správca na stránke každého modelu. Na pridanie vlastných potrebujete oprávnenie vyberať si vlastné modely pomocníkov.', + 'May send helpers to': 'Smie posielať pomocníkov k', + "It cannot run beside this model's reply: one of them serves one request at a time, or their connection holds one model at a time.": 'Nemôže bežať popri odpovedi tohto modelu: jeden z nich obsluhuje naraz jednu požiadavku alebo ich spojenie drží naraz jeden model.', + } +) diff --git a/src/lembas/web/templates/admin/_connection_row.html b/src/lembas/web/templates/admin/_connection_row.html index 2fc18d9..5d27f14 100644 --- a/src/lembas/web/templates/admin/_connection_row.html +++ b/src/lembas/web/templates/admin/_connection_row.html @@ -121,6 +121,15 @@

+
+ +

{{ t("Tick this for llama-swap in front of one GPU, or anything else that unloads one model to serve another. A model here may then be its own helper, but never send a helper to another model on this connection: loading it would unload the model whose reply is waiting.") }}

+
+
+
+ +

{{ t("Tick this for a model with a single slot. It then cannot be its own helper, because the helper would wait behind the reply that is waiting for it. Other models can still be its helpers.") }}

+
+