Custom HTTP tools an administrator defines

A row in custom_tools becomes a ToolDef like any built-in, offered beside
the thirteen. The registry had to stop being an import-time constant for
that: `resolve_tools` now returns the schemas *and* the runners together,
carried to the loop on the ToolContext.

That closes a hole on the way. `run_tool` looked names up in the global
REGISTRY with no reference to what had been offered, so a model naming a
tool its chat was gated out of -- a family switched off, a permission the
reader lacks -- had it run anyway. The resolved set is now authoritative.

Arguments come from a model, so an argument may fill a hole but never move
the target: the scheme and host of a URL template are literal, values are
escaped for where they land, and the origin is pinned afterwards. Every
redirect hop is checked the way services/fetch.py checks one, and the
secret is dropped if a hop leaves the origin it was issued for.

Also fixes the tool-activity block claiming every library tool had
"searched the web", which it has done since the second family landed.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Jaroslav Beneš
2026-08-01 16:26:47 +02:00
co-authored by Claude Opus 5
parent 4ee7d3db7d
commit d4cefb066a
26 changed files with 2771 additions and 60 deletions
+180 -18
View File
@@ -29,10 +29,12 @@ from collections.abc import Awaitable, Callable
from dataclasses import dataclass, field
from typing import Any
from sqlalchemy import select
from sqlalchemy.orm import Session as DBSession
from lembas.db.models import AUTHOR_MODEL, Chat, User
from lembas.db.session import session_scope
from lembas.services import prompts as prompts_service
from lembas.services import search as search_service
from lembas.services import settings_store
from lembas.services.library import documents as documents_service
@@ -58,8 +60,24 @@ FAMILY_NOTES = "notes"
FAMILY_MEMORY = "memory"
FAMILY_SKILLS = "skills"
# A tool that is a database row gets a family of its own, so that it can carry
# its own guidance -- "custom:weather", "mcp:github". Everything before the
# colon is the *gate*: the capability flag and the permission are per gate, not
# per row, because a server advertising forty tools must not mean forty
# checkboxes on every model.
FAMILY_CUSTOM = "custom"
FAMILY_MCP = "mcp"
# The built-in families, in the order they are offered.
FAMILIES = (FAMILY_SEARCH, FAMILY_KNOWLEDGE, FAMILY_NOTES, FAMILY_MEMORY, FAMILY_SKILLS)
GATES = (*FAMILIES, FAMILY_CUSTOM, FAMILY_MCP)
def gate_of(family: str) -> str:
"""The part a capability flag and a permission are named after."""
return family.split(":", 1)[0]
@dataclass
class ToolContext:
@@ -72,10 +90,14 @@ class ToolContext:
owner_id: str
search_config: dict[str, Any] = field(default_factory=dict)
allow_private_fetch: bool = False
# Which knowledge bases this chat is scoped to. Empty means "everything the
# owner can see", which is what a chat with none attached should do.
base_ids: list[str] = field(default_factory=list)
# Name -> definition for the tools actually offered on this request. None
# means nobody resolved a set, and only then does `run_tool` fall back to
# the import-time registry. A dict, *even an empty one*, is authoritative:
# a model naming a tool it was not offered must not get it run.
tools: dict[str, ToolDef] | None = None
@dataclass
@@ -114,6 +136,31 @@ class ToolDef:
}
@dataclass(frozen=True)
class ToolSet:
"""What one request may call: the schemas to send, and how to run them.
The two halves have to travel together. `enabled_tools` used to return
schemas alone, which worked only because every runner was reachable through
the import-time `REGISTRY`. A tool that is a database row is not, so the
resolution has to be carried from the session that made it to the loop that
uses it.
"""
defs: tuple[ToolDef, ...] = ()
@property
def schemas(self) -> list[dict[str, Any]]:
return [tool.schema for tool in self.defs]
@property
def by_name(self) -> dict[str, ToolDef]:
return {tool.name: tool for tool in self.defs}
def __bool__(self) -> bool:
return bool(self.defs)
def _object(properties: dict[str, Any], required: list[str]) -> dict[str, Any]:
return {"type": "object", "properties": properties, "required": required}
@@ -645,21 +692,69 @@ def _family_allowed(
Absent counts as on when `tools` is on, so an upgrade does not silently take
web search away from every model already set up for it.
"""
gate = gate_of(family)
default = bool(capabilities.get("tools"))
if not capabilities.get(f"tool_{family}", default):
if not capabilities.get(f"tool_{gate}", default):
return False
if family == FAMILY_SEARCH:
if gate == FAMILY_SEARCH:
return bool(
allowed.get("tools.web_search")
and config.get("enabled")
and not search_service.availability(str(config.get("provider") or "ddgs"))
)
return bool(allowed.get(f"tools.{family}") and allowed.get("library.use"))
if gate in (FAMILY_CUSTOM, FAMILY_MCP):
# Deliberately without `library.use`: an HTTP endpoint an administrator
# wrote has nothing to do with this person's own documents and notes,
# and requiring the library permission for it would be a coincidence of
# naming rather than a rule.
return bool(allowed.get(f"tools.{gate}"))
return bool(allowed.get(f"tools.{gate}") and allowed.get("library.use"))
def enabled_tools(db: DBSession, chat: Chat, user: User | None) -> list[dict[str, Any]]:
"""The tool schemas to offer for this chat."""
def _row_defs(db: DBSession, user: User | None, *, everything: bool = False) -> list[ToolDef]:
"""Tool definitions built from rows, in the order they claim names.
Imported here rather than at the top because `custom_tools` needs `ToolDef`
from this module.
"""
from lembas.services import custom_tools
return custom_tools.tool_defs(db, user, everything=everything)
def _book(defs: list[ToolDef]) -> dict[str, ToolDef]:
"""Keyed by name, first claim winning.
The built-ins are laid down first, so a row can never shadow one -- a tool
called `notes_delete` that turns out to be somebody's HTTP endpoint is the
kind of surprise that has no good failure mode.
"""
book = dict(REGISTRY)
for tool in defs:
book.setdefault(tool.name, tool)
return book
def registry(db: DBSession) -> dict[str, ToolDef]:
"""Every tool that exists on this instance, keyed by name, ungated.
`REGISTRY` holds the built-ins alone, because it is built at import time and
an administrator-defined tool is a row. Callers that only need to map a name
back to a family use this; callers deciding what to *offer* use
`resolve_tools`, which applies the gates as well.
"""
return _book(_row_defs(db, None, everything=True))
def families(db: DBSession) -> tuple[str, ...]:
"""Every family that exists, the built-ins in their fixed order first."""
rows = tuple(tool.family for tool in _row_defs(db, None, everything=True))
return (*FAMILIES, *rows)
def resolve_tools(db: DBSession, chat: Chat, user: User | None) -> ToolSet:
"""Every tool this chat may call right now, with its runner attached."""
from lembas.security import permissions
from lembas.services import chat as chat_service
@@ -669,25 +764,47 @@ def enabled_tools(db: DBSession, chat: Chat, user: User | None) -> list[dict[str
capabilities = model.capabilities_json or {}
if not capabilities.get("tools"):
return []
return ToolSet()
allowed = permissions.resolve(db, user)
config = settings_store.search(db)
families = {
family
for family in FAMILIES
if _family_allowed(family, config=config, capabilities=capabilities, allowed=allowed)
}
return [tool.schema for tool in REGISTRY.values() if tool.family in families]
# Resolved against what this reader may see, not against everything that
# exists: a tool restricted to a group is not offered outside it.
book = _book(_row_defs(db, user))
return ToolSet(
tuple(
tool
for tool in book.values()
if _family_allowed(
tool.family, config=config, capabilities=capabilities, allowed=allowed
)
)
)
def context_for(db: DBSession, user: User | None, chat: Chat | None = None) -> ToolContext:
def enabled_tools(db: DBSession, chat: Chat, user: User | None) -> list[dict[str, Any]]:
"""The tool schemas to offer for this chat.
The shape is unchanged on purpose: the inspector and `build_request` want
exactly this. Anything that will also *run* a tool wants `resolve_tools`.
"""
return resolve_tools(db, chat, user).schemas
def context_for(
db: DBSession,
user: User | None,
chat: Chat | None = None,
*,
tools: ToolSet | None = None,
) -> ToolContext:
"""The snapshot a running tool needs, taken while the session is open."""
return ToolContext(
owner_id=user.id if user else "",
search_config=settings_store.search(db),
base_ids=[base.id for base in chat.knowledge_bases] if chat is not None else [],
tools=tools.by_name if tools is not None else None,
)
@@ -697,8 +814,15 @@ async def run_tool(context: ToolContext, name: str, arguments: str) -> ToolOutco
Never raises. A tool that fails hands the model an explanation and lets it
carry on -- a failed lookup should produce "I could not find that" rather
than killing the whole reply.
The lookup is against what was *offered*, not against everything that
exists. Reaching for the registry directly meant a model naming a tool its
chat was gated out of -- a family switched off for the model, a permission
the reader does not have -- had it run anyway, because only the offer was
ever filtered.
"""
tool = REGISTRY.get(name)
book = REGISTRY if context.tools is None else context.tools
tool = book.get(name)
if tool is None:
return ToolOutcome(
f"There is no tool called {name!r}.",
@@ -710,9 +834,12 @@ async def run_tool(context: ToolContext, name: str, arguments: str) -> ToolOutco
except json.JSONDecodeError:
# Small models emit malformed argument JSON often enough that this is a
# normal path, not an exceptional one. Treat the whole string as the
# first required argument rather than giving up.
required = tool.parameters.get("required") or ["query"]
parsed = {required[0]: arguments.strip()}
# tool's first argument rather than giving up: what it says is required,
# else the first thing it declares, and only then a guess -- a schema
# somebody else wrote need not have either.
properties = tool.parameters.get("properties") or {}
names = tool.parameters.get("required") or list(properties) or ["query"]
parsed = {str(names[0]): arguments.strip()}
if not isinstance(parsed, dict):
parsed = {"query": str(parsed)}
@@ -808,6 +935,31 @@ def tool_turn(call: dict[str, Any], content: str) -> dict[str, Any]:
}
def _row_source(db: DBSession):
"""One harness fragment per administrator-defined tool.
The seam `prompts.register_source` exists for. The row supplies the default
text and the admin page supplies the override, which is why a tool deleted
and recreated under the same slug keeps whatever wording somebody chose for
it -- the override outlives the row.
Gated on the tool's own family, so the guidance appears exactly when the
tool it describes is offered and never otherwise.
"""
from lembas.db.models import CustomTool
for row in db.scalars(select(CustomTool).order_by(CustomTool.position, CustomTool.slug)):
yield prompts_service.Fragment(
key=f"tool.custom_{row.slug}",
label=row.name or row.slug,
group=prompts_service.GROUP_TOOLS,
order=500 + row.position,
families=(f"{FAMILY_CUSTOM}:{row.slug}",),
hint=f"Appears when the {row.slug} tool is offered.",
default=row.guidance or "",
)
__all__ = [
"FAMILIES",
"MAX_ROUNDS",
@@ -816,9 +968,19 @@ __all__ = [
"ToolContext",
"ToolDef",
"ToolOutcome",
"ToolSet",
"assistant_turn",
"context_for",
"enabled_tools",
"families",
"registry",
"resolve_tools",
"run_tool",
"tool_turn",
]
# Registered at import. `services.tools` is imported by the chat routes, the
# generation service and the prompts admin, so the source is in place before
# anything renders a catalogue.
prompts_service.register_source(_row_source)