e970f10cca
The first audit pass: everything from 0.8.1 to 0.9.8 read as a whole rather than one feature at a time, starting with what a model is actually told. Four of these had shipped as correct. The date line carried a timezone variable that resolves to nothing until somebody chooses one -- so every default account was told times were "in unless they say otherwise", while two comments asserted the line disappeared instead. The prompt preview built its variables without a chat, which is what eleven fragments are gated on, so the whole agent surface was absent from it whatever was ticked. Plan mode was instructed to keep its plan current with a tool that mode withdraws. And knowledge_get returned a document whole where every sibling reader caps and says so, its description promising exactly that. The subagent guidance was wrong in both directions at once: it denied a documented parameter and named seven of twenty-three allowed commands. Both halves are pinned by tests against the real list and the real schema now, because prose and a constant drift the moment one is edited alone. docs/notes/audit-0.9.md carries the findings that are not fixed here, with why -- the ones whose fix would change what a feature does are the user's call, not this pass's. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
522 lines
20 KiB
Python
522 lines
20 KiB
Python
"""The operational preamble, and how it sits beside the authored prompt."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
from lembas.db.models import Chat, Connection, Model, User
|
|
from lembas.security.passwords import hash_password
|
|
from lembas.services import chat as chat_service
|
|
from lembas.services import harness, prompts, settings_store
|
|
from lembas.services import tools as tools_service
|
|
from lembas.services.library import memories as memories_service
|
|
from lembas.services.library import skills as skills_service
|
|
|
|
|
|
@pytest.fixture
|
|
def owner(db):
|
|
user = User(name="Frodo", email="f@shire.test", password_hash=hash_password("x"))
|
|
db.add(user)
|
|
db.commit()
|
|
return user
|
|
|
|
|
|
def _tools(*names):
|
|
return [tools_service.REGISTRY[name].schema for name in names]
|
|
|
|
|
|
# --- Composition -------------------------------------------------------------
|
|
def test_no_tools_means_no_tool_guidance(db, owner):
|
|
"""The core fragments still go out -- a model with no tools has no clock
|
|
either, and telling it the date is not "tokens that say nothing". What it
|
|
must not get is instructions about tools it was never offered."""
|
|
text = harness.compose(db, owner, [])
|
|
assert "Today is" in text
|
|
assert "You have tools" not in text
|
|
assert "Look things up" not in text
|
|
assert harness.compose(db, owner, None) == text
|
|
|
|
|
|
def test_the_date_line_names_a_zone_when_nobody_has_chosen_one(db, owner):
|
|
"""The shipped default, on the shipped configuration.
|
|
|
|
`clock.name_for` returns "" for an account that has never chosen a zone,
|
|
which is every account until somebody visits the settings page. The variable
|
|
sits *inside* a sentence, and `substitute` drops a line only when the whole
|
|
line is blank after expansion -- so an unresolved zone was never a dropped
|
|
line, it was "in unless", shipped on every request. Both the comment in
|
|
`harness.py` and the fragment's own hint claimed otherwise, and no test
|
|
asked.
|
|
"""
|
|
assert not (owner.settings_json or {}).get("timezone")
|
|
text = harness.compose(db, owner, [])
|
|
|
|
line = next(ln for ln in text.splitlines() if "Times the person gives you" in ln)
|
|
assert "in unless" not in line
|
|
assert " " not in line
|
|
# Whatever the host is set to, it has to be *something*.
|
|
zone = line.split(" are in ", 1)[1].split(" unless", 1)[0]
|
|
assert zone.strip()
|
|
|
|
|
|
def test_the_date_line_uses_the_readers_own_zone_when_they_have_one(db, owner):
|
|
"""And a chosen zone is preferred to the server's, which is the whole point
|
|
of the variable -- the fallback must not have flattened it."""
|
|
owner.settings_json = {**(owner.settings_json or {}), "timezone": "Pacific/Auckland"}
|
|
db.commit()
|
|
text = harness.compose(db, owner, [])
|
|
assert "are in Pacific/Auckland unless" in text
|
|
|
|
|
|
def test_clearing_the_core_fragments_restores_an_empty_harness(db, owner):
|
|
"""The behaviour change is a default, not a rule: an administrator who wants
|
|
nothing sent to a tool-less model can still have exactly that."""
|
|
prompts.save(
|
|
db,
|
|
{f.key: "" for f in prompts.BUILTIN if f.group == prompts.GROUP_CORE},
|
|
)
|
|
assert harness.compose(db, owner, []) == ""
|
|
|
|
|
|
def test_only_the_guidance_for_offered_tools_appears(db, owner):
|
|
text = harness.compose(db, owner, _tools("web_search"))
|
|
assert "Look things up" in text
|
|
assert "You keep notes" not in text
|
|
assert "Skills are procedures" not in text
|
|
|
|
|
|
def test_a_custom_tools_guidance_appears_only_when_it_is_offered(db, owner):
|
|
"""The row supplies the default, and the fragment is gated on the tool's own
|
|
family -- which is why the registry had to stop being a module constant."""
|
|
from lembas.db.models import CustomTool
|
|
|
|
db.add(
|
|
CustomTool(
|
|
slug="weather",
|
|
name="Weather",
|
|
description="Look up the weather.",
|
|
url_template="https://api.test/{{city}}",
|
|
guidance="- Check the weather rather than guessing at it.",
|
|
)
|
|
)
|
|
db.commit()
|
|
|
|
offered = tools_service.registry(db)["weather"].schema
|
|
assert "Check the weather" in harness.compose(db, owner, [offered])
|
|
assert "Check the weather" not in harness.compose(db, owner, _tools("web_search"))
|
|
|
|
|
|
def test_the_memory_block_is_included_when_memory_is_offered(db, owner):
|
|
memories_service.add(db, owner=owner, content="Prefers metric units.")
|
|
text = harness.compose(db, owner, _tools("memory_add"))
|
|
assert "What you know about this person" in text
|
|
assert "Prefers metric units." in text
|
|
|
|
|
|
def test_memories_are_absent_without_the_memory_tool(db, owner):
|
|
"""A model not given the memory tool has no business being told them."""
|
|
memories_service.add(db, owner=owner, content="Prefers metric units.")
|
|
text = harness.compose(db, owner, _tools("web_search"))
|
|
assert "Prefers metric units." not in text
|
|
|
|
|
|
def test_the_skill_index_is_names_and_descriptions_only(db, owner):
|
|
skills_service.create(
|
|
db, owner=owner, name="weekly-report", description="When asked.", body="SECRET"
|
|
)
|
|
text = harness.compose(db, owner, _tools("skill_get"))
|
|
assert "weekly-report: When asked." in text
|
|
assert "SECRET" not in text
|
|
|
|
|
|
def test_an_empty_store_contributes_no_heading(db, owner):
|
|
"""And the guidance must not point at a heading that is not there: telling a
|
|
model to consult an absent section is a good way to make it invent one."""
|
|
text = harness.compose(db, owner, _tools("memory_add", "skill_get"))
|
|
assert "### What you know about this person" not in text
|
|
assert "### Skills available" not in text
|
|
assert "was remembered earlier" not in text
|
|
assert "You can remember durable facts" in text
|
|
|
|
|
|
def test_the_harness_is_capped(db, owner, monkeypatch):
|
|
monkeypatch.setattr(harness, "MAX_HARNESS_CHARS", 200)
|
|
for index in range(50):
|
|
skills_service.create(
|
|
db, owner=owner, name=f"skill-{index}", description="x" * 200, body="y"
|
|
)
|
|
assert len(harness.compose(db, owner, _tools("skill_get"))) <= 202
|
|
|
|
|
|
# --- Joining -----------------------------------------------------------------
|
|
def test_the_authored_prompt_comes_last():
|
|
"""It is closest to the conversation, and it is what the user actually
|
|
wrote."""
|
|
joined = harness.join("HARNESS", "AUTHORED")
|
|
assert joined.index("HARNESS") < joined.index("AUTHORED")
|
|
|
|
|
|
def test_either_half_alone_is_returned_unchanged():
|
|
assert harness.join("", "AUTHORED") == "AUTHORED"
|
|
assert harness.join("HARNESS", "") == "HARNESS"
|
|
assert harness.join("", "") == ""
|
|
|
|
|
|
# --- Through build_request ---------------------------------------------------
|
|
def _chat(db, owner, *, capabilities, model_prompt="", chat_prompt=""):
|
|
connection = Connection(name="c", base_url="http://h", api_key_encrypted="")
|
|
db.add(connection)
|
|
db.commit()
|
|
db.add(
|
|
Model(
|
|
connection_id=connection.id,
|
|
model_id="m",
|
|
capabilities_json=capabilities,
|
|
system_prompt=model_prompt,
|
|
)
|
|
)
|
|
db.commit()
|
|
chat = Chat(
|
|
user_id=owner.id, model_id="m", connection_id=connection.id, system_prompt=chat_prompt
|
|
)
|
|
db.add(chat)
|
|
db.commit()
|
|
return chat
|
|
|
|
|
|
def _system(body):
|
|
first = body["messages"][0] if body["messages"] else {}
|
|
return first.get("content", "") if first.get("role") == "system" else ""
|
|
|
|
|
|
def test_a_request_without_tools_carries_no_tool_guidance_and_no_tools_key(db, owner):
|
|
chat = _chat(db, owner, capabilities={})
|
|
body = chat_service.build_request(db, chat, tools=[], user=owner)
|
|
assert "tools" not in body
|
|
assert "You have tools" not in _system(body)
|
|
|
|
|
|
def test_the_harness_precedes_the_authored_prompt(db, owner):
|
|
settings_store.update(db, {"enabled": True}, key=settings_store.SEARCH)
|
|
chat = _chat(db, owner, capabilities={"tools": True}, chat_prompt="Speak as Gandalf.")
|
|
offered = tools_service.enabled_tools(db, chat, owner)
|
|
|
|
system = _system(chat_service.build_request(db, chat, tools=offered, user=owner))
|
|
assert system.index("How to work") < system.index("Speak as Gandalf.")
|
|
|
|
|
|
def test_precedence_between_the_authored_layers_is_untouched(db, owner):
|
|
"""The harness is a different axis. Exactly one authored layer still wins,
|
|
and it is still the most specific one."""
|
|
settings_store.update(db, {"system_prompt": "Instance."})
|
|
settings_store.update(db, {"enabled": True}, key=settings_store.SEARCH)
|
|
|
|
chat = _chat(
|
|
db, owner, capabilities={"tools": True}, model_prompt="Model.", chat_prompt="Chat."
|
|
)
|
|
offered = tools_service.enabled_tools(db, chat, owner)
|
|
system = _system(chat_service.build_request(db, chat, tools=offered, user=owner))
|
|
|
|
assert "Chat." in system
|
|
assert "Model." not in system
|
|
assert "Instance." not in system
|
|
# And the resolver on its own is unchanged.
|
|
assert chat_service.effective_system_prompt(db, chat) == "Chat."
|
|
|
|
|
|
def test_a_model_prompt_wins_when_the_chat_has_none(db, owner):
|
|
settings_store.update(db, {"system_prompt": "Instance."})
|
|
chat = _chat(db, owner, capabilities={}, model_prompt="Model.")
|
|
system = _system(chat_service.build_request(db, chat, user=owner))
|
|
assert system.endswith("Model.")
|
|
assert "Instance." not in system
|
|
assert chat_service.effective_system_prompt(db, chat) == "Model."
|
|
|
|
|
|
def test_the_seam_line_only_appears_when_there_is_something_to_hand_over_to(db, owner):
|
|
"""It introduces the authored prompt. With no authored prompt it would be
|
|
pointing at nothing, which is the failure the whole `requires` idea exists
|
|
to avoid."""
|
|
bare = _chat(db, owner, capabilities={})
|
|
assert "was written by whoever set up" not in _system(
|
|
chat_service.build_request(db, bare, user=owner)
|
|
)
|
|
|
|
authored = _chat(db, owner, capabilities={}, chat_prompt="Speak as Gandalf.")
|
|
assert "was written by whoever set up" in _system(
|
|
chat_service.build_request(db, authored, user=owner)
|
|
)
|
|
|
|
|
|
def test_an_administrators_wording_replaces_the_default(db, owner):
|
|
prompts.save(db, {"core.today": "The date is {{today}}, more or less."})
|
|
text = harness.compose(db, owner, [])
|
|
assert "more or less." in text
|
|
assert "Your training data stops well before this" not in text
|
|
|
|
|
|
def test_the_attached_files_are_named_and_explained(db, owner):
|
|
from lembas.db.models import Attachment
|
|
|
|
chat = _chat(db, owner, capabilities={})
|
|
db.add(
|
|
Attachment(
|
|
user_id=owner.id,
|
|
chat_id=chat.id,
|
|
filename="report.txt",
|
|
stored_name="x.txt",
|
|
media_type="text/plain",
|
|
size_bytes=10,
|
|
kind="text",
|
|
)
|
|
)
|
|
db.commit()
|
|
|
|
text = harness.compose(db, owner, [], chat)
|
|
assert "report.txt" in text
|
|
assert "<document name=" in text
|
|
|
|
|
|
def test_a_chat_with_no_attachments_says_nothing_about_documents(db, owner):
|
|
chat = _chat(db, owner, capabilities={})
|
|
assert "<document name=" not in harness.compose(db, owner, [], chat)
|
|
|
|
|
|
def test_the_tools_array_rides_along(db, owner):
|
|
settings_store.update(db, {"enabled": True}, key=settings_store.SEARCH)
|
|
chat = _chat(db, owner, capabilities={"tools": True})
|
|
offered = tools_service.enabled_tools(db, chat, owner)
|
|
body = chat_service.build_request(db, chat, tools=offered, user=owner)
|
|
assert body["tools"] == offered
|
|
|
|
|
|
# --- The project listing -----------------------------------------------------
|
|
# Injected from a cache that something else fills, because this module runs
|
|
# synchronously on the request path and an SFTP round trip here would hold a
|
|
# request open while somebody's machine thought about it.
|
|
def _agent_chat(db, owner):
|
|
from lembas.db.models import KIND_AGENT, SshProfile
|
|
|
|
settings_store.update(db, {"enabled": True}, key=settings_store.AGENTS)
|
|
profile = SshProfile(
|
|
owner_id=owner.id,
|
|
name="Test box",
|
|
host="127.0.0.1",
|
|
port=22,
|
|
username="tester",
|
|
host_key="host key",
|
|
host_fingerprint="SHA256:x",
|
|
default_dir="/work",
|
|
)
|
|
db.add(profile)
|
|
db.commit()
|
|
chat = Chat(
|
|
user_id=owner.id,
|
|
kind=KIND_AGENT,
|
|
ssh_profile_id=profile.id,
|
|
project_dir="/work",
|
|
)
|
|
db.add(chat)
|
|
db.commit()
|
|
return chat, profile
|
|
|
|
|
|
def _agent_tools(db):
|
|
"""The agent tools as offered, which REGISTRY does not carry.
|
|
|
|
`REGISTRY` is built at import time and holds the built-ins alone; the agent
|
|
tools are listed by `registry(db)`, unbound to any chat. That is the same
|
|
lookup the harness does to map `shell_run` back to the `agent` family, and
|
|
the reason it exists at all.
|
|
"""
|
|
book = tools_service.registry(db)
|
|
return [book["shell_run"].schema]
|
|
|
|
|
|
def _cache(profile_id, paths):
|
|
import time
|
|
|
|
from lembas.services.agent import index as index_service
|
|
|
|
index_service._CACHE[(profile_id, "/work")] = index_service.ProjectIndex(
|
|
paths=tuple(paths), total=len(paths), source="git", built_at=time.monotonic()
|
|
)
|
|
|
|
|
|
def test_the_project_listing_reaches_the_model(db, owner):
|
|
chat, profile = _agent_chat(db, owner)
|
|
_cache(profile.id, ["README.md", "src/main.py"])
|
|
|
|
text = harness.compose(db, owner, _agent_tools(db), chat=chat)
|
|
|
|
assert "Files in /work" in text
|
|
assert "README.md" in text
|
|
|
|
|
|
def test_nothing_cached_means_no_section_at_all(db, owner):
|
|
"""Not an empty heading. `Fragment.requires` makes the whole thing vanish,
|
|
which is what lets the first reply in a new chat outrun the first walk
|
|
without saying anything strange."""
|
|
chat, _profile = _agent_chat(db, owner)
|
|
|
|
text = harness.compose(db, owner, _agent_tools(db), chat=chat)
|
|
|
|
assert "Files in" not in text
|
|
|
|
|
|
def test_a_budget_of_zero_keeps_the_listing_out_of_the_prompt(db, owner):
|
|
"""The listing is still built and the file picker still uses it. This is
|
|
the only way to say "index it, but do not spend context on it"."""
|
|
chat, profile = _agent_chat(db, owner)
|
|
_cache(profile.id, ["README.md"])
|
|
settings_store.update(db, {"index_chars": 0}, key=settings_store.AGENTS)
|
|
|
|
assert "Files in" not in harness.compose(db, owner, _agent_tools(db), chat=chat)
|
|
|
|
|
|
def test_switching_the_listing_off_keeps_it_out(db, owner):
|
|
chat, profile = _agent_chat(db, owner)
|
|
_cache(profile.id, ["README.md"])
|
|
settings_store.update(db, {"index_enabled": False}, key=settings_store.AGENTS)
|
|
|
|
assert "Files in" not in harness.compose(db, owner, _agent_tools(db), chat=chat)
|
|
|
|
|
|
def test_a_plain_chat_is_told_nothing_about_files(db, owner):
|
|
chat = Chat(user_id=owner.id)
|
|
db.add(chat)
|
|
db.commit()
|
|
|
|
assert "Files in" not in harness.compose(db, owner, _tools("web_search"), chat=chat)
|
|
|
|
|
|
# --- One round, or as many as it takes ----------------------------------------
|
|
def test_a_plain_chat_is_told_its_real_ceiling(db, owner):
|
|
"""The number it is actually given, not a constant -- a model told it has
|
|
five rounds and cut off after three has been lied to about its own budget."""
|
|
settings_store.update(db, {"max_chat_rounds": 3})
|
|
text = harness.compose(db, owner, _tools("web_search"))
|
|
|
|
assert "at most 3 rounds" in text
|
|
assert "Keep working until the task is actually done" not in text
|
|
|
|
|
|
def test_no_ceiling_means_no_round_budget_is_claimed(db, owner):
|
|
"""Zero is "no ceiling", and a fragment promising zero rounds would be worse
|
|
than none at all."""
|
|
settings_store.update(db, {"max_chat_rounds": 0})
|
|
text = harness.compose(db, owner, _tools("web_search"))
|
|
|
|
# `core.interjection` also mentions rounds, so this asserts on the budget
|
|
# sentence rather than on the word.
|
|
assert "at most" not in text
|
|
|
|
|
|
def test_an_agent_chat_is_told_to_keep_going_instead(db, owner):
|
|
"""The two cannot be one fragment with a number in it. A model told it has
|
|
a budget rations it; the step count is a runaway backstop, and rationing
|
|
against it is exactly the behaviour that stops a long piece of work
|
|
halfway."""
|
|
chat, _profile = _agent_chat(db, owner)
|
|
|
|
text = harness.compose(db, owner, _agent_tools(db), chat=chat)
|
|
|
|
assert "Keep working until the task is actually done" in text
|
|
assert "one round of tool calls" not in text
|
|
|
|
|
|
def test_the_fetch_guidance_appears_only_with_the_tool(db, owner):
|
|
assert "read one web page at a time" in harness.compose(db, owner, _tools("fetch"))
|
|
assert "read one web page at a time" not in harness.compose(db, owner, _tools("web_search"))
|
|
|
|
|
|
def test_the_background_guidance_appears_only_when_enabled(db, owner):
|
|
"""Gated on the feature, so an agent chat without background commands is not
|
|
told about a tool it does not have."""
|
|
from lembas.db.models import Chat, Connection, Model, SshProfile
|
|
|
|
owner.role = "admin" # agent tools need tools.agent, which admins pass
|
|
db.commit()
|
|
profile = SshProfile(
|
|
owner_id=owner.id, name="Box", host="127.0.0.1", port=22, username="t",
|
|
host_key="k", host_fingerprint="f", default_dir="/work",
|
|
)
|
|
connection = Connection(name="cbg", base_url="http://127.0.0.1:1", api_key_encrypted="")
|
|
db.add_all([profile, connection])
|
|
db.commit()
|
|
db.add(Model(connection_id=connection.id, model_id="mbg", capabilities_json={"tools": True}))
|
|
db.commit()
|
|
settings_store.update(db, {"enabled": True}, key=settings_store.AGENTS)
|
|
chat = Chat(
|
|
user_id=owner.id, model_id="mbg", connection_id=connection.id, kind="agent",
|
|
ssh_profile_id=profile.id, project_dir="/work",
|
|
)
|
|
db.add(chat)
|
|
db.commit()
|
|
|
|
def _text():
|
|
offered = tools_service.resolve_tools(db, chat, owner).schemas
|
|
return harness.compose(db, owner, offered, chat)
|
|
|
|
settings_store.update(db, {"background_enabled": False}, key=settings_store.AGENTS)
|
|
assert "run in the background" not in _text()
|
|
|
|
settings_store.update(db, {"background_enabled": True}, key=settings_store.AGENTS)
|
|
assert "run in the background" in _text()
|
|
|
|
|
|
# --- The ceiling has to fit what the defaults already grant --------------------
|
|
def test_the_shipped_defaults_fit_under_the_ceiling(db, owner):
|
|
"""An agent chat's whole preamble, at the budgets this ships with.
|
|
|
|
It did not fit. The fragments alone are about 7,900 characters, and on top
|
|
of them `index_chars` grants a 2,000 character project listing and
|
|
`instructions_chars` a 4,000 character AGENTS.md -- both on by default. The
|
|
ceiling was 8,000, and `assemble` cuts the tail, which by fragment order is
|
|
the context worth having: the listing was severed mid-tree and
|
|
`context.agent_instructions` was dropped whole. So on a default install the
|
|
one path by which a project's own instructions reach a model did not.
|
|
"""
|
|
values = settings_store.agents(db)
|
|
# Every name the harness knows about, so a variable added later is covered
|
|
# here without anybody remembering to add it.
|
|
variables = dict.fromkeys(harness.context_variables(db, owner, [], None), "")
|
|
variables.update(
|
|
{
|
|
"today": "Monday 3 August 2026",
|
|
"instance_name": "LLeMbas",
|
|
"user_name": "Frodo",
|
|
"agent_target": "homeserver",
|
|
"agent_dir": "/srv/project",
|
|
"agent_mode": "You are in **Edit** mode.",
|
|
"tool_names": "shell_run, file_read, file_write, file_edit, file_list",
|
|
"background": "on",
|
|
"max_rounds": "200",
|
|
# Each at exactly the budget its own setting allows.
|
|
"project_files": "L" * int(values["index_chars"]),
|
|
"agent_instructions": "A" * int(values["instructions_chars"]),
|
|
"agent_instructions_file": "AGENTS.md",
|
|
"plan": "P" * 600,
|
|
"memories": "M" * 400,
|
|
}
|
|
)
|
|
|
|
out = harness.compose_from(
|
|
db, variables=variables, families=["agent"], has_tools=True, overrides={}
|
|
)
|
|
|
|
assert not out.endswith("…"), f"the preamble was truncated at {len(out):,} characters"
|
|
assert "A" * 100 in out, "the project's own instructions were cut off entirely"
|
|
assert "L" * 100 in out, "the project listing was cut off"
|
|
|
|
# Fitting is not enough. It fitted with 1,300 characters to spare once, and
|
|
# a ceiling that close to the content is one the next fragment crosses --
|
|
# silently, and by cutting the tail, which is the project's own AGENTS.md.
|
|
# The margin is also what an administrator's own wording goes in: an
|
|
# override is usually longer than the default it replaces.
|
|
room = harness.MAX_HARNESS_CHARS - len(out)
|
|
assert room >= harness.MAX_HARNESS_CHARS * harness.HARNESS_MARGIN, (
|
|
f"only {room:,} characters of headroom left under "
|
|
f"{harness.MAX_HARNESS_CHARS:,}; raise the ceiling or shorten a fragment"
|
|
)
|