Files
LLeMbas/tests/test_agent_jobs.py
T
Jaroslav Beneš a9aa89b2c1 Wake the model when a background job finishes
The other half of background execution: a job that finishes while nobody is
looking prompts the model back with its result, rather than sitting unread until
the model happens to run again.

The vehicle is the queue, because it is the only wiring that already delivers a
turn into or after a reply. A per-job poller notices completion and calls
jobs.wake. If a reply is being written the completion is left queued for that
reply's _inject/_drain; if the chat is idle a fresh reply is started to answer
it -- the send_queued_now move. All of it under a per-chat lock with no await
between the running-check and ensure, so two jobs finishing at once cannot each
spin up a generation: the second sees the first's reply already live and leaves
its completion for it. That is the invariant the queue exists to hold, reached
from outside a request for the first time.

The completion is a user-role turn whose content names itself a machine event --
"A background job you started has finished" -- not a bare person turn. _inject
sends a queued turn verbatim, so the framing cannot live there; it lives in the
words, the way execute_plan quotes the plan, and a tool.background fragment tells
the model these arrive and are a machine event rather than the person speaking.

The poller reconnects a fresh connection each tick rather than holding one open
-- holding one is the exact live-connection state the whole ssh.py/base.py design
forbids, and poll is self-healing besides. Bounded by background_max_jobs and a
six-hour ceiling, after which the remote job may keep running but we stop
watching it.

A Job table, and here the terminal/generation "lost on restart" precedent does
NOT transfer: those are seconds long with a human watching, a background job is
hours long with nobody watching -- the one case a restart forgetting it would
silently break the feature's whole promise. So the row lets a lifespan startup
hook rehydrate the watcher and wake as if nothing happened. Cancelling a watcher
never stops the detached remote job; it runs on and is picked back up.

Tested end to end against a real local shell: launch a detached command, poll it
to completion through a watcher, and assert the model was woken with the exit
code and output -- plus the lock proving two simultaneous completions start one
reply, not two.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-03 14:30:44 +02:00

415 lines
15 KiB
Python

"""Background jobs, exercised against a real local shell.
The sshd fixture elsewhere is a *fake* shell (it echoes the command), which
cannot run `setsid`, `base64`, a wait loop, or an exit-file. So these run the
wrappers through the machine's own `/bin/sh` with the same contract
`SshExecutor.run` has — cd-prefix, interleaved output, a timeout that leaves the
detached child alive — because the shell logic is the whole of the risk here.
"""
from __future__ import annotations
import asyncio
import shutil
import time
import pytest
from lembas.services import tools as tools_service
from lembas.services.agent import jobs, policy
from lembas.services.agent import ssh as ssh_service
from lembas.services.agent.base import ExecRequest, ExecResult, clean_output
from lembas.services.agent.session import AgentContext
from lembas.services.agent.tools import _run_shell
pytestmark = pytest.mark.skipif(
shutil.which("setsid") is None or shutil.which("base64") is None,
reason="needs setsid and base64 (Linux)",
)
class LocalExecutor:
"""`SshExecutor.run`'s contract, run against the local shell.
On timeout it kills only the outer shell -- exactly what an SSH channel
teardown does -- so a `setsid`-detached child survives, which is the whole
behaviour a background job depends on.
"""
def __init__(self, project_dir: str) -> None:
self.project_dir = project_dir
async def run(self, request: ExecRequest) -> ExecResult:
directory = request.cwd or self.project_dir
command = request.command
if directory:
command = f"cd {ssh_service._quote(directory)} && {command}"
started = time.monotonic()
proc = await asyncio.create_subprocess_shell(
command,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.STDOUT,
stdin=asyncio.subprocess.DEVNULL,
)
try:
out, _ = await asyncio.wait_for(proc.communicate(), timeout=request.timeout)
except TimeoutError:
proc.kill()
await proc.wait()
return ExecResult(
exit_status=-1,
output=f"The command was still running after {request.timeout:g}s.",
timed_out=True,
duration_ms=int((time.monotonic() - started) * 1000),
)
output, truncated = clean_output(out.decode("utf-8", "replace"), limit=request.max_bytes)
return ExecResult(
exit_status=proc.returncode if proc.returncode is not None else -1,
output=output,
truncated=truncated,
duration_ms=int((time.monotonic() - started) * 1000),
)
@pytest.fixture(autouse=True)
def _clear_jobs():
jobs.clear()
yield
jobs.clear()
def _agent(tmp_path, **over) -> AgentContext:
fields = {
"chat_id": "chat0001",
"label": "Box",
"project_dir": str(tmp_path),
"mode": policy.MODE_AUTO,
"max_output": 65536,
"timeout": 60.0,
"max_timeout": 600.0,
"background": True,
"background_on_timeout": True,
"background_notify": True,
}
fields.update(over)
ctx = AgentContext(**fields)
ctx.executor = lambda: LocalExecutor(fields["project_dir"]) # type: ignore[method-assign]
return ctx
def _context(agent):
return tools_service.ToolContext(owner_id="u", agent=agent)
# --- The wrappers, as strings --------------------------------------------------
def test_the_command_is_never_in_a_quoted_context():
"""git commit -m 'fix' would shatter sh -c '<cmd>'. It is base64'd instead."""
import base64
cmd = "git commit -m 'fix: it'"
wrapper = jobs.launch_and_wait_command("chatx", "abc123abc123", cmd, 4096)
assert cmd not in wrapper, "the command leaked into the wrapper as shell text"
blob = wrapper.split("printf %s '", 1)[1].split("'", 1)[0]
inner = base64.b64decode(blob).decode()
assert cmd in inner, "the command was lost in the base64 round-trip"
def test_parse_reads_the_last_sentinel():
out = ("line one\n__LEMBAS_jobjobjob01__:0\nmore\n__LEMBAS_jobjobjob01__:0")
done = jobs.parse_completed(out, "jobjobjob01")
assert done.exit_status == 0
def test_valid_id_refuses_a_path():
assert jobs.valid_id("deadbeef0000")
assert not jobs.valid_id("../../etc/passwd")
assert not jobs.valid_id("")
# --- Real shell: launch-and-wait -----------------------------------------------
async def test_a_fast_command_completes_like_a_foreground_one(tmp_path):
agent = _agent(tmp_path)
outcome = await _run_shell(_context(agent), {"command": "echo hello"})
assert "hello" in outcome.content
assert outcome.event["status"] == "ok"
assert not jobs.for_chat(agent.chat_id), "a finished command is not a job"
async def test_a_nonzero_exit_is_read_from_the_exit_file_not_the_wrapper(tmp_path):
"""The wrapper's own status is ~0 from its trailing rm; the command's real
status is in the exit-file."""
agent = _agent(tmp_path)
outcome = await _run_shell(_context(agent), {"command": "sh -c 'exit 3'"})
assert "exited 3" in outcome.content
assert outcome.event["status"] == "error"
async def test_a_command_with_single_quotes_runs(tmp_path):
agent = _agent(tmp_path)
outcome = await _run_shell(_context(agent), {"command": "echo 'a'\\''b'"})
assert "a'b" in outcome.content
async def test_output_is_captured(tmp_path):
agent = _agent(tmp_path)
outcome = await _run_shell(_context(agent), {"command": "printf 'one\\ntwo\\n'"})
assert "one" in outcome.content and "two" in outcome.content
# --- Real shell: auto-background on timeout -------------------------------------
async def test_a_slow_command_is_kept_running_when_it_times_out(tmp_path):
"""The apt-install case. It is not killed; it becomes a job that finishes
on its own, and its result is readable afterwards."""
agent = _agent(tmp_path)
outcome = await _run_shell(
_context(agent), {"command": "sleep 2; echo done-late", "timeout": 1}
)
assert "background" in outcome.content.lower()
running = jobs.for_chat(agent.chat_id)
assert len(running) == 1
job_id = running[0].id
# The detached child survived the wait being cut off; give it time to finish.
for _ in range(40):
reading = await jobs.read(agent, job_id)
if reading.status == "done":
break
await asyncio.sleep(0.2)
assert reading.status == "done"
assert "done-late" in reading.body
async def test_off_timeout_keeps_the_hard_stop(tmp_path):
"""With the auto-convert off, a timed-out command is killed as before and no
job is left running."""
agent = _agent(tmp_path, background_on_timeout=False)
outcome = await _run_shell(
_context(agent), {"command": "sleep 2; echo late", "timeout": 1}
)
assert "stopped after" in outcome.content
assert not jobs.for_chat(agent.chat_id)
async def test_feature_off_is_the_plain_path(tmp_path):
"""No wrapper, no job, byte-for-byte the old wording."""
agent = _agent(tmp_path, background=False)
outcome = await _run_shell(_context(agent), {"command": "echo hi"})
assert "hi" in outcome.content
assert not jobs.for_chat(agent.chat_id)
# --- Explicit background=true --------------------------------------------------
async def test_background_true_returns_at_once_with_a_job(tmp_path):
agent = _agent(tmp_path)
started = time.monotonic()
outcome = await _run_shell(
_context(agent), {"command": "sleep 5", "background": True}
)
assert time.monotonic() - started < 3, "it should not wait for the command"
running = jobs.for_chat(agent.chat_id)
assert len(running) == 1
assert "job " in outcome.content
await jobs.stop(agent, running[0].id)
# --- The job tools -------------------------------------------------------------
async def test_job_output_reads_a_running_job(tmp_path):
from lembas.services.agent.tools import _run_job_output
agent = _agent(tmp_path)
job = await jobs.launch(agent, "sleep 5")
outcome = await _run_job_output(_context(agent), {"id": job.id})
assert "running" in outcome.content.lower()
await jobs.stop(agent, job.id)
async def test_job_stop_ends_it(tmp_path):
from lembas.services.agent.tools import _run_job_output, _run_job_stop
agent = _agent(tmp_path)
job = await jobs.launch(agent, "sleep 30")
stopped = await _run_job_stop(_context(agent), {"id": job.id})
assert "Stopped" in stopped.content
reading = await _run_job_output(_context(agent), {"id": job.id})
assert "running" not in reading.content.lower()
async def test_job_output_refuses_a_bad_id(tmp_path):
from lembas.services.agent.tools import _run_job_output
agent = _agent(tmp_path)
outcome = await _run_job_output(_context(agent), {"id": "../../etc/passwd"})
assert outcome.event["status"] == "error"
async def test_a_job_is_only_visible_to_its_own_chat(tmp_path):
"""for_chat is keyed on chat id; the file paths are too, so a model in
another chat cannot even name it."""
agent_a = _agent(tmp_path, chat_id="chataaaa")
await jobs.launch(agent_a, "sleep 5")
assert jobs.for_chat("chatbbbb") == []
for job in jobs.for_chat("chataaaa"):
await jobs.stop(agent_a, job.id)
# --- Gating --------------------------------------------------------------------
def test_job_tools_appear_only_when_enabled(tmp_path):
from lembas.services.agent.tools import tool_defs
on = {t.name for t in tool_defs(_agent(tmp_path, background=True))}
off = {t.name for t in tool_defs(_agent(tmp_path, background=False))}
assert {"job_output", "job_list", "job_stop"} <= on
assert not ({"job_output", "job_list", "job_stop"} & off)
def test_background_param_appears_only_when_enabled(tmp_path):
from lembas.services.agent.tools import tool_defs
def shell(agent):
return next(t for t in tool_defs(agent) if t.name == "shell_run")
on = shell(_agent(tmp_path, background=True)).parameters["properties"]
off = shell(_agent(tmp_path, background=False)).parameters["properties"]
assert "background" in on
assert "background" not in off
# --- Waking the model when a job finishes --------------------------------------
def _chat(db, user_id) -> str:
from lembas.db.models import Chat
chat = Chat(user_id=user_id, model_id="m")
db.add(chat)
db.commit()
return chat.id
def test_the_completion_names_itself_a_machine_event():
text = jobs._completion_text("abc123abc123", "apt-get install -y x", "done", 0, "ok\n")
assert "machine event" in text
assert "[job abc123abc123]" in text
assert "apt-get install" in text
assert "finished successfully" in text.lower()
def test_a_nonzero_completion_reports_the_code():
text = jobs._completion_text("j", "build", "done", 2, "")
assert "exited 2" in text
def test_backticks_in_output_cannot_close_the_fence():
text = jobs._completion_text("j", "c", "done", 0, "see ```code``` here")
assert "```code```" not in text
async def test_an_idle_chat_gets_a_fresh_reply(db, user_id, registered, monkeypatch):
from lembas.services import generation as generation_service
chat_id = _chat(db, user_id)
monkeypatch.setattr(generation_service, "running_for", lambda _c: None)
started: list[str] = []
monkeypatch.setattr(generation_service, "ensure", lambda c, m: started.append(m))
await jobs.wake(chat_id, "abc123abc123", "sleep 1", "done", 0, "hi")
from lembas.db.models import Message
db.expire_all()
msgs = db.query(Message).filter(Message.chat_id == chat_id).all()
users = [m for m in msgs if m.role == "user"]
assert len(users) == 1
assert users[0].queued is False, "an idle chat's completion is delivered, not queued"
assert any(m.role == "assistant" and not m.complete for m in msgs)
assert len(started) == 1, "a reply was started"
async def test_a_busy_chat_gets_a_queued_turn_and_no_new_reply(
db, user_id, registered, monkeypatch
):
from lembas.services import generation as generation_service
chat_id = _chat(db, user_id)
monkeypatch.setattr(generation_service, "running_for", lambda _c: object())
started: list[str] = []
monkeypatch.setattr(generation_service, "ensure", lambda c, m: started.append(m))
await jobs.wake(chat_id, "abc123abc123", "sleep 1", "done", 0, "hi")
from lembas.db.models import Message
db.expire_all()
users = [
m for m in db.query(Message).filter(Message.chat_id == chat_id).all() if m.role == "user"
]
assert len(users) == 1
assert users[0].queued is True, "a running reply's _inject/_drain will deliver it"
assert started == [], "no second reply for a chat already writing one"
async def test_two_jobs_finishing_at_once_start_one_reply(db, user_id, registered, monkeypatch):
"""The per-chat lock. Without it, both wakes would each start a generation."""
from lembas.services import generation as generation_service
chat_id = _chat(db, user_id)
live = {"running": False}
monkeypatch.setattr(
generation_service, "running_for", lambda _c: object() if live["running"] else None
)
started: list[str] = []
def _ensure(c, m):
live["running"] = True # what running_for will now see
started.append(m)
monkeypatch.setattr(generation_service, "ensure", _ensure)
await asyncio.gather(
jobs.wake(chat_id, "aaaaaaaaaaaa", "a", "done", 0, "x"),
jobs.wake(chat_id, "bbbbbbbbbbbb", "b", "done", 0, "y"),
)
assert len(started) == 1, "the lock made the second wake see the first's reply"
# --- The watcher, end to end ---------------------------------------------------
async def test_the_watcher_end_to_end(db, user_id, registered, monkeypatch, tmp_path):
from lembas.services.agent import ssh as ssh_service
chat_id = _chat(db, user_id)
class _Local(LocalExecutor):
def __init__(self, spec, project_dir):
super().__init__(str(tmp_path))
monkeypatch.setattr(ssh_service, "SshExecutor", _Local)
monkeypatch.setattr(jobs, "_poll_interval", lambda _e: 0.1)
woken: list = []
async def _wake(chat, job_id, command, status, code, body):
woken.append((status, code, body))
monkeypatch.setattr(jobs, "wake", _wake)
agent = _agent(tmp_path, chat_id=chat_id)
agent.executor = lambda: _Local(None, str(tmp_path))
job = await jobs.launch(agent, "sleep 0.5; echo finished")
jobs.start_watch(agent, job)
for _ in range(50):
if woken:
break
await asyncio.sleep(0.1)
assert woken, "the watcher never woke the model"
status, code, body = woken[0]
assert status == "done"
assert "finished" in body