LLeMbas CLI 1.0.0
ci / check (push) Waiting to run

The first public release of LLeMbas CLI: a terminal coding agent and project manager for any LLM
API, with permission modes, git snapshots, memory and skills, knowledge bases, MCP, voice, and a
link to a LLeMbas instance whose web UI can work its sessions too. Signed Linux binaries for x64
and arm64.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
HomerandClaude Opus 5.5 committed 2026-10-09 21:59:03 +00:00
commit f9bad01ed7
355 files changed
+47028

No files matched your search

+160
View File
@@ -0,0 +1,160 @@
// What every dialect shares: the retry rule, the part builder, the stream recorder, and
// llama-swap's loading banner.
import type { ReasoningPart, StreamEvent, TextPart, ToolCallPart, Message } from "./types.ts"
import { ProviderError } from "./types.ts"
import { env } from "../config/paths.ts"
type Assistant = Extract<Message, { role: "assistant" }>
/** Builds an assistant message as deltas arrive, and knows whether anything real has been said —
* the line after which a retry would duplicate output. */
export class Parts {
readonly parts: Assistant["parts"] = []
output = false
add(kind: "text" | "reasoning", text: string) {
if (!text) return
if (text.trim()) this.output = true
const last = this.parts[this.parts.length - 1]
if (last && last.type === kind && !(last as ReasoningPart).signature && !(last as ReasoningPart).opaque) (last as TextPart | ReasoningPart).text += text
else this.parts.push({ type: kind, text } as TextPart | ReasoningPart)
}
push(p: ReasoningPart | ToolCallPart) {
this.output = true
this.parts.push(p)
}
message(): Assistant {
return { role: "assistant", parts: this.parts }
}
}
/** One decision per failed attempt: retry (optionally saying why), or give up. */
export type RetryHandler = (e: ProviderError) => { retry: boolean; notice?: string }
/** Run attempts until one succeeds. A retry is only ever considered while nothing of the reply has
* been passed on (`beforeOutput`); a server error gets two retries on top of whatever `handle` allows. */
export async function* retrying(attempt: () => AsyncGenerator<StreamEvent>, handle?: RetryHandler): AsyncGenerator<StreamEvent> {
let retried5xx = 0
for (let n = 0; ; n++) {
try {
yield* attempt()
return
} catch (e) {
if (!(e instanceof ProviderError) || !e.beforeOutput || n >= 3) throw e
const h = handle?.(e)
if (h?.retry) {
if (h.notice) yield { type: "notice", message: h.notice }
continue
}
// Often the model's own output failing the server's parser (llama.cpp + gpt-oss), or the
// model swapped out under us (llama-swap): sampling differs on a second go.
// Twice: the first at once, the second after a pause (a model still loading, a busy server).
if ((e.status ?? 0) >= 500 && retried5xx < 2) {
retried5xx++
yield { type: "notice", message: `server error, retrying (${retried5xx} of 2) — ${e.message.split("\n")[0]!.slice(0, 160)}` }
if (retried5xx === 2) await new Promise((r) => setTimeout(r, 2000))
continue
}
throw e
}
}
}
/** Status for an error object inside a stream: a numeric code is HTTP; llama-swap's
* `type: "server_error"` / `code: "internal_error"` is a 500 too. */
export function statusOf(err: any): number | undefined {
if (typeof err?.code === "number") return err.code
if (typeof err?.status === "number") return err.status
if (err?.type === "server_error" || err?.type === "overloaded_error" || err?.type === "api_error" || /internal/i.test(String(err?.code ?? ""))) return 500
return undefined
}
export function recordName(ref: string) {
return `${env("RECORD")}/${new Date().toISOString().replace(/[:.]/g, "-")}-${ref.replace(/[^\w.-]+/g, "_")}`
}
export function recordFailure(ref: string, request: unknown, e: unknown) {
if (!env("RECORD")) return
const name = recordName(ref)
void Bun.write(`${name}.request.json`, JSON.stringify(request, null, 2))
const err = e as ProviderError
void Bun.write(`${name}.error.json`, JSON.stringify({ status: err.status, message: err.message, body: err.body }, null, 2))
}
/** LEMBAS_RECORD=<dir>: keep each request body and its raw SSE, to become test fixtures. */
export function record(body: ReadableStream<Uint8Array>, ref: string, request: unknown): ReadableStream<Uint8Array> {
const dir = env("RECORD")
if (!dir) return body
const [a, b] = body.tee()
const name = recordName(ref)
void Bun.write(`${name}.request.json`, JSON.stringify(request, null, 2))
void new Response(b).text().then((raw) => Bun.write(`${name}.sse`, raw))
return a
}
/** llama-swap streams its model-loading progress as reasoning, fenced by "━━━━━" lines, before the
* model's own first token. That is not the model thinking: it becomes one notice. */
export class SwapBanner {
private buf = ""
private state: "start" | "inside" | "done" = "start"
private skipSpace = false
feed(text: string): { text: string; notice?: string } {
if (this.state === "done") {
// The banner is followed by blank lines, streamed token by token; they are its, not the model's.
if (this.skipSpace) {
text = text.trimStart()
if (text) this.skipSpace = false
}
return { text }
}
this.buf += text
if (this.state === "start") {
const probe = this.buf.trimStart()
if (probe.length < 5 && "━━━━━".startsWith(probe)) return { text: "" }
if (!probe.startsWith("━━━━━")) {
this.state = "done"
const out = this.buf
this.buf = ""
return { text: out }
}
this.state = "inside"
}
const open = this.buf.indexOf("━━━━━")
const close = this.buf.indexOf("━━━━━", open + 5)
if (close === -1) {
if (this.buf.length > 4000) {
this.state = "done"
const out = this.buf
this.buf = ""
return { text: out }
}
return { text: "" }
}
const inner = this.buf.slice(open + 5, close).replace(/━+/g, "").trim()
let rest = this.buf.slice(close).replace(/^━+\s*/, "")
this.state = "done"
this.buf = ""
rest = rest.trimStart()
this.skipSpace = rest === ""
const lines = inner.split("\n").map((l) => l.trim()).filter(Boolean)
const done = lines.find((l) => /^Done!/.test(l)) ?? ""
const what = lines[0] ?? "model loading"
return { text: rest, notice: `${what}${done ? " — " + done.toLowerCase() : ""}` }
}
}
/** A tool call's arguments as they go back to the server in the history. A call cut off mid-way
* (the output limit) or garbled by the model has arguments that are not JSON, and llama.cpp parses
* the history's arguments again on every request — one broken call there fails each request after
* it with a 500. The tool result beside it already told the model what went wrong. */
export function historyArgs(args: string): string {
try {
const v = JSON.parse(args)
if (v && typeof v === "object" && !Array.isArray(v)) return args
} catch {}
return "{}"
}