The first public release of LLeMbas CLI: a terminal coding agent and project manager for any LLM API, with permission modes, git snapshots, memory and skills, knowledge bases, MCP, voice, and a link to a LLeMbas instance whose web UI can work its sessions too. Signed Linux binaries for x64 and arm64. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
355 files changed
+47028
No files matched your search
@@ -0,0 +1,160 @@
|
||||
// What every dialect shares: the retry rule, the part builder, the stream recorder, and
|
||||
// llama-swap's loading banner.
|
||||
import type { ReasoningPart, StreamEvent, TextPart, ToolCallPart, Message } from "./types.ts"
|
||||
import { ProviderError } from "./types.ts"
|
||||
import { env } from "../config/paths.ts"
|
||||
|
||||
type Assistant = Extract<Message, { role: "assistant" }>
|
||||
|
||||
/** Builds an assistant message as deltas arrive, and knows whether anything real has been said —
|
||||
* the line after which a retry would duplicate output. */
|
||||
export class Parts {
|
||||
readonly parts: Assistant["parts"] = []
|
||||
output = false
|
||||
|
||||
add(kind: "text" | "reasoning", text: string) {
|
||||
if (!text) return
|
||||
if (text.trim()) this.output = true
|
||||
const last = this.parts[this.parts.length - 1]
|
||||
if (last && last.type === kind && !(last as ReasoningPart).signature && !(last as ReasoningPart).opaque) (last as TextPart | ReasoningPart).text += text
|
||||
else this.parts.push({ type: kind, text } as TextPart | ReasoningPart)
|
||||
}
|
||||
|
||||
push(p: ReasoningPart | ToolCallPart) {
|
||||
this.output = true
|
||||
this.parts.push(p)
|
||||
}
|
||||
|
||||
message(): Assistant {
|
||||
return { role: "assistant", parts: this.parts }
|
||||
}
|
||||
}
|
||||
|
||||
/** One decision per failed attempt: retry (optionally saying why), or give up. */
|
||||
export type RetryHandler = (e: ProviderError) => { retry: boolean; notice?: string }
|
||||
|
||||
/** Run attempts until one succeeds. A retry is only ever considered while nothing of the reply has
|
||||
* been passed on (`beforeOutput`); a server error gets two retries on top of whatever `handle` allows. */
|
||||
export async function* retrying(attempt: () => AsyncGenerator<StreamEvent>, handle?: RetryHandler): AsyncGenerator<StreamEvent> {
|
||||
let retried5xx = 0
|
||||
for (let n = 0; ; n++) {
|
||||
try {
|
||||
yield* attempt()
|
||||
return
|
||||
} catch (e) {
|
||||
if (!(e instanceof ProviderError) || !e.beforeOutput || n >= 3) throw e
|
||||
const h = handle?.(e)
|
||||
if (h?.retry) {
|
||||
if (h.notice) yield { type: "notice", message: h.notice }
|
||||
continue
|
||||
}
|
||||
// Often the model's own output failing the server's parser (llama.cpp + gpt-oss), or the
|
||||
// model swapped out under us (llama-swap): sampling differs on a second go.
|
||||
// Twice: the first at once, the second after a pause (a model still loading, a busy server).
|
||||
if ((e.status ?? 0) >= 500 && retried5xx < 2) {
|
||||
retried5xx++
|
||||
yield { type: "notice", message: `server error, retrying (${retried5xx} of 2) — ${e.message.split("\n")[0]!.slice(0, 160)}` }
|
||||
if (retried5xx === 2) await new Promise((r) => setTimeout(r, 2000))
|
||||
continue
|
||||
}
|
||||
throw e
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Status for an error object inside a stream: a numeric code is HTTP; llama-swap's
|
||||
* `type: "server_error"` / `code: "internal_error"` is a 500 too. */
|
||||
export function statusOf(err: any): number | undefined {
|
||||
if (typeof err?.code === "number") return err.code
|
||||
if (typeof err?.status === "number") return err.status
|
||||
if (err?.type === "server_error" || err?.type === "overloaded_error" || err?.type === "api_error" || /internal/i.test(String(err?.code ?? ""))) return 500
|
||||
return undefined
|
||||
}
|
||||
|
||||
export function recordName(ref: string) {
|
||||
return `${env("RECORD")}/${new Date().toISOString().replace(/[:.]/g, "-")}-${ref.replace(/[^\w.-]+/g, "_")}`
|
||||
}
|
||||
|
||||
export function recordFailure(ref: string, request: unknown, e: unknown) {
|
||||
if (!env("RECORD")) return
|
||||
const name = recordName(ref)
|
||||
void Bun.write(`${name}.request.json`, JSON.stringify(request, null, 2))
|
||||
const err = e as ProviderError
|
||||
void Bun.write(`${name}.error.json`, JSON.stringify({ status: err.status, message: err.message, body: err.body }, null, 2))
|
||||
}
|
||||
|
||||
/** LEMBAS_RECORD=<dir>: keep each request body and its raw SSE, to become test fixtures. */
|
||||
export function record(body: ReadableStream<Uint8Array>, ref: string, request: unknown): ReadableStream<Uint8Array> {
|
||||
const dir = env("RECORD")
|
||||
if (!dir) return body
|
||||
const [a, b] = body.tee()
|
||||
const name = recordName(ref)
|
||||
void Bun.write(`${name}.request.json`, JSON.stringify(request, null, 2))
|
||||
void new Response(b).text().then((raw) => Bun.write(`${name}.sse`, raw))
|
||||
return a
|
||||
}
|
||||
|
||||
/** llama-swap streams its model-loading progress as reasoning, fenced by "━━━━━" lines, before the
|
||||
* model's own first token. That is not the model thinking: it becomes one notice. */
|
||||
export class SwapBanner {
|
||||
private buf = ""
|
||||
private state: "start" | "inside" | "done" = "start"
|
||||
|
||||
private skipSpace = false
|
||||
|
||||
feed(text: string): { text: string; notice?: string } {
|
||||
if (this.state === "done") {
|
||||
// The banner is followed by blank lines, streamed token by token; they are its, not the model's.
|
||||
if (this.skipSpace) {
|
||||
text = text.trimStart()
|
||||
if (text) this.skipSpace = false
|
||||
}
|
||||
return { text }
|
||||
}
|
||||
this.buf += text
|
||||
if (this.state === "start") {
|
||||
const probe = this.buf.trimStart()
|
||||
if (probe.length < 5 && "━━━━━".startsWith(probe)) return { text: "" }
|
||||
if (!probe.startsWith("━━━━━")) {
|
||||
this.state = "done"
|
||||
const out = this.buf
|
||||
this.buf = ""
|
||||
return { text: out }
|
||||
}
|
||||
this.state = "inside"
|
||||
}
|
||||
const open = this.buf.indexOf("━━━━━")
|
||||
const close = this.buf.indexOf("━━━━━", open + 5)
|
||||
if (close === -1) {
|
||||
if (this.buf.length > 4000) {
|
||||
this.state = "done"
|
||||
const out = this.buf
|
||||
this.buf = ""
|
||||
return { text: out }
|
||||
}
|
||||
return { text: "" }
|
||||
}
|
||||
const inner = this.buf.slice(open + 5, close).replace(/━+/g, "").trim()
|
||||
let rest = this.buf.slice(close).replace(/^━+\s*/, "")
|
||||
this.state = "done"
|
||||
this.buf = ""
|
||||
rest = rest.trimStart()
|
||||
this.skipSpace = rest === ""
|
||||
const lines = inner.split("\n").map((l) => l.trim()).filter(Boolean)
|
||||
const done = lines.find((l) => /^Done!/.test(l)) ?? ""
|
||||
const what = lines[0] ?? "model loading"
|
||||
return { text: rest, notice: `${what}${done ? " — " + done.toLowerCase() : ""}` }
|
||||
}
|
||||
}
|
||||
|
||||
/** A tool call's arguments as they go back to the server in the history. A call cut off mid-way
|
||||
* (the output limit) or garbled by the model has arguments that are not JSON, and llama.cpp parses
|
||||
* the history's arguments again on every request — one broken call there fails each request after
|
||||
* it with a 500. The tool result beside it already told the model what went wrong. */
|
||||
export function historyArgs(args: string): string {
|
||||
try {
|
||||
const v = JSON.parse(args)
|
||||
if (v && typeof v === "object" && !Array.isArray(v)) return args
|
||||
} catch {}
|
||||
return "{}"
|
||||
}
|
||||
Reference in new issue
Block a user