Files
LLeMbas-CLI/src/library/embed.ts
T
HomerandClaude Opus 5.5 f9bad01ed7
ci / check (push) Waiting to run
LLeMbas CLI 1.0.0
The first public release of LLeMbas CLI: a terminal coding agent and project manager for any LLM
API, with permission modes, git snapshots, memory and skills, knowledge bases, MCP, voice, and a
link to a LLeMbas instance whose web UI can work its sessions too. Signed Linux binaries for x64
and arm64.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-09 21:59:03 +00:00

61 lines
2.9 KiB
TypeScript

// Embeddings for the library: `embedding: connection/model` in config.yaml, over the connection's
// OpenAI-shaped /embeddings (llama.cpp with --embedding, vLLM, LM Studio, OpenAI…) or Ollama's
// /api/embed. Vectors are made unit length here, so a dot product is the cosine.
import { resolveKey, type Loaded } from "../config/load.ts"
import { authHeaders, joinUrl, request, tlsFor } from "../provider/http.ts"
import { resolveModel } from "../provider/index.ts"
import type { Embedder } from "./store.ts"
function unit(v: number[]): Float32Array {
let n = 0
for (const x of v) n += x * x
n = Math.sqrt(n) || 1
return Float32Array.from(v, (x) => x / n)
}
export function embedderFor(loaded: Loaded, ref: string | undefined): Embedder | undefined {
if (!ref) return undefined
// Through the same resolver as every model ref: an instance's model by its provider
// (`llama/nomic-embed`) or in the older form (`example/nomic-embed`). The name stored with the
// vectors stays the ref as written, so a library indexed under it is still found.
// An embedding model need not be listed on its connection (`connection/<any id>`, as before).
let name: string
let id: string
try {
const resolved = resolveModel(loaded, ref)
name = resolved.connectionName
id = resolved.id
} catch (e) {
// An instance's embedding model by its served id (`llama/nomic-embed`): not a
// model ref, since only chat models are (lembas/webui.ts, isChatModel), so found by the ids
// its connection says it serves.
const served = Object.entries(loaded.connections).find(([, c]) => (c as { webui?: { embeddings?: string[] } }).webui?.embeddings?.includes(ref))
const slash = ref.indexOf("/")
if (served) {
name = served[0]
id = ref
} else {
if (slash <= 0) throw new Error(`embedding: "${ref}" must be written connection/model`)
name = ref.slice(0, slash)
id = ref.slice(slash + 1)
if (!loaded.connections[name]) throw new Error(`embedding: ${(e as Error).message}`)
}
}
const c = loaded.connections[name]!
if (c.dialect === "anthropic" || c.dialect === "gemini") throw new Error(`embedding: the ${c.dialect} dialect has no embeddings here; use an OpenAI-compatible or Ollama connection`)
const ollama = c.dialect === "ollama"
return {
model: ref,
async embed(texts, signal) {
const res = await request(
joinUrl(c.base_url, ollama ? "api/embed" : "embeddings"),
{ method: "POST", headers: authHeaders(c, resolveKey(name, c)), body: JSON.stringify({ model: id, input: texts }), signal, timeoutMs: (c.timeout ?? 120) * 1000, tls: tlsFor(c) },
`${ref} (embeddings)`,
)
const j = (await res.json()) as { data?: { embedding: number[]; index?: number }[]; embeddings?: number[][] }
const rows = ollama ? (j.embeddings ?? []) : [...(j.data ?? [])].sort((a, b) => (a.index ?? 0) - (b.index ?? 0)).map((d) => d.embedding)
return rows.map(unit)
},
}
}