The first public release of LLeMbas CLI: a terminal coding agent and project manager for any LLM API, with permission modes, git snapshots, memory and skills, knowledge bases, MCP, voice, and a link to a LLeMbas instance whose web UI can work its sessions too. Signed Linux binaries for x64 and arm64. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
355 files changed
+47028
No files matched your search
@@ -0,0 +1,567 @@
|
||||
{
|
||||
"$schema": "http://json-schema.org/draft-07/schema#",
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"connections": {
|
||||
"default": {},
|
||||
"description": "Each LLM endpoint by a name of your choosing; a model is then `name/model-id`. `type: webui` is a LLeMbas instance, whose models come from it.",
|
||||
"type": "object",
|
||||
"propertyNames": {
|
||||
"type": "string"
|
||||
},
|
||||
"additionalProperties": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"type": {
|
||||
"type": "string",
|
||||
"const": "webui",
|
||||
"description": "A LLeMbas instance: its models, their settings, its voice and its web search."
|
||||
},
|
||||
"url": {
|
||||
"type": "string",
|
||||
"description": "The instance's address, e.g. https://ai.example.org (no /v1)."
|
||||
},
|
||||
"api_key": {
|
||||
"description": "This machine's token, as login wrote it: {file:~/.config/lembas/lembas/<name>.key}. Or key_cmd.",
|
||||
"type": "string"
|
||||
},
|
||||
"key_cmd": {
|
||||
"type": "string"
|
||||
},
|
||||
"tls": {
|
||||
"description": "For an instance behind a private CA: `ca`, the PEM file that signed its certificate.",
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"ca": {
|
||||
"type": "string"
|
||||
},
|
||||
"insecure": {
|
||||
"type": "boolean"
|
||||
}
|
||||
},
|
||||
"additionalProperties": false
|
||||
},
|
||||
"timeout": {
|
||||
"description": "Seconds of silence allowed in a reply, as for any connection (default 600).",
|
||||
"type": "number",
|
||||
"exclusiveMinimum": 0
|
||||
},
|
||||
"quirks": {
|
||||
"description": "Server oddities; auto is right almost always.",
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"stream_usage": {
|
||||
"description": "Ask for token usage in the stream (stream_options); auto learns when a server refuses it.",
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"auto",
|
||||
"on",
|
||||
"off"
|
||||
]
|
||||
},
|
||||
"prompt_progress": {
|
||||
"description": "Ask how far the server is through reading the prompt (llama.cpp's return_progress), shown while it reads; auto learns when a server refuses it, and one that ignores it shows nothing.",
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"auto",
|
||||
"on",
|
||||
"off"
|
||||
]
|
||||
},
|
||||
"think_tags": {
|
||||
"description": "<think>…</think> in a reply is taken as reasoning; off leaves it as text.",
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"auto",
|
||||
"on",
|
||||
"off"
|
||||
]
|
||||
},
|
||||
"max_tokens_field": {
|
||||
"description": "OpenAI reasoning models refuse `max_tokens`; most local servers only know it.",
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"max_tokens",
|
||||
"max_completion_tokens"
|
||||
]
|
||||
}
|
||||
},
|
||||
"additionalProperties": false
|
||||
},
|
||||
"models": {
|
||||
"description": "Only to change what the instance says about a model, here: each entry goes over the instance's. Usually absent.",
|
||||
"type": "object",
|
||||
"propertyNames": {
|
||||
"type": "string"
|
||||
},
|
||||
"additionalProperties": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"name": {
|
||||
"description": "A display name; the id is what is sent.",
|
||||
"type": "string"
|
||||
},
|
||||
"family": {
|
||||
"description": "Picks the prompt overlay: anthropic | gpt | gemini | local. Guessed from the id when absent.",
|
||||
"type": "string"
|
||||
},
|
||||
"context": {
|
||||
"description": "Total context window in tokens. 0 / absent = unknown: no percentage, no auto-compaction.",
|
||||
"type": "integer",
|
||||
"minimum": 0,
|
||||
"maximum": 9007199254740991
|
||||
},
|
||||
"max_output": {
|
||||
"description": "Max tokens per reply.",
|
||||
"type": "integer",
|
||||
"exclusiveMinimum": 0,
|
||||
"maximum": 9007199254740991
|
||||
},
|
||||
"temperature": {
|
||||
"description": "Sampling temperature; absent = the server's default.",
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
"maximum": 2
|
||||
},
|
||||
"top_p": {
|
||||
"description": "Nucleus sampling; absent = the server's default.",
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
"maximum": 1
|
||||
},
|
||||
"efforts": {
|
||||
"description": "The model's own effort vocabulary, in offering order.",
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
]
|
||||
}
|
||||
},
|
||||
"effort": {
|
||||
"description": "Default effort; `off` sends none.",
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
]
|
||||
},
|
||||
{
|
||||
"type": "string",
|
||||
"const": "off"
|
||||
}
|
||||
]
|
||||
},
|
||||
"effort_style": {
|
||||
"description": "openai-chat only: where the effort goes. llama.cpp drops top-level, so `both` is the default.",
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"top",
|
||||
"kwargs",
|
||||
"both"
|
||||
]
|
||||
},
|
||||
"effort_map": {
|
||||
"description": "anthropic/gemini: effort → thinking budget tokens.",
|
||||
"type": "object",
|
||||
"propertyNames": {
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
]
|
||||
},
|
||||
"additionalProperties": {
|
||||
"type": "integer",
|
||||
"exclusiveMinimum": 0,
|
||||
"maximum": 9007199254740991
|
||||
}
|
||||
},
|
||||
"vision": {
|
||||
"description": "The model reads images: @image files, pasted paths and view_image.",
|
||||
"type": "boolean"
|
||||
},
|
||||
"tools": {
|
||||
"description": "false: a model without tool calls — it can only talk.",
|
||||
"type": "boolean"
|
||||
},
|
||||
"cache": {
|
||||
"description": "anthropic: prompt caching breakpoints.",
|
||||
"type": "boolean"
|
||||
},
|
||||
"headers": {
|
||||
"description": "Extra HTTP headers for this model's requests.",
|
||||
"type": "object",
|
||||
"propertyNames": {
|
||||
"type": "string"
|
||||
},
|
||||
"additionalProperties": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"body": {
|
||||
"description": "Merged into every request body for this model.",
|
||||
"type": "object",
|
||||
"propertyNames": {
|
||||
"type": "string"
|
||||
},
|
||||
"additionalProperties": {}
|
||||
},
|
||||
"notes": {
|
||||
"description": "What this model is good at, in a line: the model in use is told which others there are, for handing work over (task) or switching.",
|
||||
"type": "string",
|
||||
"maxLength": 300
|
||||
},
|
||||
"fallback": {
|
||||
"description": "Other models (connection/model), in order, to switch to when this one's server cannot be reached at all — never once a reply has begun.",
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"single_session": {
|
||||
"description": "The server serves one request at a time: a subagent on this same model would take its only slot and push the session's cached prompt out, so it is refused.",
|
||||
"type": "boolean"
|
||||
},
|
||||
"group": {
|
||||
"description": "Models sharing one server that holds one model at a time (llama-swap), named by any string: a subagent on another model of the same group would unload this one, so it is refused. One connection's one_model_at_a_time is the same rule for all its models; this is for models one connection serves from several such servers (a LLeMbas instance).",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"additionalProperties": false
|
||||
}
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"type",
|
||||
"url"
|
||||
],
|
||||
"additionalProperties": false
|
||||
},
|
||||
{
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"dialect": {
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"openai-chat",
|
||||
"responses",
|
||||
"anthropic",
|
||||
"gemini",
|
||||
"ollama"
|
||||
],
|
||||
"description": "The API it speaks: openai-chat (llama.cpp, vLLM, LM Studio, OpenAI…), responses (OpenAI Responses), anthropic, gemini, ollama (its native /api/chat)."
|
||||
},
|
||||
"base_url": {
|
||||
"type": "string",
|
||||
"description": "Where the API is, e.g. https://api.openai.com/v1 or http://localhost:8080/v1."
|
||||
},
|
||||
"api_key": {
|
||||
"description": "Never pasted: {env:NAME} or {file:~/path}. Or use key_cmd.",
|
||||
"type": "string"
|
||||
},
|
||||
"key_cmd": {
|
||||
"description": "A command whose output is the key, e.g. pass show openai.",
|
||||
"type": "string"
|
||||
},
|
||||
"auth": {
|
||||
"description": "How the key is sent. Default per dialect: anthropic x-api-key, gemini x-goog-api-key, others bearer.",
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"bearer",
|
||||
"x-api-key",
|
||||
"x-goog-api-key",
|
||||
"none"
|
||||
]
|
||||
},
|
||||
"headers": {
|
||||
"description": "Extra HTTP headers for every request.",
|
||||
"type": "object",
|
||||
"propertyNames": {
|
||||
"type": "string"
|
||||
},
|
||||
"additionalProperties": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"body": {
|
||||
"description": "Merged into every request body.",
|
||||
"type": "object",
|
||||
"propertyNames": {
|
||||
"type": "string"
|
||||
},
|
||||
"additionalProperties": {}
|
||||
},
|
||||
"tls": {
|
||||
"description": "TLS for endpoints behind a private CA. `ca` is a PEM file added to the trusted roots; `insecure` skips verification entirely (last resort). The system store is always trusted.",
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"ca": {
|
||||
"type": "string"
|
||||
},
|
||||
"insecure": {
|
||||
"type": "boolean"
|
||||
}
|
||||
},
|
||||
"additionalProperties": false
|
||||
},
|
||||
"timeout": {
|
||||
"description": "Seconds of silence allowed: waiting for the response, then between two pieces of a streamed reply. Not a limit on how long a reply may take. Default 600.",
|
||||
"type": "number",
|
||||
"exclusiveMinimum": 0
|
||||
},
|
||||
"discover": {
|
||||
"description": "Merge ids and context lengths from GET /models.",
|
||||
"type": "boolean"
|
||||
},
|
||||
"unload_url": {
|
||||
"description": "llama-swap style: frees VRAM when switching away.",
|
||||
"type": "string"
|
||||
},
|
||||
"one_model_at_a_time": {
|
||||
"description": "The server holds one model at a time (llama-swap in front of one GPU): a subagent on another of its models would unload the session's, so it is refused. The session's own model may still be its own subagent.",
|
||||
"type": "boolean"
|
||||
},
|
||||
"unload_method": {
|
||||
"description": "How unload_url is called (default POST).",
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"GET",
|
||||
"POST"
|
||||
]
|
||||
},
|
||||
"quirks": {
|
||||
"description": "Server oddities; auto is right almost always.",
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"stream_usage": {
|
||||
"description": "Ask for token usage in the stream (stream_options); auto learns when a server refuses it.",
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"auto",
|
||||
"on",
|
||||
"off"
|
||||
]
|
||||
},
|
||||
"prompt_progress": {
|
||||
"description": "Ask how far the server is through reading the prompt (llama.cpp's return_progress), shown while it reads; auto learns when a server refuses it, and one that ignores it shows nothing.",
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"auto",
|
||||
"on",
|
||||
"off"
|
||||
]
|
||||
},
|
||||
"think_tags": {
|
||||
"description": "<think>…</think> in a reply is taken as reasoning; off leaves it as text.",
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"auto",
|
||||
"on",
|
||||
"off"
|
||||
]
|
||||
},
|
||||
"max_tokens_field": {
|
||||
"description": "OpenAI reasoning models refuse `max_tokens`; most local servers only know it.",
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"max_tokens",
|
||||
"max_completion_tokens"
|
||||
]
|
||||
}
|
||||
},
|
||||
"additionalProperties": false
|
||||
},
|
||||
"models": {
|
||||
"default": {},
|
||||
"description": "The models to offer, by the id the server knows them by.",
|
||||
"type": "object",
|
||||
"propertyNames": {
|
||||
"type": "string"
|
||||
},
|
||||
"additionalProperties": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"name": {
|
||||
"description": "A display name; the id is what is sent.",
|
||||
"type": "string"
|
||||
},
|
||||
"family": {
|
||||
"description": "Picks the prompt overlay: anthropic | gpt | gemini | local. Guessed from the id when absent.",
|
||||
"type": "string"
|
||||
},
|
||||
"context": {
|
||||
"description": "Total context window in tokens. 0 / absent = unknown: no percentage, no auto-compaction.",
|
||||
"type": "integer",
|
||||
"minimum": 0,
|
||||
"maximum": 9007199254740991
|
||||
},
|
||||
"max_output": {
|
||||
"description": "Max tokens per reply.",
|
||||
"type": "integer",
|
||||
"exclusiveMinimum": 0,
|
||||
"maximum": 9007199254740991
|
||||
},
|
||||
"temperature": {
|
||||
"description": "Sampling temperature; absent = the server's default.",
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
"maximum": 2
|
||||
},
|
||||
"top_p": {
|
||||
"description": "Nucleus sampling; absent = the server's default.",
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
"maximum": 1
|
||||
},
|
||||
"efforts": {
|
||||
"description": "The model's own effort vocabulary, in offering order.",
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
]
|
||||
}
|
||||
},
|
||||
"effort": {
|
||||
"description": "Default effort; `off` sends none.",
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
]
|
||||
},
|
||||
{
|
||||
"type": "string",
|
||||
"const": "off"
|
||||
}
|
||||
]
|
||||
},
|
||||
"effort_style": {
|
||||
"description": "openai-chat only: where the effort goes. llama.cpp drops top-level, so `both` is the default.",
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"top",
|
||||
"kwargs",
|
||||
"both"
|
||||
]
|
||||
},
|
||||
"effort_map": {
|
||||
"description": "anthropic/gemini: effort → thinking budget tokens.",
|
||||
"type": "object",
|
||||
"propertyNames": {
|
||||
"type": "string",
|
||||
"enum": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max"
|
||||
]
|
||||
},
|
||||
"additionalProperties": {
|
||||
"type": "integer",
|
||||
"exclusiveMinimum": 0,
|
||||
"maximum": 9007199254740991
|
||||
}
|
||||
},
|
||||
"vision": {
|
||||
"description": "The model reads images: @image files, pasted paths and view_image.",
|
||||
"type": "boolean"
|
||||
},
|
||||
"tools": {
|
||||
"description": "false: a model without tool calls — it can only talk.",
|
||||
"type": "boolean"
|
||||
},
|
||||
"cache": {
|
||||
"description": "anthropic: prompt caching breakpoints.",
|
||||
"type": "boolean"
|
||||
},
|
||||
"headers": {
|
||||
"description": "Extra HTTP headers for this model's requests.",
|
||||
"type": "object",
|
||||
"propertyNames": {
|
||||
"type": "string"
|
||||
},
|
||||
"additionalProperties": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"body": {
|
||||
"description": "Merged into every request body for this model.",
|
||||
"type": "object",
|
||||
"propertyNames": {
|
||||
"type": "string"
|
||||
},
|
||||
"additionalProperties": {}
|
||||
},
|
||||
"notes": {
|
||||
"description": "What this model is good at, in a line: the model in use is told which others there are, for handing work over (task) or switching.",
|
||||
"type": "string",
|
||||
"maxLength": 300
|
||||
},
|
||||
"fallback": {
|
||||
"description": "Other models (connection/model), in order, to switch to when this one's server cannot be reached at all — never once a reply has begun.",
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"single_session": {
|
||||
"description": "The server serves one request at a time: a subagent on this same model would take its only slot and push the session's cached prompt out, so it is refused.",
|
||||
"type": "boolean"
|
||||
},
|
||||
"group": {
|
||||
"description": "Models sharing one server that holds one model at a time (llama-swap), named by any string: a subagent on another model of the same group would unload this one, so it is refused. One connection's one_model_at_a_time is the same rule for all its models; this is for models one connection serves from several such servers (a LLeMbas instance).",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"additionalProperties": false
|
||||
}
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"dialect",
|
||||
"base_url"
|
||||
],
|
||||
"additionalProperties": false
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"additionalProperties": false,
|
||||
"title": "LLeMbas CLI connections.yaml",
|
||||
"description": "~/.config/lembas/connections.yaml: the LLM endpoints. Global only. Keys as {env:NAME}, {file:path} or key_cmd — never pasted."
|
||||
}
|
||||
Reference in new issue
Block a user