Files
LLeMbas-CLI/harness/models/schema.json
T
HomerandClaude Opus 5.5 f9bad01ed7
ci / check (push) Waiting to run
LLeMbas CLI 1.0.0
The first public release of LLeMbas CLI: a terminal coding agent and project manager for any LLM
API, with permission modes, git snapshots, memory and skills, knowledge bases, MCP, voice, and a
link to a LLeMbas instance whose web UI can work its sessions too. Signed Linux binaries for x64
and arm64.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-09 21:59:03 +00:00

34 lines
3.3 KiB
JSON

{
"$schema": "http://json-schema.org/draft-07/schema#",
"$id": "harness/models/schema.json",
"title": "Model metadata",
"description": "What both harnesses need to know about a model, in one shape: a LLeMbas CLI connections.yaml model entry, a LLeMbas model row, and each model LLeMbas serves at /v1/models (where a logged-in CLI configures itself from it). Unknown is not zero: an absent field means nobody has said.",
"type": "object",
"properties": {
"id": { "type": "string", "description": "What is sent as `model` in a request." },
"name": { "type": "string", "description": "A display name." },
"family": { "type": "string", "enum": ["anthropic", "gpt", "gemini", "local"], "description": "Which prompt overlay the model gets (prompts/family/). Guessed from the id when absent." },
"context": { "type": "integer", "minimum": 0, "description": "The whole context window in tokens. 0 or absent: unknown — no percentage, no automatic compaction." },
"max_output": { "type": "integer", "minimum": 1, "description": "The most tokens one reply may have." },
"temperature": { "type": "number", "minimum": 0, "maximum": 2, "description": "The sampling temperature the model is set up with. LLeMbas: the administrator's, which it also applies to a request that does not set one." },
"top_p": { "type": "number", "minimum": 0, "maximum": 1, "description": "Nucleus sampling, as temperature." },
"efforts": { "type": "array", "items": { "enum": ["minimal", "low", "medium", "high", "xhigh", "max"] }, "description": "The reasoning efforts the model takes, in offering order. Narrowed when a server refuses one (quirks.json effort-refused)." },
"effort": { "enum": ["minimal", "low", "medium", "high", "xhigh", "max", "off"], "description": "The effort a new conversation starts with; off sends none." },
"effort_style": { "enum": ["top", "kwargs", "both"], "description": "OpenAI Chat dialect: where the effort goes — top-level `reasoning_effort`, `chat_template_kwargs.reasoning_effort` (all llama.cpp reads), or both (the default)." },
"vision": { "type": "boolean", "description": "Takes images. Never sent images otherwise: most servers refuse the whole request." },
"tools": { "type": "boolean", "description": "Takes tool calls. false: it can only talk; never sent a tools array." },
"notes": { "type": "string", "maxLength": 300, "description": "What the model is good at, in a line: other models are told about it in the roster, for handing work over." },
"capacity": {
"type": "object",
"description": "How much the model's server can do at once, so a helper or a subagent on it is not started where it would evict or wait for the model already working.",
"properties": {
"group": { "type": "string", "description": "Models that share one server holding one model at a time (llama-swap): starting another model of the group unloads this one. The CLI: a model's `group`, or its connection when the connection sets one_model_at_a_time. LLeMbas: the connection's name when it sets one_model_at_a_time, which is what /v1/models says." },
"single_session": { "type": "boolean", "description": "The model serves one request at a time: a second one waits behind the first." }
},
"additionalProperties": false
}
},
"required": ["id"],
"additionalProperties": true
}