ci / check (push) Waiting to run
The first public release of LLeMbas CLI: a terminal coding agent and project manager for any LLM API, with permission modes, git snapshots, memory and skills, knowledge bases, MCP, voice, and a link to a LLeMbas instance whose web UI can work its sessions too. Signed Linux binaries for x64 and arm64. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
34 lines
3.3 KiB
JSON
34 lines
3.3 KiB
JSON
{
|
|
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
"$id": "harness/models/schema.json",
|
|
"title": "Model metadata",
|
|
"description": "What both harnesses need to know about a model, in one shape: a LLeMbas CLI connections.yaml model entry, a LLeMbas model row, and each model LLeMbas serves at /v1/models (where a logged-in CLI configures itself from it). Unknown is not zero: an absent field means nobody has said.",
|
|
"type": "object",
|
|
"properties": {
|
|
"id": { "type": "string", "description": "What is sent as `model` in a request." },
|
|
"name": { "type": "string", "description": "A display name." },
|
|
"family": { "type": "string", "enum": ["anthropic", "gpt", "gemini", "local"], "description": "Which prompt overlay the model gets (prompts/family/). Guessed from the id when absent." },
|
|
"context": { "type": "integer", "minimum": 0, "description": "The whole context window in tokens. 0 or absent: unknown — no percentage, no automatic compaction." },
|
|
"max_output": { "type": "integer", "minimum": 1, "description": "The most tokens one reply may have." },
|
|
"temperature": { "type": "number", "minimum": 0, "maximum": 2, "description": "The sampling temperature the model is set up with. LLeMbas: the administrator's, which it also applies to a request that does not set one." },
|
|
"top_p": { "type": "number", "minimum": 0, "maximum": 1, "description": "Nucleus sampling, as temperature." },
|
|
"efforts": { "type": "array", "items": { "enum": ["minimal", "low", "medium", "high", "xhigh", "max"] }, "description": "The reasoning efforts the model takes, in offering order. Narrowed when a server refuses one (quirks.json effort-refused)." },
|
|
"effort": { "enum": ["minimal", "low", "medium", "high", "xhigh", "max", "off"], "description": "The effort a new conversation starts with; off sends none." },
|
|
"effort_style": { "enum": ["top", "kwargs", "both"], "description": "OpenAI Chat dialect: where the effort goes — top-level `reasoning_effort`, `chat_template_kwargs.reasoning_effort` (all llama.cpp reads), or both (the default)." },
|
|
"vision": { "type": "boolean", "description": "Takes images. Never sent images otherwise: most servers refuse the whole request." },
|
|
"tools": { "type": "boolean", "description": "Takes tool calls. false: it can only talk; never sent a tools array." },
|
|
"notes": { "type": "string", "maxLength": 300, "description": "What the model is good at, in a line: other models are told about it in the roster, for handing work over." },
|
|
"capacity": {
|
|
"type": "object",
|
|
"description": "How much the model's server can do at once, so a helper or a subagent on it is not started where it would evict or wait for the model already working.",
|
|
"properties": {
|
|
"group": { "type": "string", "description": "Models that share one server holding one model at a time (llama-swap): starting another model of the group unloads this one. The CLI: a model's `group`, or its connection when the connection sets one_model_at_a_time. LLeMbas: the connection's name when it sets one_model_at_a_time, which is what /v1/models says." },
|
|
"single_session": { "type": "boolean", "description": "The model serves one request at a time: a second one waits behind the first." }
|
|
},
|
|
"additionalProperties": false
|
|
}
|
|
},
|
|
"required": ["id"],
|
|
"additionalProperties": true
|
|
}
|