Models know how much context they hold
A column rather than a key in capabilities_json, which is rebuilt wholesale from the submitted checkboxes on every save and would destroy a number living in it. 0 means unknown, and unknown has to stay tellable from small: the context percentage and automatic compaction both refuse to act on a figure nobody supplied. Filled in from /v1/models where the runner advertises it -- OpenRouter, vLLM and llama.cpp each spell it differently, so context_from() reads the four spellings actually in use, accepts a quoted number but not "8192 tokens", and rejects anything outside 256..10,000,000. Applied on discovery only when nothing is set: a refresh must never undo a correction, since an administrator sets this precisely because the endpoint was wrong. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
+15
-2
@@ -14,7 +14,7 @@ from lembas.api.deps import AdminUser, Db
|
||||
from lembas.db.models import Connection, Model, User
|
||||
from lembas.services import settings_store
|
||||
from lembas.services.crypto import UNCHANGED_SENTINEL, decrypt, encrypt, mask
|
||||
from lembas.services.llm.openai_client import Endpoint, LLMError, list_models
|
||||
from lembas.services.llm.openai_client import Endpoint, LLMError, context_from, list_models
|
||||
from lembas.web.templating import render
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
@@ -198,8 +198,21 @@ async def _refresh_models(db: DBSession, connection: Connection) -> tuple[int, s
|
||||
model_id = str(entry["id"])[:300]
|
||||
seen.add(model_id)
|
||||
if model_id in existing:
|
||||
# A context length is filled in only when nobody has one yet. A
|
||||
# refresh must never overwrite a number an administrator typed --
|
||||
# they are usually correcting the endpoint.
|
||||
model = existing[model_id]
|
||||
if not model.context_length:
|
||||
model.context_length = context_from(entry)
|
||||
continue
|
||||
db.add(Model(connection_id=connection.id, model_id=model_id, position=next_position))
|
||||
db.add(
|
||||
Model(
|
||||
connection_id=connection.id,
|
||||
model_id=model_id,
|
||||
position=next_position,
|
||||
context_length=context_from(entry),
|
||||
)
|
||||
)
|
||||
next_position += 1
|
||||
|
||||
# Models that vanished upstream are dropped, so the picker never offers
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import logging
|
||||
|
||||
from fastapi import APIRouter, File, Form, HTTPException, Request, Response, UploadFile, status
|
||||
@@ -12,6 +13,7 @@ from sqlalchemy.orm import Session as DBSession
|
||||
from lembas.api.deps import AdminUser, Db, RequiredUser
|
||||
from lembas.db.models import Connection, Group, Model
|
||||
from lembas.services import settings_store, uploads
|
||||
from lembas.services.llm.openai_client import MAX_CONTEXT
|
||||
from lembas.web.templating import render
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
@@ -213,6 +215,7 @@ async def update_model(
|
||||
pinned: bool = Form(False),
|
||||
public: bool = Form(False),
|
||||
position: str = Form(""),
|
||||
context_length: str = Form(""),
|
||||
group_ids: list[str] = Form(default=[]),
|
||||
capability: list[str] = Form(default=[]),
|
||||
) -> Response:
|
||||
@@ -221,6 +224,13 @@ async def update_model(
|
||||
model.display_name = display_name.strip()[:300]
|
||||
model.description = description.strip()[:2000]
|
||||
model.system_prompt = system_prompt.strip()[:8000]
|
||||
# A string, so an emptied field is distinguishable and junk can be ignored
|
||||
# rather than becoming a 422 -- the same shape `position` uses below.
|
||||
if context_length.strip():
|
||||
with contextlib.suppress(ValueError):
|
||||
model.context_length = min(max(int(context_length), 0), MAX_CONTEXT)
|
||||
else:
|
||||
model.context_length = 0
|
||||
model.enabled = enabled
|
||||
model.pinned = pinned
|
||||
model.public = public
|
||||
|
||||
Reference in New Issue
Block a user