Migration 38 left memories.embedding in place so a rollback could still find the vectors. Production has since been verified reading from embedding_blob, so migration 42 drops it: 4 MB of a 99.6 MB database holding nothing anyone reads. Removing it surfaced a live bug. Changing your embedding model is supposed to throw the bank's vectors away and let the post-turn pass rebuild them, because two models' vectors are not comparable. The settings route did that by nulling memories.embedding -- correct until 38 moved the vectors, after which it cleared the dead column and left the blob intact with `embedded` still true. _embed_pending filters on `embedded IS FALSE`, so it never saw those rows and the bank went on ranking against the old model's vectors permanently. Nothing would have reported it. cosine returns 0.0 on a width mismatch, so a different-width model scores every memory zero and retrieval returns whichever rows happen to sort first; a same-width model scores plausible garbage. The bulk clear now sets both columns. It stays a bulk UPDATE rather than going through set_vector -- loading the rows is the cost that whole path exists to avoid -- so set_vector's docstring now names it as the one caller that legitimately writes those columns by hand. No cache invalidation is added: clearing `embedded` drops the rows out of the catalogue query, and set_vector evicts each entry as the re-embed puts it back. test_embedding_blob.py now rebuilds the pre-38 schema by hand where it tests the backfill, since create_all no longer produces the column it converts from, and asserts 42 removes it at the end of a full bootstrap -- 38 reads that column and 42 drops it, so an upgrade that reordered them would arrive with an empty bank. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017Dvvqn9ZDR4ixeFPHNbww7
118 lines
4.9 KiB
Python
118 lines
4.9 KiB
Python
import httpx
|
|
from fastapi import APIRouter, Depends, Request
|
|
from sqlalchemy.orm import Session
|
|
from starlette.concurrency import run_in_threadpool
|
|
|
|
from .. import auth, limits, models, netguard, schemas, security
|
|
from ..database import get_db
|
|
|
|
router = APIRouter(prefix="/api/settings", tags=["settings"])
|
|
|
|
|
|
def get_settings(db: Session, user: models.User) -> models.Settings:
|
|
"""Per-user settings row, created on first access (Phase 8: settings —
|
|
endpoint, key, models, memory config — are per user, not global)."""
|
|
settings = (
|
|
db.query(models.Settings).filter(models.Settings.user_id == user.id).first()
|
|
)
|
|
if settings is None:
|
|
settings = models.Settings(user_id=user.id)
|
|
db.add(settings)
|
|
db.commit()
|
|
return settings
|
|
|
|
|
|
@router.get("", response_model=schemas.SettingsOut)
|
|
def read_settings(
|
|
db: Session = Depends(get_db),
|
|
user: models.User = Depends(auth.get_current_user),
|
|
):
|
|
return get_settings(db, user)
|
|
|
|
|
|
@router.put("", response_model=schemas.SettingsOut)
|
|
def update_settings(
|
|
payload: schemas.SettingsUpdate,
|
|
db: Session = Depends(get_db),
|
|
user: models.User = Depends(auth.get_current_user),
|
|
):
|
|
settings = get_settings(db, user)
|
|
fields = payload.model_dump(exclude_unset=True)
|
|
# Write-only API key: absent = unchanged, "" = cleared, else encrypted.
|
|
if "api_key" in fields:
|
|
fields["api_key"] = security.encrypt_secret(fields["api_key"].strip())
|
|
embedding_model_changed = (
|
|
"embedding_model" in fields
|
|
and fields["embedding_model"] != settings.embedding_model
|
|
)
|
|
for field, value in fields.items():
|
|
setattr(settings, field, value)
|
|
if embedding_model_changed:
|
|
# Vectors from the old model have a different dimensionality/space;
|
|
# clear them so the post-turn task re-embeds with the new model.
|
|
# (This user's adventures only — settings are per-user now.)
|
|
#
|
|
# Both columns, and the flag. This is the one place that clears vectors
|
|
# in bulk rather than through memorybank.set_vector, and when the
|
|
# vectors moved to embedding_blob it kept nulling the old JSON column
|
|
# alone: the blob survived, `embedded` stayed true, and _embed_pending
|
|
# — which looks for embedded IS FALSE — never picked the rows up. The
|
|
# bank went on ranking against the previous model's vectors forever.
|
|
owned = (
|
|
db.query(models.Adventure.id)
|
|
.filter(models.Adventure.user_id == user.id)
|
|
.scalar_subquery()
|
|
)
|
|
db.query(models.Memory).filter(models.Memory.adventure_id.in_(owned)).update(
|
|
{"embedding_blob": None, "embedded": False}, synchronize_session=False
|
|
)
|
|
# No cache invalidation needed, and deliberately none added: clearing
|
|
# `embedded` drops these rows out of the catalogue query, so retrieval
|
|
# stops asking for them, and by the time _embed_pending puts one back
|
|
# it has gone through set_vector, which evicts that entry. The rule
|
|
# holds — anything that removes a memory from play self-corrects.
|
|
db.commit()
|
|
return settings
|
|
|
|
|
|
async def list_endpoint_models(cfg: auth.ProviderConfig) -> dict:
|
|
"""GET the endpoint's /models listing. Doubles as a connectivity check, so
|
|
failures come back as {"ok": False, "detail": ...} rather than raising."""
|
|
# SSRF guard: never probe a non-public address the user pointed us at.
|
|
reason = await run_in_threadpool(netguard.endpoint_block_reason, cfg.endpoint_url)
|
|
if reason:
|
|
return {"ok": False, "detail": f"Can't reach that endpoint — {reason}."}
|
|
url = cfg.endpoint_url.rstrip("/") + "/models"
|
|
headers = {}
|
|
if cfg.api_key:
|
|
headers["Authorization"] = f"Bearer {cfg.api_key}"
|
|
try:
|
|
async with httpx.AsyncClient(timeout=10) as client:
|
|
resp = await client.get(url, headers=headers)
|
|
except httpx.HTTPError as exc:
|
|
return {"ok": False, "detail": f"Connection failed: {exc}"}
|
|
|
|
if resp.status_code != 200:
|
|
return {"ok": False, "detail": f"HTTP {resp.status_code}: {resp.text[:300]}"}
|
|
|
|
models_available: list[str] = []
|
|
try:
|
|
data = resp.json()
|
|
models_available = [m.get("id", "?") for m in data.get("data", [])]
|
|
except (ValueError, AttributeError, TypeError):
|
|
pass # non-JSON or unexpected shape — connectivity is still confirmed
|
|
return {"ok": True, "models": models_available}
|
|
|
|
|
|
@router.post("/test")
|
|
async def test_connection(
|
|
request: Request,
|
|
db: Session = Depends(get_db),
|
|
user: models.User = Depends(auth.get_current_user),
|
|
):
|
|
"""Cheap connectivity check against whatever the turn engine would actually
|
|
use — including the shared demo endpoint when the user has no key."""
|
|
limits.rate_limit("connection-test", request, user)
|
|
settings = get_settings(db, user)
|
|
return await list_endpoint_models(auth.resolve_provider_config(settings))
|