"""AI Chat: a plain scratchpad for talking to the configured model directly. Deliberately thin. It adds no story context and no world state, and it persists nothing. The conversation lives in the browser and is posted in full on each turn. It exists for checking a model, a prompt, or an endpoint without starting an adventure — which is exactly the kind of thing a local single-user install wants a page for. Upstream gated this behind a "power user" email allowlist and pinned the model when a shared demo key was in play. M2 removed both: there is one local user, who owns the endpoint, and there is no server-funded key to protect. The model this page talks to is the one in Settings, or one the user names per request — either way it is their own Ollama. """ from fastapi import APIRouter, Depends, HTTPException from fastapi.responses import StreamingResponse from sqlalchemy.orm import Session from .. import auth, models, schemas from ..database import get_db from ..providers import OpenAICompatibleProvider, ProviderError from ..sse import SSE_HEADERS, sse from .settings import get_settings, list_endpoint_models router = APIRouter(prefix="/api/chat", tags=["chat"]) @router.get("/config") async def chat_config( db: Session = Depends(get_db), user: models.User = Depends(auth.get_current_user), ): """Returns what this page can talk to. The model listing is best effort: an unreachable endpoint returns an empty list and the reason, rather than failing the page. """ settings = get_settings(db, user) listing = await list_endpoint_models(settings.endpoint_url) return { "endpoint_url": settings.endpoint_url, "model": settings.model, "api_mode": settings.api_mode, "temperature": settings.temperature, "max_tokens": settings.max_output_tokens, # Suggestions from the endpoint, not a restriction. "models": listing.get("models", []), "models_error": None if listing.get("ok") else listing.get("detail"), } async def run_chat( settings: models.Settings, model: str, payload: schemas.ChatRequest ): """Streams the reply as SSE, using the turn stream's event shape. The generator emits `reasoning` and `chunk` events while generating and then a `done` event, so the frontend reuses the same code. """ provider = OpenAICompatibleProvider( settings.endpoint_url, model, settings.api_mode, settings.model_timeout_seconds, ) messages = [m.model_dump() for m in payload.messages] chunks: list[str] = [] reasoning_chunks: list[str] = [] try: async for kind, chunk in provider.chat( messages, temperature=( payload.temperature if payload.temperature is not None else settings.temperature ), max_tokens=payload.max_tokens or settings.max_output_tokens, ): if kind == "reasoning": reasoning_chunks.append(chunk) yield sse({"type": "reasoning", "text": chunk}) else: chunks.append(chunk) yield sse({"type": "chunk", "text": chunk}) except ProviderError as exc: yield sse({"type": "error", "detail": str(exc)}) return text = "".join(chunks).strip() if not text: detail = ( "The model used its entire token budget on reasoning and returned no " "reply — raise max tokens or use a non-reasoning model." if reasoning_chunks else "The AI returned an empty response." ) yield sse({"type": "error", "detail": detail}) return yield sse({ "type": "done", "text": text, "reasoning": "".join(reasoning_chunks).strip() or None, "model": model, }) @router.post("/stream") def chat_stream( payload: schemas.ChatRequest, db: Session = Depends(get_db), user: models.User = Depends(auth.get_current_user), ): total = sum(len(m.content) for m in payload.messages) if total > schemas.CHAT_TOTAL_MAX: raise HTTPException( 413, f"This conversation is too long to send ({total:,} characters) — " "clear it or start a new one." ) settings = get_settings(db, user) model = (payload.model or "").strip() or settings.model if not model: raise HTTPException(400, "No model configured — set one in Settings or pick one here.") return StreamingResponse( run_chat(settings, model, payload), media_type="text/event-stream", headers=SSE_HEADERS, )