diff --git a/README.md b/README.md index e93b9f8..d17fae6 100644 --- a/README.md +++ b/README.md @@ -5,6 +5,10 @@ with your own AI model. Create scenarios, play open-ended adventures where an LL world, and extend the engine with **JavaScript scripts compatible with real AI Dungeon scripting**. +> ### ▶️ Try it live: **[ai-dnd-1gmp.onrender.com](https://ai-dnd-1gmp.onrender.com)** +> Play a demo scenario as a guest — no sign-up, no API key needed. (Hosted on Render's free +> tier, so the first load after it's been idle takes ~30–60s to wake up.) + Built with FastAPI + SQLite on the backend and React (Vite) on the frontend. Works with **any OpenAI-compatible endpoint**: Ollama and LM Studio locally, or OpenRouter / OpenAI / Groq / vLLM in the cloud — endpoint, key, and model are all runtime settings, and OpenRouter's free-tier diff --git a/backend/app/models.py b/backend/app/models.py index ab6d7bf..b4ba3f2 100644 --- a/backend/app/models.py +++ b/backend/app/models.py @@ -239,7 +239,9 @@ class Settings(Base): model: Mapped[str] = mapped_column(String(200), default="") api_mode: Mapped[str] = mapped_column(String(20), default="chat") # chat|completion temperature: Mapped[float] = mapped_column(Float, default=0.8) - max_output_tokens: Mapped[int] = mapped_column(Integer, default=400) + # 800 leaves room for a full scene; 400 tended to truncate mid-paragraph + # and left reasoning models with nothing after their thinking. + max_output_tokens: Mapped[int] = mapped_column(Integer, default=800) # Separate thinking budget for reasoning models (OpenRouter-style # `reasoning: {max_tokens}`); 0 = param not sent. Added on top of # max_output_tokens so story output keeps its full budget. diff --git a/backend/app/routers/adventures.py b/backend/app/routers/adventures.py index 71a64da..6a58761 100644 --- a/backend/app/routers/adventures.py +++ b/backend/app/routers/adventures.py @@ -288,7 +288,17 @@ async def generate_turn( text = "".join(chunks).strip() if not text: - yield sse({"type": "error", "detail": "The AI returned an empty response."}) + # If the model streamed reasoning but no story text, it spent its whole + # budget thinking — say so instead of a mysterious "empty response". + if reasoning_chunks: + detail = ( + "The model used its entire token budget on reasoning and returned no " + 'story text. Raise "Max output tokens" in Settings, set a "Reasoning ' + 'max tokens" cap, or switch to a non-reasoning model.' + ) + else: + detail = "The AI returned an empty response." + yield sse({"type": "error", "detail": detail}) return # onOutput