Store embeddings as packed float32 instead of a JSON list
A 1536-dimension vector spelled out as JSON decimals is ~31 KB. The same
numbers packed as float32 are 6,144 bytes, and the whole bank is read on
every turn, so those bytes are paid over and over.
It is a format change, not a precision trade: the endpoints compute in
float32 and render that into JSON, so converting back recovers the original
bits exactly. Nothing is re-embedded and no API call is made -- migration 38
is a pure repack of what is already stored.
Unlike migrations 36 and 37 this backfill cannot be expressed in portable
SQL, so it comes through Python, batched, and pays a one-time read of every
vector to stop paying three megabytes a turn.
The JSON column stays, still written through set_vector, so a rollback finds
the vectors intact. Reading from the blob comes next; a follow-up migration
drops the old column once that is verified.
Migration SQL can now be a {dialect: sql} map -- BLOB and BYTEA have no
common spelling, and every Postgres deploy replays this one.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_015CYEJKobJ2Re4Dv7qUoSA7
This commit is contained in:
co-authored by
Claude Opus 5
parent
7ee5ceea6c
commit
c56864877a
+14
-13
@@ -20,14 +20,14 @@ on a later turn because the cursors only advance on success.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import math
|
||||
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from . import models
|
||||
from . import models, vectors
|
||||
from .context import history, story_actions, truncate_to_last_tokens
|
||||
from .database import SessionLocal
|
||||
from .providers import OpenAICompatibleProvider, ProviderError
|
||||
from .vectors import cosine # re-exported: the ranking lives here, the maths there
|
||||
|
||||
MEMORY_INTERVAL = 6 # actions per memory
|
||||
MEMORY_START = 12 # first memory once the adventure reaches this many actions
|
||||
@@ -79,14 +79,15 @@ def embedding_provider(settings: models.Settings) -> OpenAICompatibleProvider:
|
||||
)
|
||||
|
||||
|
||||
def cosine(a: list[float], b: list[float]) -> float:
|
||||
# Different lengths means the embedding model changed since this vector was
|
||||
# stored; zip() would silently score garbage.
|
||||
if len(a) != len(b):
|
||||
return 0.0
|
||||
dot = sum(x * y for x, y in zip(a, b))
|
||||
norm = math.sqrt(sum(x * x for x in a)) * math.sqrt(sum(y * y for y in b))
|
||||
return dot / norm if norm else 0.0
|
||||
def set_vector(memory: models.Memory, vector: list[float] | None) -> None:
|
||||
"""Store (or clear) a memory's embedding.
|
||||
|
||||
Both columns, always together: `embedding_blob` is what will be read, and
|
||||
the JSON `embedding` stays correct behind it until the follow-up migration
|
||||
drops it. Going through one function is what keeps them from drifting.
|
||||
"""
|
||||
memory.embedding = vector
|
||||
memory.embedding_blob = None if vector is None else vectors.pack(vector)
|
||||
|
||||
|
||||
def settled_count(adventure: models.Adventure) -> int:
|
||||
@@ -395,11 +396,11 @@ async def _embed_pending(
|
||||
if not pending:
|
||||
return
|
||||
try:
|
||||
vectors = await embedding_provider(settings).embed([m.text for m in pending])
|
||||
new = await embedding_provider(settings).embed([m.text for m in pending])
|
||||
except ProviderError:
|
||||
return
|
||||
for memory, vector in zip(pending, vectors):
|
||||
memory.embedding = vector
|
||||
for memory, vector in zip(pending, new):
|
||||
set_vector(memory, vector)
|
||||
db.commit()
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user