Initial commit: standalone vector memory server

REST API for per-agent semantic memory retrieval.

- FastAPI server with /api/{agent}/save and /api/{agent}/query
- ChromaDB for persistent vector storage
- all-MiniLM-L6-v2 via sentence-transformers for embeddings
- Per-agent collections for clean separation
- Config through env vars or config.py
- .venv ready with all dependencies
This commit is contained in:
2026-06-25 14:50:53 +02:00
commit 82328b0a45
6 changed files with 301 additions and 0 deletions

30
.gitignore vendored Normal file
View File

@@ -0,0 +1,30 @@
# Python
__pycache__/
*.py[cod]
*.egg-info/
dist/
build/
# Virtual environment
.venv/
venv/
env/
# IDE / editors
.vscode/
.idea/
*.swp
*.swo
*~
# OS
.DS_Store
Thumbs.db
# ChromaDB data (persisted vectors — sacred, not in repo)
*.sqlite3
chroma_data/
# Environment overrides
.env
.env.local

17
config.py Normal file
View File

@@ -0,0 +1,17 @@
"""Configuration for the Vector Memory Server."""
import os
class Config:
DATA_DIR = os.environ.get(
"MEMORY_SERVER_DATA",
"/home/admin/agent-dir/vector_memory"
)
EMBEDDING_MODEL = os.environ.get(
"MEMORY_SERVER_MODEL",
"all-MiniLM-L6-v2"
)
DEFAULT_TOP_N = int(os.environ.get("MEMORY_SERVER_TOP_N", "5"))
HOST = os.environ.get("MEMORY_SERVER_HOST", "127.0.0.1")
PORT = int(os.environ.get("MEMORY_SERVER_PORT", "8000"))

29
embedder.py Normal file
View File

@@ -0,0 +1,29 @@
"""Embedding model wrapper — singleton pattern.
Uses sentence-transformers with all-MiniLM-L6-v2 (384-dim, ~80MB).
Loaded once, reused across all requests.
"""
from sentence_transformers import SentenceTransformer
from config import Config
class Embedder:
"""Thread-safe singleton wrapper for the embedding model."""
_instance = None
@classmethod
def get_instance(cls) -> "Embedder":
if cls._instance is None:
cls._instance = cls()
return cls._instance
def __init__(self) -> None:
self.model = SentenceTransformer(Config.EMBEDDING_MODEL)
def embed(self, text: str) -> list[float]:
return self.model.encode(text).tolist()
def embed_batch(self, texts: list[str]) -> list[list[float]]:
return self.model.encode(texts).tolist()

109
main.py Normal file
View File

@@ -0,0 +1,109 @@
"""Vector Memory Server — REST API.
Endpoints
---------
- ``POST /api/{agent_name}/save`` — Save a memory
- ``POST /api/{agent_name}/query`` — Query memories by semantic similarity
- ``GET /health`` — Health check
"""
from __future__ import annotations
from typing import Optional
import uvicorn
from fastapi import FastAPI, HTTPException
from pydantic import BaseModel
from config import Config
from store import MemoryStore
# ---------------------------------------------------------------------------
# App
# ---------------------------------------------------------------------------
app = FastAPI(
title="Vector Memory Server",
description="Semantic memory retrieval via REST. "
"Per-agent collections, powered by ChromaDB + all-MiniLM-L6-v2.",
version="0.1.0",
)
store = MemoryStore(Config.DATA_DIR)
# ---------------------------------------------------------------------------
# Request / Response models
# ---------------------------------------------------------------------------
class SaveRequest(BaseModel):
text: str
type: Optional[str] = "fact"
tags: Optional[list[str]] = None
date: Optional[str] = None
class SaveResponse(BaseModel):
success: bool
id: str
class QueryRequest(BaseModel):
query: str
top_n: Optional[int] = Config.DEFAULT_TOP_N
class QueryResponse(BaseModel):
results: list[dict]
# ---------------------------------------------------------------------------
# Routes
# ---------------------------------------------------------------------------
@app.post("/api/{agent_name}/save", response_model=SaveResponse)
async def save_memory(agent_name: str, request: SaveRequest) -> SaveResponse:
"""Embed and store a memory for the given agent."""
try:
metadata: dict = {}
if request.type:
metadata["type"] = request.type
if request.tags:
metadata["tags"] = ",".join(request.tags)
if request.date:
metadata["date"] = request.date
memory_id = store.save(agent_name, request.text, metadata)
return SaveResponse(success=True, id=memory_id)
except Exception as exc:
raise HTTPException(status_code=500, detail=str(exc))
@app.post("/api/{agent_name}/query", response_model=QueryResponse)
async def query_memory(
agent_name: str, request: QueryRequest
) -> QueryResponse:
"""Retrieve semantically similar memories for the given agent."""
try:
results = store.query(agent_name, request.query, request.top_n)
return QueryResponse(results=results)
except Exception as exc:
raise HTTPException(status_code=500, detail=str(exc))
@app.get("/health")
async def health() -> dict:
return {"status": "ok"}
# ---------------------------------------------------------------------------
# Entrypoint
# ---------------------------------------------------------------------------
if __name__ == "__main__":
uvicorn.run(
"main:app",
host=Config.HOST,
port=Config.PORT,
reload=False,
)

9
requirements.txt Normal file
View File

@@ -0,0 +1,9 @@
# Vector Memory Server
# Install: pip install -r requirements.txt
fastapi>=0.115.0
uvicorn[standard]>=0.34.0
chromadb>=0.6.0
sentence-transformers>=3.4.0
numpy>=2.0.0
pydantic>=2.0.0

107
store.py Normal file
View File

@@ -0,0 +1,107 @@
"""ChromaDB-backed memory store with per-agent collections."""
import uuid
import chromadb
from chromadb.config import Settings
from embedder import Embedder
class MemoryStore:
"""Persistent vector store using ChromaDB.
Each agent gets its own collection (``agent_{name}``) so memories
are cleanly separated and can be queried independently.
"""
def __init__(self, data_dir: str) -> None:
self.client = chromadb.PersistentClient(
path=data_dir,
settings=Settings(anonymized_telemetry=False),
)
self.embedder = Embedder.get_instance()
# ------------------------------------------------------------------
# Internal helpers
# ------------------------------------------------------------------
def _get_collection(self, agent_name: str):
"""Get or create a ChromaDB collection for *agent_name*."""
return self.client.get_or_create_collection(
name=f"agent_{agent_name}",
metadata={"agent": agent_name},
)
# ------------------------------------------------------------------
# Public API
# ------------------------------------------------------------------
def save(
self,
agent_name: str,
text: str,
metadata: dict | None = None,
) -> str:
"""Embed *text* and store it in the agent's collection.
Returns the auto-generated memory ID.
"""
collection = self._get_collection(agent_name)
memory_id = str(uuid.uuid4())
embedding = self.embedder.embed(text)
collection.add(
embeddings=[embedding],
documents=[text],
metadatas=[metadata or {}],
ids=[memory_id],
)
return memory_id
def query(
self,
agent_name: str,
query_text: str,
top_n: int = 5,
) -> list[dict]:
"""Return the *top_n* most semantically similar memories as text.
Each result dict contains:
- ``text`` the stored memory text
- ``metadata`` any additional metadata (original ``text``
field in metadata is stripped)
- ``distance`` cosine distance from the query
"""
collection = self._get_collection(agent_name)
query_embedding = self.embedder.embed(query_text)
results = collection.query(
query_embeddings=[query_embedding],
n_results=top_n,
)
formatted: list[dict] = []
if results["documents"] and results["documents"][0]:
for i, doc in enumerate(results["documents"][0]):
distance = (
results["distances"][0][i]
if results.get("distances")
else None
)
meta = (
{k: v for k, v in results["metadatas"][0][i].items()}
if results.get("metadatas")
else {}
)
# Don't return the raw document text in metadata (it's
# already the top-level "text" field).
meta.pop("text", None)
formatted.append({
"text": doc,
"metadata": meta,
"distance": distance,
})
return formatted