feat: Phase 2 — embeddings + hybrid memory.search
Adds a small embedder sidecar (Xenova/bge-small-en-v1.5, ONNX, CPU-only) that the web app calls inline on memory.write and memory.update, and on demand from the new memory.search tool. memory.search performs three candidate fetches in parallel — pgvector cosine similarity, Postgres full-text via plainto_tsquery + ts_rank_cd, and tag-set overlap — then fuses them with Reciprocal Rank Fusion (k=60). Each result carries its per-source rank so the model can see *why* a memory surfaced. The migrator boot step gained an idempotent embedding backfill: any row with embedding IS NULL is batched (32 at a time) through the embedder after SQL migrations apply. Safe to run on every boot. New tool memory.update fixes the missing edit path; centralises the re-embed-on-content-change rule alongside write. Stack additions: - apps/embedder/ — Fastify server, persistent /data/models volume so the ~30 MB model only downloads once - apps/web/lib/embedder.ts — typed HTTP client with batched embed + health probe - packages/schemas — MemoryUpdateInput, MemorySearchInput - docker-compose — embedder service, healthcheck, app + migrator both depend_on it healthy; EMBEDDER_URL promoted to a required env var Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,62 @@
|
||||
# syntax=docker/dockerfile:1.7
|
||||
# -----------------------------------------------------------------------------
|
||||
# Embedder sidecar.
|
||||
#
|
||||
# Builds from the repo root: docker build -f apps/embedder/Dockerfile .
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
FROM node:20-alpine AS base
|
||||
RUN corepack enable
|
||||
WORKDIR /app
|
||||
|
||||
# ---------- deps ----------
|
||||
FROM base AS deps
|
||||
COPY package.json pnpm-workspace.yaml pnpm-lock.yaml .npmrc ./
|
||||
COPY apps/embedder/package.json ./apps/embedder/
|
||||
# Other workspace package.json files needed so pnpm install can resolve the
|
||||
# workspace before --filter narrows things down.
|
||||
COPY apps/web/package.json ./apps/web/
|
||||
COPY packages/schemas/package.json ./packages/schemas/
|
||||
RUN --mount=type=cache,id=pnpm,target=/root/.local/share/pnpm/store \
|
||||
pnpm install --frozen-lockfile --filter @shared-memory/embedder...
|
||||
|
||||
# ---------- builder ----------
|
||||
FROM base AS builder
|
||||
COPY --from=deps /app/node_modules ./node_modules
|
||||
COPY --from=deps /app/apps/embedder/node_modules ./apps/embedder/node_modules
|
||||
COPY . .
|
||||
|
||||
# Compile TS to JS.
|
||||
RUN cd apps/embedder \
|
||||
&& pnpm exec tsc -p tsconfig.json --noEmit false --outDir dist
|
||||
|
||||
# Prune devDependencies so the runtime image only ships production deps.
|
||||
RUN cd apps/embedder \
|
||||
&& pnpm install --prod --frozen-lockfile --filter @shared-memory/embedder...
|
||||
|
||||
# ---------- runner ----------
|
||||
FROM node:20-alpine AS runner
|
||||
WORKDIR /app
|
||||
ENV NODE_ENV=production \
|
||||
PORT=8080 \
|
||||
HOST=0.0.0.0 \
|
||||
MODEL_CACHE_DIR=/data/models
|
||||
|
||||
RUN apk add --no-cache wget \
|
||||
&& addgroup --system --gid 1001 nodejs \
|
||||
&& adduser --system --uid 1001 --ingroup nodejs node-embedder \
|
||||
&& mkdir -p /data/models \
|
||||
&& chown -R node-embedder:nodejs /data
|
||||
|
||||
COPY --from=builder --chown=node-embedder:nodejs /app/apps/embedder/dist ./dist
|
||||
COPY --from=builder --chown=node-embedder:nodejs /app/apps/embedder/node_modules ./node_modules
|
||||
COPY --from=builder --chown=node-embedder:nodejs /app/apps/embedder/package.json ./package.json
|
||||
|
||||
USER node-embedder
|
||||
EXPOSE 8080
|
||||
VOLUME ["/data/models"]
|
||||
|
||||
HEALTHCHECK --interval=15s --timeout=5s --start-period=120s --retries=5 \
|
||||
CMD wget -q -O - http://127.0.0.1:8080/health | grep -q '"ready":true' || exit 1
|
||||
|
||||
CMD ["node", "--enable-source-maps", "dist/index.js"]
|
||||
@@ -0,0 +1,21 @@
|
||||
{
|
||||
"name": "@shared-memory/embedder",
|
||||
"version": "0.1.0",
|
||||
"private": true,
|
||||
"type": "module",
|
||||
"scripts": {
|
||||
"dev": "tsx watch src/index.ts",
|
||||
"build": "tsc --noEmit",
|
||||
"start": "node --enable-source-maps dist/index.js",
|
||||
"typecheck": "tsc --noEmit"
|
||||
},
|
||||
"dependencies": {
|
||||
"@xenova/transformers": "^2.17.2",
|
||||
"fastify": "^5.2.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@types/node": "^22.10.2",
|
||||
"tsx": "^4.19.2",
|
||||
"typescript": "^5.7.2"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,118 @@
|
||||
/**
|
||||
* Embedder sidecar — loads a small ONNX model once at boot and serves
|
||||
* mean-pooled, L2-normalized sentence embeddings over HTTP.
|
||||
*
|
||||
* Endpoints:
|
||||
* GET /health → { status, ready, model, dim }
|
||||
* POST /embed → { vectors: number[][] } given { texts: string[] }
|
||||
*
|
||||
* Used by the web app's memory.write / memory.update / memory.search and
|
||||
* by the migrator's one-shot backfill step.
|
||||
*/
|
||||
import Fastify from "fastify";
|
||||
import { pipeline, env as txEnv } from "@xenova/transformers";
|
||||
|
||||
// Persist the downloaded model on a named docker volume so subsequent
|
||||
// boots don't re-fetch ~30 MB.
|
||||
txEnv.cacheDir = process.env.MODEL_CACHE_DIR ?? "/data/models";
|
||||
txEnv.allowLocalModels = true;
|
||||
txEnv.allowRemoteModels = true;
|
||||
|
||||
const MODEL_NAME = process.env.EMBEDDING_MODEL ?? "Xenova/bge-small-en-v1.5";
|
||||
const EXPECTED_DIM = Number.parseInt(process.env.EMBEDDING_DIM ?? "384", 10);
|
||||
const PORT = Number.parseInt(process.env.PORT ?? "8080", 10);
|
||||
const HOST = process.env.HOST ?? "0.0.0.0";
|
||||
|
||||
// The pipeline()'s return type is a giant union covering every task; we
|
||||
// only use feature-extraction, so a narrower call signature is much easier
|
||||
// to work with than the upstream typing.
|
||||
interface FeatureExtractor {
|
||||
(
|
||||
texts: string[],
|
||||
options: { pooling: "mean" | "cls"; normalize: boolean },
|
||||
): Promise<{ tolist: () => number[] | number[][] }>;
|
||||
}
|
||||
let extractor: FeatureExtractor | null = null;
|
||||
|
||||
async function loadModel() {
|
||||
const start = Date.now();
|
||||
console.log(`[embedder] loading ${MODEL_NAME}…`);
|
||||
// Quantized=true is the @xenova default and is fast enough; flip via env if
|
||||
// we ever need the full-precision model.
|
||||
extractor = (await pipeline("feature-extraction", MODEL_NAME, {
|
||||
quantized: process.env.EMBEDDER_QUANTIZED !== "false",
|
||||
})) as unknown as FeatureExtractor;
|
||||
console.log(`[embedder] model ready in ${Date.now() - start}ms`);
|
||||
}
|
||||
|
||||
const app = Fastify({
|
||||
logger: { level: process.env.LOG_LEVEL ?? "info" },
|
||||
bodyLimit: 5 * 1024 * 1024, // 5 MB — generous for batched embeds
|
||||
});
|
||||
|
||||
app.get("/health", async () => ({
|
||||
status: "ok",
|
||||
ready: extractor !== null,
|
||||
model: MODEL_NAME,
|
||||
dim: EXPECTED_DIM,
|
||||
}));
|
||||
|
||||
interface EmbedRequest {
|
||||
texts: string[];
|
||||
}
|
||||
|
||||
app.post("/embed", async (req, reply) => {
|
||||
if (!extractor) {
|
||||
return reply.code(503).send({ error: "model not loaded yet" });
|
||||
}
|
||||
|
||||
const body = req.body as EmbedRequest | null;
|
||||
if (!body || !Array.isArray(body.texts)) {
|
||||
return reply.code(400).send({ error: "body must be { texts: string[] }" });
|
||||
}
|
||||
if (body.texts.length === 0) {
|
||||
return { vectors: [] };
|
||||
}
|
||||
if (body.texts.length > 256) {
|
||||
return reply.code(400).send({ error: "max 256 texts per request" });
|
||||
}
|
||||
if (body.texts.some((t) => typeof t !== "string")) {
|
||||
return reply.code(400).send({ error: "every entry in texts must be a string" });
|
||||
}
|
||||
|
||||
// Mean-pool the per-token hidden states and L2-normalize so cosine sim
|
||||
// matches the inner-product distance we'll feed into pgvector.
|
||||
const output = await extractor(body.texts, {
|
||||
pooling: "mean",
|
||||
normalize: true,
|
||||
});
|
||||
|
||||
// Transformers.js returns a Tensor; .tolist() gives nested JS arrays.
|
||||
// For batches the shape is [batch, dim]; for a single input the wrapper
|
||||
// may collapse to [dim] — defensively re-wrap.
|
||||
const raw = output.tolist();
|
||||
const vectors: number[][] = Array.isArray(raw[0])
|
||||
? (raw as number[][])
|
||||
: [raw as number[]];
|
||||
|
||||
// Sanity-check the dimension once at runtime — catches a model swap that
|
||||
// wasn't accompanied by an EMBEDDING_DIM bump.
|
||||
if (vectors[0] && vectors[0].length !== EXPECTED_DIM) {
|
||||
return reply.code(500).send({
|
||||
error: `model produced dim=${vectors[0].length}, expected ${EXPECTED_DIM}`,
|
||||
});
|
||||
}
|
||||
|
||||
return { vectors };
|
||||
});
|
||||
|
||||
async function start() {
|
||||
await loadModel();
|
||||
await app.listen({ host: HOST, port: PORT });
|
||||
console.log(`[embedder] listening on http://${HOST}:${PORT}`);
|
||||
}
|
||||
|
||||
start().catch((err) => {
|
||||
console.error("[embedder] startup failed:", err);
|
||||
process.exit(1);
|
||||
});
|
||||
@@ -0,0 +1,13 @@
|
||||
{
|
||||
"extends": "../../tsconfig.base.json",
|
||||
"compilerOptions": {
|
||||
"rootDir": "./src",
|
||||
"outDir": "./dist",
|
||||
"noEmit": false,
|
||||
"declaration": false,
|
||||
"module": "ESNext",
|
||||
"moduleResolution": "Bundler",
|
||||
"lib": ["ES2022"]
|
||||
},
|
||||
"include": ["src/**/*.ts"]
|
||||
}
|
||||
Reference in New Issue
Block a user