feat(mcpedia): Phase 2 — semantic + hybrid search, tRPC/Hono API

- @mcpedia/embeddings: OpenRouter provider (9router /v1, encoding_format float),
  chunkText + embedChunks; EMBED_DIM=2048
- document_chunks table (real[] embedding) — pgvector NOT available on shared
  imrnes Postgres, so cosine is computed in-app (KB-scale fine); pgvector deferred
- indexer: chunk + embed + upsert per document
- @mcpedia/search: semanticSearch (cosine) + hybridSearch (FTS+cosine RRF)
- apps/api: Hono + tRPC v11 (6 procedures), serve on :4020
- MCP: semantic_search + hybrid_search tools (6 total)
- web: keyword/hybrid toggle; .env.example + README + PHASES updated
This commit is contained in:
asepharyana
2026-08-19 19:00:29 +07:00
parent ec3a2867d4
commit 9397303f01
32 changed files with 1120 additions and 42 deletions
+4
View File
@@ -36,6 +36,10 @@ export const CONTENT_ROOT =
export const DATABASE_URL = process.env.DATABASE_URL ?? "";
export const EMBED_BASE_URL = process.env.EMBED_BASE_URL ?? "";
export const EMBED_API_KEY = process.env.EMBED_API_KEY ?? "";
export const EMBED_MODEL = process.env.EMBED_MODEL ?? "";
if (!DATABASE_URL) {
// Fail fast with an explicit message instead of a cryptic driver error.
throw new Error(
+1
View File
@@ -10,6 +10,7 @@
"dependencies": {
"@mcpedia/config": "workspace:*",
"@mcpedia/db": "workspace:*",
"@mcpedia/embeddings": "workspace:*",
"@mcpedia/parser": "workspace:*",
"@mcpedia/search": "workspace:*",
"@mcpedia/types": "workspace:*",
+41 -1
View File
@@ -1,6 +1,6 @@
import { and, eq, sql } from "drizzle-orm";
import { db } from "@mcpedia/db";
import { documents } from "@mcpedia/db/schema";
import { documentChunks, documents } from "@mcpedia/db/schema";
import { CONTENT_ROOT } from "@mcpedia/config";
import { existsSync, readFileSync } from "node:fs";
import { join } from "node:path";
@@ -8,9 +8,12 @@ import type {
Document,
DocumentMeta,
} from "@mcpedia/types";
import { chunkText, embedChunks, createEmbeddingProvider } from "@mcpedia/embeddings";
import { readContentFile } from "./content.service";
import { toMeta } from "./row-map";
const embedder = createEmbeddingProvider();
export async function listDocuments(opts: {
section?: string;
status?: string;
@@ -56,3 +59,40 @@ export async function getRelated(slug: string, limit = 5): Promise<DocumentMeta[
}
export { readContentFile };
/**
* Chunk a document body, embed the chunks, and upsert them into
* `document_chunks` (replacing any prior chunks for the same slug).
* Failures are thrown so the caller can decide whether to abort the index.
*/
export async function indexChunks(slug: string, body: string): Promise<number> {
const [doc] = await db
.select({ id: documents.id })
.from(documents)
.where(eq(documents.slug, slug));
if (!doc) return 0;
const chunks = chunkText(body, { size: 1000, overlap: 150 });
if (chunks.length === 0) return 0;
const vectors = await embedChunks(embedder, chunks, 16);
if (vectors.length !== chunks.length) {
throw new Error(
`chunk/embedding count mismatch for ${slug}: ${chunks.length} vs ${vectors.length}`,
);
}
// Replace existing chunks for this doc in one transaction.
await db.delete(documentChunks).where(eq(documentChunks.slug, slug));
await db.insert(documentChunks).values(
chunks.map((content: string, i: number) => ({
documentId: doc.id,
slug,
chunkIndex: i,
content,
embedding: vectors[i],
})),
);
return chunks.length;
}
+8 -1
View File
@@ -1 +1,8 @@
export { keywordSearch, toTsQuery } from "@mcpedia/search";
export {
keywordSearch,
semanticSearch,
hybridSearch,
toTsQuery,
cosine,
} from "@mcpedia/search";
export type { ChunkHit } from "@mcpedia/search";
@@ -0,0 +1,13 @@
CREATE TABLE "document_chunks" (
"id" uuid PRIMARY KEY DEFAULT gen_random_uuid() NOT NULL,
"document_id" text NOT NULL,
"slug" text NOT NULL,
"chunk_index" integer NOT NULL,
"content" text NOT NULL,
"embedding" real[],
"created_at" timestamp with time zone DEFAULT now() NOT NULL
);
--> statement-breakpoint
CREATE INDEX "document_chunks_slug_idx" ON "document_chunks" USING btree ("slug");
--> statement-breakpoint
ALTER TABLE "document_chunks" ADD CONSTRAINT "document_chunks_document_id_documents_id_fk" FOREIGN KEY ("document_id") REFERENCES "public"."documents"("id") ON DELETE cascade;
+260
View File
@@ -0,0 +1,260 @@
{
"id": "fd210ce3-31f0-434d-9e80-6f5a0bb07504",
"prevId": "249f80c0-d953-42f2-a98b-6701e2115856",
"version": "7",
"dialect": "postgresql",
"tables": {
"public.document_chunks": {
"name": "document_chunks",
"schema": "",
"columns": {
"id": {
"name": "id",
"type": "uuid",
"primaryKey": true,
"notNull": true,
"default": "gen_random_uuid()"
},
"document_id": {
"name": "document_id",
"type": "text",
"primaryKey": false,
"notNull": true
},
"slug": {
"name": "slug",
"type": "text",
"primaryKey": false,
"notNull": true
},
"chunk_index": {
"name": "chunk_index",
"type": "integer",
"primaryKey": false,
"notNull": true
},
"content": {
"name": "content",
"type": "text",
"primaryKey": false,
"notNull": true
},
"embedding": {
"name": "embedding",
"type": "vector(2048)",
"primaryKey": false,
"notNull": false
},
"created_at": {
"name": "created_at",
"type": "timestamp with time zone",
"primaryKey": false,
"notNull": true,
"default": "now()"
}
},
"indexes": {
"document_chunks_embedding_idx": {
"name": "document_chunks_embedding_idx",
"columns": [
{
"expression": "embedding",
"isExpression": false,
"asc": true,
"nulls": "last",
"opclass": "vector_cosine_ops"
}
],
"isUnique": false,
"concurrently": false,
"method": "hnsw",
"with": {}
},
"document_chunks_slug_idx": {
"name": "document_chunks_slug_idx",
"columns": [
{
"expression": "slug",
"isExpression": false,
"asc": true,
"nulls": "last"
}
],
"isUnique": false,
"concurrently": false,
"method": "btree",
"with": {}
}
},
"foreignKeys": {
"document_chunks_document_id_documents_id_fk": {
"name": "document_chunks_document_id_documents_id_fk",
"tableFrom": "document_chunks",
"tableTo": "documents",
"columnsFrom": [
"document_id"
],
"columnsTo": [
"id"
],
"onDelete": "cascade",
"onUpdate": "no action"
}
},
"compositePrimaryKeys": {},
"uniqueConstraints": {},
"policies": {},
"checkConstraints": {},
"isRLSEnabled": false
},
"public.documents": {
"name": "documents",
"schema": "",
"columns": {
"id": {
"name": "id",
"type": "text",
"primaryKey": true,
"notNull": true
},
"slug": {
"name": "slug",
"type": "text",
"primaryKey": false,
"notNull": true
},
"title": {
"name": "title",
"type": "text",
"primaryKey": false,
"notNull": true
},
"type": {
"name": "type",
"type": "text",
"primaryKey": false,
"notNull": true
},
"section": {
"name": "section",
"type": "text",
"primaryKey": false,
"notNull": true
},
"status": {
"name": "status",
"type": "text",
"primaryKey": false,
"notNull": true,
"default": "'published'"
},
"author": {
"name": "author",
"type": "text",
"primaryKey": false,
"notNull": true,
"default": "''"
},
"tags": {
"name": "tags",
"type": "text[]",
"primaryKey": false,
"notNull": true,
"default": "'{}'"
},
"path": {
"name": "path",
"type": "text",
"primaryKey": false,
"notNull": true
},
"body": {
"name": "body",
"type": "text",
"primaryKey": false,
"notNull": true,
"default": "''"
},
"search_vector": {
"name": "search_vector",
"type": "tsvector",
"primaryKey": false,
"notNull": true,
"generated": {
"as": "setweight(to_tsvector('simple', coalesce(\"documents\".\"title\", '')), 'A') || setweight(to_tsvector('simple', coalesce(\"documents\".\"body\", '')), 'B')",
"type": "stored"
}
},
"created_at": {
"name": "created_at",
"type": "timestamp with time zone",
"primaryKey": false,
"notNull": true
},
"updated_at": {
"name": "updated_at",
"type": "timestamp with time zone",
"primaryKey": false,
"notNull": true
}
},
"indexes": {
"documents_search_idx": {
"name": "documents_search_idx",
"columns": [
{
"expression": "search_vector",
"isExpression": false,
"asc": true,
"nulls": "last"
}
],
"isUnique": false,
"concurrently": false,
"method": "gin",
"with": {}
},
"documents_section_idx": {
"name": "documents_section_idx",
"columns": [
{
"expression": "section",
"isExpression": false,
"asc": true,
"nulls": "last"
}
],
"isUnique": false,
"concurrently": false,
"method": "btree",
"with": {}
}
},
"foreignKeys": {},
"compositePrimaryKeys": {},
"uniqueConstraints": {
"documents_slug_unique": {
"name": "documents_slug_unique",
"nullsNotDistinct": false,
"columns": [
"slug"
]
}
},
"policies": {},
"checkConstraints": {},
"isRLSEnabled": false
}
},
"enums": {},
"schemas": {},
"sequences": {},
"roles": {},
"policies": {},
"views": {},
"_meta": {
"columns": {},
"schemas": {},
"tables": {}
}
}
+8 -1
View File
@@ -8,6 +8,13 @@
"when": 1787133375079,
"tag": "0000_grey_toro",
"breakpoints": true
},
{
"idx": 1,
"version": "7",
"when": 1787137149735,
"tag": "0001_document_chunks",
"breakpoints": true
}
]
}
}
+1 -1
View File
@@ -4,7 +4,7 @@
"private": true,
"type": "module",
"exports": {
".": "./src/client.ts",
".": "./src/index.ts",
"./schema": "./src/schema.ts"
},
"dependencies": {
+2
View File
@@ -0,0 +1,2 @@
export { db, schema, client } from "./client";
export * from "./schema";
+30 -8
View File
@@ -4,8 +4,11 @@ import {
index,
integer,
pgTable,
real,
text,
timestamp,
uuid,
vector,
} from "drizzle-orm/pg-core";
// tsvector isn't a first-class drizzle type; wrap the raw Postgres type.
@@ -46,14 +49,33 @@ export const documents = pgTable(
}),
);
// Phase 2 (semantic search) — defined here for reference, NOT created yet:
// export const documentChunks = pgTable("document_chunks", {
// id: text("id").primaryKey(),
// documentId: text("document_id").notNull().references(() => documents.id, { onDelete: "cascade" }),
// content: text("content").notNull(),
// position: integer("position").notNull(),
// embedding: customType<{ data: number[] }>({ dataType: () => "vector(1536)" })("embedding"),
// });
// Phase 2: semantic search chunks. Each row is an embedded slice of a document
// body. `embedding` is a plain float array (real[]). We compute cosine
// similarity in the application layer — pgvector isn't available on the shared
// imrnes Postgres, and brute-force cosine is instant for a KB-sized corpus.
// (pgvector/HNSW is the Phase-4 scale-out path.)
export const documentChunks = pgTable(
"document_chunks",
{
id: uuid("id").primaryKey().defaultRandom(),
documentId: text("document_id")
.notNull()
.references(() => documents.id, { onDelete: "cascade" }),
slug: text("slug").notNull(),
chunkIndex: integer("chunk_index").notNull(),
content: text("content").notNull(),
embedding: real("embedding").array(),
createdAt: timestamp("created_at", { withTimezone: true })
.notNull()
.defaultNow(),
},
(t) => ({
slugIdx: index("document_chunks_slug_idx").on(t.slug),
}),
);
export type DocumentChunkRow = typeof documentChunks.$inferSelect;
export type NewDocumentChunkRow = typeof documentChunks.$inferInsert;
export type DocumentRow = typeof documents.$inferSelect;
export type NewDocumentRow = typeof documents.$inferInsert;
+17
View File
@@ -0,0 +1,17 @@
{
"name": "@mcpedia/embeddings",
"version": "0.1.0",
"private": true,
"type": "module",
"main": "./src/index.ts",
"exports": {
".": "./src/index.ts"
},
"dependencies": {
"@mcpedia/config": "workspace:*",
"@mcpedia/types": "workspace:*"
},
"devDependencies": {
"typescript": "^5.6.0"
}
}
+48
View File
@@ -0,0 +1,48 @@
import type { EmbeddingProvider } from "./provider";
/**
* Split text into overlapping chunks for embedding. Keeps paragraphs/words
* intact where possible; never splits a chunk mid-word by more than `overlap`.
*/
export function chunkText(
text: string,
opts: { size?: number; overlap?: number } = {},
): string[] {
const size = opts.size ?? 1000;
const overlap = opts.overlap ?? 150;
const clean = text.replace(/\r\n/g, "\n").trim();
if (!clean) return [];
if (clean.length <= size) return [clean];
const chunks: string[] = [];
let start = 0;
while (start < clean.length) {
let end = Math.min(start + size, clean.length);
// Prefer to break on a newline/space near the boundary.
if (end < clean.length) {
const nl = clean.lastIndexOf("\n", end);
const sp = clean.lastIndexOf(" ", end);
const breakAt = nl > start + size * 0.5 ? nl : sp > start + size * 0.5 ? sp : end;
if (breakAt > start) end = breakAt;
}
chunks.push(clean.slice(start, end).trim());
if (end >= clean.length) break;
start = Math.max(end - overlap, start + 1);
}
return chunks.filter(Boolean);
}
/** Embed a list of chunks in batches to avoid oversized requests. */
export async function embedChunks(
provider: EmbeddingProvider,
chunks: string[],
batchSize = 16,
): Promise<number[][]> {
const out: number[][] = [];
for (let i = 0; i < chunks.length; i += batchSize) {
const batch = chunks.slice(i, i + batchSize);
const vecs = await provider.embed(batch);
out.push(...vecs);
}
return out;
}
+2
View File
@@ -0,0 +1,2 @@
export * from "./provider";
export * from "./chunk";
+82
View File
@@ -0,0 +1,82 @@
import {
EMBED_API_KEY,
EMBED_BASE_URL,
EMBED_MODEL,
} from "@mcpedia/config";
export interface EmbeddingProvider {
/** Embed a batch of texts into vectors of fixed dimension. */
embed(texts: string[]): Promise<number[][]>;
readonly model: string;
readonly dimensions: number;
}
/** Pinned embedding dimension for the configured OpenRouter model. */
export const EMBED_DIM = 2048;
/**
* OpenRouter embeddings provider (we route through 9router's OpenAI-compatible
* /v1 endpoint). `encoding_format: "float"` is REQUIRED — the Nvidia-backed
* model rejects base64.
*/
export class OpenRouterEmbeddingProvider implements EmbeddingProvider {
readonly model: string;
private readonly baseUrl: string;
private readonly apiKey: string;
constructor(opts?: {
baseUrl?: string;
apiKey?: string;
model?: string;
}) {
this.baseUrl = (opts?.baseUrl ?? EMBED_BASE_URL).replace(/\/$/, "");
this.apiKey = opts?.apiKey ?? EMBED_API_KEY;
this.model = opts?.model ?? EMBED_MODEL;
if (!this.baseUrl || !this.apiKey || !this.model) {
throw new Error(
"OpenRouterEmbeddingProvider: missing EMBED_BASE_URL / EMBED_API_KEY / EMBED_MODEL",
);
}
}
get dimensions(): number {
return EMBED_DIM;
}
async embed(texts: string[]): Promise<number[][]> {
if (texts.length === 0) return [];
const res = await fetch(`${this.baseUrl}/embeddings`, {
method: "POST",
headers: {
"Content-Type": "application/json",
Authorization: `Bearer ${this.apiKey}`,
},
body: JSON.stringify({
model: this.model,
input: texts,
encoding_format: "float",
}),
});
if (!res.ok) {
const body = await res.text().catch(() => "");
throw new Error(
`embedding request failed (${res.status}): ${body.slice(0, 300)}`,
);
}
const json = (await res.json()) as {
data?: { embedding: number[] }[];
};
const data = json.data;
if (!data || data.length !== texts.length) {
throw new Error(
`embedding response mismatch: expected ${texts.length}, got ${data?.length ?? 0}`,
);
}
return data.map((d) => d.embedding);
}
}
/** Default singleton provider. */
export function createEmbeddingProvider(): EmbeddingProvider {
return new OpenRouterEmbeddingProvider();
}
+1
View File
@@ -8,6 +8,7 @@
},
"dependencies": {
"@mcpedia/db": "workspace:*",
"@mcpedia/embeddings": "workspace:*",
"@mcpedia/types": "workspace:*",
"drizzle-orm": "^0.38.0"
}
+112 -1
View File
@@ -1,6 +1,7 @@
import { db } from "@mcpedia/db";
import { documents, type DocumentRow } from "@mcpedia/db/schema";
import { documents, documentChunks, type DocumentRow } from "@mcpedia/db/schema";
import { and, eq, sql } from "drizzle-orm";
import { createEmbeddingProvider } from "@mcpedia/embeddings";
import type {
DocSection,
DocStatus,
@@ -9,6 +10,23 @@ import type {
SearchHit,
} from "@mcpedia/types";
const embedder = createEmbeddingProvider();
/** Cosine similarity between two equal-length vectors. */
export function cosine(a: number[], b: number[]): number {
if (a.length === 0 || a.length !== b.length) return 0;
let dot = 0;
let na = 0;
let nb = 0;
for (let i = 0; i < a.length; i++) {
dot += a[i] * b[i];
na += a[i] * a[i];
nb += b[i] * b[i];
}
const denom = Math.sqrt(na) * Math.sqrt(nb);
return denom === 0 ? 0 : dot / denom;
}
const VALID_SECTIONS: DocSection[] = ["docs", "writeups", "research", "notes"];
const VALID_TYPES: DocType[] = ["documentation", "writeup", "research", "note"];
@@ -43,6 +61,99 @@ export function toTsQuery(q: string): string {
return terms.map((t) => `${t}:*`).join(" & ");
}
export interface ChunkHit {
slug: string;
chunkIndex: number;
content: string;
score: number;
}
/**
* Semantic search: embed the query, then rank document chunks by cosine
* similarity. Cosine is computed in the app layer (pgvector isn't available on
* the shared imrnes Postgres); for a KB-sized corpus this is instant.
*/
export async function semanticSearch(q: string, limit = 10): Promise<ChunkHit[]> {
const query = q.trim();
if (!query) return [];
const [vec] = await embedder.embed([query]);
if (!vec || vec.length === 0) return [];
const rows = await db
.select({
slug: documentChunks.slug,
chunkIndex: documentChunks.chunkIndex,
content: documentChunks.content,
embedding: documentChunks.embedding,
})
.from(documentChunks)
.where(sql`${documentChunks.embedding} IS NOT NULL`);
return rows
.map((r) => ({
slug: r.slug,
chunkIndex: r.chunkIndex,
content: r.content,
score: cosine(vec, (r.embedding ?? []) as number[]),
}))
.filter((h) => h.score > 0)
.sort((a, b) => b.score - a.score)
.slice(0, limit);
}
/**
* Hybrid search: run FTS (ts_rank) and semantic (cosine) in parallel, then fuse
* with Reciprocal Rank Fusion (RRF, k=60). Returns merged document-level hits.
*/
export async function hybridSearch(q: string, limit = 10): Promise<SearchHit[]> {
const [fts, sem] = await Promise.all([keywordSearch(q, limit * 2), semanticSearch(q, limit * 2)]);
const k = 60;
const fused = new Map<string, { score: number; snippet: string; chunk: string }>();
fts.forEach((hit, i) => {
const rrf = 1 / (k + i + 1);
fused.set(hit.doc.slug, {
score: (fused.get(hit.doc.slug)?.score ?? 0) + rrf,
snippet: hit.snippet,
chunk: "",
});
});
sem.forEach((hit, i) => {
const rrf = 1 / (k + i + 1);
const prev = fused.get(hit.slug);
fused.set(hit.slug, {
score: (prev?.score ?? 0) + rrf,
snippet: prev?.snippet ?? hit.content.slice(0, 160),
chunk: prev?.chunk || hit.content,
});
});
const slugs = [...fused.entries()]
.sort((a, b) => b[1].score - a[1].score)
.slice(0, limit)
.map(([slug]) => slug);
if (slugs.length === 0) return [];
const rows = await db
.select()
.from(documents)
.where(and(eq(documents.status, "published"), sql`${documents.slug} IN ${slugs}`));
const bySlug = new Map(rows.map((r) => [r.slug, r]));
return slugs
.map((slug, i) => {
const row = bySlug.get(slug);
if (!row) return null;
const m = fused.get(slug)!;
return {
doc: toMeta(row),
rank: m.score,
snippet: m.snippet,
} as SearchHit;
})
.filter((x): x is SearchHit => x !== null);
}
/**
* Postgres FTS keyword search over published documents.
* Ranks by ts_rank and returns a headline snippet for display.