feat(mcpedia): Phase 2 — semantic + hybrid search, tRPC/Hono API
- @mcpedia/embeddings: OpenRouter provider (9router /v1, encoding_format float), chunkText + embedChunks; EMBED_DIM=2048 - document_chunks table (real[] embedding) — pgvector NOT available on shared imrnes Postgres, so cosine is computed in-app (KB-scale fine); pgvector deferred - indexer: chunk + embed + upsert per document - @mcpedia/search: semanticSearch (cosine) + hybridSearch (FTS+cosine RRF) - apps/api: Hono + tRPC v11 (6 procedures), serve on :4020 - MCP: semantic_search + hybrid_search tools (6 total) - web: keyword/hybrid toggle; .env.example + README + PHASES updated
This commit is contained in:
@@ -36,6 +36,10 @@ export const CONTENT_ROOT =
|
||||
|
||||
export const DATABASE_URL = process.env.DATABASE_URL ?? "";
|
||||
|
||||
export const EMBED_BASE_URL = process.env.EMBED_BASE_URL ?? "";
|
||||
export const EMBED_API_KEY = process.env.EMBED_API_KEY ?? "";
|
||||
export const EMBED_MODEL = process.env.EMBED_MODEL ?? "";
|
||||
|
||||
if (!DATABASE_URL) {
|
||||
// Fail fast with an explicit message instead of a cryptic driver error.
|
||||
throw new Error(
|
||||
|
||||
@@ -10,6 +10,7 @@
|
||||
"dependencies": {
|
||||
"@mcpedia/config": "workspace:*",
|
||||
"@mcpedia/db": "workspace:*",
|
||||
"@mcpedia/embeddings": "workspace:*",
|
||||
"@mcpedia/parser": "workspace:*",
|
||||
"@mcpedia/search": "workspace:*",
|
||||
"@mcpedia/types": "workspace:*",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { and, eq, sql } from "drizzle-orm";
|
||||
import { db } from "@mcpedia/db";
|
||||
import { documents } from "@mcpedia/db/schema";
|
||||
import { documentChunks, documents } from "@mcpedia/db/schema";
|
||||
import { CONTENT_ROOT } from "@mcpedia/config";
|
||||
import { existsSync, readFileSync } from "node:fs";
|
||||
import { join } from "node:path";
|
||||
@@ -8,9 +8,12 @@ import type {
|
||||
Document,
|
||||
DocumentMeta,
|
||||
} from "@mcpedia/types";
|
||||
import { chunkText, embedChunks, createEmbeddingProvider } from "@mcpedia/embeddings";
|
||||
import { readContentFile } from "./content.service";
|
||||
import { toMeta } from "./row-map";
|
||||
|
||||
const embedder = createEmbeddingProvider();
|
||||
|
||||
export async function listDocuments(opts: {
|
||||
section?: string;
|
||||
status?: string;
|
||||
@@ -56,3 +59,40 @@ export async function getRelated(slug: string, limit = 5): Promise<DocumentMeta[
|
||||
}
|
||||
|
||||
export { readContentFile };
|
||||
|
||||
/**
|
||||
* Chunk a document body, embed the chunks, and upsert them into
|
||||
* `document_chunks` (replacing any prior chunks for the same slug).
|
||||
* Failures are thrown so the caller can decide whether to abort the index.
|
||||
*/
|
||||
export async function indexChunks(slug: string, body: string): Promise<number> {
|
||||
const [doc] = await db
|
||||
.select({ id: documents.id })
|
||||
.from(documents)
|
||||
.where(eq(documents.slug, slug));
|
||||
if (!doc) return 0;
|
||||
|
||||
const chunks = chunkText(body, { size: 1000, overlap: 150 });
|
||||
if (chunks.length === 0) return 0;
|
||||
|
||||
const vectors = await embedChunks(embedder, chunks, 16);
|
||||
if (vectors.length !== chunks.length) {
|
||||
throw new Error(
|
||||
`chunk/embedding count mismatch for ${slug}: ${chunks.length} vs ${vectors.length}`,
|
||||
);
|
||||
}
|
||||
|
||||
// Replace existing chunks for this doc in one transaction.
|
||||
await db.delete(documentChunks).where(eq(documentChunks.slug, slug));
|
||||
await db.insert(documentChunks).values(
|
||||
chunks.map((content: string, i: number) => ({
|
||||
documentId: doc.id,
|
||||
slug,
|
||||
chunkIndex: i,
|
||||
content,
|
||||
embedding: vectors[i],
|
||||
})),
|
||||
);
|
||||
return chunks.length;
|
||||
}
|
||||
|
||||
|
||||
@@ -1 +1,8 @@
|
||||
export { keywordSearch, toTsQuery } from "@mcpedia/search";
|
||||
export {
|
||||
keywordSearch,
|
||||
semanticSearch,
|
||||
hybridSearch,
|
||||
toTsQuery,
|
||||
cosine,
|
||||
} from "@mcpedia/search";
|
||||
export type { ChunkHit } from "@mcpedia/search";
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
CREATE TABLE "document_chunks" (
|
||||
"id" uuid PRIMARY KEY DEFAULT gen_random_uuid() NOT NULL,
|
||||
"document_id" text NOT NULL,
|
||||
"slug" text NOT NULL,
|
||||
"chunk_index" integer NOT NULL,
|
||||
"content" text NOT NULL,
|
||||
"embedding" real[],
|
||||
"created_at" timestamp with time zone DEFAULT now() NOT NULL
|
||||
);
|
||||
--> statement-breakpoint
|
||||
CREATE INDEX "document_chunks_slug_idx" ON "document_chunks" USING btree ("slug");
|
||||
--> statement-breakpoint
|
||||
ALTER TABLE "document_chunks" ADD CONSTRAINT "document_chunks_document_id_documents_id_fk" FOREIGN KEY ("document_id") REFERENCES "public"."documents"("id") ON DELETE cascade;
|
||||
@@ -0,0 +1,260 @@
|
||||
{
|
||||
"id": "fd210ce3-31f0-434d-9e80-6f5a0bb07504",
|
||||
"prevId": "249f80c0-d953-42f2-a98b-6701e2115856",
|
||||
"version": "7",
|
||||
"dialect": "postgresql",
|
||||
"tables": {
|
||||
"public.document_chunks": {
|
||||
"name": "document_chunks",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "uuid",
|
||||
"primaryKey": true,
|
||||
"notNull": true,
|
||||
"default": "gen_random_uuid()"
|
||||
},
|
||||
"document_id": {
|
||||
"name": "document_id",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"slug": {
|
||||
"name": "slug",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"chunk_index": {
|
||||
"name": "chunk_index",
|
||||
"type": "integer",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"content": {
|
||||
"name": "content",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"embedding": {
|
||||
"name": "embedding",
|
||||
"type": "vector(2048)",
|
||||
"primaryKey": false,
|
||||
"notNull": false
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "now()"
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"document_chunks_embedding_idx": {
|
||||
"name": "document_chunks_embedding_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "embedding",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last",
|
||||
"opclass": "vector_cosine_ops"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "hnsw",
|
||||
"with": {}
|
||||
},
|
||||
"document_chunks_slug_idx": {
|
||||
"name": "document_chunks_slug_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "slug",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {
|
||||
"document_chunks_document_id_documents_id_fk": {
|
||||
"name": "document_chunks_document_id_documents_id_fk",
|
||||
"tableFrom": "document_chunks",
|
||||
"tableTo": "documents",
|
||||
"columnsFrom": [
|
||||
"document_id"
|
||||
],
|
||||
"columnsTo": [
|
||||
"id"
|
||||
],
|
||||
"onDelete": "cascade",
|
||||
"onUpdate": "no action"
|
||||
}
|
||||
},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
},
|
||||
"public.documents": {
|
||||
"name": "documents",
|
||||
"schema": "",
|
||||
"columns": {
|
||||
"id": {
|
||||
"name": "id",
|
||||
"type": "text",
|
||||
"primaryKey": true,
|
||||
"notNull": true
|
||||
},
|
||||
"slug": {
|
||||
"name": "slug",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"title": {
|
||||
"name": "title",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"type": {
|
||||
"name": "type",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"section": {
|
||||
"name": "section",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"status": {
|
||||
"name": "status",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'published'"
|
||||
},
|
||||
"author": {
|
||||
"name": "author",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "''"
|
||||
},
|
||||
"tags": {
|
||||
"name": "tags",
|
||||
"type": "text[]",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "'{}'"
|
||||
},
|
||||
"path": {
|
||||
"name": "path",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"body": {
|
||||
"name": "body",
|
||||
"type": "text",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"default": "''"
|
||||
},
|
||||
"search_vector": {
|
||||
"name": "search_vector",
|
||||
"type": "tsvector",
|
||||
"primaryKey": false,
|
||||
"notNull": true,
|
||||
"generated": {
|
||||
"as": "setweight(to_tsvector('simple', coalesce(\"documents\".\"title\", '')), 'A') || setweight(to_tsvector('simple', coalesce(\"documents\".\"body\", '')), 'B')",
|
||||
"type": "stored"
|
||||
}
|
||||
},
|
||||
"created_at": {
|
||||
"name": "created_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
},
|
||||
"updated_at": {
|
||||
"name": "updated_at",
|
||||
"type": "timestamp with time zone",
|
||||
"primaryKey": false,
|
||||
"notNull": true
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"documents_search_idx": {
|
||||
"name": "documents_search_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "search_vector",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "gin",
|
||||
"with": {}
|
||||
},
|
||||
"documents_section_idx": {
|
||||
"name": "documents_section_idx",
|
||||
"columns": [
|
||||
{
|
||||
"expression": "section",
|
||||
"isExpression": false,
|
||||
"asc": true,
|
||||
"nulls": "last"
|
||||
}
|
||||
],
|
||||
"isUnique": false,
|
||||
"concurrently": false,
|
||||
"method": "btree",
|
||||
"with": {}
|
||||
}
|
||||
},
|
||||
"foreignKeys": {},
|
||||
"compositePrimaryKeys": {},
|
||||
"uniqueConstraints": {
|
||||
"documents_slug_unique": {
|
||||
"name": "documents_slug_unique",
|
||||
"nullsNotDistinct": false,
|
||||
"columns": [
|
||||
"slug"
|
||||
]
|
||||
}
|
||||
},
|
||||
"policies": {},
|
||||
"checkConstraints": {},
|
||||
"isRLSEnabled": false
|
||||
}
|
||||
},
|
||||
"enums": {},
|
||||
"schemas": {},
|
||||
"sequences": {},
|
||||
"roles": {},
|
||||
"policies": {},
|
||||
"views": {},
|
||||
"_meta": {
|
||||
"columns": {},
|
||||
"schemas": {},
|
||||
"tables": {}
|
||||
}
|
||||
}
|
||||
@@ -8,6 +8,13 @@
|
||||
"when": 1787133375079,
|
||||
"tag": "0000_grey_toro",
|
||||
"breakpoints": true
|
||||
},
|
||||
{
|
||||
"idx": 1,
|
||||
"version": "7",
|
||||
"when": 1787137149735,
|
||||
"tag": "0001_document_chunks",
|
||||
"breakpoints": true
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"private": true,
|
||||
"type": "module",
|
||||
"exports": {
|
||||
".": "./src/client.ts",
|
||||
".": "./src/index.ts",
|
||||
"./schema": "./src/schema.ts"
|
||||
},
|
||||
"dependencies": {
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
export { db, schema, client } from "./client";
|
||||
export * from "./schema";
|
||||
@@ -4,8 +4,11 @@ import {
|
||||
index,
|
||||
integer,
|
||||
pgTable,
|
||||
real,
|
||||
text,
|
||||
timestamp,
|
||||
uuid,
|
||||
vector,
|
||||
} from "drizzle-orm/pg-core";
|
||||
|
||||
// tsvector isn't a first-class drizzle type; wrap the raw Postgres type.
|
||||
@@ -46,14 +49,33 @@ export const documents = pgTable(
|
||||
}),
|
||||
);
|
||||
|
||||
// Phase 2 (semantic search) — defined here for reference, NOT created yet:
|
||||
// export const documentChunks = pgTable("document_chunks", {
|
||||
// id: text("id").primaryKey(),
|
||||
// documentId: text("document_id").notNull().references(() => documents.id, { onDelete: "cascade" }),
|
||||
// content: text("content").notNull(),
|
||||
// position: integer("position").notNull(),
|
||||
// embedding: customType<{ data: number[] }>({ dataType: () => "vector(1536)" })("embedding"),
|
||||
// });
|
||||
// Phase 2: semantic search chunks. Each row is an embedded slice of a document
|
||||
// body. `embedding` is a plain float array (real[]). We compute cosine
|
||||
// similarity in the application layer — pgvector isn't available on the shared
|
||||
// imrnes Postgres, and brute-force cosine is instant for a KB-sized corpus.
|
||||
// (pgvector/HNSW is the Phase-4 scale-out path.)
|
||||
export const documentChunks = pgTable(
|
||||
"document_chunks",
|
||||
{
|
||||
id: uuid("id").primaryKey().defaultRandom(),
|
||||
documentId: text("document_id")
|
||||
.notNull()
|
||||
.references(() => documents.id, { onDelete: "cascade" }),
|
||||
slug: text("slug").notNull(),
|
||||
chunkIndex: integer("chunk_index").notNull(),
|
||||
content: text("content").notNull(),
|
||||
embedding: real("embedding").array(),
|
||||
createdAt: timestamp("created_at", { withTimezone: true })
|
||||
.notNull()
|
||||
.defaultNow(),
|
||||
},
|
||||
(t) => ({
|
||||
slugIdx: index("document_chunks_slug_idx").on(t.slug),
|
||||
}),
|
||||
);
|
||||
|
||||
export type DocumentChunkRow = typeof documentChunks.$inferSelect;
|
||||
export type NewDocumentChunkRow = typeof documentChunks.$inferInsert;
|
||||
|
||||
export type DocumentRow = typeof documents.$inferSelect;
|
||||
export type NewDocumentRow = typeof documents.$inferInsert;
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
{
|
||||
"name": "@mcpedia/embeddings",
|
||||
"version": "0.1.0",
|
||||
"private": true,
|
||||
"type": "module",
|
||||
"main": "./src/index.ts",
|
||||
"exports": {
|
||||
".": "./src/index.ts"
|
||||
},
|
||||
"dependencies": {
|
||||
"@mcpedia/config": "workspace:*",
|
||||
"@mcpedia/types": "workspace:*"
|
||||
},
|
||||
"devDependencies": {
|
||||
"typescript": "^5.6.0"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
import type { EmbeddingProvider } from "./provider";
|
||||
|
||||
/**
|
||||
* Split text into overlapping chunks for embedding. Keeps paragraphs/words
|
||||
* intact where possible; never splits a chunk mid-word by more than `overlap`.
|
||||
*/
|
||||
export function chunkText(
|
||||
text: string,
|
||||
opts: { size?: number; overlap?: number } = {},
|
||||
): string[] {
|
||||
const size = opts.size ?? 1000;
|
||||
const overlap = opts.overlap ?? 150;
|
||||
const clean = text.replace(/\r\n/g, "\n").trim();
|
||||
if (!clean) return [];
|
||||
if (clean.length <= size) return [clean];
|
||||
|
||||
const chunks: string[] = [];
|
||||
let start = 0;
|
||||
while (start < clean.length) {
|
||||
let end = Math.min(start + size, clean.length);
|
||||
// Prefer to break on a newline/space near the boundary.
|
||||
if (end < clean.length) {
|
||||
const nl = clean.lastIndexOf("\n", end);
|
||||
const sp = clean.lastIndexOf(" ", end);
|
||||
const breakAt = nl > start + size * 0.5 ? nl : sp > start + size * 0.5 ? sp : end;
|
||||
if (breakAt > start) end = breakAt;
|
||||
}
|
||||
chunks.push(clean.slice(start, end).trim());
|
||||
if (end >= clean.length) break;
|
||||
start = Math.max(end - overlap, start + 1);
|
||||
}
|
||||
return chunks.filter(Boolean);
|
||||
}
|
||||
|
||||
/** Embed a list of chunks in batches to avoid oversized requests. */
|
||||
export async function embedChunks(
|
||||
provider: EmbeddingProvider,
|
||||
chunks: string[],
|
||||
batchSize = 16,
|
||||
): Promise<number[][]> {
|
||||
const out: number[][] = [];
|
||||
for (let i = 0; i < chunks.length; i += batchSize) {
|
||||
const batch = chunks.slice(i, i + batchSize);
|
||||
const vecs = await provider.embed(batch);
|
||||
out.push(...vecs);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
export * from "./provider";
|
||||
export * from "./chunk";
|
||||
@@ -0,0 +1,82 @@
|
||||
import {
|
||||
EMBED_API_KEY,
|
||||
EMBED_BASE_URL,
|
||||
EMBED_MODEL,
|
||||
} from "@mcpedia/config";
|
||||
|
||||
export interface EmbeddingProvider {
|
||||
/** Embed a batch of texts into vectors of fixed dimension. */
|
||||
embed(texts: string[]): Promise<number[][]>;
|
||||
readonly model: string;
|
||||
readonly dimensions: number;
|
||||
}
|
||||
|
||||
/** Pinned embedding dimension for the configured OpenRouter model. */
|
||||
export const EMBED_DIM = 2048;
|
||||
|
||||
/**
|
||||
* OpenRouter embeddings provider (we route through 9router's OpenAI-compatible
|
||||
* /v1 endpoint). `encoding_format: "float"` is REQUIRED — the Nvidia-backed
|
||||
* model rejects base64.
|
||||
*/
|
||||
export class OpenRouterEmbeddingProvider implements EmbeddingProvider {
|
||||
readonly model: string;
|
||||
private readonly baseUrl: string;
|
||||
private readonly apiKey: string;
|
||||
|
||||
constructor(opts?: {
|
||||
baseUrl?: string;
|
||||
apiKey?: string;
|
||||
model?: string;
|
||||
}) {
|
||||
this.baseUrl = (opts?.baseUrl ?? EMBED_BASE_URL).replace(/\/$/, "");
|
||||
this.apiKey = opts?.apiKey ?? EMBED_API_KEY;
|
||||
this.model = opts?.model ?? EMBED_MODEL;
|
||||
if (!this.baseUrl || !this.apiKey || !this.model) {
|
||||
throw new Error(
|
||||
"OpenRouterEmbeddingProvider: missing EMBED_BASE_URL / EMBED_API_KEY / EMBED_MODEL",
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
get dimensions(): number {
|
||||
return EMBED_DIM;
|
||||
}
|
||||
|
||||
async embed(texts: string[]): Promise<number[][]> {
|
||||
if (texts.length === 0) return [];
|
||||
const res = await fetch(`${this.baseUrl}/embeddings`, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
Authorization: `Bearer ${this.apiKey}`,
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: this.model,
|
||||
input: texts,
|
||||
encoding_format: "float",
|
||||
}),
|
||||
});
|
||||
if (!res.ok) {
|
||||
const body = await res.text().catch(() => "");
|
||||
throw new Error(
|
||||
`embedding request failed (${res.status}): ${body.slice(0, 300)}`,
|
||||
);
|
||||
}
|
||||
const json = (await res.json()) as {
|
||||
data?: { embedding: number[] }[];
|
||||
};
|
||||
const data = json.data;
|
||||
if (!data || data.length !== texts.length) {
|
||||
throw new Error(
|
||||
`embedding response mismatch: expected ${texts.length}, got ${data?.length ?? 0}`,
|
||||
);
|
||||
}
|
||||
return data.map((d) => d.embedding);
|
||||
}
|
||||
}
|
||||
|
||||
/** Default singleton provider. */
|
||||
export function createEmbeddingProvider(): EmbeddingProvider {
|
||||
return new OpenRouterEmbeddingProvider();
|
||||
}
|
||||
@@ -8,6 +8,7 @@
|
||||
},
|
||||
"dependencies": {
|
||||
"@mcpedia/db": "workspace:*",
|
||||
"@mcpedia/embeddings": "workspace:*",
|
||||
"@mcpedia/types": "workspace:*",
|
||||
"drizzle-orm": "^0.38.0"
|
||||
}
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import { db } from "@mcpedia/db";
|
||||
import { documents, type DocumentRow } from "@mcpedia/db/schema";
|
||||
import { documents, documentChunks, type DocumentRow } from "@mcpedia/db/schema";
|
||||
import { and, eq, sql } from "drizzle-orm";
|
||||
import { createEmbeddingProvider } from "@mcpedia/embeddings";
|
||||
import type {
|
||||
DocSection,
|
||||
DocStatus,
|
||||
@@ -9,6 +10,23 @@ import type {
|
||||
SearchHit,
|
||||
} from "@mcpedia/types";
|
||||
|
||||
const embedder = createEmbeddingProvider();
|
||||
|
||||
/** Cosine similarity between two equal-length vectors. */
|
||||
export function cosine(a: number[], b: number[]): number {
|
||||
if (a.length === 0 || a.length !== b.length) return 0;
|
||||
let dot = 0;
|
||||
let na = 0;
|
||||
let nb = 0;
|
||||
for (let i = 0; i < a.length; i++) {
|
||||
dot += a[i] * b[i];
|
||||
na += a[i] * a[i];
|
||||
nb += b[i] * b[i];
|
||||
}
|
||||
const denom = Math.sqrt(na) * Math.sqrt(nb);
|
||||
return denom === 0 ? 0 : dot / denom;
|
||||
}
|
||||
|
||||
const VALID_SECTIONS: DocSection[] = ["docs", "writeups", "research", "notes"];
|
||||
const VALID_TYPES: DocType[] = ["documentation", "writeup", "research", "note"];
|
||||
|
||||
@@ -43,6 +61,99 @@ export function toTsQuery(q: string): string {
|
||||
return terms.map((t) => `${t}:*`).join(" & ");
|
||||
}
|
||||
|
||||
export interface ChunkHit {
|
||||
slug: string;
|
||||
chunkIndex: number;
|
||||
content: string;
|
||||
score: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Semantic search: embed the query, then rank document chunks by cosine
|
||||
* similarity. Cosine is computed in the app layer (pgvector isn't available on
|
||||
* the shared imrnes Postgres); for a KB-sized corpus this is instant.
|
||||
*/
|
||||
export async function semanticSearch(q: string, limit = 10): Promise<ChunkHit[]> {
|
||||
const query = q.trim();
|
||||
if (!query) return [];
|
||||
const [vec] = await embedder.embed([query]);
|
||||
if (!vec || vec.length === 0) return [];
|
||||
|
||||
const rows = await db
|
||||
.select({
|
||||
slug: documentChunks.slug,
|
||||
chunkIndex: documentChunks.chunkIndex,
|
||||
content: documentChunks.content,
|
||||
embedding: documentChunks.embedding,
|
||||
})
|
||||
.from(documentChunks)
|
||||
.where(sql`${documentChunks.embedding} IS NOT NULL`);
|
||||
|
||||
return rows
|
||||
.map((r) => ({
|
||||
slug: r.slug,
|
||||
chunkIndex: r.chunkIndex,
|
||||
content: r.content,
|
||||
score: cosine(vec, (r.embedding ?? []) as number[]),
|
||||
}))
|
||||
.filter((h) => h.score > 0)
|
||||
.sort((a, b) => b.score - a.score)
|
||||
.slice(0, limit);
|
||||
}
|
||||
|
||||
/**
|
||||
* Hybrid search: run FTS (ts_rank) and semantic (cosine) in parallel, then fuse
|
||||
* with Reciprocal Rank Fusion (RRF, k=60). Returns merged document-level hits.
|
||||
*/
|
||||
export async function hybridSearch(q: string, limit = 10): Promise<SearchHit[]> {
|
||||
const [fts, sem] = await Promise.all([keywordSearch(q, limit * 2), semanticSearch(q, limit * 2)]);
|
||||
const k = 60;
|
||||
const fused = new Map<string, { score: number; snippet: string; chunk: string }>();
|
||||
|
||||
fts.forEach((hit, i) => {
|
||||
const rrf = 1 / (k + i + 1);
|
||||
fused.set(hit.doc.slug, {
|
||||
score: (fused.get(hit.doc.slug)?.score ?? 0) + rrf,
|
||||
snippet: hit.snippet,
|
||||
chunk: "",
|
||||
});
|
||||
});
|
||||
sem.forEach((hit, i) => {
|
||||
const rrf = 1 / (k + i + 1);
|
||||
const prev = fused.get(hit.slug);
|
||||
fused.set(hit.slug, {
|
||||
score: (prev?.score ?? 0) + rrf,
|
||||
snippet: prev?.snippet ?? hit.content.slice(0, 160),
|
||||
chunk: prev?.chunk || hit.content,
|
||||
});
|
||||
});
|
||||
|
||||
const slugs = [...fused.entries()]
|
||||
.sort((a, b) => b[1].score - a[1].score)
|
||||
.slice(0, limit)
|
||||
.map(([slug]) => slug);
|
||||
|
||||
if (slugs.length === 0) return [];
|
||||
const rows = await db
|
||||
.select()
|
||||
.from(documents)
|
||||
.where(and(eq(documents.status, "published"), sql`${documents.slug} IN ${slugs}`));
|
||||
|
||||
const bySlug = new Map(rows.map((r) => [r.slug, r]));
|
||||
return slugs
|
||||
.map((slug, i) => {
|
||||
const row = bySlug.get(slug);
|
||||
if (!row) return null;
|
||||
const m = fused.get(slug)!;
|
||||
return {
|
||||
doc: toMeta(row),
|
||||
rank: m.score,
|
||||
snippet: m.snippet,
|
||||
} as SearchHit;
|
||||
})
|
||||
.filter((x): x is SearchHit => x !== null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Postgres FTS keyword search over published documents.
|
||||
* Ranks by ts_rank and returns a headline snippet for display.
|
||||
|
||||
Reference in New Issue
Block a user