chore(testing): Phase 9 — bun:test suite + CI gating with no-DB fakes
CI / typecheck + build (turbo) (push) Canceled after 0s
CI / typecheck + build (turbo) (push) Canceled after 0s
Added a real test suite (32 tests, 0 external services) using bun:test with in-process module mocking for @mcpedia/db, @mcpedia/queue, @mcpedia/core. Enablers: - apps/api: extracted createApp(deps?) factory + dashboard.ts module from index.ts so the HTTP surface is unit-testable (real queue is lazy-imported). - packages/core: exported shouldCreateRevision pure predicate; restoreRevision gained an opts.reindex seam for the chunk-rebuild contract. - apps/mcp: renamed smoke.test.ts -> smoke.ts (bun test now owns .test.ts), updated stale assertions (10 tools, 4 docs in docs section). - infra: turbo test task (cache:false), test scripts across packages, @types/bun + tsconfig base types, CI 'Test' step after Build. Packages with tests: embeddings(5), parser(5), search(8), core(4), mcp(6 auth-gates), api(8 contracts). All green: typecheck(4/4), test(6/6 pkgs), build(web). Live API verified /health, /metrics, /dashboard, /hooks/* auth gate on temp port.
This commit is contained in:
@@ -15,5 +15,8 @@
|
||||
"@mcpedia/search": "workspace:*",
|
||||
"@mcpedia/types": "workspace:*",
|
||||
"drizzle-orm": "^0.38.0"
|
||||
},
|
||||
"scripts": {
|
||||
"test": "bun test"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
import { test, expect } from "bun:test";
|
||||
import { shouldCreateRevision } from "../src/index.service";
|
||||
|
||||
/**
|
||||
* Phase 9: unit tests for the revision-dedup decision rule.
|
||||
*
|
||||
* `shouldCreateRevision` is the pure predicate that `indexContentFile` consults
|
||||
* before writing a new row to `document_revisions`. It's extracted because the
|
||||
* dedup correctness is the single most important guarantee of the revision
|
||||
* system ("metadata-only edits don't bloat history"), and it must hold without
|
||||
* a database.
|
||||
*
|
||||
* The DB-backed paths (`snapshotRevision`, `restoreRevision`) are exercised
|
||||
* end-to-end by the existing manual e2e (`bun run index` + restore via the web
|
||||
* /api/revisions/restore route, see PHASES.md Phase 4 verification). Here we
|
||||
* lock the decision invariant in CI.
|
||||
*/
|
||||
|
||||
test("shouldCreateRevision: first snapshot when no prior revision exists", () => {
|
||||
// No prior revision row → always snapshot the first version.
|
||||
expect(shouldCreateRevision(null, "body text")).toBe(true);
|
||||
expect(shouldCreateRevision(undefined, "body text")).toBe(true);
|
||||
});
|
||||
|
||||
test("shouldCreateRevision: identical body creates no new revision (dedup)", () => {
|
||||
// The exact dedup rule that prevents metadata-only edits from bloating
|
||||
// history: if the body is byte-identical to the latest revision's body,
|
||||
// skip the snapshot.
|
||||
expect(shouldCreateRevision("same body", "same body")).toBe(false);
|
||||
// The dedup rule applies regardless of body length — large identical bodies
|
||||
// also skip the snapshot.
|
||||
expect(shouldCreateRevision("a".repeat(5000), "a".repeat(5000))).toBe(false);
|
||||
});
|
||||
|
||||
test("shouldCreateRevision: changed body creates a new revision", () => {
|
||||
expect(shouldCreateRevision("old body", "new body")).toBe(true);
|
||||
// Whitespace / trailing newline changes count as a real body change.
|
||||
expect(shouldCreateRevision("body", "body\n")).toBe(true);
|
||||
});
|
||||
|
||||
test("shouldCreateRevision: empty-string vs non-empty counts as a change", () => {
|
||||
expect(shouldCreateRevision("", "content")).toBe(true);
|
||||
expect(shouldCreateRevision("content", "")).toBe(true);
|
||||
});
|
||||
@@ -84,6 +84,22 @@ export async function indexContentFile(
|
||||
return { indexed: true, chunks, revision };
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure decision rule for the revision system: create a new revision only when
|
||||
* the body genuinely changed vs the latest snapshot.
|
||||
* - no prior revision (latestBody null) -> true (first snapshot)
|
||||
* - identical body -> false (no noise)
|
||||
* - different body -> true
|
||||
*
|
||||
* Exported separately so it can be unit-tested without a database.
|
||||
*/
|
||||
export function shouldCreateRevision(
|
||||
latestBody: string | null | undefined,
|
||||
body: string,
|
||||
): boolean {
|
||||
return latestBody == null || latestBody !== body;
|
||||
}
|
||||
|
||||
/**
|
||||
* Compare the incoming body against the latest revision's body; if different
|
||||
* (or no prior revision exists), create a new revision with an incremented
|
||||
|
||||
@@ -76,9 +76,18 @@ export async function getRevision(
|
||||
};
|
||||
}
|
||||
|
||||
/** Restore a revision: write its body+metadata back into the live `documents` row. */
|
||||
/**
|
||||
* Restore a revision: write its body+metadata back into the live `documents` row.
|
||||
*
|
||||
* @param id revision UUID
|
||||
* @param opts optional seam for testing — override the chunk-rebuild step so
|
||||
* tests can assert it's invoked without touching embeddings.
|
||||
*/
|
||||
export async function restoreRevision(
|
||||
id: string,
|
||||
opts?: {
|
||||
reindex?: (slug: string) => Promise<number>;
|
||||
},
|
||||
): Promise<{ slug: string; documentId: string } | null> {
|
||||
const [rev] = await db
|
||||
.select({
|
||||
@@ -118,7 +127,8 @@ export async function restoreRevision(
|
||||
// Rebuild semantic chunks + embeddings from the restored body so semantic
|
||||
// and hybrid search stay consistent (otherwise document_chunks would hold
|
||||
// the NEW body's chunks while documents.body holds the OLD/restore body).
|
||||
await reindexChunks(rev.slug);
|
||||
const reindex = opts?.reindex ?? reindexChunks;
|
||||
await reindex(rev.slug);
|
||||
|
||||
return { slug: rev.slug, documentId: rev.documentId };
|
||||
}
|
||||
|
||||
@@ -13,5 +13,8 @@
|
||||
},
|
||||
"devDependencies": {
|
||||
"typescript": "^5.6.0"
|
||||
},
|
||||
"scripts": {
|
||||
"test": "bun test"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,57 @@
|
||||
import { test, expect } from "bun:test";
|
||||
import { chunkText } from "../src/chunk";
|
||||
|
||||
test("empty / whitespace input returns empty array", () => {
|
||||
expect(chunkText("")).toEqual([]);
|
||||
expect(chunkText(" \n ")).toEqual([]);
|
||||
});
|
||||
|
||||
test("short text (<= size) returns a single chunk", () => {
|
||||
const text = "hello world this is short";
|
||||
const chunks = chunkText(text, { size: 1000, overlap: 150 });
|
||||
expect(chunks).toHaveLength(1);
|
||||
expect(chunks[0]).toBe(text);
|
||||
});
|
||||
|
||||
test("long text splits into multiple chunks with overlap honored", () => {
|
||||
// Build ~3000 chars of words so we get >1 chunk at default size 1000.
|
||||
const word = "lorem";
|
||||
const text = Array.from({ length: 600 }, () => word).join(" ");
|
||||
const chunks = chunkText(text, { size: 1000, overlap: 150 });
|
||||
expect(chunks.length).toBeGreaterThan(1);
|
||||
|
||||
// Every chunk must respect the size upper bound (trimmed).
|
||||
for (const c of chunks) {
|
||||
expect(c.length).toBeLessThanOrEqual(1000);
|
||||
}
|
||||
|
||||
// The overlap region: second chunk should start near the end of the first
|
||||
// minus the overlap window. We just assert they share some suffix/prefix
|
||||
// overlap roughly, i.e. the join doesn't lose content boundaries badly.
|
||||
const joined = chunks.join(" ");
|
||||
// Most words are preserved across the split (at least the bulk).
|
||||
expect(joined.length).toBeGreaterThan(text.length * 0.9);
|
||||
});
|
||||
|
||||
test("chunkText never splits a chunk mid-word past the boundary (no truncation mid-token)", () => {
|
||||
const text = "alpha beta gamma delta epsilon zeta eta theta iota kappa lambda mu nu xi";
|
||||
const chunks = chunkText(text, { size: 20, overlap: 4 });
|
||||
// No chunk should contain a partial word boundary that corrupts tokens —
|
||||
// i.e. every resulting piece still reassembles into the original words set.
|
||||
const reassembled = chunks
|
||||
.flatMap((c) => c.split(/\s+/))
|
||||
.filter(Boolean)
|
||||
.sort();
|
||||
const original = text.split(/\s+/).sort();
|
||||
// Overlap means some words repeat — assert all original words are present.
|
||||
for (const w of original) {
|
||||
expect(reassembled).toContain(w);
|
||||
}
|
||||
});
|
||||
|
||||
test("default options produce reasonable chunking", () => {
|
||||
const text = "x".repeat(2500);
|
||||
const chunks = chunkText(text); // defaults: size 1000, overlap 150
|
||||
expect(chunks.length).toBeGreaterThanOrEqual(2);
|
||||
expect(chunks[chunks.length - 1].length).toBeLessThanOrEqual(1000);
|
||||
});
|
||||
@@ -9,5 +9,8 @@
|
||||
"dependencies": {
|
||||
"@mcpedia/types": "workspace:*",
|
||||
"gray-matter": "^4.0.3"
|
||||
},
|
||||
"scripts": {
|
||||
"test": "bun test"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
import { test, expect, afterEach, beforeEach } from "bun:test";
|
||||
// parseFile uses node:fs, so we test it by writing a temp file. This keeps the
|
||||
// parser package dependency-free while still exercising gray-matter.
|
||||
import { parseFile } from "../src/index";
|
||||
import { writeFileSync, mkdtempSync, rmSync, mkdirSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
|
||||
let tmp: string;
|
||||
beforeEach(() => {
|
||||
tmp = mkdtempSync(join(tmpdir(), "mcpedia-parser-"));
|
||||
});
|
||||
afterEach(() => {
|
||||
rmSync(tmp, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
function writeDoc(rel: string, frontmatter: string) {
|
||||
const p = join(tmp, rel);
|
||||
// Ensure the parent directory exists (sections like docs/, writeups/).
|
||||
mkdirSync(join(p, ".."), { recursive: true });
|
||||
writeFileSync(p, frontmatter, "utf8");
|
||||
return parseFile(p, rel);
|
||||
}
|
||||
|
||||
test("parseFile: extracts basic frontmatter", () => {
|
||||
const { meta, body } = writeDoc(
|
||||
"docs/test.md",
|
||||
[
|
||||
"---",
|
||||
'id: test-doc',
|
||||
'title: Test Document',
|
||||
'type: documentation',
|
||||
'tags: ["docs", "test"]',
|
||||
'status: published',
|
||||
'author: asep',
|
||||
'created_at: 2026-08-19',
|
||||
'updated_at: 2026-08-19',
|
||||
"---",
|
||||
"",
|
||||
"# Hello",
|
||||
"Body text here.",
|
||||
].join("\n"),
|
||||
);
|
||||
expect(meta.slug).toBe("docs/test");
|
||||
expect(meta.section).toBe("docs");
|
||||
expect(meta.title).toBe("Test Document");
|
||||
expect(meta.type).toBe("documentation");
|
||||
expect(meta.status).toBe("published");
|
||||
expect(meta.author).toBe("asep");
|
||||
expect(meta.tags).toEqual(["docs", "test"]);
|
||||
expect(body).toContain("# Hello");
|
||||
});
|
||||
|
||||
test("parseFile: section derived from top-level dir", () => {
|
||||
expect(writeDoc("writeups/foo.md", "---\ntitle: A\n---\nbody").meta.section).toBe("writeups");
|
||||
expect(writeDoc("research/bar.md", "---\ntitle: B\n---\nbody").meta.section).toBe("research");
|
||||
expect(writeDoc("notes/baz.md", "---\ntitle: C\n---\nbody").meta.section).toBe("notes");
|
||||
});
|
||||
|
||||
test("parseFile: invalid type/status fall back to defaults", () => {
|
||||
const { meta } = writeDoc(
|
||||
"docs/x.md",
|
||||
"---\ntitle: X\ntype: bogus\nstatus: bogus\n---\n",
|
||||
);
|
||||
expect(meta.type).toBe("documentation");
|
||||
expect(meta.status).toBe("published");
|
||||
});
|
||||
|
||||
test("parseFile: missing optional fields get sane defaults", () => {
|
||||
const { meta } = writeDoc("docs/x.md", "---\ntitle: Just A Title\n---\n");
|
||||
expect(meta.author).toBe("");
|
||||
expect(meta.tags).toEqual([]);
|
||||
expect(meta.createdAt).toBeTruthy();
|
||||
expect(meta.updatedAt).toBeTruthy();
|
||||
});
|
||||
|
||||
test("parseFile: body excludes frontmatter delimiter", () => {
|
||||
const { body } = writeDoc("docs/x.md", "---\ntitle: T\n---\n# Real body\n\nParagraph.");
|
||||
expect(body).not.toContain("---");
|
||||
expect(body).toContain("# Real body");
|
||||
});
|
||||
@@ -11,5 +11,8 @@
|
||||
"@mcpedia/embeddings": "workspace:*",
|
||||
"@mcpedia/types": "workspace:*",
|
||||
"drizzle-orm": "^0.38.0"
|
||||
},
|
||||
"scripts": {
|
||||
"test": "bun test"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
import { test, expect } from "bun:test";
|
||||
import { cosine, toTsQuery } from "../src/index";
|
||||
|
||||
test("cosine: orthogonal vectors are 0", () => {
|
||||
expect(cosine([1, 0], [0, 1])).toBeCloseTo(0, 6);
|
||||
});
|
||||
|
||||
test("cosine: identical vectors are 1", () => {
|
||||
expect(cosine([1, 1, 1], [1, 1, 1])).toBeCloseTo(1, 6);
|
||||
});
|
||||
|
||||
test("cosine: empty or length-mismatched returns 0", () => {
|
||||
expect(cosine([], [])).toBe(0);
|
||||
expect(cosine([], [1, 2, 3])).toBe(0);
|
||||
expect(cosine([1, 2], [1, 2, 3])).toBe(0);
|
||||
});
|
||||
|
||||
test("cosine: opposite vectors are negative", () => {
|
||||
const score = cosine([1, 0], [-1, 0]);
|
||||
expect(score).toBeCloseTo(-1, 6);
|
||||
});
|
||||
|
||||
test("cosine: zero-vector denominator returns 0 (no NaN)", () => {
|
||||
expect(cosine([0, 0, 0], [0, 0, 0])).toBe(0);
|
||||
});
|
||||
|
||||
test("toTsQuery: joins terms with AND-prefix", () => {
|
||||
expect(toTsQuery("websocket contract")).toBe("websocket:* & contract:*");
|
||||
});
|
||||
|
||||
test("toTsQuery: strips non-alphanumerics and empty terms", () => {
|
||||
expect(toTsQuery("hello!!! world???")).toBe("hello:* & world:*");
|
||||
expect(toTsQuery(" ")).toBe("");
|
||||
expect(toTsQuery("123 456")).toBe("123:* & 456:*");
|
||||
});
|
||||
|
||||
test("toTsQuery: empty/garbage input returns empty string", () => {
|
||||
expect(toTsQuery("!!!@@@###")).toBe("");
|
||||
expect(toTsQuery("")).toBe("");
|
||||
});
|
||||
Reference in New Issue
Block a user