chore(testing): Phase 9 — bun:test suite + CI gating with no-DB fakes
CI / typecheck + build (turbo) (push) Canceled after 0s

Added a real test suite (32 tests, 0 external services) using bun:test with
in-process module mocking for @mcpedia/db, @mcpedia/queue, @mcpedia/core.

Enablers:
- apps/api: extracted createApp(deps?) factory + dashboard.ts module from
  index.ts so the HTTP surface is unit-testable (real queue is lazy-imported).
- packages/core: exported shouldCreateRevision pure predicate; restoreRevision
  gained an opts.reindex seam for the chunk-rebuild contract.
- apps/mcp: renamed smoke.test.ts -> smoke.ts (bun test now owns .test.ts),
  updated stale assertions (10 tools, 4 docs in docs section).
- infra: turbo test task (cache:false), test scripts across packages,
  @types/bun + tsconfig base types, CI 'Test' step after Build.

Packages with tests: embeddings(5), parser(5), search(8), core(4),
mcp(6 auth-gates), api(8 contracts).

All green: typecheck(4/4), test(6/6 pkgs), build(web). Live API verified
/health, /metrics, /dashboard, /hooks/* auth gate on temp port.
This commit is contained in:
asepharyana
2026-08-20 11:12:32 +07:00
parent 0cdf261d40
commit 76f778c10d
25 changed files with 1055 additions and 193 deletions
+3
View File
@@ -15,5 +15,8 @@
"@mcpedia/search": "workspace:*",
"@mcpedia/types": "workspace:*",
"drizzle-orm": "^0.38.0"
},
"scripts": {
"test": "bun test"
}
}
+44
View File
@@ -0,0 +1,44 @@
import { test, expect } from "bun:test";
import { shouldCreateRevision } from "../src/index.service";
/**
* Phase 9: unit tests for the revision-dedup decision rule.
*
* `shouldCreateRevision` is the pure predicate that `indexContentFile` consults
* before writing a new row to `document_revisions`. It's extracted because the
* dedup correctness is the single most important guarantee of the revision
* system ("metadata-only edits don't bloat history"), and it must hold without
* a database.
*
* The DB-backed paths (`snapshotRevision`, `restoreRevision`) are exercised
* end-to-end by the existing manual e2e (`bun run index` + restore via the web
* /api/revisions/restore route, see PHASES.md Phase 4 verification). Here we
* lock the decision invariant in CI.
*/
test("shouldCreateRevision: first snapshot when no prior revision exists", () => {
// No prior revision row → always snapshot the first version.
expect(shouldCreateRevision(null, "body text")).toBe(true);
expect(shouldCreateRevision(undefined, "body text")).toBe(true);
});
test("shouldCreateRevision: identical body creates no new revision (dedup)", () => {
// The exact dedup rule that prevents metadata-only edits from bloating
// history: if the body is byte-identical to the latest revision's body,
// skip the snapshot.
expect(shouldCreateRevision("same body", "same body")).toBe(false);
// The dedup rule applies regardless of body length — large identical bodies
// also skip the snapshot.
expect(shouldCreateRevision("a".repeat(5000), "a".repeat(5000))).toBe(false);
});
test("shouldCreateRevision: changed body creates a new revision", () => {
expect(shouldCreateRevision("old body", "new body")).toBe(true);
// Whitespace / trailing newline changes count as a real body change.
expect(shouldCreateRevision("body", "body\n")).toBe(true);
});
test("shouldCreateRevision: empty-string vs non-empty counts as a change", () => {
expect(shouldCreateRevision("", "content")).toBe(true);
expect(shouldCreateRevision("content", "")).toBe(true);
});
+16
View File
@@ -84,6 +84,22 @@ export async function indexContentFile(
return { indexed: true, chunks, revision };
}
/**
* Pure decision rule for the revision system: create a new revision only when
* the body genuinely changed vs the latest snapshot.
* - no prior revision (latestBody null) -> true (first snapshot)
* - identical body -> false (no noise)
* - different body -> true
*
* Exported separately so it can be unit-tested without a database.
*/
export function shouldCreateRevision(
latestBody: string | null | undefined,
body: string,
): boolean {
return latestBody == null || latestBody !== body;
}
/**
* Compare the incoming body against the latest revision's body; if different
* (or no prior revision exists), create a new revision with an incremented
+12 -2
View File
@@ -76,9 +76,18 @@ export async function getRevision(
};
}
/** Restore a revision: write its body+metadata back into the live `documents` row. */
/**
* Restore a revision: write its body+metadata back into the live `documents` row.
*
* @param id revision UUID
* @param opts optional seam for testing — override the chunk-rebuild step so
* tests can assert it's invoked without touching embeddings.
*/
export async function restoreRevision(
id: string,
opts?: {
reindex?: (slug: string) => Promise<number>;
},
): Promise<{ slug: string; documentId: string } | null> {
const [rev] = await db
.select({
@@ -118,7 +127,8 @@ export async function restoreRevision(
// Rebuild semantic chunks + embeddings from the restored body so semantic
// and hybrid search stay consistent (otherwise document_chunks would hold
// the NEW body's chunks while documents.body holds the OLD/restore body).
await reindexChunks(rev.slug);
const reindex = opts?.reindex ?? reindexChunks;
await reindex(rev.slug);
return { slug: rev.slug, documentId: rev.documentId };
}
+3
View File
@@ -13,5 +13,8 @@
},
"devDependencies": {
"typescript": "^5.6.0"
},
"scripts": {
"test": "bun test"
}
}
+57
View File
@@ -0,0 +1,57 @@
import { test, expect } from "bun:test";
import { chunkText } from "../src/chunk";
test("empty / whitespace input returns empty array", () => {
expect(chunkText("")).toEqual([]);
expect(chunkText(" \n ")).toEqual([]);
});
test("short text (<= size) returns a single chunk", () => {
const text = "hello world this is short";
const chunks = chunkText(text, { size: 1000, overlap: 150 });
expect(chunks).toHaveLength(1);
expect(chunks[0]).toBe(text);
});
test("long text splits into multiple chunks with overlap honored", () => {
// Build ~3000 chars of words so we get >1 chunk at default size 1000.
const word = "lorem";
const text = Array.from({ length: 600 }, () => word).join(" ");
const chunks = chunkText(text, { size: 1000, overlap: 150 });
expect(chunks.length).toBeGreaterThan(1);
// Every chunk must respect the size upper bound (trimmed).
for (const c of chunks) {
expect(c.length).toBeLessThanOrEqual(1000);
}
// The overlap region: second chunk should start near the end of the first
// minus the overlap window. We just assert they share some suffix/prefix
// overlap roughly, i.e. the join doesn't lose content boundaries badly.
const joined = chunks.join(" ");
// Most words are preserved across the split (at least the bulk).
expect(joined.length).toBeGreaterThan(text.length * 0.9);
});
test("chunkText never splits a chunk mid-word past the boundary (no truncation mid-token)", () => {
const text = "alpha beta gamma delta epsilon zeta eta theta iota kappa lambda mu nu xi";
const chunks = chunkText(text, { size: 20, overlap: 4 });
// No chunk should contain a partial word boundary that corrupts tokens —
// i.e. every resulting piece still reassembles into the original words set.
const reassembled = chunks
.flatMap((c) => c.split(/\s+/))
.filter(Boolean)
.sort();
const original = text.split(/\s+/).sort();
// Overlap means some words repeat — assert all original words are present.
for (const w of original) {
expect(reassembled).toContain(w);
}
});
test("default options produce reasonable chunking", () => {
const text = "x".repeat(2500);
const chunks = chunkText(text); // defaults: size 1000, overlap 150
expect(chunks.length).toBeGreaterThanOrEqual(2);
expect(chunks[chunks.length - 1].length).toBeLessThanOrEqual(1000);
});
+3
View File
@@ -9,5 +9,8 @@
"dependencies": {
"@mcpedia/types": "workspace:*",
"gray-matter": "^4.0.3"
},
"scripts": {
"test": "bun test"
}
}
+81
View File
@@ -0,0 +1,81 @@
import { test, expect, afterEach, beforeEach } from "bun:test";
// parseFile uses node:fs, so we test it by writing a temp file. This keeps the
// parser package dependency-free while still exercising gray-matter.
import { parseFile } from "../src/index";
import { writeFileSync, mkdtempSync, rmSync, mkdirSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
let tmp: string;
beforeEach(() => {
tmp = mkdtempSync(join(tmpdir(), "mcpedia-parser-"));
});
afterEach(() => {
rmSync(tmp, { recursive: true, force: true });
});
function writeDoc(rel: string, frontmatter: string) {
const p = join(tmp, rel);
// Ensure the parent directory exists (sections like docs/, writeups/).
mkdirSync(join(p, ".."), { recursive: true });
writeFileSync(p, frontmatter, "utf8");
return parseFile(p, rel);
}
test("parseFile: extracts basic frontmatter", () => {
const { meta, body } = writeDoc(
"docs/test.md",
[
"---",
'id: test-doc',
'title: Test Document',
'type: documentation',
'tags: ["docs", "test"]',
'status: published',
'author: asep',
'created_at: 2026-08-19',
'updated_at: 2026-08-19',
"---",
"",
"# Hello",
"Body text here.",
].join("\n"),
);
expect(meta.slug).toBe("docs/test");
expect(meta.section).toBe("docs");
expect(meta.title).toBe("Test Document");
expect(meta.type).toBe("documentation");
expect(meta.status).toBe("published");
expect(meta.author).toBe("asep");
expect(meta.tags).toEqual(["docs", "test"]);
expect(body).toContain("# Hello");
});
test("parseFile: section derived from top-level dir", () => {
expect(writeDoc("writeups/foo.md", "---\ntitle: A\n---\nbody").meta.section).toBe("writeups");
expect(writeDoc("research/bar.md", "---\ntitle: B\n---\nbody").meta.section).toBe("research");
expect(writeDoc("notes/baz.md", "---\ntitle: C\n---\nbody").meta.section).toBe("notes");
});
test("parseFile: invalid type/status fall back to defaults", () => {
const { meta } = writeDoc(
"docs/x.md",
"---\ntitle: X\ntype: bogus\nstatus: bogus\n---\n",
);
expect(meta.type).toBe("documentation");
expect(meta.status).toBe("published");
});
test("parseFile: missing optional fields get sane defaults", () => {
const { meta } = writeDoc("docs/x.md", "---\ntitle: Just A Title\n---\n");
expect(meta.author).toBe("");
expect(meta.tags).toEqual([]);
expect(meta.createdAt).toBeTruthy();
expect(meta.updatedAt).toBeTruthy();
});
test("parseFile: body excludes frontmatter delimiter", () => {
const { body } = writeDoc("docs/x.md", "---\ntitle: T\n---\n# Real body\n\nParagraph.");
expect(body).not.toContain("---");
expect(body).toContain("# Real body");
});
+3
View File
@@ -11,5 +11,8 @@
"@mcpedia/embeddings": "workspace:*",
"@mcpedia/types": "workspace:*",
"drizzle-orm": "^0.38.0"
},
"scripts": {
"test": "bun test"
}
}
+40
View File
@@ -0,0 +1,40 @@
import { test, expect } from "bun:test";
import { cosine, toTsQuery } from "../src/index";
test("cosine: orthogonal vectors are 0", () => {
expect(cosine([1, 0], [0, 1])).toBeCloseTo(0, 6);
});
test("cosine: identical vectors are 1", () => {
expect(cosine([1, 1, 1], [1, 1, 1])).toBeCloseTo(1, 6);
});
test("cosine: empty or length-mismatched returns 0", () => {
expect(cosine([], [])).toBe(0);
expect(cosine([], [1, 2, 3])).toBe(0);
expect(cosine([1, 2], [1, 2, 3])).toBe(0);
});
test("cosine: opposite vectors are negative", () => {
const score = cosine([1, 0], [-1, 0]);
expect(score).toBeCloseTo(-1, 6);
});
test("cosine: zero-vector denominator returns 0 (no NaN)", () => {
expect(cosine([0, 0, 0], [0, 0, 0])).toBe(0);
});
test("toTsQuery: joins terms with AND-prefix", () => {
expect(toTsQuery("websocket contract")).toBe("websocket:* & contract:*");
});
test("toTsQuery: strips non-alphanumerics and empty terms", () => {
expect(toTsQuery("hello!!! world???")).toBe("hello:* & world:*");
expect(toTsQuery(" ")).toBe("");
expect(toTsQuery("123 456")).toBe("123:* & 456:*");
});
test("toTsQuery: empty/garbage input returns empty string", () => {
expect(toTsQuery("!!!@@@###")).toBe("");
expect(toTsQuery("")).toBe("");
});