feat(mcpedia): Phase 3 — async indexing (BullMQ), git-sync webhook, revisions, MCP Resources

- packages/queue: ioredis singleton + BullMQ Queue/Worker (prefix mcpedia:
  on shared imrnes Redis :6379); apps/worker runs startWorker()
- @mcpedia/core: indexContentFile/runFullIndex (single indexing entry point
  shared by script/worker/hook) + revision.service (list/get/restore)
- document_revisions table (migration 0002) — snapshots only on body change
- apps/api: POST /hooks/reindex + /hooks/index webhooks; tRPC revisions,
  getRevision, restoreRevision, jobStatus, queueStatus
- apps/mcp: register MCP Resources mcpedia://docs{/,+slug/chunks/revisions}
  ({+slug} RFC6570 reserved expansion for slugs containing /)
- apps/mcp zod pinned to ^4 to match MCP SDK 1.30 compiled types
  (resolves registerTool TS2589/ShapeOutput skew)
- scripts/enqueue.ts one-shot job enqueue helper; indexer refactored to runFullIndex
- PHASES.md/README/.env.example/docs updated
This commit is contained in:
asepharyana
2026-08-19 20:18:28 +07:00
parent 9397303f01
commit 8f2229d447
30 changed files with 1233 additions and 78 deletions
+123 -1
View File
@@ -1,7 +1,19 @@
import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
import { ResourceTemplate } from "@modelcontextprotocol/sdk/server/mcp.js";
import { z } from "zod";
import { listDocuments, getDocument, getRelated, semanticSearch, hybridSearch, keywordSearch } from "@mcpedia/core";
import {
listDocuments,
getDocument,
getRelated,
semanticSearch,
hybridSearch,
keywordSearch,
listRevisions,
readContentFile,
} from "@mcpedia/core";
import { CONTENT_ROOT } from "@mcpedia/config";
import { join } from "node:path";
export function createMcpServer(): McpServer {
const server = new McpServer({
@@ -119,6 +131,116 @@ export function createMcpServer(): McpServer {
},
);
// --- Phase 3: MCP Resources (read-only knowledge base surfaced via URIs) ---
// mcpedia://docs -> list all published documents
// mcpedia://docs/{slug} -> full markdown body (from disk)
// mcpedia://docs/{slug}/chunks -> chunked preview (semantic slices)
// mcpedia://docs/{slug}/revisions -> revision history summary
server.registerResource(
"mcpedia-docs-list",
"mcpedia://docs",
{
title: "MCPedia document index",
description: "List of all published documents in the knowledge base.",
mimeType: "application/json",
},
async (uri) => {
const docs = await listDocuments();
return {
contents: [
{
uri: uri.href,
mimeType: "application/json",
text: JSON.stringify(docs, null, 2),
},
],
};
},
);
server.registerResource(
"mcpedia-doc-chunks",
new ResourceTemplate("mcpedia://docs/{+slug}/chunks", { list: undefined }),
{
title: "MCPedia document chunks",
description: "Preview of the embedded semantic chunks for a document.",
mimeType: "application/json",
},
async (uri, vars) => {
const slug = String(vars.slug);
const doc = await getDocument(slug);
if (!doc) throw new Error(`Document not found: ${slug}`);
// Chunk the body the same way the indexer does (size 1000 / overlap 150)
// so the resource mirrors what semantic search actually sees.
const { chunkText } = await import("@mcpedia/embeddings");
const chunks = chunkText(doc.body, { size: 1000, overlap: 150 });
return {
contents: [
{
uri: uri.href,
mimeType: "application/json",
text: JSON.stringify(
chunks.map((c, i) => ({ index: i, length: c.length, preview: c.slice(0, 200) })),
null,
2,
),
},
],
};
},
);
server.registerResource(
"mcpedia-doc-revisions",
new ResourceTemplate("mcpedia://docs/{+slug}/revisions", { list: undefined }),
{
title: "MCPedia document revisions",
description: "Revision history summary for a document.",
mimeType: "application/json",
},
async (uri, vars) => {
const slug = String(vars.slug);
const revs = await listRevisions(slug, 20);
return {
contents: [
{
uri: uri.href,
mimeType: "application/json",
text: JSON.stringify(revs, null, 2),
},
],
};
},
);
// Registered LAST: the bare {+slug} template is greedy and would otherwise
// swallow /chunks and /revisions URIs. Specific templates must match first.
server.registerResource(
"mcpedia-doc",
new ResourceTemplate("mcpedia://docs/{+slug}", { list: undefined }),
{
title: "MCPedia document",
description: "Full markdown body of a single document, read from disk (source of truth).",
mimeType: "text/markdown",
},
async (uri, vars) => {
const slug = String(vars.slug);
const doc = await getDocument(slug);
if (!doc) {
throw new Error(`Document not found: ${slug}`);
}
return {
contents: [
{
uri: uri.href,
mimeType: "text/markdown",
text: doc.body,
},
],
};
},
);
return server;
}
+29
View File
@@ -92,6 +92,35 @@ async function main() {
}
console.log(`hybrid_search => ${hybHits.length} docs, top: ${hybHits[0].doc.slug}`);
// 8) resources: list
const resList = await client.listResources();
const resNames = resList.resources.map((r: any) => r.name).sort();
console.log("resources:", resNames.join(", "));
if (!resNames.includes("mcpedia-docs-list")) {
throw new Error("expected mcpedia-docs-list resource");
}
// 9) resource: read the docs list (must not throw, returns JSON content)
const readList = await client.readResource({ uri: "mcpedia://docs" });
const listText = (readList.contents as any)[0].text;
if (!listText.includes("docs/websocket/contract")) {
throw new Error("mcpedia://docs did not list the websocket contract doc");
}
console.log("readResource(mcpedia://docs) => ok");
// 10) resource: read a single doc body + revisions
const readDoc = await client.readResource({ uri: "mcpedia://docs/docs/websocket/contract" });
const docText = (readDoc.contents as any)[0].text;
if (!docText.includes("WebSocket Contract")) {
throw new Error("mcpedia://docs/{slug} returned unexpected body");
}
console.log("readResource(mcpedia://docs/docs/websocket/contract) => ok");
const readRev = await client.readResource({
uri: "mcpedia://docs/docs/websocket/contract/revisions",
});
console.log("readResource(.../revisions) => ok");
await client.close();
await server.close();
console.log("\nSMOKE OK");