feat(mcpedia): Phase 3 — async indexing (BullMQ), git-sync webhook, revisions, MCP Resources

- packages/queue: ioredis singleton + BullMQ Queue/Worker (prefix mcpedia:
  on shared imrnes Redis :6379); apps/worker runs startWorker()
- @mcpedia/core: indexContentFile/runFullIndex (single indexing entry point
  shared by script/worker/hook) + revision.service (list/get/restore)
- document_revisions table (migration 0002) — snapshots only on body change
- apps/api: POST /hooks/reindex + /hooks/index webhooks; tRPC revisions,
  getRevision, restoreRevision, jobStatus, queueStatus
- apps/mcp: register MCP Resources mcpedia://docs{/,+slug/chunks/revisions}
  ({+slug} RFC6570 reserved expansion for slugs containing /)
- apps/mcp zod pinned to ^4 to match MCP SDK 1.30 compiled types
  (resolves registerTool TS2589/ShapeOutput skew)
- scripts/enqueue.ts one-shot job enqueue helper; indexer refactored to runFullIndex
- PHASES.md/README/.env.example/docs updated
This commit is contained in:
asepharyana
2026-08-19 20:18:28 +07:00
parent 9397303f01
commit 8f2229d447
30 changed files with 1233 additions and 78 deletions
+9 -62
View File
@@ -1,67 +1,14 @@
import { db } from "@mcpedia/db";
import { documents } from "@mcpedia/db/schema";
import { parseFile } from "@mcpedia/parser";
import { CONTENT_ROOT } from "@mcpedia/config";
import { listContentFiles, indexChunks } from "@mcpedia/core";
import { join } from "node:path";
import { runFullIndex } from "@mcpedia/core";
// Phase 3: the indexer now goes through `runFullIndex`, the single indexing
// entry point shared with the BullMQ worker and the git-sync hook. It parses
// each content file, upserts `documents`, chunks+embeds, and snapshots a
// revision when the body changed.
async function main() {
const files = listContentFiles();
let indexed = 0;
let chunked = 0;
for (const rel of files) {
const abs = join(CONTENT_ROOT, rel);
const { meta, body } = parseFile(abs, rel);
const nowIso =
meta.updatedAt && meta.updatedAt !== ""
? meta.updatedAt
: new Date().toISOString();
await db
.insert(documents)
.values({
id: meta.id,
slug: meta.slug,
title: meta.title,
type: meta.type,
section: meta.section,
status: meta.status,
author: meta.author,
tags: meta.tags,
path: meta.path,
body,
createdAt: new Date(meta.createdAt || nowIso),
updatedAt: new Date(nowIso),
})
.onConflictDoUpdate({
target: documents.slug,
set: {
title: meta.title,
type: meta.type,
section: meta.section,
status: meta.status,
author: meta.author,
tags: meta.tags,
path: meta.path,
body,
updatedAt: new Date(nowIso),
},
});
indexed++;
console.log(` indexed ${rel}`);
// Phase 2: chunk + embed for semantic search.
try {
const n = await indexChunks(meta.slug, body);
chunked += n;
console.log(` embedded ${n} chunks`);
} catch (err) {
console.error(
` embed FAILED for ${meta.slug}: ${err instanceof Error ? err.message : err}`,
);
// Don't abort the whole index over one doc's embedding failure.
}
}
console.log(`indexed ${indexed} documents, ${chunked} chunks embedded`);
const reason = process.argv[2] && process.argv[2].startsWith("--reason=")
? process.argv[2].slice("--reason=".length)
: "index";
await runFullIndex(reason);
}
main()