feat(mcpedia): Phase 3 — async indexing (BullMQ), git-sync webhook, revisions, MCP Resources
- packages/queue: ioredis singleton + BullMQ Queue/Worker (prefix mcpedia:
on shared imrnes Redis :6379); apps/worker runs startWorker()
- @mcpedia/core: indexContentFile/runFullIndex (single indexing entry point
shared by script/worker/hook) + revision.service (list/get/restore)
- document_revisions table (migration 0002) — snapshots only on body change
- apps/api: POST /hooks/reindex + /hooks/index webhooks; tRPC revisions,
getRevision, restoreRevision, jobStatus, queueStatus
- apps/mcp: register MCP Resources mcpedia://docs{/,+slug/chunks/revisions}
({+slug} RFC6570 reserved expansion for slugs containing /)
- apps/mcp zod pinned to ^4 to match MCP SDK 1.30 compiled types
(resolves registerTool TS2589/ShapeOutput skew)
- scripts/enqueue.ts one-shot job enqueue helper; indexer refactored to runFullIndex
- PHASES.md/README/.env.example/docs updated
This commit is contained in:
+9
-62
@@ -1,67 +1,14 @@
|
||||
import { db } from "@mcpedia/db";
|
||||
import { documents } from "@mcpedia/db/schema";
|
||||
import { parseFile } from "@mcpedia/parser";
|
||||
import { CONTENT_ROOT } from "@mcpedia/config";
|
||||
import { listContentFiles, indexChunks } from "@mcpedia/core";
|
||||
import { join } from "node:path";
|
||||
import { runFullIndex } from "@mcpedia/core";
|
||||
|
||||
// Phase 3: the indexer now goes through `runFullIndex`, the single indexing
|
||||
// entry point shared with the BullMQ worker and the git-sync hook. It parses
|
||||
// each content file, upserts `documents`, chunks+embeds, and snapshots a
|
||||
// revision when the body changed.
|
||||
async function main() {
|
||||
const files = listContentFiles();
|
||||
let indexed = 0;
|
||||
let chunked = 0;
|
||||
for (const rel of files) {
|
||||
const abs = join(CONTENT_ROOT, rel);
|
||||
const { meta, body } = parseFile(abs, rel);
|
||||
const nowIso =
|
||||
meta.updatedAt && meta.updatedAt !== ""
|
||||
? meta.updatedAt
|
||||
: new Date().toISOString();
|
||||
await db
|
||||
.insert(documents)
|
||||
.values({
|
||||
id: meta.id,
|
||||
slug: meta.slug,
|
||||
title: meta.title,
|
||||
type: meta.type,
|
||||
section: meta.section,
|
||||
status: meta.status,
|
||||
author: meta.author,
|
||||
tags: meta.tags,
|
||||
path: meta.path,
|
||||
body,
|
||||
createdAt: new Date(meta.createdAt || nowIso),
|
||||
updatedAt: new Date(nowIso),
|
||||
})
|
||||
.onConflictDoUpdate({
|
||||
target: documents.slug,
|
||||
set: {
|
||||
title: meta.title,
|
||||
type: meta.type,
|
||||
section: meta.section,
|
||||
status: meta.status,
|
||||
author: meta.author,
|
||||
tags: meta.tags,
|
||||
path: meta.path,
|
||||
body,
|
||||
updatedAt: new Date(nowIso),
|
||||
},
|
||||
});
|
||||
indexed++;
|
||||
console.log(` indexed ${rel}`);
|
||||
|
||||
// Phase 2: chunk + embed for semantic search.
|
||||
try {
|
||||
const n = await indexChunks(meta.slug, body);
|
||||
chunked += n;
|
||||
console.log(` embedded ${n} chunks`);
|
||||
} catch (err) {
|
||||
console.error(
|
||||
` embed FAILED for ${meta.slug}: ${err instanceof Error ? err.message : err}`,
|
||||
);
|
||||
// Don't abort the whole index over one doc's embedding failure.
|
||||
}
|
||||
}
|
||||
console.log(`indexed ${indexed} documents, ${chunked} chunks embedded`);
|
||||
const reason = process.argv[2] && process.argv[2].startsWith("--reason=")
|
||||
? process.argv[2].slice("--reason=".length)
|
||||
: "index";
|
||||
await runFullIndex(reason);
|
||||
}
|
||||
|
||||
main()
|
||||
|
||||
Reference in New Issue
Block a user