feat(mcpedia): Phase 3 — async indexing (BullMQ), git-sync webhook, revisions, MCP Resources

- packages/queue: ioredis singleton + BullMQ Queue/Worker (prefix mcpedia:
  on shared imrnes Redis :6379); apps/worker runs startWorker()
- @mcpedia/core: indexContentFile/runFullIndex (single indexing entry point
  shared by script/worker/hook) + revision.service (list/get/restore)
- document_revisions table (migration 0002) — snapshots only on body change
- apps/api: POST /hooks/reindex + /hooks/index webhooks; tRPC revisions,
  getRevision, restoreRevision, jobStatus, queueStatus
- apps/mcp: register MCP Resources mcpedia://docs{/,+slug/chunks/revisions}
  ({+slug} RFC6570 reserved expansion for slugs containing /)
- apps/mcp zod pinned to ^4 to match MCP SDK 1.30 compiled types
  (resolves registerTool TS2589/ShapeOutput skew)
- scripts/enqueue.ts one-shot job enqueue helper; indexer refactored to runFullIndex
- PHASES.md/README/.env.example/docs updated
This commit is contained in:
asepharyana
2026-08-19 20:18:28 +07:00
parent 9397303f01
commit 8f2229d447
30 changed files with 1233 additions and 78 deletions
+155
View File
@@ -0,0 +1,155 @@
import { db } from "@mcpedia/db";
import { documents, documentRevisions, documentChunks } from "@mcpedia/db/schema";
import { parseFile } from "@mcpedia/parser";
import { CONTENT_ROOT } from "@mcpedia/config";
import { listContentFiles } from "./content.service";
import { indexChunks } from "./document.service";
import { toMeta } from "./row-map";
import { eq, desc, and, sql } from "drizzle-orm";
import { join } from "node:path";
export interface IndexResult {
indexed: number;
chunks: number;
revisions: number;
}
/**
* Index a single content file: parse → upsert `documents` → chunk+embed →
* snapshot a revision if the body changed since the last indexed revision.
*
* This is THE single indexing entry point shared by the CLI script, the
* BullMQ worker, and the git-sync hook — no business logic is duplicated.
*
* @param relPath path relative to CONTENT_ROOT (e.g. "docs/websocket/contract")
* @param reason provenance tag for the revision ("index" | "git-push" | "reindex")
*/
export async function indexContentFile(
relPath: string,
reason = "index",
): Promise<{ indexed: boolean; chunks: number; revision: boolean }> {
const abs = join(CONTENT_ROOT, relPath);
const { meta, body } = parseFile(abs, relPath);
const nowIso =
meta.updatedAt && meta.updatedAt !== ""
? meta.updatedAt
: new Date().toISOString();
await db
.insert(documents)
.values({
id: meta.id,
slug: meta.slug,
title: meta.title,
type: meta.type,
section: meta.section,
status: meta.status,
author: meta.author,
tags: meta.tags,
path: meta.path,
body,
createdAt: new Date(meta.createdAt || nowIso),
updatedAt: new Date(nowIso),
})
.onConflictDoUpdate({
target: documents.slug,
set: {
title: meta.title,
type: meta.type,
section: meta.section,
status: meta.status,
author: meta.author,
tags: meta.tags,
path: meta.path,
body,
updatedAt: new Date(nowIso),
},
});
// Semantic chunks (embedding). A failure here must not abort the whole
// index — log and continue; FTS still works without embeddings.
let chunks = 0;
try {
chunks = await indexChunks(meta.slug, body);
} catch (err) {
console.error(
` embed FAILED for ${meta.slug}: ${err instanceof Error ? err.message : err}`,
);
}
// Snapshot a revision only when the body actually changed vs the latest
// revision. Pure metadata/index changes (tags/title) won't create noise.
const revision = await snapshotRevision(meta.slug, meta, body, reason);
return { indexed: true, chunks, revision };
}
/**
* Compare the incoming body against the latest revision's body; if different
* (or no prior revision exists), create a new revision with an incremented
* per-document revisionNo.
*/
async function snapshotRevision(
slug: string,
meta: ReturnType<typeof parseFile>["meta"],
body: string,
reason: string,
): Promise<boolean> {
const [doc] = await db
.select({ id: documents.id })
.from(documents)
.where(eq(documents.slug, slug));
if (!doc) return false;
const [latest] = await db
.select({ body: documentRevisions.body, revisionNo: documentRevisions.revisionNo })
.from(documentRevisions)
.where(eq(documentRevisions.documentId, doc.id))
.orderBy(desc(documentRevisions.revisionNo))
.limit(1);
if (latest && latest.body === body) {
return false; // unchanged → no new revision
}
const nextNo = (latest?.revisionNo ?? 0) + 1;
await db.insert(documentRevisions).values({
documentId: doc.id,
slug,
revisionNo: nextNo,
title: meta.title,
body,
meta: {
type: meta.type,
section: meta.section,
status: meta.status,
author: meta.author,
tags: meta.tags,
},
reason,
});
return true;
}
/**
* Walk the entire content tree and index every file. Returns aggregate counts.
*/
export async function runFullIndex(reason = "index"): Promise<IndexResult> {
const files = listContentFiles();
let indexed = 0;
let chunks = 0;
let revisions = 0;
for (const rel of files) {
const r = await indexContentFile(rel, reason);
indexed++;
chunks += r.chunks;
if (r.revision) revisions++;
console.log(
` indexed ${rel}${r.chunks ? ` (${r.chunks} chunks)` : ""}${r.revision ? " [revision]" : ""}`,
);
}
console.log(
`indexed ${indexed} documents, ${chunks} chunks, ${revisions} new revisions`,
);
return { indexed, chunks, revisions };
}
+2
View File
@@ -1,6 +1,8 @@
export * from "./content.service";
export * from "./document.service";
export * from "./search.service";
export * from "./index.service";
export * from "./revision.service";
export { toMeta } from "./row-map";
export type {
+116
View File
@@ -0,0 +1,116 @@
import { db } from "@mcpedia/db";
import { documents, documentRevisions, documentChunks } from "@mcpedia/db/schema";
import { eq, desc, and, sql } from "drizzle-orm";
import { toMeta } from "./row-map";
import type { DocumentMeta } from "@mcpedia/types";
export interface RevisionSummary {
id: string;
slug: string;
revisionNo: number;
title: string;
reason: string;
createdAt: string;
bodyLength: number;
}
/** List revisions for a slug, newest first. */
export async function listRevisions(
slug: string,
limit = 20,
): Promise<RevisionSummary[]> {
const [doc] = await db
.select({ id: documents.id })
.from(documents)
.where(eq(documents.slug, slug));
if (!doc) return [];
const rows = await db
.select({
id: documentRevisions.id,
slug: documentRevisions.slug,
revisionNo: documentRevisions.revisionNo,
title: documentRevisions.title,
reason: documentRevisions.reason,
createdAt: documentRevisions.createdAt,
bodyLength: sql<number>`length(${documentRevisions.body})`,
})
.from(documentRevisions)
.where(eq(documentRevisions.documentId, doc.id))
.orderBy(desc(documentRevisions.revisionNo))
.limit(limit);
return rows.map((r) => ({
id: r.id,
slug: r.slug,
revisionNo: r.revisionNo,
title: r.title,
reason: r.reason,
createdAt: r.createdAt.toISOString(),
bodyLength: r.bodyLength,
}));
}
/** Fetch a single revision's full body. */
export async function getRevision(
id: string,
): Promise<{ id: string; revisionNo: number; body: string; meta: unknown } | null> {
const [row] = await db
.select({
id: documentRevisions.id,
revisionNo: documentRevisions.revisionNo,
body: documentRevisions.body,
meta: documentRevisions.meta,
})
.from(documentRevisions)
.where(eq(documentRevisions.id, id));
if (!row) return null;
return {
id: row.id,
revisionNo: row.revisionNo,
body: row.body,
meta: row.meta,
};
}
/** Restore a revision: write its body+metadata back into the live `documents` row. */
export async function restoreRevision(
id: string,
): Promise<{ slug: string; documentId: string } | null> {
const [rev] = await db
.select({
id: documentRevisions.id,
documentId: documentRevisions.documentId,
slug: documentRevisions.slug,
title: documentRevisions.title,
body: documentRevisions.body,
meta: documentRevisions.meta,
})
.from(documentRevisions)
.where(eq(documentRevisions.id, id));
if (!rev) return null;
const m = rev.meta as {
type?: string;
section?: string;
status?: string;
author?: string;
tags?: string[];
};
await db
.update(documents)
.set({
title: rev.title,
type: (m.type as any) ?? "documentation",
section: (m.section as any) ?? "docs",
status: (m.status as any) ?? "published",
author: m.author ?? "",
tags: m.tags ?? [],
body: rev.body,
updatedAt: new Date(),
})
.where(eq(documents.id, rev.documentId));
return { slug: rev.slug, documentId: rev.documentId };
}