feat(mcpedia): Phase 1 MVP — monorepo, Core, Web UI, MCP server, Postgres FTS

- bun workspaces + Turborepo monorepo (apps/web, apps/mcp; packages/*, scripts)
- @mcpedia/core single business-logic layer (Document/Content/Search services)
- @mcpedia/db Drizzle schema: documents + weighted tsvector (GIN) for FTS
- @mcpedia/parser frontmatter, @mcpedia/search Postgres FTS (ts_rank+ts_headline)
- Next.js 16 Web UI (home/doc SSG, search dynamic) + react-markdown render
- MCP server (stdio) with 4 tools + in-memory smoke test
- scripts/indexer walks content/ -> upserts into Postgres
- 4 seed docs; README + PHASES status
This commit is contained in:
asepharyana
2026-08-19 17:40:14 +07:00
commit ec3a2867d4
60 changed files with 3723 additions and 0 deletions
+12
View File
@@ -0,0 +1,12 @@
{
"name": "@mcpedia/config",
"version": "0.1.0",
"private": true,
"type": "module",
"exports": {
".": "./src/index.ts"
},
"dependencies": {
"@mcpedia/types": "workspace:*"
}
}
+44
View File
@@ -0,0 +1,44 @@
import { fileURLToPath } from "node:url";
import { dirname, resolve } from "node:path";
import { existsSync, readFileSync } from "node:fs";
const here = dirname(fileURLToPath(import.meta.url));
// packages/config -> repo root (../../..)
export const REPO_ROOT = resolve(here, "../../..");
/**
* Load .env (repo root) as the authoritative dev config and apply it to
* process.env. We intentionally OVERRIDE any inherited DATABASE_URL so a stray
* shell env var can never point the app at the wrong database. .env is
* gitignored; for deploy, set the real vars in the environment and omit .env.
*/
function loadDotEnv() {
const dotEnv = resolve(REPO_ROOT, ".env");
if (!existsSync(dotEnv)) return;
for (const line of readFileSync(dotEnv, "utf8").split("\n")) {
const m = line.match(/^\s*([A-Z0-9_]+)\s*=\s*(.*)\s*$/);
if (!m) continue;
const key = m[1];
let val = m[2];
if (
(val.startsWith('"') && val.endsWith('"')) ||
(val.startsWith("'") && val.endsWith("'"))
) {
val = val.slice(1, -1);
}
process.env[key] = val;
}
}
loadDotEnv();
export const CONTENT_ROOT =
process.env.CONTENT_ROOT ?? resolve(REPO_ROOT, "content");
export const DATABASE_URL = process.env.DATABASE_URL ?? "";
if (!DATABASE_URL) {
// Fail fast with an explicit message instead of a cryptic driver error.
throw new Error(
"DATABASE_URL is not set. Copy .env.example to .env and set it (dev uses imrnes Postgres :6432).",
);
}
+18
View File
@@ -0,0 +1,18 @@
{
"name": "@mcpedia/core",
"version": "0.1.0",
"private": true,
"type": "module",
"exports": {
".": "./src/index.ts",
"./row-map": "./src/row-map.ts"
},
"dependencies": {
"@mcpedia/config": "workspace:*",
"@mcpedia/db": "workspace:*",
"@mcpedia/parser": "workspace:*",
"@mcpedia/search": "workspace:*",
"@mcpedia/types": "workspace:*",
"drizzle-orm": "^0.38.0"
}
}
+23
View File
@@ -0,0 +1,23 @@
import { readdirSync, readFileSync, statSync } from "node:fs";
import { join, relative } from "node:path";
import { CONTENT_ROOT } from "@mcpedia/config";
/** List all markdown/mdx content files relative to CONTENT_ROOT. */
export function listContentFiles(): string[] {
const out: string[] = [];
const walk = (dir: string) => {
for (const entry of readdirSync(dir)) {
const full = join(dir, entry);
const st = statSync(full);
if (st.isDirectory()) walk(full);
else if (/\.mdx?$/.test(entry)) out.push(relative(CONTENT_ROOT, full));
}
};
walk(CONTENT_ROOT);
return out;
}
/** Read a content file's raw text by relative path. */
export function readContentFile(relPath: string): string {
return readFileSync(join(CONTENT_ROOT, relPath), "utf8");
}
+58
View File
@@ -0,0 +1,58 @@
import { and, eq, sql } from "drizzle-orm";
import { db } from "@mcpedia/db";
import { documents } from "@mcpedia/db/schema";
import { CONTENT_ROOT } from "@mcpedia/config";
import { existsSync, readFileSync } from "node:fs";
import { join } from "node:path";
import type {
Document,
DocumentMeta,
} from "@mcpedia/types";
import { readContentFile } from "./content.service";
import { toMeta } from "./row-map";
export async function listDocuments(opts: {
section?: string;
status?: string;
} = {}): Promise<DocumentMeta[]> {
const status = opts.status ?? "published";
const where = [eq(documents.status, status)];
if (opts.section) where.push(eq(documents.section, opts.section));
const rows = await db
.select()
.from(documents)
.where(and(...where))
.orderBy(documents.updatedAt);
return rows.map(toMeta);
}
export async function getDocument(slug: string): Promise<Document | null> {
const [row] = await db.select().from(documents).where(eq(documents.slug, slug));
if (!row) return null;
// Prefer the on-disk file (source of truth); fall back to stored body.
const abs = join(CONTENT_ROOT, row.path);
const body = existsSync(abs) ? readFileSync(abs, "utf8") : row.body;
return { ...toMeta(row), body };
}
export async function getRelated(slug: string, limit = 5): Promise<DocumentMeta[]> {
const [row] = await db
.select({ tags: documents.tags })
.from(documents)
.where(eq(documents.slug, slug));
if (!row || row.tags.length === 0) return [];
// Build a text[] array literal for the && (overlap) operator, binding each
// tag as a parameter to avoid SQL injection from frontmatter content.
const arrLit = sql`ARRAY[${sql.join(
row.tags.map((t) => sql.param(t)),
sql`, `,
)}]::text[]`;
const related = await db
.select()
.from(documents)
.where(and(eq(documents.status, "published"), sql`${documents.tags} && ${arrLit}`))
.limit(limit + 1);
return related.filter((d) => d.slug !== slug).map(toMeta).slice(0, limit);
}
export { readContentFile };
+13
View File
@@ -0,0 +1,13 @@
export * from "./content.service";
export * from "./document.service";
export * from "./search.service";
export { toMeta } from "./row-map";
export type {
DocSection,
DocStatus,
DocType,
Document,
DocumentMeta,
SearchHit,
} from "@mcpedia/types";
+3
View File
@@ -0,0 +1,3 @@
// Single source of truth for row→domain mapping lives in @mcpedia/search.
export { toMeta } from "@mcpedia/search";
export type { DocumentRow } from "@mcpedia/db/schema";
+1
View File
@@ -0,0 +1 @@
export { keywordSearch, toTsQuery } from "@mcpedia/search";
+11
View File
@@ -0,0 +1,11 @@
import { defineConfig } from "drizzle-kit";
import { DATABASE_URL } from "@mcpedia/config";
export default defineConfig({
dialect: "postgresql",
schema: "./src/schema.ts",
out: "./drizzle",
dbCredentials: { url: DATABASE_URL },
verbose: true,
strict: true,
});
+19
View File
@@ -0,0 +1,19 @@
CREATE TABLE "documents" (
"id" text PRIMARY KEY NOT NULL,
"slug" text NOT NULL,
"title" text NOT NULL,
"type" text NOT NULL,
"section" text NOT NULL,
"status" text DEFAULT 'published' NOT NULL,
"author" text DEFAULT '' NOT NULL,
"tags" text[] DEFAULT '{}' NOT NULL,
"path" text NOT NULL,
"body" text DEFAULT '' NOT NULL,
"search_vector" "tsvector" GENERATED ALWAYS AS (setweight(to_tsvector('simple', coalesce("documents"."title", '')), 'A') || setweight(to_tsvector('simple', coalesce("documents"."body", '')), 'B')) STORED NOT NULL,
"created_at" timestamp with time zone NOT NULL,
"updated_at" timestamp with time zone NOT NULL,
CONSTRAINT "documents_slug_unique" UNIQUE("slug")
);
--> statement-breakpoint
CREATE INDEX "documents_search_idx" ON "documents" USING gin ("search_vector");--> statement-breakpoint
CREATE INDEX "documents_section_idx" ON "documents" USING btree ("section");
+157
View File
@@ -0,0 +1,157 @@
{
"id": "249f80c0-d953-42f2-a98b-6701e2115856",
"prevId": "00000000-0000-0000-0000-000000000000",
"version": "7",
"dialect": "postgresql",
"tables": {
"public.documents": {
"name": "documents",
"schema": "",
"columns": {
"id": {
"name": "id",
"type": "text",
"primaryKey": true,
"notNull": true
},
"slug": {
"name": "slug",
"type": "text",
"primaryKey": false,
"notNull": true
},
"title": {
"name": "title",
"type": "text",
"primaryKey": false,
"notNull": true
},
"type": {
"name": "type",
"type": "text",
"primaryKey": false,
"notNull": true
},
"section": {
"name": "section",
"type": "text",
"primaryKey": false,
"notNull": true
},
"status": {
"name": "status",
"type": "text",
"primaryKey": false,
"notNull": true,
"default": "'published'"
},
"author": {
"name": "author",
"type": "text",
"primaryKey": false,
"notNull": true,
"default": "''"
},
"tags": {
"name": "tags",
"type": "text[]",
"primaryKey": false,
"notNull": true,
"default": "'{}'"
},
"path": {
"name": "path",
"type": "text",
"primaryKey": false,
"notNull": true
},
"body": {
"name": "body",
"type": "text",
"primaryKey": false,
"notNull": true,
"default": "''"
},
"search_vector": {
"name": "search_vector",
"type": "tsvector",
"primaryKey": false,
"notNull": true,
"generated": {
"as": "setweight(to_tsvector('simple', coalesce(\"documents\".\"title\", '')), 'A') || setweight(to_tsvector('simple', coalesce(\"documents\".\"body\", '')), 'B')",
"type": "stored"
}
},
"created_at": {
"name": "created_at",
"type": "timestamp with time zone",
"primaryKey": false,
"notNull": true
},
"updated_at": {
"name": "updated_at",
"type": "timestamp with time zone",
"primaryKey": false,
"notNull": true
}
},
"indexes": {
"documents_search_idx": {
"name": "documents_search_idx",
"columns": [
{
"expression": "search_vector",
"isExpression": false,
"asc": true,
"nulls": "last"
}
],
"isUnique": false,
"concurrently": false,
"method": "gin",
"with": {}
},
"documents_section_idx": {
"name": "documents_section_idx",
"columns": [
{
"expression": "section",
"isExpression": false,
"asc": true,
"nulls": "last"
}
],
"isUnique": false,
"concurrently": false,
"method": "btree",
"with": {}
}
},
"foreignKeys": {},
"compositePrimaryKeys": {},
"uniqueConstraints": {
"documents_slug_unique": {
"name": "documents_slug_unique",
"nullsNotDistinct": false,
"columns": [
"slug"
]
}
},
"policies": {},
"checkConstraints": {},
"isRLSEnabled": false
}
},
"enums": {},
"schemas": {},
"sequences": {},
"roles": {},
"policies": {},
"views": {},
"_meta": {
"columns": {},
"schemas": {},
"tables": {}
}
}
+13
View File
@@ -0,0 +1,13 @@
{
"version": "7",
"dialect": "postgresql",
"entries": [
{
"idx": 0,
"version": "7",
"when": 1787133375079,
"tag": "0000_grey_toro",
"breakpoints": true
}
]
}
+19
View File
@@ -0,0 +1,19 @@
{
"name": "@mcpedia/db",
"version": "0.1.0",
"private": true,
"type": "module",
"exports": {
".": "./src/client.ts",
"./schema": "./src/schema.ts"
},
"dependencies": {
"@mcpedia/config": "workspace:*",
"drizzle-orm": "^0.38.0",
"postgres": "^3.4.5"
},
"devDependencies": {
"drizzle-kit": "^0.30.0",
"@types/node": "^20"
}
}
+15
View File
@@ -0,0 +1,15 @@
import { drizzle } from "drizzle-orm/postgres-js";
import postgres from "postgres";
import { DATABASE_URL } from "@mcpedia/config";
import * as schema from "./schema";
// PgBouncer (imrnes :6432) uses transaction pooling, which rejects protocol
//-level prepared statements. prepare:false makes postgres-js use simple queries.
const client = postgres(DATABASE_URL, {
prepare: false,
max: 5,
onnotice: () => {},
});
export const db = drizzle(client, { schema });
export { schema, client };
+59
View File
@@ -0,0 +1,59 @@
import { sql, SQL } from "drizzle-orm";
import {
customType,
index,
integer,
pgTable,
text,
timestamp,
} from "drizzle-orm/pg-core";
// tsvector isn't a first-class drizzle type; wrap the raw Postgres type.
const tsvector = customType<{ data: string }>({
dataType() {
return "tsvector";
},
});
export const documents = pgTable(
"documents",
{
id: text("id").primaryKey(), // slug
slug: text("slug").notNull().unique(),
title: text("title").notNull(),
type: text("type").notNull(),
section: text("section").notNull(),
status: text("status").notNull().default("published"),
author: text("author").notNull().default(""),
tags: text("tags").array().notNull().default(sql`'{}'`),
path: text("path").notNull(),
body: text("body").notNull().default(""),
// Weighted search vector: title (A) + body (B), using the 'simple' config so
// mixed ID/EN queries match literally without stemming. Lazy closure over
// `documents` (must NOT reference the table eagerly — it is in TDZ here).
searchVector: tsvector("search_vector")
.notNull()
.generatedAlwaysAs(
(): SQL =>
sql`setweight(to_tsvector('simple', coalesce(${documents.title}, '')), 'A') || setweight(to_tsvector('simple', coalesce(${documents.body}, '')), 'B')`,
),
createdAt: timestamp("created_at", { withTimezone: true }).notNull(),
updatedAt: timestamp("updated_at", { withTimezone: true }).notNull(),
},
(t) => ({
searchIdx: index("documents_search_idx").using("gin", t.searchVector),
sectionIdx: index("documents_section_idx").on(t.section),
}),
);
// Phase 2 (semantic search) — defined here for reference, NOT created yet:
// export const documentChunks = pgTable("document_chunks", {
// id: text("id").primaryKey(),
// documentId: text("document_id").notNull().references(() => documents.id, { onDelete: "cascade" }),
// content: text("content").notNull(),
// position: integer("position").notNull(),
// embedding: customType<{ data: number[] }>({ dataType: () => "vector(1536)" })("embedding"),
// });
export type DocumentRow = typeof documents.$inferSelect;
export type NewDocumentRow = typeof documents.$inferInsert;
+13
View File
@@ -0,0 +1,13 @@
{
"name": "@mcpedia/parser",
"version": "0.1.0",
"private": true,
"type": "module",
"exports": {
".": "./src/index.ts"
},
"dependencies": {
"@mcpedia/types": "workspace:*",
"gray-matter": "^4.0.3"
}
}
+65
View File
@@ -0,0 +1,65 @@
import matter from "gray-matter";
import { readFileSync } from "node:fs";
import type {
DocSection,
DocStatus,
DocType,
DocumentMeta,
} from "@mcpedia/types";
const SECTIONS: DocSection[] = ["docs", "writeups", "research", "notes"];
const VALID_TYPES: DocType[] = [
"documentation",
"writeup",
"research",
"note",
];
export interface ParsedFile {
meta: DocumentMeta;
body: string;
}
/**
* Parse a content markdown file into metadata + body.
* @param absPath absolute path on disk
* @param relPath path relative to content root (e.g. "docs/websocket/contract.md")
*/
export function parseFile(absPath: string, relPath: string): ParsedFile {
const raw = readFileSync(absPath, "utf8");
const { data, content } = matter(raw);
const section: DocSection =
(SECTIONS.find((s) => relPath.startsWith(s + "/")) as DocSection | undefined) ??
"docs";
const slug = relPath.replace(/\.mdx?$/, "");
const type = (VALID_TYPES.includes(data.type) ? data.type : "documentation") as DocType;
const status = (data.status === "draft" ? "draft" : "published") as DocStatus;
const tags: string[] = Array.isArray(data.tags)
? data.tags.map((t: unknown) => String(t))
: [];
const nowIso = new Date().toISOString();
const createdAt = data.created_at ?? nowIso;
const updatedAt = data.updated_at ?? data.created_at ?? nowIso;
const meta: DocumentMeta = {
id: slug,
slug,
title: typeof data.title === "string" ? data.title : slug,
type,
section,
status,
author: typeof data.author === "string" ? data.author : "",
tags,
path: relPath,
createdAt: String(createdAt),
updatedAt: String(updatedAt),
};
return { meta, body: content };
}
+14
View File
@@ -0,0 +1,14 @@
{
"name": "@mcpedia/search",
"version": "0.1.0",
"private": true,
"type": "module",
"exports": {
".": "./src/index.ts"
},
"dependencies": {
"@mcpedia/db": "workspace:*",
"@mcpedia/types": "workspace:*",
"drizzle-orm": "^0.38.0"
}
}
+79
View File
@@ -0,0 +1,79 @@
import { db } from "@mcpedia/db";
import { documents, type DocumentRow } from "@mcpedia/db/schema";
import { and, eq, sql } from "drizzle-orm";
import type {
DocSection,
DocStatus,
DocType,
DocumentMeta,
SearchHit,
} from "@mcpedia/types";
const VALID_SECTIONS: DocSection[] = ["docs", "writeups", "research", "notes"];
const VALID_TYPES: DocType[] = ["documentation", "writeup", "research", "note"];
/** Map a Drizzle row (text columns, Date timestamps) into the strict types. */
function toMeta(row: DocumentRow): DocumentMeta {
return {
id: row.id,
slug: row.slug,
title: row.title,
type: (VALID_TYPES.includes(row.type as DocType) ? row.type : "documentation") as DocType,
section: (VALID_SECTIONS.includes(row.section as DocSection)
? row.section
: "docs") as DocSection,
status: (row.status === "draft" ? "draft" : "published") as DocStatus,
author: row.author,
tags: row.tags,
path: row.path,
createdAt: row.createdAt.toISOString(),
updatedAt: row.updatedAt.toISOString(),
};
}
/** Build a prefix tsquery from free text (AND of terms, each as prefix). */
export function toTsQuery(q: string): string {
const terms = q
.trim()
.split(/\s+/)
.filter(Boolean)
.map((t) => t.replace(/[^a-zA-Z0-9_]/g, ""))
.filter(Boolean);
if (terms.length === 0) return "";
return terms.map((t) => `${t}:*`).join(" & ");
}
/**
* Postgres FTS keyword search over published documents.
* Ranks by ts_rank and returns a headline snippet for display.
*/
export async function keywordSearch(q: string, limit = 20): Promise<SearchHit[]> {
const query = toTsQuery(q);
if (!query) return [];
const rows = await db
.select({
doc: documents,
rank: sql<number>`ts_rank(${documents.searchVector}, to_tsquery('simple', ${query}))`,
snippet: sql<string>`ts_headline('simple', ${documents.body}, to_tsquery('simple', ${query}), 'MaxWords=25, MinWords=5, StartSel=<mark>, StopSel=</mark>')`,
})
.from(documents)
.where(
and(
eq(documents.status, "published"),
sql`${documents.searchVector} @@ to_tsquery('simple', ${query})`,
),
)
.orderBy(
sql`ts_rank(${documents.searchVector}, to_tsquery('simple', ${query})) desc`,
)
.limit(limit);
return rows.map((r) => ({
doc: toMeta(r.doc),
rank: r.rank,
snippet: r.snippet,
}));
}
export { toMeta };
+9
View File
@@ -0,0 +1,9 @@
{
"name": "@mcpedia/types",
"version": "0.1.0",
"private": true,
"type": "module",
"exports": {
".": "./src/index.ts"
}
}
+27
View File
@@ -0,0 +1,27 @@
export type DocSection = "docs" | "writeups" | "research" | "notes";
export type DocType = "documentation" | "writeup" | "research" | "note";
export type DocStatus = "published" | "draft";
export interface DocumentMeta {
id: string; // slug
slug: string;
title: string;
type: DocType;
section: DocSection;
status: DocStatus;
author: string;
tags: string[];
path: string; // relative path under content/
createdAt: string; // ISO
updatedAt: string; // ISO
}
export interface Document extends DocumentMeta {
body: string; // raw markdown (read from disk or stored)
}
export interface SearchHit {
doc: DocumentMeta;
rank: number;
snippet: string;
}