refactor(ai): remove semantic embedding cache + Qdrant vector store
Hapus seluruh fitur embedding/Qdrant (tidak dipakai lagi): - gateway: drop embeddingClient.ts, qdrantClient.ts, archiveEmbedder.ts dan tes qdrantEnsure.test.ts; moderationOrchestrator kembali ke exact-hash cache -> LLM (tanpa phase-2 semantic lookup); textCacheStore kehilangan findSimilarTextModeration / parseQdrantVerdict / isSemanticBandAccepted / upsertBareKeyToQdrant; cache-prune hanya menyapu Postgres. - backend: drop embed.ts + qdrant.ts, endpoint messages.semanticSearch dan schema/type terkait; kolom embedding dilepas dari schema text_analysis_cache. - frontend: hapus toggle EXACT/SEMANTIC, hook useSemanticSearch, API client + tipe SemanticSearchResult. - config: buang AI_LLM_EMBEDDING_* dan QDRANT_* (env + .env.example). - docs: ARCHITECTURE.md / AGENTS.md / README.md / diagram arsitektur disesuaikan (LLM caller - vision, cache = exact-hash saja). Verifikasi: tsc 0 (backend, gateway, frontend); bun test 135 pass + 37 pass, 0 fail; biome 0 error.
This commit is contained in:
@@ -60,7 +60,7 @@ src/
|
||||
| moderation | Moderation actions & metrics | `ai_moderations`, `moderation_actions` |
|
||||
| media | Media file management | `media_attachments` |
|
||||
| dashboard | Stats aggregation | Various (read-only) |
|
||||
| knowledge | Semantic search | Qdrant vector DB |
|
||||
| knowledge | Channel cultures & glossary browser | `channel_cultures`, `term_glossary_cache` |
|
||||
| chatbot | AI chatbot with tools | `chatbot_history` |
|
||||
| health | Health checks + metrics | Various |
|
||||
| analysis | Text analysis cache | `text_analysis_cache` |
|
||||
|
||||
@@ -1,65 +0,0 @@
|
||||
import { config } from "@/shared/config/index";
|
||||
import { createChildLogger } from "@/shared/logger/index";
|
||||
|
||||
const logger = createChildLogger("messages-embed");
|
||||
|
||||
/** Max chars for a search query fed to the embedding model. */
|
||||
const MAX_QUERY_CHARS = 300;
|
||||
|
||||
/**
|
||||
* Normalize a user search query before embedding so it lands in the same
|
||||
* vector space as the archived content (which is normalized the same way on
|
||||
* write). Mirrors the gateway's normalizer: strip control/zero-width chars,
|
||||
* lowercase, collapse whitespace, cap length. Readable punctuation is kept —
|
||||
* a search query is already compact.
|
||||
*/
|
||||
export function normalizeEmbeddingQuery(raw: string): string {
|
||||
if (!raw) return "";
|
||||
return raw
|
||||
.replace(/[\p{Cc}\p{Cf}]/gu, " ")
|
||||
.toLowerCase()
|
||||
.replace(/\s+/g, " ")
|
||||
.trim()
|
||||
.slice(0, MAX_QUERY_CHARS);
|
||||
}
|
||||
|
||||
/**
|
||||
* Embed a search query with the configured OpenAI-compatible embedding model.
|
||||
* Uses raw fetch (the backend has no openai SDK dependency) and returns null
|
||||
* when embeddings are not configured (search unavailable).
|
||||
*
|
||||
* encoding_format: "float" is REQUIRED — Nvidia-backed models reject base64.
|
||||
*/
|
||||
export async function embedQuery(rawQuery: string): Promise<number[] | null> {
|
||||
if (!config.AI_LLM_API_KEY || !config.AI_LLM_EMBEDDING_MODEL) return null;
|
||||
const text = normalizeEmbeddingQuery(rawQuery);
|
||||
if (!text) return null;
|
||||
try {
|
||||
const res = await fetch(`${config.AI_LLM_BASE_URL}/embeddings`, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
"Content-Type": "application/json",
|
||||
Authorization: `Bearer ${config.AI_LLM_API_KEY}`,
|
||||
},
|
||||
body: JSON.stringify({
|
||||
model: config.AI_LLM_EMBEDDING_MODEL,
|
||||
input: text,
|
||||
encoding_format: "float",
|
||||
}),
|
||||
});
|
||||
if (!res.ok) {
|
||||
logger.warn({ status: res.status }, "query embed HTTP error");
|
||||
return null;
|
||||
}
|
||||
const json = (await res.json()) as {
|
||||
data?: Array<{ embedding?: number[] }>;
|
||||
};
|
||||
return json.data?.[0]?.embedding ?? null;
|
||||
} catch (error) {
|
||||
logger.warn(
|
||||
{ error: error instanceof Error ? error.message : String(error) },
|
||||
"query embed failed",
|
||||
);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
@@ -41,11 +41,3 @@ export const messageUpdateSchema = z.object({
|
||||
export type MessageQuery = z.infer<typeof messageQuerySchema>;
|
||||
export type MessageCreate = z.infer<typeof messageCreateSchema>;
|
||||
export type MessageUpdate = z.infer<typeof messageUpdateSchema>;
|
||||
|
||||
export const semanticSearchSchema = z.object({
|
||||
query: z.string().min(1).max(500),
|
||||
limit: z.coerce.number().int().positive().max(50).default(10),
|
||||
guildId: z.string().optional(),
|
||||
});
|
||||
|
||||
export type SemanticSearchQuery = z.infer<typeof semanticSearchSchema>;
|
||||
|
||||
@@ -1,10 +1,7 @@
|
||||
import { config } from "@/shared/config/index";
|
||||
import { NotFoundError, ValidationError } from "@/shared/errors/index";
|
||||
import { createChildLogger } from "@/shared/logger/index";
|
||||
import { embedQuery } from "./embed.js";
|
||||
import { type MessageRow, messagesRepository } from "./messages.repository.js";
|
||||
import type { MessageQuery, SemanticSearchQuery } from "./messages.schema.js";
|
||||
import { searchArchive } from "./qdrant.js";
|
||||
import type { MessageQuery } from "./messages.schema.js";
|
||||
|
||||
const logger = createChildLogger("messages.service");
|
||||
|
||||
@@ -102,32 +99,6 @@ export class MessagesService {
|
||||
return messagesRepository.getReviewMessages(channelId, limit);
|
||||
}
|
||||
|
||||
/**
|
||||
* Public, read-only semantic search over the persistent message archive.
|
||||
* Embeds the query, searches Qdrant, returns text + metadata. Best-effort:
|
||||
* if embeddings/Qdrant are unavailable, returns an empty result set.
|
||||
*/
|
||||
async semanticSearch(
|
||||
input: SemanticSearchQuery,
|
||||
): Promise<{ results: ReturnType<typeof mapSearchHit>[]; nextCursor: null }> {
|
||||
const vector = await embedQuery(input.query);
|
||||
if (!vector) {
|
||||
logger.debug(
|
||||
{ query: input.query },
|
||||
"semantic search skipped: no embedder",
|
||||
);
|
||||
return { results: [], nextCursor: null };
|
||||
}
|
||||
const hits = await searchArchive(
|
||||
vector,
|
||||
input.limit,
|
||||
config.AI_LLM_EMBEDDING_ARCHIVE_MIN_SIMILARITY,
|
||||
input.guildId,
|
||||
);
|
||||
const results = hits.map((h) => mapSearchHit(h));
|
||||
return { results, nextCursor: null };
|
||||
}
|
||||
|
||||
async getActivity(
|
||||
days = 30,
|
||||
): Promise<Awaited<ReturnType<typeof messagesRepository.getActivity>>> {
|
||||
@@ -157,36 +128,4 @@ export class MessagesService {
|
||||
}
|
||||
}
|
||||
|
||||
/** Shape returned to the frontend (text + rich metadata from the archive payload). */
|
||||
function mapSearchHit(hit: {
|
||||
score: number;
|
||||
payload: {
|
||||
text: string;
|
||||
content_hash?: string;
|
||||
analyzed_at: number;
|
||||
username?: string;
|
||||
channel_id?: string;
|
||||
guild_id?: string;
|
||||
thread_id?: string | null;
|
||||
channel_name?: string | null;
|
||||
thread_name?: string | null;
|
||||
created_at?: number;
|
||||
};
|
||||
}) {
|
||||
return {
|
||||
message_id: hit.payload.content_hash ?? null,
|
||||
content: hit.payload.text,
|
||||
score: hit.score,
|
||||
// Prefer the real message timestamp; fall back to embed time for old
|
||||
// points that predate rich metadata.
|
||||
created_at: hit.payload.created_at ?? hit.payload.analyzed_at,
|
||||
username: hit.payload.username ?? null,
|
||||
channel_id: hit.payload.channel_id ?? null,
|
||||
guild_id: hit.payload.guild_id ?? null,
|
||||
thread_id: hit.payload.thread_id ?? null,
|
||||
channel_name: hit.payload.channel_name ?? null,
|
||||
thread_name: hit.payload.thread_name ?? null,
|
||||
};
|
||||
}
|
||||
|
||||
export const messagesService = new MessagesService();
|
||||
|
||||
@@ -1,114 +0,0 @@
|
||||
import { config } from "@/shared/config/index";
|
||||
import { createChildLogger } from "@/shared/logger/index";
|
||||
|
||||
const logger = createChildLogger("messages-qdrant");
|
||||
|
||||
export interface ArchiveHit {
|
||||
score: number;
|
||||
payload: {
|
||||
text: string;
|
||||
content_hash?: string;
|
||||
analyzed_at: number;
|
||||
expires_at: number;
|
||||
username?: string;
|
||||
channel_id?: string;
|
||||
guild_id?: string;
|
||||
thread_id?: string | null;
|
||||
channel_name?: string | null;
|
||||
thread_name?: string | null;
|
||||
created_at?: number;
|
||||
};
|
||||
}
|
||||
|
||||
function baseUrl(): string {
|
||||
return (config.QDRANT_URL ?? "http://100.121.180.82:6333").replace(
|
||||
/\/+$/,
|
||||
"",
|
||||
);
|
||||
}
|
||||
|
||||
function headers(): Record<string, string> {
|
||||
const h: Record<string, string> = { "Content-Type": "application/json" };
|
||||
if (config.QDRANT_API_KEY) h["api-key"] = config.QDRANT_API_KEY;
|
||||
return h;
|
||||
}
|
||||
|
||||
export const ARCHIVE_COLLECTION =
|
||||
config.QDRANT_ARCHIVE_COLLECTION ?? "gmw_message_archive";
|
||||
|
||||
async function request(
|
||||
method: string,
|
||||
path: string,
|
||||
body?: unknown,
|
||||
timeoutMs = 10_000,
|
||||
): Promise<unknown> {
|
||||
const controller = new AbortController();
|
||||
const timer = setTimeout(() => controller.abort(), timeoutMs);
|
||||
try {
|
||||
const res = await fetch(`${baseUrl()}${path}`, {
|
||||
method,
|
||||
headers: headers(),
|
||||
body: body === undefined ? undefined : JSON.stringify(body),
|
||||
signal: controller.signal,
|
||||
});
|
||||
const text = await res.text();
|
||||
if (!res.ok) {
|
||||
throw new Error(
|
||||
`Qdrant ${method} ${path} -> ${res.status}: ${text.slice(0, 200)}`,
|
||||
);
|
||||
}
|
||||
return text ? JSON.parse(text) : null;
|
||||
} finally {
|
||||
clearTimeout(timer);
|
||||
}
|
||||
}
|
||||
|
||||
/** Search the archive collection for the nearest vectors to `vector`. */
|
||||
export async function searchArchive(
|
||||
vector: number[],
|
||||
limit: number,
|
||||
scoreThreshold: number,
|
||||
guildId?: string,
|
||||
): Promise<ArchiveHit[]> {
|
||||
if (!config.QDRANT_URL) return [];
|
||||
try {
|
||||
const json = (await request(
|
||||
"POST",
|
||||
`/collections/${ARCHIVE_COLLECTION}/points/search`,
|
||||
{
|
||||
vector,
|
||||
limit,
|
||||
score_threshold: scoreThreshold,
|
||||
with_payload: true,
|
||||
// Optional scope: only return vectors from a specific guild's archive.
|
||||
// Old points (embedded before rich metadata) have no guild_id payload —
|
||||
// the `must` match simply excludes them, which is the correct behavior
|
||||
// for a guild-scoped search.
|
||||
...(guildId
|
||||
? {
|
||||
filter: {
|
||||
must: [{ key: "guild_id", match: { value: guildId } }],
|
||||
},
|
||||
}
|
||||
: {}),
|
||||
},
|
||||
)) as {
|
||||
result?: Array<{
|
||||
score?: number;
|
||||
payload?: ArchiveHit["payload"];
|
||||
}>;
|
||||
};
|
||||
return (json.result ?? [])
|
||||
.filter((h) => h.payload?.text)
|
||||
.map((h) => ({
|
||||
score: h.score ?? 0,
|
||||
payload: h.payload as ArchiveHit["payload"],
|
||||
}));
|
||||
} catch (error) {
|
||||
logger.warn(
|
||||
{ error: error instanceof Error ? error.message : String(error) },
|
||||
"archive search failed",
|
||||
);
|
||||
return [];
|
||||
}
|
||||
}
|
||||
@@ -5,10 +5,7 @@ import { chatRequestSchema } from "../modules/chatbot/chatbot.schema";
|
||||
import { chatbotService } from "../modules/chatbot/chatbot.service";
|
||||
import { dashboardService } from "../modules/dashboard/dashboard.service";
|
||||
import { knowledgeService } from "../modules/knowledge/knowledge.service";
|
||||
import {
|
||||
messageQuerySchema,
|
||||
semanticSearchSchema,
|
||||
} from "../modules/messages/messages.schema";
|
||||
import { messageQuerySchema } from "../modules/messages/messages.schema";
|
||||
import { messagesService } from "../modules/messages/messages.service";
|
||||
import { moderationService } from "../modules/moderation/moderation.service";
|
||||
import { uiStateService } from "../modules/ui-state/ui-state.service";
|
||||
@@ -123,10 +120,6 @@ const messagesRouter = {
|
||||
);
|
||||
return { results: rows, limit: input.limit, cursor: null };
|
||||
}),
|
||||
// Public, read-only semantic search over the message archive.
|
||||
semanticSearch: os
|
||||
.input(semanticSearchSchema)
|
||||
.handler(({ input }) => messagesService.semanticSearch(input)),
|
||||
// Public, read-only activity heatmap data (per-hour volume by channel).
|
||||
activity: os
|
||||
.input(
|
||||
|
||||
@@ -102,15 +102,6 @@ export const configSchema = z
|
||||
AI_LLM_BASE_URL: z.string().url().default("http://127.0.0.1:4014/v1"),
|
||||
AI_LLM_MODEL: z.string().default("text"),
|
||||
AI_LLM_VISION_MODEL: z.string().optional(),
|
||||
AI_LLM_EMBEDDING_MODEL: z.string().optional(),
|
||||
// Minimum cosine similarity for the public archive semantic search. Lower
|
||||
// = more (noisier) results; raise it to tighten precision. Tuned for a 1B
|
||||
// embedding model — re-tune if the model's dimensionality changes.
|
||||
AI_LLM_EMBEDDING_ARCHIVE_MIN_SIMILARITY: z.coerce
|
||||
.number()
|
||||
.min(0)
|
||||
.max(1)
|
||||
.default(0.6),
|
||||
AI_LLM_MAX_CONCURRENT: z.coerce.number().int().positive().default(5),
|
||||
AI_LLM_IMAGE_MAX_DIMENSION: z.coerce
|
||||
.number()
|
||||
@@ -179,11 +170,6 @@ export const configSchema = z
|
||||
.default("https://api.openai.com/v1"),
|
||||
OPENAI_MODERATION_MODEL: z.string().default("omni-moderation-latest"),
|
||||
|
||||
// ── Qdrant (message archive for semantic search) ──────────────────
|
||||
QDRANT_URL: z.string().optional(),
|
||||
QDRANT_API_KEY: z.string().optional(),
|
||||
QDRANT_ARCHIVE_COLLECTION: z.string().default("gmw_message_archive"),
|
||||
|
||||
// ── Auto Delete ─────────────────────────────────────────────────────
|
||||
AUTO_DELETE_FLAGGED_ENABLED: z
|
||||
.string()
|
||||
|
||||
Reference in New Issue
Block a user