refactor(ai): remove semantic embedding cache + Qdrant vector store

Hapus seluruh fitur embedding/Qdrant (tidak dipakai lagi):

- gateway: drop embeddingClient.ts, qdrantClient.ts, archiveEmbedder.ts
  dan tes qdrantEnsure.test.ts; moderationOrchestrator kembali ke
  exact-hash cache -> LLM (tanpa phase-2 semantic lookup); textCacheStore
  kehilangan findSimilarTextModeration / parseQdrantVerdict /
  isSemanticBandAccepted / upsertBareKeyToQdrant; cache-prune hanya
  menyapu Postgres.
- backend: drop embed.ts + qdrant.ts, endpoint messages.semanticSearch
  dan schema/type terkait; kolom embedding dilepas dari schema
  text_analysis_cache.
- frontend: hapus toggle EXACT/SEMANTIC, hook useSemanticSearch,
  API client + tipe SemanticSearchResult.
- config: buang AI_LLM_EMBEDDING_* dan QDRANT_* (env + .env.example).
- docs: ARCHITECTURE.md / AGENTS.md / README.md / diagram arsitektur
  disesuaikan (LLM caller - vision, cache = exact-hash saja).

Verifikasi: tsc 0 (backend, gateway, frontend); bun test 135 pass +
37 pass, 0 fail; biome 0 error.
This commit is contained in:
mytheclipsebotreview
2026-09-25 01:38:51 +07:00
parent fdc7f01c26
commit 2b6ec59286
30 changed files with 41 additions and 1987 deletions
-5
View File
@@ -64,11 +64,6 @@ AI_LLM_API_KEY= # REQUIRED if AI_ANALYSIS_ENABLED=true. L
AI_LLM_BASE_URL=http://100.121.180.82:20128/api/v1 # LLM API base URL (omniroute — OpenAI-compatible router on imrnes, Tailscale 100.121.180.82) AI_LLM_BASE_URL=http://100.121.180.82:20128/api/v1 # LLM API base URL (omniroute — OpenAI-compatible router on imrnes, Tailscale 100.121.180.82)
AI_LLM_MODEL=text # LLM text model name (default: text) AI_LLM_MODEL=text # LLM text model name (default: text)
# AI_LLM_VISION_MODEL= # Vision model for image analysis (falls back to AI_LLM_MODEL) # AI_LLM_VISION_MODEL= # Vision model for image analysis (falls back to AI_LLM_MODEL)
# AI_LLM_EMBEDDING_MODEL= # Embedding model for semantic moderation cache (optional; enables near-duplicate text reuse to save LLM calls)
# AI_LLM_EMBEDDING_MIN_SIMILARITY=0.97 # Min cosine similarity to reuse a cached verdict (default: 0.97)
QDRANT_URL=http://100.121.180.82:6333 # Qdrant vector store for embeddings (semantic cache); when set, vectors are stored/searched in Qdrant instead of Postgres
# QDRANT_COLLECTION=gmw_text_moderation # Qdrant collection name (default: gmw_text_moderation)
# QDRANT_API_KEY= # Qdrant API key (optional)
AI_LLM_MAX_CONCURRENT=5 # Max concurrent LLM API calls (default: 5) AI_LLM_MAX_CONCURRENT=5 # Max concurrent LLM API calls (default: 5)
AI_LLM_IMAGE_MAX_DIMENSION=1024 # Max image dimension in pixels before resize (default: 1024) AI_LLM_IMAGE_MAX_DIMENSION=1024 # Max image dimension in pixels before resize (default: 1024)
AI_LLM_TEXT_BATCH_SIZE=20 # Max messages per text-only moderation batch (default: 20) AI_LLM_TEXT_BATCH_SIZE=20 # Max messages per text-only moderation batch (default: 20)
+1 -1
View File
@@ -14,7 +14,7 @@
{ "id": "gateway", "type": "backend", "label": "discord-gateway", "sublabel": "selfbot :4016", "pos": [260, 240], "size": [140, 60], "tag": "discord.js-selfbot-v13" }, { "id": "gateway", "type": "backend", "label": "discord-gateway", "sublabel": "selfbot :4016", "pos": [260, 240], "size": [140, 60], "tag": "discord.js-selfbot-v13" },
{ "id": "backend", "type": "backend", "label": "gmw-backend", "sublabel": "Express · oRPC · WS :4001", "pos": [640, 240], "size": [140, 60], "tag": "Drizzle ORM" }, { "id": "backend", "type": "backend", "label": "gmw-backend", "sublabel": "Express · oRPC · WS :4001", "pos": [640, 240], "size": [140, 60], "tag": "Drizzle ORM" },
{ "id": "proxy", "type": "cloud", "label": "nginx proxy", "sublabel": "reverse proxy :4009", "pos": [960, 240], "size": [140, 60], "tag": "nginx" }, { "id": "proxy", "type": "cloud", "label": "nginx proxy", "sublabel": "reverse proxy :4009", "pos": [960, 240], "size": [140, 60], "tag": "nginx" },
{ "id": "ai", "type": "backend", "label": "AI Moderation", "sublabel": "LLM caller · embeddings", "pos": [260, 400], "size": [140, 60] }, { "id": "ai", "type": "backend", "label": "AI Moderation", "sublabel": "LLM caller · vision", "pos": [260, 400], "size": [140, 60] },
{ "id": "frontend", "type": "frontend", "label": "gmw-frontend", "sublabel": "Next.js 16 SSR :4017", "pos": [960, 400], "size": [140, 60], "tag": "React 19 · Tailwind v4" }, { "id": "frontend", "type": "frontend", "label": "gmw-frontend", "sublabel": "Next.js 16 SSR :4017", "pos": [960, 400], "size": [140, 60], "tag": "React 19 · Tailwind v4" },
{ "id": "llm", "type": "cloud", "label": "9router LLM", "sublabel": "text + vision API", "pos": [260, 540], "size": [140, 60], "tag": "AI_LLM_BASE_URL" }, { "id": "llm", "type": "cloud", "label": "9router LLM", "sublabel": "text + vision API", "pos": [260, 540], "size": [140, 60], "tag": "AI_LLM_BASE_URL" },
{ "id": "browser", "type": "external", "label": "Dashboard Users", "sublabel": "browser · partysocket", "pos": [960, 540], "size": [140, 60] } { "id": "browser", "type": "external", "label": "Dashboard Users", "sublabel": "browser · partysocket", "pos": [960, 540], "size": [140, 60] }
+1 -1
View File
@@ -60,7 +60,7 @@ src/
| moderation | Moderation actions & metrics | `ai_moderations`, `moderation_actions` | | moderation | Moderation actions & metrics | `ai_moderations`, `moderation_actions` |
| media | Media file management | `media_attachments` | | media | Media file management | `media_attachments` |
| dashboard | Stats aggregation | Various (read-only) | | dashboard | Stats aggregation | Various (read-only) |
| knowledge | Semantic search | Qdrant vector DB | | knowledge | Channel cultures & glossary browser | `channel_cultures`, `term_glossary_cache` |
| chatbot | AI chatbot with tools | `chatbot_history` | | chatbot | AI chatbot with tools | `chatbot_history` |
| health | Health checks + metrics | Various | | health | Health checks + metrics | Various |
| analysis | Text analysis cache | `text_analysis_cache` | | analysis | Text analysis cache | `text_analysis_cache` |
@@ -1,65 +0,0 @@
import { config } from "@/shared/config/index";
import { createChildLogger } from "@/shared/logger/index";
const logger = createChildLogger("messages-embed");
/** Max chars for a search query fed to the embedding model. */
const MAX_QUERY_CHARS = 300;
/**
* Normalize a user search query before embedding so it lands in the same
* vector space as the archived content (which is normalized the same way on
* write). Mirrors the gateway's normalizer: strip control/zero-width chars,
* lowercase, collapse whitespace, cap length. Readable punctuation is kept —
* a search query is already compact.
*/
export function normalizeEmbeddingQuery(raw: string): string {
if (!raw) return "";
return raw
.replace(/[\p{Cc}\p{Cf}]/gu, " ")
.toLowerCase()
.replace(/\s+/g, " ")
.trim()
.slice(0, MAX_QUERY_CHARS);
}
/**
* Embed a search query with the configured OpenAI-compatible embedding model.
* Uses raw fetch (the backend has no openai SDK dependency) and returns null
* when embeddings are not configured (search unavailable).
*
* encoding_format: "float" is REQUIRED — Nvidia-backed models reject base64.
*/
export async function embedQuery(rawQuery: string): Promise<number[] | null> {
if (!config.AI_LLM_API_KEY || !config.AI_LLM_EMBEDDING_MODEL) return null;
const text = normalizeEmbeddingQuery(rawQuery);
if (!text) return null;
try {
const res = await fetch(`${config.AI_LLM_BASE_URL}/embeddings`, {
method: "POST",
headers: {
"Content-Type": "application/json",
Authorization: `Bearer ${config.AI_LLM_API_KEY}`,
},
body: JSON.stringify({
model: config.AI_LLM_EMBEDDING_MODEL,
input: text,
encoding_format: "float",
}),
});
if (!res.ok) {
logger.warn({ status: res.status }, "query embed HTTP error");
return null;
}
const json = (await res.json()) as {
data?: Array<{ embedding?: number[] }>;
};
return json.data?.[0]?.embedding ?? null;
} catch (error) {
logger.warn(
{ error: error instanceof Error ? error.message : String(error) },
"query embed failed",
);
return null;
}
}
@@ -41,11 +41,3 @@ export const messageUpdateSchema = z.object({
export type MessageQuery = z.infer<typeof messageQuerySchema>; export type MessageQuery = z.infer<typeof messageQuerySchema>;
export type MessageCreate = z.infer<typeof messageCreateSchema>; export type MessageCreate = z.infer<typeof messageCreateSchema>;
export type MessageUpdate = z.infer<typeof messageUpdateSchema>; export type MessageUpdate = z.infer<typeof messageUpdateSchema>;
export const semanticSearchSchema = z.object({
query: z.string().min(1).max(500),
limit: z.coerce.number().int().positive().max(50).default(10),
guildId: z.string().optional(),
});
export type SemanticSearchQuery = z.infer<typeof semanticSearchSchema>;
@@ -1,10 +1,7 @@
import { config } from "@/shared/config/index";
import { NotFoundError, ValidationError } from "@/shared/errors/index"; import { NotFoundError, ValidationError } from "@/shared/errors/index";
import { createChildLogger } from "@/shared/logger/index"; import { createChildLogger } from "@/shared/logger/index";
import { embedQuery } from "./embed.js";
import { type MessageRow, messagesRepository } from "./messages.repository.js"; import { type MessageRow, messagesRepository } from "./messages.repository.js";
import type { MessageQuery, SemanticSearchQuery } from "./messages.schema.js"; import type { MessageQuery } from "./messages.schema.js";
import { searchArchive } from "./qdrant.js";
const logger = createChildLogger("messages.service"); const logger = createChildLogger("messages.service");
@@ -102,32 +99,6 @@ export class MessagesService {
return messagesRepository.getReviewMessages(channelId, limit); return messagesRepository.getReviewMessages(channelId, limit);
} }
/**
* Public, read-only semantic search over the persistent message archive.
* Embeds the query, searches Qdrant, returns text + metadata. Best-effort:
* if embeddings/Qdrant are unavailable, returns an empty result set.
*/
async semanticSearch(
input: SemanticSearchQuery,
): Promise<{ results: ReturnType<typeof mapSearchHit>[]; nextCursor: null }> {
const vector = await embedQuery(input.query);
if (!vector) {
logger.debug(
{ query: input.query },
"semantic search skipped: no embedder",
);
return { results: [], nextCursor: null };
}
const hits = await searchArchive(
vector,
input.limit,
config.AI_LLM_EMBEDDING_ARCHIVE_MIN_SIMILARITY,
input.guildId,
);
const results = hits.map((h) => mapSearchHit(h));
return { results, nextCursor: null };
}
async getActivity( async getActivity(
days = 30, days = 30,
): Promise<Awaited<ReturnType<typeof messagesRepository.getActivity>>> { ): Promise<Awaited<ReturnType<typeof messagesRepository.getActivity>>> {
@@ -157,36 +128,4 @@ export class MessagesService {
} }
} }
/** Shape returned to the frontend (text + rich metadata from the archive payload). */
function mapSearchHit(hit: {
score: number;
payload: {
text: string;
content_hash?: string;
analyzed_at: number;
username?: string;
channel_id?: string;
guild_id?: string;
thread_id?: string | null;
channel_name?: string | null;
thread_name?: string | null;
created_at?: number;
};
}) {
return {
message_id: hit.payload.content_hash ?? null,
content: hit.payload.text,
score: hit.score,
// Prefer the real message timestamp; fall back to embed time for old
// points that predate rich metadata.
created_at: hit.payload.created_at ?? hit.payload.analyzed_at,
username: hit.payload.username ?? null,
channel_id: hit.payload.channel_id ?? null,
guild_id: hit.payload.guild_id ?? null,
thread_id: hit.payload.thread_id ?? null,
channel_name: hit.payload.channel_name ?? null,
thread_name: hit.payload.thread_name ?? null,
};
}
export const messagesService = new MessagesService(); export const messagesService = new MessagesService();
@@ -1,114 +0,0 @@
import { config } from "@/shared/config/index";
import { createChildLogger } from "@/shared/logger/index";
const logger = createChildLogger("messages-qdrant");
export interface ArchiveHit {
score: number;
payload: {
text: string;
content_hash?: string;
analyzed_at: number;
expires_at: number;
username?: string;
channel_id?: string;
guild_id?: string;
thread_id?: string | null;
channel_name?: string | null;
thread_name?: string | null;
created_at?: number;
};
}
function baseUrl(): string {
return (config.QDRANT_URL ?? "http://100.121.180.82:6333").replace(
/\/+$/,
"",
);
}
function headers(): Record<string, string> {
const h: Record<string, string> = { "Content-Type": "application/json" };
if (config.QDRANT_API_KEY) h["api-key"] = config.QDRANT_API_KEY;
return h;
}
export const ARCHIVE_COLLECTION =
config.QDRANT_ARCHIVE_COLLECTION ?? "gmw_message_archive";
async function request(
method: string,
path: string,
body?: unknown,
timeoutMs = 10_000,
): Promise<unknown> {
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeoutMs);
try {
const res = await fetch(`${baseUrl()}${path}`, {
method,
headers: headers(),
body: body === undefined ? undefined : JSON.stringify(body),
signal: controller.signal,
});
const text = await res.text();
if (!res.ok) {
throw new Error(
`Qdrant ${method} ${path} -> ${res.status}: ${text.slice(0, 200)}`,
);
}
return text ? JSON.parse(text) : null;
} finally {
clearTimeout(timer);
}
}
/** Search the archive collection for the nearest vectors to `vector`. */
export async function searchArchive(
vector: number[],
limit: number,
scoreThreshold: number,
guildId?: string,
): Promise<ArchiveHit[]> {
if (!config.QDRANT_URL) return [];
try {
const json = (await request(
"POST",
`/collections/${ARCHIVE_COLLECTION}/points/search`,
{
vector,
limit,
score_threshold: scoreThreshold,
with_payload: true,
// Optional scope: only return vectors from a specific guild's archive.
// Old points (embedded before rich metadata) have no guild_id payload —
// the `must` match simply excludes them, which is the correct behavior
// for a guild-scoped search.
...(guildId
? {
filter: {
must: [{ key: "guild_id", match: { value: guildId } }],
},
}
: {}),
},
)) as {
result?: Array<{
score?: number;
payload?: ArchiveHit["payload"];
}>;
};
return (json.result ?? [])
.filter((h) => h.payload?.text)
.map((h) => ({
score: h.score ?? 0,
payload: h.payload as ArchiveHit["payload"],
}));
} catch (error) {
logger.warn(
{ error: error instanceof Error ? error.message : String(error) },
"archive search failed",
);
return [];
}
}
+1 -8
View File
@@ -5,10 +5,7 @@ import { chatRequestSchema } from "../modules/chatbot/chatbot.schema";
import { chatbotService } from "../modules/chatbot/chatbot.service"; import { chatbotService } from "../modules/chatbot/chatbot.service";
import { dashboardService } from "../modules/dashboard/dashboard.service"; import { dashboardService } from "../modules/dashboard/dashboard.service";
import { knowledgeService } from "../modules/knowledge/knowledge.service"; import { knowledgeService } from "../modules/knowledge/knowledge.service";
import { import { messageQuerySchema } from "../modules/messages/messages.schema";
messageQuerySchema,
semanticSearchSchema,
} from "../modules/messages/messages.schema";
import { messagesService } from "../modules/messages/messages.service"; import { messagesService } from "../modules/messages/messages.service";
import { moderationService } from "../modules/moderation/moderation.service"; import { moderationService } from "../modules/moderation/moderation.service";
import { uiStateService } from "../modules/ui-state/ui-state.service"; import { uiStateService } from "../modules/ui-state/ui-state.service";
@@ -123,10 +120,6 @@ const messagesRouter = {
); );
return { results: rows, limit: input.limit, cursor: null }; return { results: rows, limit: input.limit, cursor: null };
}), }),
// Public, read-only semantic search over the message archive.
semanticSearch: os
.input(semanticSearchSchema)
.handler(({ input }) => messagesService.semanticSearch(input)),
// Public, read-only activity heatmap data (per-hour volume by channel). // Public, read-only activity heatmap data (per-hour volume by channel).
activity: os activity: os
.input( .input(
@@ -102,15 +102,6 @@ export const configSchema = z
AI_LLM_BASE_URL: z.string().url().default("http://127.0.0.1:4014/v1"), AI_LLM_BASE_URL: z.string().url().default("http://127.0.0.1:4014/v1"),
AI_LLM_MODEL: z.string().default("text"), AI_LLM_MODEL: z.string().default("text"),
AI_LLM_VISION_MODEL: z.string().optional(), AI_LLM_VISION_MODEL: z.string().optional(),
AI_LLM_EMBEDDING_MODEL: z.string().optional(),
// Minimum cosine similarity for the public archive semantic search. Lower
// = more (noisier) results; raise it to tighten precision. Tuned for a 1B
// embedding model — re-tune if the model's dimensionality changes.
AI_LLM_EMBEDDING_ARCHIVE_MIN_SIMILARITY: z.coerce
.number()
.min(0)
.max(1)
.default(0.6),
AI_LLM_MAX_CONCURRENT: z.coerce.number().int().positive().default(5), AI_LLM_MAX_CONCURRENT: z.coerce.number().int().positive().default(5),
AI_LLM_IMAGE_MAX_DIMENSION: z.coerce AI_LLM_IMAGE_MAX_DIMENSION: z.coerce
.number() .number()
@@ -179,11 +170,6 @@ export const configSchema = z
.default("https://api.openai.com/v1"), .default("https://api.openai.com/v1"),
OPENAI_MODERATION_MODEL: z.string().default("omni-moderation-latest"), OPENAI_MODERATION_MODEL: z.string().default("omni-moderation-latest"),
// ── Qdrant (message archive for semantic search) ──────────────────
QDRANT_URL: z.string().optional(),
QDRANT_API_KEY: z.string().optional(),
QDRANT_ARCHIVE_COLLECTION: z.string().default("gmw_message_archive"),
// ── Auto Delete ───────────────────────────────────────────────────── // ── Auto Delete ─────────────────────────────────────────────────────
AUTO_DELETE_FLAGGED_ENABLED: z AUTO_DELETE_FLAGGED_ENABLED: z
.string() .string()
+2 -6
View File
@@ -53,19 +53,16 @@ src/
1. **LLM is the only judge.** Failed LLM → `status:"error"` + recovery retry. 1. **LLM is the only judge.** Failed LLM → `status:"error"` + recovery retry.
**Never** reintroduce regex/heuristic content classification. **Never** reintroduce regex/heuristic content classification.
2. **Discord tokens sanitized** before reaching LLM (`discordTokens.ts`). 2. **Discord tokens sanitized** before reaching LLM (`discordTokens.ts`).
3. **Semantic cache is batched** — one embed call + one Qdrant batch search. 3. **Streaming is mandatory** against the router base URL.
4. **Streaming is mandatory** against the router base URL.
## AI moderation pipeline ## AI moderation pipeline
``` ```
aiAnalyzer.ts → batchScheduler.ts → batchProcessor.ts → individualFallbackProcessor.ts aiAnalyzer.ts → batchScheduler.ts → batchProcessor.ts → individualFallbackProcessor.ts
↓ ↓ ↓ ↓ ↓ ↓ ↓ ↓
moderationOrchestrator.ts → (hash cache → Qdrant → LLM) moderationOrchestrator.ts → (hash cache → LLM)
↓ ↓ ↓ ↓ ↓ ↓
textBatchProcessor.ts mediaBatchProcessor.ts llmClient.ts textBatchProcessor.ts mediaBatchProcessor.ts llmClient.ts
embeddingClient.ts
qdrantClient.ts
``` ```
- Entry: `aiAnalyzer.ts` (`queueMessageAnalysis`, `startPendingAIAnalysisWorker`) - Entry: `aiAnalyzer.ts` (`queueMessageAnalysis`, `startPendingAIAnalysisWorker`)
@@ -85,7 +82,6 @@ textBatchProcessor.ts mediaBatchProcessor.ts llmClient.ts
- `messageStore.ts` — DB operations - `messageStore.ts` — DB operations
- `messageMetadata.ts` — metadata extraction - `messageMetadata.ts` — metadata extraction
- `messagesDb.ts` / `messagesCrud.ts` — DB schema operations - `messagesDb.ts` / `messagesCrud.ts` — DB schema operations
- `archiveEmbedder.ts` — Qdrant embedding (respect age-restricted guard)
- `retentionDb.ts` / `reviewsDb.ts` / `attachmentsDb.ts` — auxiliary tables - `retentionDb.ts` / `reviewsDb.ts` / `attachmentsDb.ts` — auxiliary tables
## Redis channels (outbound to backend) ## Redis channels (outbound to backend)
+4 -8
View File
@@ -68,8 +68,8 @@ of the same conversation, and vice versa.
(re-scheduled per lane) and `error`/`analysis_incomplete` messages (re-scheduled per lane) and `error`/`analysis_incomplete` messages
(individual fallback queue); prunes stale lane locks, per-conversation CB (individual fallback queue); prunes stale lane locks, per-conversation CB
counters and individual in-flight markers. counters and individual in-flight markers.
- `cache-prune.ts` — throttled (6h) expired-verdict sweep across Postgres and - `cache-prune.ts` — throttled (6h) expired-verdict sweep across Postgres,
Qdrant, driven from the recovery interval. driven from the recovery interval.
- `batchScheduler.ts` — per-conversation per-LANE debounce → `processBatch` - `batchScheduler.ts` — per-conversation per-LANE debounce → `processBatch`
(lane-aware). `splitMessagesByLane` / `laneOfMessage` live in (lane-aware). `splitMessagesByLane` / `laneOfMessage` live in
`analysisLanes.ts` (pure, unit-testable). `analysisLanes.ts` (pure, unit-testable).
@@ -82,8 +82,8 @@ of the same conversation, and vice versa.
Piscina `textWorkerPool`/`mediaWorkerPool`, `getConversationKey`. Piscina `textWorkerPool`/`mediaWorkerPool`, `getConversationKey`.
- `ai-analysis-worker.ts` — Piscina entry point (`batch` (lane) / - `ai-analysis-worker.ts` — Piscina entry point (`batch` (lane) /
`individual` jobs). Runs `runModerationAnalysis` off the main thread. `individual` jobs). Runs `runModerationAnalysis` off the main thread.
- `moderationOrchestrator.ts` — exact-hash cache → batched semantic (Qdrant) - `moderationOrchestrator.ts` — exact-hash cache → LLM. Text and media paths
cache → LLM. Text and media paths run in parallel. run in parallel.
- `textBatchProcessor.ts` / `mediaBatchProcessor.ts` — actual LLM calls - `textBatchProcessor.ts` / `mediaBatchProcessor.ts` — actual LLM calls
(one call per sub-batch, not per message). `mediaBatchProcessor` routes its (one call per sub-batch, not per message). `mediaBatchProcessor` routes its
moderation LLM call through the MEDIA semaphore. moderation LLM call through the MEDIA semaphore.
@@ -93,8 +93,6 @@ of the same conversation, and vice versa.
`AI_LLM_MEDIA_MAX_CONCURRENT` (media lane, default 4) — a vision backlog `AI_LLM_MEDIA_MAX_CONCURRENT` (media lane, default 4) — a vision backlog
can never consume text slots. `visionAnalyzer.ts` / `mediaAnalysisClient.ts` can never consume text slots. `visionAnalyzer.ts` / `mediaAnalysisClient.ts`
share the same router/base URL (different model alias for vision). share the same router/base URL (different model alias for vision).
- `embeddingClient.ts` + `qdrantClient.ts` — semantic cache (one embed call +
one batched Qdrant search for all uncached targets).
- `textCacheStore.ts` / `channelCultureStore.ts` / `userProfileStore.ts` / - `textCacheStore.ts` / `channelCultureStore.ts` / `userProfileStore.ts` /
`userProfileStore.ts` — caches learned user profile summaries (optional). `userProfileStore.ts` — caches learned user profile summaries (optional).
@@ -192,7 +190,5 @@ pipeline gauges registered by `app/metrics-collector.ts` —
- **Discord tokens are sanitized** (`discordTokens.ts`: `<:emoji:id>` → - **Discord tokens are sanitized** (`discordTokens.ts`: `<:emoji:id>` →
`[emoji:name]`, `<@id>` → `@user`, etc.) before content reaches the LLM, so `[emoji:name]`, `<@id>` → `@user`, etc.) before content reaches the LLM, so
numeric snowflake IDs never trigger false positives. numeric snowflake IDs never trigger false positives.
- **Semantic cache is batched** (one embed call + one Qdrant batch search),
not N sequential round-trips. `ensureQdrantCollection` is memoized.
- **Streaming is mandatory** against the router base URL (non-stream waits for - **Streaming is mandatory** against the router base URL (non-stream waits for
the full body and times out). `llmClient` aggregates SSE chunks. the full body and times out). `llmClient` aggregates SSE chunks.
+1 -1
View File
@@ -59,5 +59,5 @@ Callers outside a module import its `index.ts` facade, never an internal file.
## Testing ## Testing
Vitest, tests in `tests/`. Config supplies dummy env vars so the suite runs Vitest, tests in `tests/`. Config supplies dummy env vars so the suite runs
without live Postgres/Redis/Qdrant; external services are mocked. `llmE2e.test.ts` without live Postgres/Redis; external services are mocked. `llmE2e.test.ts`
is skipped by default and needs real credentials (`pnpm test:e2e:live`). is skipped by default and needs real credentials (`pnpm test:e2e:live`).
@@ -1,6 +1,4 @@
import { createChildLogger } from "@/shared/logger/index"; import { createChildLogger } from "@/shared/logger/index";
import { config } from "../../shared/config/index.js";
import { deleteExpiredQdrantPoints } from "./qdrantClient.js";
import { pruneExpiredTexts } from "./textCacheStore.js"; import { pruneExpiredTexts } from "./textCacheStore.js";
const logger = createChildLogger("cache-prune"); const logger = createChildLogger("cache-prune");
@@ -11,7 +9,7 @@ const CACHE_PRUNE_INTERVAL_MS = 6 * 60 * 60 * 1000; // every 6 hours
let lastCachePruneAt = 0; let lastCachePruneAt = 0;
/** /**
* Cache hygiene: purge expired moderation verdicts from Postgres and Qdrant. * Cache hygiene: purge expired moderation verdicts from Postgres.
* *
* Expired entries are never reused (read filters check `expires_at`) but they * Expired entries are never reused (read filters check `expires_at`) but they
* accumulate forever without a sweep. Called from the recovery interval; the * accumulate forever without a sweep. Called from the recovery interval; the
@@ -21,13 +19,10 @@ export function runCachePruneIfDue(now: number = Date.now()): void {
if (now - lastCachePruneAt < CACHE_PRUNE_INTERVAL_MS) return; if (now - lastCachePruneAt < CACHE_PRUNE_INTERVAL_MS) return;
lastCachePruneAt = now; lastCachePruneAt = now;
Promise.all([pruneExpiredTexts(), deleteExpiredQdrantPoints()]) Promise.resolve(pruneExpiredTexts())
.then(([pgDeleted, qdDeleted]) => { .then((pgDeleted) => {
if (pgDeleted > 0 || qdDeleted > 0) { if (pgDeleted > 0) {
logger.info( logger.info({ pgDeleted }, "Expired moderation cache pruned");
{ pgDeleted, qdDeleted },
"Expired moderation cache pruned",
);
} }
}) })
.catch((err: unknown) => { .catch((err: unknown) => {
@@ -1,200 +0,0 @@
/**
* embeddingClient.ts
*
* OpenAI-compatible embeddings helper used by the semantic moderation
* cache. When AI_LLM_EMBEDDING_MODEL is configured, near-duplicate
* messages can reuse a stored verdict (cosine similarity) instead of
* paying for a full chat-completion call — the main cost-saver.
*
* Every function degrades gracefully: if the embedding model is not
* configured or the API fails, callers fall back to the exact-hash cache
* and then the LLM, so moderation quality is never reduced.
*/
import OpenAI from "openai";
import { createChildLogger } from "@/shared/logger/index";
import { config } from "../../shared/config/index.js";
import { cleanContent } from "./textSignals.js";
const log = createChildLogger("embedding-client");
// ---------------------------------------------------------------------------
// Text normalization (shared by moderation + archive embedding)
// ---------------------------------------------------------------------------
/**
* Max characters fed to the embedding model for a single document. Embedding
* models have a hard token ceiling; embedding past it throws / wastes tokens.
* Content messages are truncated; search queries have their own (smaller) cap.
*/
export const EMBEDDING_MAX_CHARS = 1200;
export const EMBEDDING_MAX_QUERY_CHARS = 300;
/**
* Normalize free-form Discord text before embedding.
*
* Raw messages are full of signal-hostile noise: @mentions, channels, custom
* emoji, URLs, markdown and control chars. Embedding that noise directly
* dilutes the vector (sentences that differ only in an @mention or a link
* land far apart) and inflates token cost. The same cleanup is applied to the
* user's search query so archived vectors and the query share one space.
*
* - `cleanContent` (from textSignals) strips URLs/@mentions/emoji/markdown and
* collapses whitespace — good for embeddings, not just term extraction.
* - Control characters / zero-width joiners are removed (Discord pastes these).
* - Lowercase is applied so "Discord" and "discord" embed identically (embed
* models are case-sensitive; this measurably improves near-duplicate recall).
*
* Falls back to the raw input if normalization empties a string (e.g. a
* message that was only a URL) so index alignment is preserved by callers.
*/
export function normalizeEmbeddingText(raw: string, maxChars: number): string {
if (!raw) return "";
const cleaned = cleanContent(
raw.replace(/[\p{Cc}\p{Cf}]/gu, " ").toLowerCase(),
);
if (!cleaned) return raw.slice(0, maxChars); // preserve original if stripped
return cleaned.slice(0, maxChars);
}
/** Normalize a single embedded document/message. */
export function normalizeEmbeddingContent(raw: string): string {
return normalizeEmbeddingText(raw, EMBEDDING_MAX_CHARS);
}
/** Normalize a user-provided search query before embedding it. */
export function normalizeEmbeddingQuery(raw: string): string {
return normalizeEmbeddingText(raw, EMBEDDING_MAX_QUERY_CHARS);
}
// ---------------------------------------------------------------------------
// Client (lazy singleton — same base URL as the chat client)
// ---------------------------------------------------------------------------
let openaiClient: OpenAI | null = null;
function getClient(): OpenAI | null {
if (!config.AI_LLM_API_KEY || !config.AI_LLM_EMBEDDING_MODEL) return null;
if (!openaiClient) {
openaiClient = new OpenAI({
apiKey: config.AI_LLM_API_KEY,
baseURL: config.AI_LLM_BASE_URL,
// Embeddings are cheap and idempotent — a transient network blip should
// NOT silently disable the whole semantic cache for a batch. Let the SDK
// retry (2 retries, jittered) instead of failing open immediately.
maxRetries: 2,
timeout: 60_000,
});
}
return openaiClient;
}
/** True when the semantic cache is usable (key + model configured). */
export function isEmbeddingEnabled(): boolean {
return Boolean(config.AI_LLM_API_KEY && config.AI_LLM_EMBEDDING_MODEL);
}
// ---------------------------------------------------------------------------
// Embedding calls
// ---------------------------------------------------------------------------
/**
* Embed a batch of texts with the configured model.
* Returns null on any failure so callers can skip semantic lookup.
*
* Each input is normalized (noise stripped, lowercased, length-capped) before
* embedding — see normalizeEmbeddingText. Index alignment with `texts` is
* preserved: if a text normalizes to empty we embed the raw original so the
* caller's `embeddings[i] ↔ texts[i]` mapping never shifts.
*/
export async function embedTexts(texts: string[]): Promise<number[][] | null> {
if (!isEmbeddingEnabled()) return null;
if (texts.length === 0) return [];
const client = getClient();
if (!client) return null;
const normalized = texts.map((t) => normalizeEmbeddingContent(t));
try {
const response = await client.embeddings.create({
model: config.AI_LLM_EMBEDDING_MODEL as string,
input: normalized,
// OpenAI SDK v6 defaults to base64; Nvidia-backed embedding models
// (e.g. llama-nemotron-embed) reject it with 400. Always float.
encoding_format: "float",
});
// All vectors in one response must share the model's dimension. If they
// don't (shouldn't happen, but guards against a misconfigured/mismatched
// model), fail the batch rather than feed garbage to cosine + Qdrant.
const vectors = response.data.map((item) => item.embedding);
const firstLen = vectors[0]?.length ?? 0;
const consistent = vectors.every((v) => v.length === firstLen);
if (!consistent || firstLen === 0) {
log.error(
{
model: config.AI_LLM_EMBEDDING_MODEL,
dims: vectors.map((v) => v.length),
},
"Embedding response had inconsistent/empty dimensions — treating as failure",
);
return null;
}
return vectors;
} catch (error) {
log.warn(
{ error: error instanceof Error ? error.message : String(error) },
"Embedding request failed — semantic cache disabled for this call",
);
return null;
}
}
/** Embed a single text; returns null on failure. */
export async function embedText(text: string): Promise<number[] | null> {
const vectors = await embedTexts([text]);
return vectors?.[0] ?? null;
}
// ---------------------------------------------------------------------------
// Similarity
// ---------------------------------------------------------------------------
/** Cosine similarity between two equal-length vectors. */
export function cosineSimilarity(a: number[], b: number[]): number {
if (a.length === 0 || a.length !== b.length) return 0;
let dot = 0;
let normA = 0;
let normB = 0;
for (let i = 0; i < a.length; i++) {
dot += a[i] * b[i];
normA += a[i] * a[i];
normB += b[i] * b[i];
}
if (normA === 0 || normB === 0) return 0;
return dot / (Math.sqrt(normA) * Math.sqrt(normB));
}
/**
* Pick the best match above `minSimilarity`, or null.
* Returns { index, similarity } relative to the candidates array.
*/
export function findBestEmbeddingMatch(
vector: number[],
candidates: number[][],
minSimilarity: number,
): { index: number; similarity: number } | null {
let bestIndex = -1;
let bestSimilarity = minSimilarity;
for (let i = 0; i < candidates.length; i++) {
const sim = cosineSimilarity(vector, candidates[i]);
if (sim > bestSimilarity) {
bestSimilarity = sim;
bestIndex = i;
}
}
return bestIndex >= 0
? { index: bestIndex, similarity: bestSimilarity }
: null;
}
@@ -16,25 +16,19 @@ import type {
MessageRecord, MessageRecord,
} from "../message-capture/types.js"; } from "../message-capture/types.js";
import { initCacheStore } from "./cacheStore.js"; import { initCacheStore } from "./cacheStore.js";
import { embedTexts, isEmbeddingEnabled } from "./embeddingClient.js";
import { hasMediaContent } from "./mediaAnalysisClient.js"; import { hasMediaContent } from "./mediaAnalysisClient.js";
import { runMediaBatch } from "./mediaBatchProcessor.js"; import { runMediaBatch } from "./mediaBatchProcessor.js";
import { isQdrantConfigured, searchQdrantBatch } from "./qdrantClient.js";
import { logCacheEvent } from "./responseLogger.js"; import { logCacheEvent } from "./responseLogger.js";
import { runTextOnlyBatch } from "./textBatchProcessor.js"; import { runTextOnlyBatch } from "./textBatchProcessor.js";
import { import {
bumpTextModerationHitCounts, bumpTextModerationHitCounts,
ERROR_ARTIFACT_FLAGS, ERROR_ARTIFACT_FLAGS,
findSimilarTextModeration,
getCachedTextModerations, getCachedTextModerations,
isGloballyReusableCleanVerdict, isGloballyReusableCleanVerdict,
isSemanticBandAccepted,
makeModerationContextKey, makeModerationContextKey,
makeTextModerationCacheKey, makeTextModerationCacheKey,
parseQdrantVerdict,
type StoredModerationVerdict, type StoredModerationVerdict,
setCachedTextModeration, setCachedTextModeration,
upsertBareKeyToQdrant,
} from "./textCacheStore.js"; } from "./textCacheStore.js";
const log = createChildLogger("moderationOrchestrator"); const log = createChildLogger("moderationOrchestrator");
@@ -74,12 +68,9 @@ export interface ModerationOutput {
* Runs LLM-based moderation analysis on messages. * Runs LLM-based moderation analysis on messages.
* Splits text-only vs media, runs both paths in parallel, applies caching. * Splits text-only vs media, runs both paths in parallel, applies caching.
* *
* Cache strategy (two-phase, batched): * Cache strategy: exact-hash lookups only (no API) — key is content +
* 1. Exact-hash lookups (no API) — key is content + conversation context * conversation context (channel/thread) because LLM verdicts depend on
* (channel/thread) because LLM verdicts depend on context. * context. Every miss goes to the LLM.
* 2. Semantic near-duplicate lookup — ONE embeddings call for all uncached
* text targets, then ONE Qdrant batch search (index-aligned), instead of
* N sequential embed→search round-trips.
*/ */
export async function runModerationAnalysis( export async function runModerationAnalysis(
input: ModerationInput, input: ModerationInput,
@@ -100,9 +91,6 @@ export async function runModerationAnalysis(
const uncachedTargets: MessageRecord[] = []; const uncachedTargets: MessageRecord[] = [];
// cacheKey → representative result for identical-content dedupe // cacheKey → representative result for identical-content dedupe
const hitByKey = new Map<string, AnalysisResult>(); const hitByKey = new Map<string, AnalysisResult>();
// Embedding per exact cache key — computed once during lookup, reused
// when the fresh LLM verdict is written back to the semantic cache.
const embeddingsByKey = new Map<string, number[]>();
interface ExactCandidate { interface ExactCandidate {
target: MessageRecord; target: MessageRecord;
@@ -251,142 +239,6 @@ export async function runModerationAnalysis(
} }
incrementCounterBy("moderation_cache_misses", uncachedTargets.length); incrementCounterBy("moderation_cache_misses", uncachedTargets.length);
// ── Phase 2: semantic cache — batched (one embed call + one Qdrant
// batch search for ALL uncached text targets) ─────────────────────────
if (isEmbeddingEnabled()) {
const semanticCandidates = uncachedTargets
.map((t) => ({
target: t,
cacheKey: makeTextModerationCacheKey(
t.edited_content ?? t.content,
makeModerationContextKey(t),
),
}))
.filter(({ target }) => {
const raw = (target.edited_content ?? target.content).trim();
if (raw.length < 5) return false;
if (hasMediaContent(target, attachments)) return false;
return !hitByKey.has(
makeTextModerationCacheKey(raw, makeModerationContextKey(target)),
);
});
if (semanticCandidates.length > 0) {
const texts = semanticCandidates.map(
({ target }) => target.edited_content ?? target.content,
);
const embeddings = await embedTexts(texts);
if (embeddings && embeddings.length === texts.length) {
// index-aligned with semanticCandidates
for (let i = 0; i < semanticCandidates.length; i++) {
const { cacheKey } = semanticCandidates[i];
embeddingsByKey.set(cacheKey, embeddings[i]);
}
if (isQdrantConfigured()) {
// ONE batch search at the LOOSER threshold; per-hit re-classification
// enforces the strict band for actionable verdicts.
const batchHits = await searchQdrantBatch(
embeddings,
config.AI_LLM_EMBEDDING_MAX_CANDIDATES,
config.AI_LLM_EMBEDDING_MIN_SIMILARITY_CLEAN,
);
for (let i = 0; i < semanticCandidates.length; i++) {
const { target, cacheKey } = semanticCandidates[i];
const hits = batchHits[i] ?? [];
if (hits.length === 0) continue;
const verdict = parseQdrantVerdict(hits[0].payload, hits[0].score);
if (!verdict) continue;
if (!isSemanticBandAccepted(verdict, verdict.similarity)) continue;
log.debug(
{
messageId: target.id,
similarity: Number(verdict.similarity.toFixed(4)),
status: verdict.status,
},
"Semantic moderation cache hit — reusing stored verdict",
);
const hit: AnalysisResult = {
messageId: target.id,
status: verdict.status,
flags: verdict.flags,
score: verdict.score,
analysis: verdict.analysis,
categories: verdict.categories,
severity: verdict.severity as AnalysisResult["severity"],
confidence: verdict.confidence,
recommendedAction:
verdict.recommendedAction as AnalysisResult["recommendedAction"],
policyVersion: "semantic-cache-2026-07",
evidence: [],
};
cacheHits.push(hit);
hitByKey.set(cacheKey, hit);
servedCacheKeys.add(cacheKey); // bump hit_count for metrics
logCacheEvent("hit", cacheKey, "text");
incrementCounterBy("moderation_cache_hits", 1, {
type: "semantic-qdrant",
});
}
} else {
// Legacy Postgres fallback path (no Qdrant): per-candidate scan.
for (let i = 0; i < semanticCandidates.length; i++) {
const { target, cacheKey } = semanticCandidates[i];
const semantic = await findSimilarTextModeration(
embeddings[i],
config.AI_LLM_EMBEDDING_MIN_SIMILARITY_CLEAN,
config.AI_LLM_EMBEDDING_MAX_CANDIDATES,
);
if (!semantic) continue;
if (!isSemanticBandAccepted(semantic, semantic.similarity))
continue;
log.debug(
{
messageId: target.id,
similarity: Number(semantic.similarity.toFixed(4)),
status: semantic.status,
},
"Semantic moderation cache hit (PG fallback) — reusing stored verdict",
);
const hit: AnalysisResult = {
messageId: target.id,
status: semantic.status,
flags: semantic.flags,
score: semantic.score,
analysis: semantic.analysis,
categories: semantic.categories,
severity: semantic.severity as AnalysisResult["severity"],
confidence: semantic.confidence,
recommendedAction:
semantic.recommendedAction as AnalysisResult["recommendedAction"],
policyVersion: "semantic-cache-2026-07",
evidence: [],
};
cacheHits.push(hit);
hitByKey.set(cacheKey, hit);
servedCacheKeys.add(cacheKey); // bump hit_count for metrics
logCacheEvent("hit", cacheKey, "text");
incrementCounterBy("moderation_cache_hits", 1, {
type: "semantic-pg",
});
}
}
// Drop semantic hits from the LLM work queue.
for (let i = uncachedTargets.length - 1; i >= 0; i--) {
const t = uncachedTargets[i];
const key = makeTextModerationCacheKey(
t.edited_content ?? t.content,
makeModerationContextKey(t),
);
if (hitByKey.has(key)) {
uncachedTargets.splice(i, 1);
}
}
}
}
}
if (cacheHits.length > 0) { if (cacheHits.length > 0) {
// Metrics: one bulk UPDATE for every exact-cache key actually served. // Metrics: one bulk UPDATE for every exact-cache key actually served.
bumpTextModerationHitCounts(Array.from(servedCacheKeys)); bumpTextModerationHitCounts(Array.from(servedCacheKeys));
@@ -466,11 +318,7 @@ export async function runModerationAnalysis(
recommendedAction: result.recommendedAction ?? "none", recommendedAction: result.recommendedAction ?? "none",
status: result.status, status: result.status,
}; };
setCachedTextModeration( setCachedTextModeration(cacheKey, stored).catch((err: unknown) => {
cacheKey,
stored,
embeddingsByKey.get(cacheKey),
).catch((err: unknown) => {
log.warn({ cacheKey, error: String(err) }, "Cache write failed"); log.warn({ cacheKey, error: String(err) }, "Cache write failed");
}); });
@@ -479,13 +327,6 @@ export async function runModerationAnalysis(
// under the context-free bare key so repeats in OTHER channels hit the // under the context-free bare key so repeats in OTHER channels hit the
// exact cache instead of paying a new LLM call. Same guard as the read // exact cache instead of paying a new LLM call. Same guard as the read
// path — only non-actionable clean verdicts may cross channels. // path — only non-actionable clean verdicts may cross channels.
//
// 2026-08-25 cache-hit fix: the bare key is ALSO upserted to Qdrant
// (via upsertBareKeyToQdrant) with the SAME embedding already computed
// at lookup time. Previously the bare key was only PG-written with
// embedding=null — bare clean verdicts were DB-only and invisible to
// searchQdrantBatch, capping the semantic hit-rate below the exact-cache
// hit-rate for cross-channel repeats.
const bareKey = makeTextModerationCacheKey(rawContent); const bareKey = makeTextModerationCacheKey(rawContent);
if ( if (
bareKey !== cacheKey && bareKey !== cacheKey &&
@@ -505,11 +346,7 @@ export async function runModerationAnalysis(
) )
) { ) {
globalBareKeysWritten.set(bareKey, true); globalBareKeysWritten.set(bareKey, true);
setCachedTextModeration(bareKey, stored, null).catch(() => {}); setCachedTextModeration(bareKey, stored).catch(() => {});
const bareEmbedding = embeddingsByKey.get(cacheKey);
if (bareEmbedding && bareEmbedding.length > 0) {
upsertBareKeyToQdrant(bareKey, stored, bareEmbedding).catch(() => {});
}
} }
} }
@@ -1,589 +0,0 @@
/**
* qdrantClient.ts
*
* Minimal Qdrant REST client (zero dependencies, fetch-based) used by the
* semantic moderation cache. Embedding vectors + verdict payloads live in
* Qdrant instead of the Postgres `embedding` column (legacy, kept for
* backward-compatible fallback reads).
*
* All functions degrade gracefully: failures return null / empty results so
* callers fall back to the LLM — moderation quality is never reduced.
*/
import { createHash } from "node:crypto";
import { createChildLogger } from "@/shared/logger/index";
import { config } from "../../shared/config/index.js";
const log = createChildLogger("qdrant");
// ensureQdrantCollection performs a network round-trip (GET, possibly
// DELETE+PUT). Running it on every upsert adds 1-3 HTTP calls per
// moderation verdict, which under Qdrant load pushes the upsert past the
// request timeout and aborts it ("This operation was aborted"). Memoise the
// result so the collection is only verified once per process lifetime.
let ensureCollectionPromise: Promise<boolean> | null = null;
/** Reset the memoised ensure result (used by tests / config reload). */
export function resetQdrantCollectionCache(): void {
ensureCollectionPromise = null;
}
export interface QdrantVerdictPayload {
text: string;
flags: string; // JSON string of the full moderation result
analyzed_at: number;
expires_at: number;
/** Bare content hash (16 hex chars) — enables content-based invalidation
* regardless of the (context-scoped) point id. */
content_hash?: string;
// Archive metadata (archiveEmbedder writes these; moderation cache doesn't).
// Optional so the same payload shape serves both the moderation cache
// collection and the public message archive collection.
username?: string;
channel_id?: string;
guild_id?: string;
thread_id?: string | null;
channel_name?: string | null;
thread_name?: string | null;
created_at?: number;
}
function baseUrl(): string {
return (config.QDRANT_URL ?? "http://100.121.180.82:6333").replace(
/\/+$/,
"",
);
}
function collectionName(): string {
return config.QDRANT_COLLECTION ?? "gmw_text_moderation";
}
/** Persistent archive collection for semantic message search (no TTL). */
export const ARCHIVE_COLLECTION =
config.QDRANT_ARCHIVE_COLLECTION ?? "gmw_message_archive";
function headers(): Record<string, string> {
const h: Record<string, string> = {
"Content-Type": "application/json",
};
if (config.QDRANT_API_KEY) {
h["api-key"] = config.QDRANT_API_KEY;
}
return h;
}
async function request(
method: string,
path: string,
body?: unknown,
timeoutMs = 10_000,
): Promise<unknown> {
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), timeoutMs);
try {
const res = await fetch(`${baseUrl()}${path}`, {
method,
headers: headers(),
body: body === undefined ? undefined : JSON.stringify(body),
signal: controller.signal,
});
const text = await res.text();
let json: unknown = null;
try {
json = text ? JSON.parse(text) : null;
} catch {
json = null;
}
if (!res.ok) {
throw new Error(
`Qdrant ${method} ${path} -> ${res.status}: ${text.slice(0, 200)}`,
);
}
return json;
} finally {
clearTimeout(timer);
}
}
/** True for transient errors worth retrying (408 timeout, ECONNRESET, aborts). */
function isTransientQdrantError(error: unknown): boolean {
if (error instanceof Error && error.name === "AbortError") return true;
const msg = error instanceof Error ? error.message : String(error);
return (
msg.includes("-> 408") ||
msg.includes("aborted") ||
msg.includes("ECONNRESET") ||
msg.includes("ETIMEDOUT") ||
msg.includes("fetch failed")
);
}
/** Retry with exponential backoff around `request`, abort-aware. */
async function requestWithRetry(
method: string,
path: string,
body?: unknown,
timeoutMs = 10_000,
attempts = 3,
): Promise<unknown> {
let lastError: unknown;
for (let attempt = 0; attempt < attempts; attempt++) {
if (attempt > 0) {
// Exponential backoff: 500ms → 1s → 2s (jittered ±20%).
const base = 500 * 2 ** (attempt - 1);
const delayMs = base + Math.floor(Math.random() * base * 0.2);
await new Promise((resolve) => setTimeout(resolve, delayMs));
}
try {
return await request(method, path, body, timeoutMs);
} catch (error) {
lastError = error;
if (!isTransientQdrantError(error)) throw error;
log.debug(
{
error: error instanceof Error ? error.message : String(error),
attempt: attempt + 1,
method,
path,
},
"Qdrant transient error — retrying with backoff",
);
}
}
throw lastError;
}
/** Deterministic uint64 point id from the exact-hash cache key. */
export function qdrantPointId(cacheKey: string): number {
const digest = createHash("sha256").update(cacheKey).digest();
// First 8 bytes as BigInt, then clamp into Qdrant's uint64 space.
const big = digest.readBigUInt64BE(0);
return Number(big & 0x7fffffffffffffffn);
}
/**
* Ensure the collection exists with the right vector size. If the size
* changed (embedding model swapped), recreate — stale vectors are useless
* anyway and cosine scores would be meaningless across dimensions.
*/
export async function ensureQdrantCollection(
vectorSize: number,
): Promise<boolean> {
if (ensureCollectionPromise) return ensureCollectionPromise;
ensureCollectionPromise = (async () => {
try {
// 404 = collection doesn't exist yet → create it.
let existing: {
result?: { config?: { params?: { vectors?: { size?: number } } } };
} | null = null;
try {
existing = (await request(
"GET",
`/collections/${collectionName()}`,
)) as {
result?: { config?: { params?: { vectors?: { size?: number } } } };
};
} catch (error) {
if (!(error instanceof Error) || !error.message.includes("-> 404")) {
throw error;
}
}
const size = existing?.result?.config?.params?.vectors?.size;
if (size === vectorSize) return true;
if (size !== undefined && size !== vectorSize) {
log.warn(
{ collection: collectionName(), oldSize: size, newSize: vectorSize },
"Qdrant collection vector size changed — recreating collection",
);
await request("DELETE", `/collections/${collectionName()}`);
}
await request("PUT", `/collections/${collectionName()}`, {
vectors: { size: vectorSize, distance: "Cosine" },
});
return true;
} catch (error) {
log.error(
{
error: error instanceof Error ? error.message : String(error),
collection: collectionName(),
},
"Failed to ensure Qdrant collection",
);
// The memoised promise is permanently sticky: once it rejects (e.g. a
// recreate aborted mid-way — DELETE done, PUT failed), every later call
// returns the same rejected promise and the collection is never
// re-created until process restart. Reset so the next call retries.
ensureCollectionPromise = null;
return false;
}
})();
return ensureCollectionPromise;
}
/** Upsert one embedding + verdict payload point. Returns false on failure. */
export async function upsertQdrantPoint(
cacheKey: string,
vector: number[],
payload: QdrantVerdictPayload,
): Promise<boolean> {
try {
if (!(await ensureQdrantCollection(vector.length))) return false;
await requestWithRetry(
"PUT",
`/collections/${collectionName()}/points`,
{
points: [{ id: qdrantPointId(cacheKey), vector, payload }],
wait: true,
},
30_000,
3,
);
return true;
} catch (error) {
log.warn(
{ error: error instanceof Error ? error.message : String(error) },
"Qdrant upsert failed — semantic entry skipped",
);
return false;
}
}
export interface QdrantSearchHit {
cacheKey: string;
score: number;
payload: QdrantVerdictPayload;
}
/**
* Search the nearest stored vector. Returns hits sorted by score desc,
* filtered to unexpired payloads. Empty array on failure.
*/
export async function searchQdrant(
vector: number[],
limit: number,
scoreThreshold: number,
): Promise<QdrantSearchHit[]> {
try {
const json = (await request(
"POST",
`/collections/${collectionName()}/points/search`,
{
vector,
limit,
score_threshold: scoreThreshold,
with_payload: true,
filter: {
must: [
{
key: "expires_at",
range: { gte: Date.now() },
},
],
},
},
)) as {
result?: Array<{
id?: number;
score?: number;
payload?: QdrantVerdictPayload;
}>;
};
return (json.result ?? [])
.filter((hit) => hit.payload?.flags)
.map((hit) => ({
cacheKey: `qdrant:${hit.id ?? "?"}`,
score: hit.score ?? 0,
payload: hit.payload as QdrantVerdictPayload,
}));
} catch (error) {
log.warn(
{ error: error instanceof Error ? error.message : String(error) },
"Qdrant search failed — semantic cache skipped",
);
return [];
}
}
/**
* Batch search: one HTTP round-trip for N vectors (Qdrant
* `/points/search/batch`). Result is index-aligned with `vectors` — each
* entry is the top hits for that vector (or [] on per-vector failure).
* Used by the orchestrator to avoid N sequential embed→search round-trips.
*/
export async function searchQdrantBatch(
vectors: number[][],
limit: number,
scoreThreshold: number,
): Promise<QdrantSearchHit[][]> {
if (vectors.length === 0) return [];
try {
const json = (await request(
"POST",
`/collections/${collectionName()}/points/search/batch`,
{
searches: vectors.map((vector) => ({
vector,
limit,
score_threshold: scoreThreshold,
with_payload: true,
filter: {
must: [
{
key: "expires_at",
range: { gte: Date.now() },
},
],
},
})),
},
)) as {
result?: Array<{
result?: Array<{
id?: number;
score?: number;
payload?: QdrantVerdictPayload;
}>;
}>;
};
return (json.result ?? []).map((entry) =>
(entry.result ?? [])
.filter((hit) => hit.payload?.flags)
.map((hit) => ({
cacheKey: `qdrant:${hit.id ?? "?"}`,
score: hit.score ?? 0,
payload: hit.payload as QdrantVerdictPayload,
})),
);
} catch (error) {
log.warn(
{ error: error instanceof Error ? error.message : String(error) },
"Qdrant batch search failed — semantic cache skipped",
);
return vectors.map(() => []);
}
}
/**
* Delete expired verdict points from the collection. Best-effort: 404
* (collection missing) and failures are swallowed — the periodic pruner
* just retries next sweep.
*/
export async function deleteExpiredQdrantPoints(): Promise<number> {
try {
const json = (await request(
"POST",
`/collections/${collectionName()}/points/delete`,
{
filter: {
must: [
{
key: "expires_at",
range: { lt: Date.now() },
},
],
},
},
)) as { result?: { deleted?: number } | null };
return json.result?.deleted ?? 0;
} catch (error) {
if (error instanceof Error && error.message.includes("-> 404")) {
log.debug({}, "Qdrant collection absent — nothing to prune");
} else {
log.warn(
{ error: error instanceof Error ? error.message : String(error) },
"Qdrant expired-point prune failed",
);
}
return 0;
}
}
/**
* Delete the verdict point for an exact cache key (used by cache
* invalidation when a moderator corrects a verdict).
*/
export async function deleteQdrantPoint(cacheKey: string): Promise<boolean> {
try {
await request("POST", `/collections/${collectionName()}/points/delete`, {
points: [qdrantPointId(cacheKey)],
});
return true;
} catch (error) {
log.warn(
{ error: error instanceof Error ? error.message : String(error) },
"Qdrant point delete failed",
);
return false;
}
}
/**
* Delete all verdict points whose payload carries a given bare content hash.
* Used by cache invalidation for corrected verdicts — matches context-scoped
* points that share the same content regardless of their point ids.
*/
export async function deleteQdrantPointsByContentHash(
bareHash: string,
): Promise<boolean> {
try {
await request("POST", `/collections/${collectionName()}/points/delete`, {
filter: {
must: [
{
key: "content_hash",
match: { value: bareHash },
},
],
},
});
return true;
} catch (error) {
log.warn(
{ error: error instanceof Error ? error.message : String(error) },
"Qdrant content-hash point delete failed",
);
return false;
}
}
/** True when Qdrant is configured (non-empty URL). */
export function isQdrantConfigured(): boolean {
return Boolean(config.QDRANT_URL);
}
// ─── Archive variants (collection-aware, for persistent message search) ───
// These mirror the cache functions but take an explicit collection name so the
// semantic-search archive (gmw_message_archive) can live alongside the
// TTL-bounded automod cache without disturbing it.
/** Ensure an arbitrary collection exists with the right vector size. */
export async function ensureQdrantCollectionV2(
name: string,
vectorSize: number,
): Promise<boolean> {
try {
let existing: {
result?: { config?: { params?: { vectors?: { size?: number } } } };
} | null = null;
try {
existing = (await request("GET", `/collections/${name}`)) as {
result?: { config?: { params?: { vectors?: { size?: number } } } };
} | null;
} catch (error) {
if (!(error instanceof Error) || !error.message.includes("-> 404")) {
throw error;
}
}
const size = existing?.result?.config?.params?.vectors?.size;
if (size === vectorSize) return true;
if (size !== undefined && size !== vectorSize) {
log.warn(
{ collection: name, oldSize: size, newSize: vectorSize },
"Qdrant archive collection vector size changed — recreating collection",
);
await request("DELETE", `/collections/${name}`);
}
await request("PUT", `/collections/${name}`, {
vectors: { size: vectorSize, distance: "Cosine" },
});
return true;
} catch (error) {
log.error(
{
error: error instanceof Error ? error.message : String(error),
collection: name,
},
"Failed to ensure Qdrant archive collection",
);
return false;
}
}
/** Upsert one embedding + payload point into a named collection. */
export async function upsertQdrantPointV2(
name: string,
pointId: number,
vector: number[],
payload: QdrantVerdictPayload,
): Promise<boolean> {
try {
if (!(await ensureQdrantCollectionV2(name, vector.length))) return false;
await requestWithRetry(
"PUT",
`/collections/${name}/points`,
{
points: [{ id: pointId, vector, payload }],
wait: true,
},
30_000,
3,
);
return true;
} catch (error) {
log.warn(
{
error: error instanceof Error ? error.message : String(error),
collection: name,
} as Record<string, unknown>,
"Qdrant archive upsert failed — entry skipped",
);
return false;
}
}
export interface QdrantArchiveHit {
pointId: number;
score: number;
payload: QdrantVerdictPayload;
}
/** Search a named collection for the nearest stored vector. */
export async function searchQdrantV2(
name: string,
vector: number[],
limit: number,
scoreThreshold: number,
): Promise<QdrantArchiveHit[]> {
try {
const json = (await request("POST", `/collections/${name}/points/search`, {
vector,
limit,
score_threshold: scoreThreshold,
with_payload: true,
})) as {
result?: Array<{
id?: number;
score?: number;
payload?: QdrantVerdictPayload;
}>;
};
return (json.result ?? [])
.filter((hit) => hit.payload?.text)
.map((hit) => ({
pointId: hit.id ?? 0,
score: hit.score ?? 0,
payload: hit.payload as QdrantVerdictPayload,
}));
} catch (error) {
log.warn(
{
error: error instanceof Error ? error.message : String(error),
collection: name,
} as Record<string, unknown>,
"Qdrant archive search failed — semantic search skipped",
);
return [];
}
}
@@ -2,15 +2,6 @@ import { createHash } from "node:crypto";
import { createChildLogger } from "@/shared/logger/index"; import { createChildLogger } from "@/shared/logger/index";
import { config } from "../../shared/config/index.js"; import { config } from "../../shared/config/index.js";
import { executeAll, executeGet } from "../../shared/database/drizzle.js"; import { executeAll, executeGet } from "../../shared/database/drizzle.js";
import { findBestEmbeddingMatch } from "./embeddingClient.js";
import {
deleteQdrantPoint,
deleteQdrantPointsByContentHash,
isQdrantConfigured,
type QdrantVerdictPayload,
searchQdrant,
upsertQdrantPoint,
} from "./qdrantClient.js";
const logger = createChildLogger("text-cache-store"); const logger = createChildLogger("text-cache-store");
@@ -263,8 +254,8 @@ export function makeModerationContextKey(message: {
/** /**
* Invalidate cached moderation verdicts for a piece of content: removes * Invalidate cached moderation verdicts for a piece of content: removes
* matching Postgres rows AND Qdrant points. Called when a moderator * matching Postgres rows. Called when a moderator corrects a verdict so a
* corrects a verdict so a stale/wrong cached decision cannot resurface. * stale/wrong cached decision cannot resurface.
* *
* Handles both key formats: * Handles both key formats:
* - legacy `text_mod:<hash>` (content-only, pre-context keys) * - legacy `text_mod:<hash>` (content-only, pre-context keys)
@@ -277,20 +268,10 @@ export async function invalidateTextModerationCache(
.update(content) .update(content)
.digest("hex") .digest("hex")
.slice(0, 16); .slice(0, 16);
const legacyKey = `text_mod:${bareHash}`;
const queries: Promise<unknown>[] = [ await executeAll(`DELETE FROM text_analysis_cache WHERE text LIKE $1`, [
executeAll(`DELETE FROM text_analysis_cache WHERE text LIKE $1`, [ `text_mod:%${bareHash}`,
`text_mod:%${bareHash}`, ]).catch(() => {});
]).catch(() => {}),
];
if (isQdrantConfigured()) {
queries.push(
deleteQdrantPoint(legacyKey).catch(() => {}),
deleteQdrantPointsByContentHash(bareHash).catch(() => {}),
);
}
await Promise.all(queries).catch(() => {});
} }
/** /**
@@ -488,31 +469,6 @@ export function bumpTextModerationHitCounts(cacheKeys: string[]): void {
).catch(() => {}); ).catch(() => {});
} }
// ---------------------------------------------------------------------------
// Semantic two-band acceptance
// ---------------------------------------------------------------------------
/**
* True when a semantic-cache hit may be reused given its verdict class.
* Two bands (2026-08-24): non-actionable verdicts (clean / flagless /
* action=none) are accepted from the LOOSER clean band; actionable verdicts
* (warn/flagged or any flags/action) keep the strict historical gate.
* Between the bands → reject → the message falls through to the LLM
* (fail-open toward accuracy).
*/
export function isSemanticBandAccepted(
verdict: StoredModerationVerdict,
similarity: number,
): boolean {
const isNonActionable =
verdict.status === "clean" &&
verdict.flags.length === 0 &&
(verdict.recommendedAction ?? "none") === "none";
return isNonActionable
? similarity >= config.AI_LLM_EMBEDDING_MIN_SIMILARITY_CLEAN
: similarity >= config.AI_LLM_EMBEDDING_MIN_SIMILARITY;
}
// --------------------------------------------------------------------------- // ---------------------------------------------------------------------------
// Global exact-cache reuse guard (context-free fallback) // Global exact-cache reuse guard (context-free fallback)
// --------------------------------------------------------------------------- // ---------------------------------------------------------------------------
@@ -542,181 +498,9 @@ export function isGloballyReusableCleanVerdict(
return true; return true;
} }
/**
* Parse a Qdrant verdict payload into the result shape shared by the
* semantic cache lookups. Returns null on malformed payloads (callers then
* fall through to the LLM).
*/
export function parseQdrantVerdict(
payload: QdrantVerdictPayload,
similarity: number,
):
| (StoredModerationVerdict & {
text: string;
similarity: number;
})
| null {
const parsed = parseStoredVerdictRow({ flags: payload.flags });
if (!parsed) return null;
return {
...parsed,
text: payload.text,
similarity,
};
}
/**
* Semantic moderation cache lookup.
*
* Primary: Qdrant vector search (when QDRANT_URL configured) — nearest
* unexpired verdict above `minSimilarity`. Fallback: Postgres embedding
* column (legacy rows written before Qdrant was wired in).
* Returns null on no match or any failure — callers then proceed to the LLM.
*/
export async function findSimilarTextModeration(
embedding: number[],
minSimilarity: number,
limit: number,
): Promise<
(StoredModerationVerdict & { text: string; similarity: number }) | null
> {
// Qdrant path (primary)
if (isQdrantConfigured()) {
const hits = await searchQdrant(embedding, limit, minSimilarity);
if (hits.length > 0) {
const hit = hits[0];
return parseQdrantVerdict(hit.payload, hit.score);
}
// No Qdrant hit — fall through to Postgres legacy rows.
}
try {
const rows = await executeAll(
`SELECT text, flags, embedding
FROM text_analysis_cache
WHERE source = 'user_moderation'
AND embedding IS NOT NULL
AND expires_at > $1
ORDER BY analyzed_at DESC
LIMIT $2`,
[Date.now(), limit],
);
if (!rows || rows.length === 0) return null;
const candidates = rows.flatMap((row) => {
let embeddingArr: number[] = [];
let parsed: Record<string, unknown>;
try {
embeddingArr = JSON.parse(row.embedding) as number[];
parsed = JSON.parse(row.flags) as Record<string, unknown>;
} catch {
return [];
}
// Skip processing locks / malformed entries — never reuse an
// in-flight or non-verdict row.
const storedStatus = parsed.status as string | undefined;
if (storedStatus === "processing" || storedStatus === undefined) {
return [];
}
if (!Array.isArray(parsed.flags)) return [];
return [
{
text: row.text,
embedding: embeddingArr,
parsed,
},
];
});
const match = findBestEmbeddingMatch(
embedding,
candidates.map((c) => c.embedding),
minSimilarity,
);
if (!match) return null;
const hit = candidates[match.index];
const parsed = hit.parsed;
const flags = (parsed.flags as string[]) ?? [];
const status = normalizeStoredStatus(
parsed.status as string | undefined,
flags,
);
return {
text: hit.text,
similarity: match.similarity,
status,
flags,
score: (parsed.score as number) ?? 0,
analysis: (parsed.analysis as string) ?? "",
categories: (parsed.categories as string[]) ?? [],
severity: (parsed.severity as string) ?? "none",
confidence: (parsed.confidence as number) ?? 0,
recommendedAction: (parsed.recommendedAction as string) ?? "none",
};
} catch (error) {
logger.error(
{ error: error instanceof Error ? error.message : String(error) },
"Failed semantic text moderation lookup",
);
return null;
}
}
/**
* Upsert a bare (context-free) clean verdict to the Qdrant vector store,
* making global-reuse clean verdicts discoverable by semantic search.
*
* Why: the main `setCachedTextModeration` writes bare-key rows to Postgres
* with embedding=null (deliberate — no duplicate PG embedding column), but
* a bare clean verdict that never reaches Qdrant is invisible to
* searchQdrantBatch. So two messages with identical clean content in
* DIFFERENT channels never match semantically — the semantic hit-rate is
* capped below the exact-cache hit-rate. This helper shares the embedding
* already computed at lookup time so the bare point is semantically
* findable.
*
* Guard: only non-actionable clean verdicts qualify (same guard as the
* read path and as the orchestrator's bare-key write-back). No-op when
* Qdrant is disabled or no embedding is available.
*/
export async function upsertBareKeyToQdrant(
bareKey: string,
result: {
status: string;
flags: string[];
score: number;
analysis: string;
categories: string[];
severity: string;
confidence: number;
recommendedAction: string;
},
embedding: number[] | null | undefined,
): Promise<void> {
if (!isQdrantConfigured() || !embedding || embedding.length === 0) return;
if (!isGloballyReusableCleanVerdict(result, undefined)) return;
const now = Date.now();
const USER_MOD_CACHE_TTL_MS = 24 * 60 * 60 * 1000;
await upsertQdrantPoint(bareKey, embedding, {
text: bareKey,
flags: JSON.stringify(result),
analyzed_at: now,
expires_at: now + USER_MOD_CACHE_TTL_MS,
content_hash: bareKey.split(":").pop() ?? "",
}).catch((err: unknown) => {
logger.error(
{ error: err instanceof Error ? err.message : String(err), bareKey },
"Failed to upsert bare-key clean verdict to Qdrant",
);
});
}
/** /**
* Store a moderation result for a (user, content) pair. * Store a moderation result for a (user, content) pair.
* The `flags` field stores the full result object as JSON. * The `flags` field stores the full result object as JSON.
* `embedding` (optional) is stored for semantic near-duplicate lookups.
*/ */
export async function setCachedTextModeration( export async function setCachedTextModeration(
cacheKey: string, cacheKey: string,
@@ -730,41 +514,25 @@ export async function setCachedTextModeration(
recommendedAction: string; recommendedAction: string;
status?: "clean" | "warn" | "flagged" | "processing"; status?: "clean" | "warn" | "flagged" | "processing";
}, },
embedding?: number[] | null,
): Promise<void> { ): Promise<void> {
const now = Date.now(); const now = Date.now();
const USER_MOD_CACHE_TTL_MS = 24 * 60 * 60 * 1000; const USER_MOD_CACHE_TTL_MS = 24 * 60 * 60 * 1000;
try { try {
// Qdrant is the primary vector store when configured: upsert the point
// with the verdict payload; skip the Postgres embedding column entirely.
if (isQdrantConfigured() && embedding && embedding.length > 0) {
await upsertQdrantPoint(cacheKey, embedding, {
text: cacheKey,
flags: JSON.stringify(result),
analyzed_at: now,
expires_at: now + USER_MOD_CACHE_TTL_MS,
content_hash: cacheKey.split(":").pop() ?? "",
});
}
await executeAll( await executeAll(
`INSERT INTO text_analysis_cache (text, flags, source, analyzed_at, expires_at, hit_count, embedding) `INSERT INTO text_analysis_cache (text, flags, source, analyzed_at, expires_at, hit_count)
VALUES ($1, $2, $3, $4, $5, 0, $6) VALUES ($1, $2, $3, $4, $5, 0)
ON CONFLICT (text) DO UPDATE SET ON CONFLICT (text) DO UPDATE SET
flags = EXCLUDED.flags, flags = EXCLUDED.flags,
source = EXCLUDED.source, source = EXCLUDED.source,
analyzed_at = EXCLUDED.analyzed_at, analyzed_at = EXCLUDED.analyzed_at,
expires_at = EXCLUDED.expires_at, expires_at = EXCLUDED.expires_at`,
embedding = COALESCE(EXCLUDED.embedding, text_analysis_cache.embedding)`,
[ [
cacheKey, cacheKey,
JSON.stringify(result), JSON.stringify(result),
"user_moderation", "user_moderation",
now, now,
now + USER_MOD_CACHE_TTL_MS, now + USER_MOD_CACHE_TTL_MS,
// Postgres embedding stays as legacy fallback; Qdrant is primary.
embedding && embedding.length > 0 ? JSON.stringify(embedding) : null,
], ],
); );
} catch (error) { } catch (error) {
@@ -833,11 +601,10 @@ export async function getRecentCorrectedModerations(
/** /**
* Store a corrected moderation entry for future few-shot injection. * Store a corrected moderation entry for future few-shot injection.
* *
* Also invalidates any cached verdicts for the corrected content (both * Also invalidates any cached verdicts for the corrected content so the
* Postgres rows and Qdrant points) so the corrected decision propagates * corrected decision propagates immediately instead of being shadowed by a
* immediately instead of being shadowed by a stale cache entry. Full * stale cache entry. Full content is looked up by message_id when available
* content is looked up by message_id when available — more precise than * — more precise than the (possibly truncated) snippet.
* the (possibly truncated) snippet.
*/ */
export async function insertCorrectedModeration(entry: { export async function insertCorrectedModeration(entry: {
messageId: string; messageId: string;
@@ -75,7 +75,7 @@ export const KNOWN_SAFE_TERMS = new Set(
( (
"discord youtube google facebook instagram twitter tiktok whatsapp telegram netflix spotify steam github gitlab bitbucket chatgpt openai anthropic claude deepseek gemini llama copilot cursor vscode vscodium jetbrains intellij pycharm webstorm sublime codeblocks" + "discord youtube google facebook instagram twitter tiktok whatsapp telegram netflix spotify steam github gitlab bitbucket chatgpt openai anthropic claude deepseek gemini llama copilot cursor vscode vscodium jetbrains intellij pycharm webstorm sublime codeblocks" +
" docker kubernetes k8s linux ubuntu debian arch fedora manjaro kali windows macos android ios chrome firefox safari edge opera brave" + " docker kubernetes k8s linux ubuntu debian arch fedora manjaro kali windows macos android ios chrome firefox safari edge opera brave" +
" react nextjs next vue svelte angular node nodejs deno bun pnpm yarn npm javascript typescript python golang go rust java kotlin swift cplusplus cpp css html json xml yaml toml regex backend frontend database mysql postgres postgresql mongodb redis qdrant sqlite nosql graphql rest websocket webhook" + " react nextjs next vue svelte angular node nodejs deno bun pnpm yarn npm javascript typescript python golang go rust java kotlin swift cplusplus cpp css html json xml yaml toml regex backend frontend database mysql postgres postgresql mongodb redis sqlite nosql graphql rest websocket webhook" +
" bug crash error debug fix issue pr merge commit push pull branch main master dev staging production server client app website web browser" + " bug crash error debug fix issue pr merge commit push pull branch main master dev staging production server client app website web browser" +
" stream streaming video audio voice call camera screen share screenshare gameplay gaming game play steam epic xbox playstation nintendo switch console" + " stream streaming video audio voice call camera screen share screenshare gameplay gaming game play steam epic xbox playstation nintendo switch console" +
" bot discordbot moderation moderator admin member user profile avatar channel server guild message chat dm reply forward embed sticker emoji role permission" + " bot discordbot moderation moderator admin member user profile avatar channel server guild message chat dm reply forward embed sticker emoji role permission" +
@@ -1,122 +0,0 @@
import {
embedText,
normalizeEmbeddingContent,
} from "@/modules/ai-moderation/embeddingClient";
import {
ARCHIVE_COLLECTION,
qdrantPointId,
upsertQdrantPointV2,
} from "@/modules/ai-moderation/qdrantClient";
import { createChildLogger } from "@/shared/logger/index";
import { config } from "../../shared/config/index.js";
const log = createChildLogger("archive-embedder");
export interface ArchiveMessage {
id: string;
content: string;
username: string;
channel_id: string;
guild_id: string;
thread_id: string | null;
created_at: number;
/** JSON string of RichMessageMetadata (parsed for channel/thread names). */
metadata?: string | null;
/** True when the message came from an age-restricted (NSFW) channel. NSFW
* content is deliberately NOT embedded into the public archive so it can't
* be found via public semantic search. Defaults to false. */
isAgeRestricted?: boolean;
}
/**
* Extract a human-readable channel label from the message's metadata JSON.
* The gateway captures `metadata.channel.{channelName,threadName}` per message;
* prefer the thread name (thread messages read better by their thread title),
* then the channel name. Returns null when unavailable (old messages without
* the metadata field, or malformed JSON).
*/
export function extractChannelLabel(metadata: string | null | undefined): {
channel_name: string | null;
thread_name: string | null;
} {
if (!metadata) return { channel_name: null, thread_name: null };
try {
const m = JSON.parse(metadata) as {
channel?: {
channelName?: string | null;
threadName?: string | null;
};
};
return {
channel_name: m?.channel?.channelName ?? null,
thread_name: m?.channel?.threadName ?? null,
};
} catch {
return { channel_name: null, thread_name: null };
}
}
/**
* Fire-and-forget: embed a captured message and upsert it into the persistent
* archive collection so the public web can semantic-search the corpus.
*
* Failures are swallowed — searching is a nice-to-have, never a precondition
* for capture or moderation. The message text is kept in the payload so the
* search endpoint can return results even for deleted messages.
*
* NSFW/age-restricted messages are skipped (never embedded) — they are stored
* in the database for the dashboard but kept out of the public search archive.
*/
export function archiveMessageEmbedded(message: ArchiveMessage): void {
if (message.isAgeRestricted) return; // never surface NSFW in public archive
if (!config.AI_LLM_EMBEDDING_MODEL) return; // embeddings disabled → skip
const text = message.content?.trim();
if (!text || text.length < 3) return;
void (async () => {
try {
// Normalize once: the vector AND the stored payload both use the clean
// text so the public search returns readable content and the vector
// isn't diluted by @mentions/URLs/emoji (see normalizeEmbeddingContent).
const normalized = normalizeEmbeddingContent(text);
if (!normalized) return; // nothing meaningful left after cleanup
const vector = await embedText(normalized);
if (!vector) return;
const { channel_name, thread_name } = extractChannelLabel(
message.metadata,
);
const ok = await upsertQdrantPointV2(
ARCHIVE_COLLECTION,
qdrantPointId(`archive:${message.id}`),
vector,
{
text: normalized.slice(0, 4000),
flags: "",
// Rich metadata so public semantic search results can be shown in
// context (who said it, where, when) instead of a bare text blob.
username: message.username ?? "",
channel_id: message.channel_id ?? "",
guild_id: message.guild_id ?? "",
thread_id: message.thread_id ?? null,
channel_name: channel_name ?? null,
thread_name: thread_name ?? null,
created_at: message.created_at ?? Date.now(),
analyzed_at: Date.now(),
// 5-year persistent window (archive is NOT a TTL cache).
expires_at: Date.now() + 1000 * 60 * 60 * 24 * 365 * 5,
content_hash: message.id,
},
);
if (!ok) return;
log.debug({ messageId: message.id }, "Archived message embedding");
} catch (err) {
log.debug(
{
messageId: message.id,
error: err instanceof Error ? err.message : String(err),
},
"archive embed skipped",
);
}
})();
}
@@ -4,12 +4,10 @@ import { config } from "../../shared/config/index.js";
import { queueMessageAnalysis } from "../ai-moderation/aiAnalyzer.js"; import { queueMessageAnalysis } from "../ai-moderation/aiAnalyzer.js";
import { processAttachmentUpload } from "../attachment-upload/attachmentUploader.js"; import { processAttachmentUpload } from "../attachment-upload/attachmentUploader.js";
import type { EventBroadcaster } from "../event-broadcaster/eventBroadcaster.js"; import type { EventBroadcaster } from "../event-broadcaster/eventBroadcaster.js";
import { archiveMessageEmbedded } from "../message-capture/archiveEmbedder.js";
import { import {
getDisplayContent, getDisplayContent,
getMessageLocation, getMessageLocation,
getMessageMetadata, getMessageMetadata,
isAgeRestrictedMessage,
} from "../message-capture/messageMetadata.js"; } from "../message-capture/messageMetadata.js";
import { messageStore } from "../message-capture/messageStore.js"; import { messageStore } from "../message-capture/messageStore.js";
import type { import type {
@@ -213,16 +211,6 @@ export async function captureMessage(
return; return;
} }
// Fire-and-forget: make the captured message searchable in the persistent
// archive (public semantic search). Never blocks capture/moderation.
// NSFW/age-restricted messages are kept OUT of the public archive.
if (!isBacklog && messageRecord.content) {
archiveMessageEmbedded({
...messageRecord,
isAgeRestricted: isAgeRestrictedMessage(message),
});
}
if (_eventBroadcaster && !isBacklog) { if (_eventBroadcaster && !isBacklog) {
_eventBroadcaster.messageCreated(messageRecord); _eventBroadcaster.messageCreated(messageRecord);
} }
@@ -167,34 +167,6 @@ export const configSchema = z
.describe( .describe(
"Disable LLM chain-of-thought (reasoning/thinking) to speed up AI analysis. Set false to restore thinking.", "Disable LLM chain-of-thought (reasoning/thinking) to speed up AI analysis. Set false to restore thinking.",
), ),
AI_LLM_EMBEDDING_MODEL: z.string().optional(),
AI_LLM_EMBEDDING_MIN_SIMILARITY: z.coerce
.number()
.min(0)
.max(1)
.default(0.97),
// Two-band semantic acceptance (2026-08-24): non-actionable verdicts
// (clean, no flags, action=none) may be reused from a LOOSER similarity
// band than actionable ones (warn/flagged). Actionable verdicts keep the
// strict gate above; anything between the two bands falls through to the
// LLM (fail-open toward accuracy).
AI_LLM_EMBEDDING_MIN_SIMILARITY_CLEAN: z.coerce
.number()
.min(0)
.max(1)
.default(0.92),
AI_LLM_EMBEDDING_MAX_CANDIDATES: z.coerce
.number()
.int()
.positive()
.default(50),
// Qdrant vector store for the semantic moderation cache. When
// QDRANT_URL is set, embeddings are stored/searched there (Postgres
// embedding column remains as a legacy fallback).
QDRANT_URL: z.string().optional(),
QDRANT_COLLECTION: z.string().default("gmw_text_moderation"),
QDRANT_ARCHIVE_COLLECTION: z.string().default("gmw_message_archive"),
QDRANT_API_KEY: z.string().optional(),
AI_LLM_MAX_CONCURRENT: z.coerce.number().int().positive().default(8), AI_LLM_MAX_CONCURRENT: z.coerce.number().int().positive().default(8),
// Media-lane LLM concurrency cap (2026-09-24): vision + media-batch calls // Media-lane LLM concurrency cap (2026-09-24): vision + media-batch calls
// use their OWN semaphore instead of sharing AI_LLM_MAX_CONCURRENT, so a // use their OWN semaphore instead of sharing AI_LLM_MAX_CONCURRENT, so a
@@ -395,9 +395,6 @@ export const pgTextAnalysisCacheTable = pgTable(
analyzed_at: pgBigint("analyzed_at", { mode: "number" }).notNull(), analyzed_at: pgBigint("analyzed_at", { mode: "number" }).notNull(),
expires_at: pgBigint("expires_at", { mode: "number" }).notNull(), expires_at: pgBigint("expires_at", { mode: "number" }).notNull(),
hit_count: pgInteger("hit_count").notNull().default(0), hit_count: pgInteger("hit_count").notNull().default(0),
// JSON-encoded embedding vector for semantic moderation cache lookups.
// Null for entries stored before embeddings were enabled.
embedding: pgText("embedding"),
}, },
(table) => ({ (table) => ({
expiresAtIdx: pgIndex("idx_text_analysis_cache_expires_at").on( expiresAtIdx: pgIndex("idx_text_analysis_cache_expires_at").on(
@@ -1,11 +1,9 @@
// ═══════════════════════════════════════════════════════════════════════════ // ═══════════════════════════════════════════════════════════════════════════
// Semantic two-band acceptance + global exact-cache reuse guard // Global exact-cache reuse guard
// ═══════════════════════════════════════════════════════════════════════════ // ═══════════════════════════════════════════════════════════════════════════
// Design (2026-08-24): cache hits may be served MORE aggressively for // Design (2026-08-24): cache hits may be served MORE aggressively for
// verdicts that cannot trigger enforcement actions, and NEVER more // verdicts that cannot trigger enforcement actions, and NEVER more
// aggressively for actionable ones. Two layers enforce this: // aggressively for actionable ones:
// - isSemanticBandAccepted: similarity thresholds differ by verdict class
// (clean band 0.92 default vs strict actionable band 0.97 default).
// - isGloballyReusableCleanVerdict: context-free (cross-channel) reuse of // - isGloballyReusableCleanVerdict: context-free (cross-channel) reuse of
// the legacy bare key only for clean / flagless / action=none verdicts // the legacy bare key only for clean / flagless / action=none verdicts
// with high confidence and bounded age. // with high confidence and bounded age.
@@ -31,64 +29,6 @@ function makeVerdict(
}; };
} }
describe("isSemanticBandAccepted", () => {
it("accepts a non-actionable clean verdict at the loose clean band", () => {
// Default AI_LLM_EMBEDDING_MIN_SIMILARITY_CLEAN = 0.92.
expect(isBandAccept(makeVerdict(), 0.93)).toBe(true);
});
it("accepts a clean verdict exactly at the clean band boundary", () => {
expect(isBandAccept(makeVerdict({ confidence: 0.99 }), 0.92)).toBe(true);
});
it("rejects a clean verdict below the clean band", () => {
expect(isBandAccept(makeVerdict(), 0.91)).toBe(false);
});
it("rejects an actionable flagged verdict between the bands", () => {
// 0.93 >= clean band BUT < strict band → must NOT be served.
expect(
isBandAccept(
makeVerdict({ status: "flagged", flags: ["hate_speech"] }),
0.93,
),
).toBe(false);
});
it("accepts a flagged verdict at the strict band", () => {
expect(
isBandAccept(
makeVerdict({ status: "flagged", flags: ["hate_speech"] }),
0.98,
),
).toBe(true);
});
it("rejects a warn verdict below the strict band", () => {
expect(
isBandAccept(
makeVerdict({ status: "warn", recommendedAction: "warn" }),
0.96,
),
).toBe(false);
});
it("treats a clean verdict WITH flags as actionable (strict band)", () => {
expect(isBandAccept(makeVerdict({ flags: ["borderline"] }), 0.93)).toBe(
false,
);
});
it("treats a clean verdict with a non-none action as actionable", () => {
expect(
isBandAccept(makeVerdict({ recommendedAction: "review" }), 0.93),
).toBe(false);
});
});
// Import indirection so the describe block reads cleanly.
import { isSemanticBandAccepted as isBandAccept } from "../src/modules/ai-moderation/textCacheStore.js";
describe("isGloballyReusableCleanVerdict", () => { describe("isGloballyReusableCleanVerdict", () => {
it("accepts a fresh, confident, flagless clean verdict", () => { it("accepts a fresh, confident, flagless clean verdict", () => {
const v = makeVerdict({ confidence: 0.9 }); const v = makeVerdict({ confidence: 0.9 });
@@ -1,82 +0,0 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
// The qdrant client reads config at import time — we only need the
// reset-on-failure behavior, so mock global fetch and import fresh.
import {
ensureQdrantCollection,
resetQdrantCollectionCache,
} from "../src/modules/ai-moderation/qdrantClient.js";
function jsonResponse(body: unknown, ok = true, status = 200): Response {
return {
ok,
status,
text: async () => JSON.stringify(body),
} as unknown as Response;
}
describe("ensureQdrantCollection retry-on-failure", () => {
beforeEach(() => {
resetQdrantCollectionCache();
vi.restoreAllMocks();
});
afterEach(() => {
resetQdrantCollectionCache();
vi.restoreAllMocks();
});
it("returns false on failure and does NOT get stuck — retries on next call", async () => {
// First call: GET collection fails hard (not a 404) → ensure rejects.
const fetchMock = vi
.spyOn(globalThis, "fetch")
.mockRejectedValueOnce(new Error("network down"))
// Second call: GET returns a matching-size collection → success.
.mockResolvedValueOnce(
jsonResponse({
result: { config: { params: { vectors: { size: 3072 } } } },
}),
);
const first = await ensureQdrantCollection(3072);
expect(first).toBe(false);
// Without the fix, the second call returns the memoised rejected promise
// and fetch is never called again. With the fix, it retries.
const second = await ensureQdrantCollection(3072);
expect(second).toBe(true);
expect(fetchMock).toHaveBeenCalledTimes(2);
});
it("recreates collection when vector size changes (DELETE + PUT)", async () => {
const fetchMock = vi
.spyOn(globalThis, "fetch")
.mockResolvedValueOnce(
jsonResponse({
result: { config: { params: { vectors: { size: 2048 } } } },
}),
) // GET old size
.mockResolvedValueOnce(jsonResponse({ result: true })) // DELETE
.mockResolvedValueOnce(jsonResponse({ result: true })); // PUT
const ok = await ensureQdrantCollection(3072);
expect(ok).toBe(true);
const methods = fetchMock.mock.calls.map(
(c) => (c[1] as RequestInit).method,
);
expect(methods).toEqual(["GET", "DELETE", "PUT"]);
});
it("is idempotent when the collection already matches", async () => {
const fetchMock = vi.spyOn(globalThis, "fetch").mockResolvedValueOnce(
jsonResponse({
result: { config: { params: { vectors: { size: 3072 } } } },
}),
);
const ok = await ensureQdrantCollection(3072);
expect(ok).toBe(true);
expect(fetchMock).toHaveBeenCalledTimes(1);
});
});
@@ -6,10 +6,10 @@
// ["conflict_instigation"]) fell into the legacy `flags.length === 0 ? // ["conflict_instigation"]) fell into the legacy `flags.length === 0 ?
// clean : flagged` branch and was read back as FLAGGED. Downstream this // clean : flagged` branch and was read back as FLAGGED. Downstream this
// broke auto-delete eligibility gating and mislabelled warnings on the // broke auto-delete eligibility gating and mislabelled warnings on the
// dashboard. parseQdrantVerdict had the same narrowing (warn → clean). // dashboard.
// //
// Fix: normalizeStoredStatus() accepts the full clean/warn/flagged union in // Fix: normalizeStoredStatus() accepts the full clean/warn/flagged union;
// BOTH readers; unknown/legacy values still derive from flags. // unknown/legacy values still derive from flags.
import { describe, expect, it } from "vitest"; import { describe, expect, it } from "vitest";
import { normalizeStoredStatus } from "../src/modules/ai-moderation/textCacheStore.js"; import { normalizeStoredStatus } from "../src/modules/ai-moderation/textCacheStore.js";
@@ -11,7 +11,6 @@ import {
Paperclip, Paperclip,
Search, Search,
ShieldAlert, ShieldAlert,
Sparkles,
} from "lucide-react"; } from "lucide-react";
import { useCallback, useEffect, useMemo, useRef, useState } from "react"; import { useCallback, useEffect, useMemo, useRef, useState } from "react";
import { ActivityHeatmap } from "@/components/ActivityHeatmap"; import { ActivityHeatmap } from "@/components/ActivityHeatmap";
@@ -42,7 +41,6 @@ import {
useMessagesWsSync, useMessagesWsSync,
useRecentEdits, useRecentEdits,
useReviewWsSync, useReviewWsSync,
useSemanticSearch,
} from "@/hooks"; } from "@/hooks";
import { useStaggerReveal } from "@/hooks/use-gsap-animation"; import { useStaggerReveal } from "@/hooks/use-gsap-animation";
import { aiTone } from "@/lib/ai-status"; import { aiTone } from "@/lib/ai-status";
@@ -85,9 +83,6 @@ export function MessagesView({
const [channelId, setChannelId] = useState<string | null>(null); const [channelId, setChannelId] = useState<string | null>(null);
const [selected, setSelected] = useState<string | null>(null); const [selected, setSelected] = useState<string | null>(null);
const [query, setQuery] = useState(""); const [query, setQuery] = useState("");
// Search mode: "exact" (substring match over captured messages) or
// "semantic" (vector similarity over the persistent Qdrant archive).
const [semanticMode, setSemanticMode] = useState(false);
// feed | timeline: "timeline" groups messages into date-grouped cards. // feed | timeline: "timeline" groups messages into date-grouped cards.
const [viewMode, setViewMode] = useState<"feed" | "timeline">("feed"); const [viewMode, setViewMode] = useState<"feed" | "timeline">("feed");
// Guard against loading the entire history on a long scroll: cap how many // Guard against loading the entire history on a long scroll: cap how many
@@ -123,15 +118,7 @@ export function MessagesView({
const loadMore = useLoadMore(); const loadMore = useLoadMore();
useMessagesWsSync(ws, guildId ?? ""); useMessagesWsSync(ws, guildId ?? "");
useReviewWsSync(ws); useReviewWsSync(ws);
const search = useMessageSearch( const search = useMessageSearch(query, query.trim().length >= 2);
query,
query.trim().length >= 2 && !semanticMode,
);
const semantic = useSemanticSearch(
query,
query.trim().length >= 2 && semanticMode,
guildId,
);
const activity = useMessageActivity(30); const activity = useMessageActivity(30);
const edits = useRecentEdits(50, undefined, initialEdits); const edits = useRecentEdits(50, undefined, initialEdits);
const detail = useMessageDetail(selected); const detail = useMessageDetail(selected);
@@ -161,8 +148,7 @@ export function MessagesView({
ambient.set(query ? "amber" : "signal", 0.3, query ? "search" : "messages"); ambient.set(query ? "amber" : "signal", 0.3, query ? "search" : "messages");
}, [query, ambient]); }, [query, ambient]);
const searching = query.trim().length >= 2 && !semanticMode; const searching = query.trim().length >= 2;
const semanticSearching = query.trim().length >= 2 && semanticMode;
const list = searching ? (search.data ?? []) : (messages ?? []); const list = searching ? (search.data ?? []) : (messages ?? []);
// Discord-style order: oldest at the top, newest at the bottom. The backend // Discord-style order: oldest at the top, newest at the bottom. The backend
// returns DESC (newest first); reverse so the feed reads top→bottom like DC. // returns DESC (newest first); reverse so the feed reads top→bottom like DC.
@@ -283,32 +269,6 @@ export function MessagesView({
onChange={(e) => setQuery(e.target.value)} onChange={(e) => setQuery(e.target.value)}
/> />
</div> </div>
<div className="flex items-center gap-1.5 rounded-[6px] border border-hairline bg-surface-2 p-0.5">
<button
type="button"
onClick={() => setSemanticMode(false)}
className={`rounded-[4px] px-2.5 py-1 font-mono text-[10px] font-medium transition-all ${
!semanticMode
? "bg-surface text-ink border border-hairline-focus shadow-xs"
: "text-ink-muted hover:text-ink"
}`}
>
EXACT
</button>
<button
type="button"
onClick={() => setSemanticMode(true)}
className={`flex items-center gap-1 rounded-[4px] px-2.5 py-1 font-mono text-[10px] font-medium transition-all ${
semanticMode
? "bg-signal/20 text-signal border border-signal/40 shadow-xs"
: "text-ink-muted hover:text-ink"
}`}
>
<Sparkles className="size-3" />
SEMANTIC
</button>
</div>
<div className="flex items-center gap-1.5 rounded-[6px] border border-hairline bg-surface-2 p-0.5"> <div className="flex items-center gap-1.5 rounded-[6px] border border-hairline bg-surface-2 p-0.5">
<button <button
type="button" type="button"
@@ -336,76 +296,6 @@ export function MessagesView({
</GlassPanel> </GlassPanel>
<div className="grid gap-3 lg:grid-cols-5"> <div className="grid gap-3 lg:grid-cols-5">
{semanticSearching && (
<GlassPanel className="lg:col-span-5">
<SectionHeader
eyebrow="semantic archive"
title={`Vector Matches for “${query}”`}
action={
<span className="mono text-xs text-[#8a8f98]">
{semantic.data?.length ?? 0} matches
</span>
}
/>
{semantic.isLoading ? (
<SkeletonRows rows={4} />
) : semantic.data && semantic.data.length > 0 ? (
<div className="max-h-[60vh] space-y-1.5 overflow-y-auto pr-1">
{semantic.data.map((r, i) => (
<div
key={r.message_id ?? i}
className="hud-card animate-stagger flex items-start gap-3 p-3"
style={staggerDelay(i)}
>
<div className="min-w-0 flex-1">
<div className="flex items-center gap-2">
<span className="font-mono text-[10px] font-semibold text-signal">
{(r.score * 100).toFixed(0)}% RELEVANCE
</span>
{r.username && (
<span className="font-mono text-[10px] font-medium text-ink">
{r.username}
</span>
)}
{r.thread_name && (
<span className="font-mono text-[10px] text-ink-muted">
▶ {r.thread_name}
</span>
)}
{!r.thread_name && r.channel_name && (
<span className="font-mono text-[10px] text-ink-muted">
#{r.channel_name}
</span>
)}
{!r.thread_name && !r.channel_name && r.channel_id && (
<span className="font-mono text-[10px] text-ink-faint">
#{r.channel_id}
</span>
)}
<span
className="ml-auto font-mono text-[10px] text-ink-muted"
suppressHydrationWarning
>
{formatRelativeTime(r.created_at)}
</span>
</div>
<div className="mt-1 text-xs text-ink-soft leading-relaxed">
{r.content}
</div>
</div>
</div>
))}
</div>
) : (
<EmptyState
icon={<Search className="size-7" />}
title="No semantic matches"
description="Try different phrasing — vector search inspects contextual semantics."
/>
)}
</GlassPanel>
)}
{/* Message Stream Deck */} {/* Message Stream Deck */}
<GlassPanel className="lg:col-span-3"> <GlassPanel className="lg:col-span-3">
<SectionHeader <SectionHeader
-1
View File
@@ -25,7 +25,6 @@ export {
useRecentEdits, useRecentEdits,
useReview, useReview,
useReviewWsSync, useReviewWsSync,
useSemanticSearch,
useTextChannels, useTextChannels,
} from "./use-messages"; } from "./use-messages";
export { export {
@@ -8,7 +8,6 @@ import type {
EditHistoryRow, EditHistoryRow,
MessageActivityBucket, MessageActivityBucket,
MessageRecord, MessageRecord,
SemanticSearchResult,
} from "@/lib/types"; } from "@/lib/types";
import type { WsHook } from "@/lib/ws-hook"; import type { WsHook } from "@/lib/ws-hook";
@@ -210,29 +209,6 @@ export function useMessageSearch(query: string, enabled: boolean) {
); );
} }
// ── Semantic Search (public archive, Qdrant) ──────
export function useSemanticSearch(
query: string,
enabled: boolean,
guildId?: string | null,
) {
return useSWR<SemanticSearchResult[]>(
enabled && query.trim().length >= 2
? ["semantic-search", query.trim(), guildId ?? ""]
: null,
async () => {
const res = await messagesApi.semanticSearch(
query.trim(),
10,
guildId ?? undefined,
);
return res.results;
},
{ keepPreviousData: true },
);
}
// ── WS sync helpers ────────────────────────────── // ── WS sync helpers ──────────────────────────────
/** /**
-12
View File
@@ -6,7 +6,6 @@ import type {
Guild, Guild,
MessageActivityBucket, MessageActivityBucket,
MessageRecord, MessageRecord,
SemanticSearchResult,
} from "@/lib/types"; } from "@/lib/types";
export const messagesApi = { export const messagesApi = {
@@ -76,17 +75,6 @@ export const messagesApi = {
results: MessageRecord[]; results: MessageRecord[];
}>, }>,
// Public semantic search over the persistent message archive (Qdrant).
semanticSearch: (query: string, limit?: number, guildId?: string) =>
orpc.messages.semanticSearch({
query,
limit,
...(guildId ? { guildId } : {}),
}) as unknown as Promise<{
results: SemanticSearchResult[];
nextCursor: null;
}>,
// Public, read-only activity heatmap data (per-hour volume by channel). // Public, read-only activity heatmap data (per-hour volume by channel).
getActivity: (days = 30) => getActivity: (days = 30) =>
orpc.messages.activity({ days }) as unknown as Promise< orpc.messages.activity({ days }) as unknown as Promise<
@@ -166,29 +166,9 @@ export interface AttachmentRecord {
uploaded_at?: number | null; uploaded_at?: number | null;
} }
// ── Semantic Search (read-only public archive search) ──────────
export interface SemanticSearchResult {
message_id: string | null;
content: string;
score: number;
created_at: number;
username: string | null;
channel_id: string | null;
guild_id: string | null;
thread_id: string | null;
channel_name: string | null;
thread_name: string | null;
}
export interface MessageActivityBucket { export interface MessageActivityBucket {
channelId: string; channelId: string;
channelName: string; channelName: string;
hour: number; hour: number;
count: number; count: number;
} }
export interface SemanticSearchResponse {
results: SemanticSearchResult[];
nextCursor: null;
}