Gateway: - Remove dead barrels (ai-moderation/index, attachment-upload/index, message-capture/index) — all consumers import files directly - Remove orphaned schema/ split dir (analytics/cache/messages/meta) — schema.ts is monolithic - Merge duplicate config singleton: delete shared/config/config.ts, point all 44 imports at shared/config/index Backend: - Remove dead commandHelper.ts (voice-era fallback), ws/index.ts barrel, health.schema.ts, moderationMetrics.ts, analysis.schema.ts (0 importers; metrics/handlers route directly) Frontend: - Remove orphaned CategoryDrilldown/CoverageTiles/TopicTrends, primitives/slot, use-mobile, use-mounted - Remove unused charts donut/sparkline (TopicTrends was only consumer) Kept (verified active): shared/database/index.ts facade (11 importers), hooks/index + lib/api/index barrels (10 importers), orpc/ws.ts, charts/index.ts barrel. Verified: tsc + biome + vitest per service (backend e2e 3 failures pre-existing on main); frontend next build 8 routes.
123 lines
4.4 KiB
TypeScript
123 lines
4.4 KiB
TypeScript
import {
|
|
embedText,
|
|
normalizeEmbeddingContent,
|
|
} from "@/modules/ai-moderation/embeddingClient";
|
|
import {
|
|
ARCHIVE_COLLECTION,
|
|
qdrantPointId,
|
|
upsertQdrantPointV2,
|
|
} from "@/modules/ai-moderation/qdrantClient";
|
|
import { createChildLogger } from "@/shared/logger/index";
|
|
import { config } from "../../shared/config/index.js";
|
|
|
|
const log = createChildLogger("archive-embedder");
|
|
|
|
export interface ArchiveMessage {
|
|
id: string;
|
|
content: string;
|
|
username: string;
|
|
channel_id: string;
|
|
guild_id: string;
|
|
thread_id: string | null;
|
|
created_at: number;
|
|
/** JSON string of RichMessageMetadata (parsed for channel/thread names). */
|
|
metadata?: string | null;
|
|
/** True when the message came from an age-restricted (NSFW) channel. NSFW
|
|
* content is deliberately NOT embedded into the public archive so it can't
|
|
* be found via public semantic search. Defaults to false. */
|
|
isAgeRestricted?: boolean;
|
|
}
|
|
|
|
/**
|
|
* Extract a human-readable channel label from the message's metadata JSON.
|
|
* The gateway captures `metadata.channel.{channelName,threadName}` per message;
|
|
* prefer the thread name (thread messages read better by their thread title),
|
|
* then the channel name. Returns null when unavailable (old messages without
|
|
* the metadata field, or malformed JSON).
|
|
*/
|
|
export function extractChannelLabel(metadata: string | null | undefined): {
|
|
channel_name: string | null;
|
|
thread_name: string | null;
|
|
} {
|
|
if (!metadata) return { channel_name: null, thread_name: null };
|
|
try {
|
|
const m = JSON.parse(metadata) as {
|
|
channel?: {
|
|
channelName?: string | null;
|
|
threadName?: string | null;
|
|
};
|
|
};
|
|
return {
|
|
channel_name: m?.channel?.channelName ?? null,
|
|
thread_name: m?.channel?.threadName ?? null,
|
|
};
|
|
} catch {
|
|
return { channel_name: null, thread_name: null };
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Fire-and-forget: embed a captured message and upsert it into the persistent
|
|
* archive collection so the public web can semantic-search the corpus.
|
|
*
|
|
* Failures are swallowed — searching is a nice-to-have, never a precondition
|
|
* for capture or moderation. The message text is kept in the payload so the
|
|
* search endpoint can return results even for deleted messages.
|
|
*
|
|
* NSFW/age-restricted messages are skipped (never embedded) — they are stored
|
|
* in the database for the dashboard but kept out of the public search archive.
|
|
*/
|
|
export function archiveMessageEmbedded(message: ArchiveMessage): void {
|
|
if (message.isAgeRestricted) return; // never surface NSFW in public archive
|
|
if (!config.AI_LLM_EMBEDDING_MODEL) return; // embeddings disabled → skip
|
|
const text = message.content?.trim();
|
|
if (!text || text.length < 3) return;
|
|
|
|
void (async () => {
|
|
try {
|
|
// Normalize once: the vector AND the stored payload both use the clean
|
|
// text so the public search returns readable content and the vector
|
|
// isn't diluted by @mentions/URLs/emoji (see normalizeEmbeddingContent).
|
|
const normalized = normalizeEmbeddingContent(text);
|
|
if (!normalized) return; // nothing meaningful left after cleanup
|
|
const vector = await embedText(normalized);
|
|
if (!vector) return;
|
|
const { channel_name, thread_name } = extractChannelLabel(
|
|
message.metadata,
|
|
);
|
|
const ok = await upsertQdrantPointV2(
|
|
ARCHIVE_COLLECTION,
|
|
qdrantPointId(`archive:${message.id}`),
|
|
vector,
|
|
{
|
|
text: normalized.slice(0, 4000),
|
|
flags: "",
|
|
// Rich metadata so public semantic search results can be shown in
|
|
// context (who said it, where, when) instead of a bare text blob.
|
|
username: message.username ?? "",
|
|
channel_id: message.channel_id ?? "",
|
|
guild_id: message.guild_id ?? "",
|
|
thread_id: message.thread_id ?? null,
|
|
channel_name: channel_name ?? null,
|
|
thread_name: thread_name ?? null,
|
|
created_at: message.created_at ?? Date.now(),
|
|
analyzed_at: Date.now(),
|
|
// 5-year persistent window (archive is NOT a TTL cache).
|
|
expires_at: Date.now() + 1000 * 60 * 60 * 24 * 365 * 5,
|
|
content_hash: message.id,
|
|
},
|
|
);
|
|
if (!ok) return;
|
|
log.debug({ messageId: message.id }, "Archived message embedding");
|
|
} catch (err) {
|
|
log.debug(
|
|
{
|
|
messageId: message.id,
|
|
error: err instanceof Error ? err.message : String(err),
|
|
},
|
|
"archive embed skipped",
|
|
);
|
|
}
|
|
})();
|
|
}
|