feat(ai): audit + harden embedding pipeline
- Normalize text before embedding (strip mentions/URLs/emoji/markdown/control chars, lowercase, truncate) on both write and query sides so vectors aren't diluted and tokens aren't wasted - embeddingClient: retry embeddings (maxRetries 2), validate batch dimension consistency, preserve index alignment for empty-normalized texts - archiveEmbedder: store normalized text in archive payload, skip empty-normalized content - backend: normalize search queries, make archive search similarity threshold configurable (AI_LLM_EMBEDDING_ARCHIVE_MIN_SIMILARITY, default 0.6)
This commit is contained in:
@@ -3,6 +3,26 @@ import { createChildLogger } from "@/shared/logger/index";
|
||||
|
||||
const logger = createChildLogger("messages-embed");
|
||||
|
||||
/** Max chars for a search query fed to the embedding model. */
|
||||
const MAX_QUERY_CHARS = 300;
|
||||
|
||||
/**
|
||||
* Normalize a user search query before embedding so it lands in the same
|
||||
* vector space as the archived content (which is normalized the same way on
|
||||
* write). Mirrors the gateway's normalizer: strip control/zero-width chars,
|
||||
* lowercase, collapse whitespace, cap length. Readable punctuation is kept —
|
||||
* a search query is already compact.
|
||||
*/
|
||||
export function normalizeEmbeddingQuery(raw: string): string {
|
||||
if (!raw) return "";
|
||||
return raw
|
||||
.replace(/[\p{Cc}\p{Cf}]/gu, " ")
|
||||
.toLowerCase()
|
||||
.replace(/\s+/g, " ")
|
||||
.trim()
|
||||
.slice(0, MAX_QUERY_CHARS);
|
||||
}
|
||||
|
||||
/**
|
||||
* Embed a search query with the configured OpenAI-compatible embedding model.
|
||||
* Uses raw fetch (the backend has no openai SDK dependency) and returns null
|
||||
@@ -10,8 +30,10 @@ const logger = createChildLogger("messages-embed");
|
||||
*
|
||||
* encoding_format: "float" is REQUIRED — Nvidia-backed models reject base64.
|
||||
*/
|
||||
export async function embedQuery(text: string): Promise<number[] | null> {
|
||||
export async function embedQuery(rawQuery: string): Promise<number[] | null> {
|
||||
if (!config.AI_LLM_API_KEY || !config.AI_LLM_EMBEDDING_MODEL) return null;
|
||||
const text = normalizeEmbeddingQuery(rawQuery);
|
||||
if (!text) return null;
|
||||
try {
|
||||
const res = await fetch(`${config.AI_LLM_BASE_URL}/embeddings`, {
|
||||
method: "POST",
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { config } from "@/shared/config/index";
|
||||
import { NotFoundError, ValidationError } from "@/shared/errors/index";
|
||||
import { createChildLogger } from "@/shared/logger/index";
|
||||
import { embedQuery } from "./embed.js";
|
||||
@@ -100,7 +101,11 @@ export class MessagesService {
|
||||
);
|
||||
return { results: [], nextCursor: null };
|
||||
}
|
||||
const hits = await searchArchive(vector, input.limit, 0.6);
|
||||
const hits = await searchArchive(
|
||||
vector,
|
||||
input.limit,
|
||||
config.AI_LLM_EMBEDDING_ARCHIVE_MIN_SIMILARITY,
|
||||
);
|
||||
const results = hits.map((h) => mapSearchHit(h));
|
||||
return { results, nextCursor: null };
|
||||
}
|
||||
|
||||
@@ -135,6 +135,14 @@ export const configSchema = z
|
||||
AI_LLM_MODEL: z.string().default("text"),
|
||||
AI_LLM_VISION_MODEL: z.string().optional(),
|
||||
AI_LLM_EMBEDDING_MODEL: z.string().optional(),
|
||||
// Minimum cosine similarity for the public archive semantic search. Lower
|
||||
// = more (noisier) results; raise it to tighten precision. Tuned for a 1B
|
||||
// embedding model — re-tune if the model's dimensionality changes.
|
||||
AI_LLM_EMBEDDING_ARCHIVE_MIN_SIMILARITY: z.coerce
|
||||
.number()
|
||||
.min(0)
|
||||
.max(1)
|
||||
.default(0.6),
|
||||
AI_LLM_MAX_CONCURRENT: z.coerce.number().int().positive().default(5),
|
||||
AI_LLM_IMAGE_MAX_DIMENSION: z.coerce
|
||||
.number()
|
||||
|
||||
Reference in New Issue
Block a user