feat(ai): audit + harden embedding pipeline

- Normalize text before embedding (strip mentions/URLs/emoji/markdown/control chars, lowercase, truncate) on both write and query sides so vectors aren't diluted and tokens aren't wasted
- embeddingClient: retry embeddings (maxRetries 2), validate batch dimension consistency, preserve index alignment for empty-normalized texts
- archiveEmbedder: store normalized text in archive payload, skip empty-normalized content
- backend: normalize search queries, make archive search similarity threshold configurable (AI_LLM_EMBEDDING_ARCHIVE_MIN_SIMILARITY, default 0.6)
This commit is contained in:
asepharyana
2026-09-01 18:01:47 +07:00
parent 0e31aa06b8
commit 7ef86c81ca
5 changed files with 127 additions and 8 deletions
@@ -135,6 +135,14 @@ export const configSchema = z
AI_LLM_MODEL: z.string().default("text"),
AI_LLM_VISION_MODEL: z.string().optional(),
AI_LLM_EMBEDDING_MODEL: z.string().optional(),
// Minimum cosine similarity for the public archive semantic search. Lower
// = more (noisier) results; raise it to tighten precision. Tuned for a 1B
// embedding model — re-tune if the model's dimensionality changes.
AI_LLM_EMBEDDING_ARCHIVE_MIN_SIMILARITY: z.coerce
.number()
.min(0)
.max(1)
.default(0.6),
AI_LLM_MAX_CONCURRENT: z.coerce.number().int().positive().default(5),
AI_LLM_IMAGE_MAX_DIMENSION: z.coerce
.number()