refactor(analytics): update topic analysis to use word frequency

Replaces the previous AI-category based topic analysis with a word-based
frequency analysis using a Common Table Expression (CTE).

- Implements `word_list` CTE to split message content into individual words.
- Adds regex filtering to exclude URLs, Discord stickers, and emojis.
- Implements a stop-word filter to remove common Indonesian and English
  conjunctions, pronouns, and prepositions.
- Filters out words shorter than 3 characters.
- Updates the aggregation to group by word and limit results to the top 10.
This commit is contained in:
MythEclipse
2026-06-03 19:37:16 +07:00
parent dc78faa65d
commit de0c7927c9
@@ -317,16 +317,45 @@ export class AnalyticsRepository {
const { rows } = await pool.query( const { rows } = await pool.query(
` `
WITH word_list AS (
SELECT
LOWER(TRIM(BOTH '.,!?;:\"'()[]{}' FROM word)) AS word,
ai_moderation_score
FROM messages,
LATERAL regexp_split_to_table(content, E'\\\\s+') AS word
${filter.where}
AND content IS NOT NULL
AND content != ''
AND LENGTH(TRIM(BOTH '.,!?;:\"()[]{}' FROM word)) >= 3
-- Skip sticker & custom emoji (<:name:id> or <a:name:id>)
AND word !~ '^<a?:.+:\\d+>$'
-- Skip URLs
AND word !~ '^https?://'
AND word !~ '^discord\\.(gg|app|com)'
-- Skip common Discord embed artifacts
AND word !~ '^cdn\\.discord'
)
SELECT SELECT
TRIM(UNNEST(STRING_TO_ARRAY(ai_categories, ','))) AS topic, word AS topic,
COUNT(*)::int AS count, COUNT(*)::int AS count,
COALESCE(AVG(ai_moderation_score), 0)::real AS score COALESCE(AVG(ai_moderation_score), 0)::real AS score
FROM messages FROM word_list
${filter.where} WHERE word NOT IN (
AND ai_categories IS NOT NULL 'yang', 'dan', 'di', 'ke', 'dari', 'dengan', 'untuk', 'pada', 'ini', 'itu',
AND ai_categories != '' 'ada', 'akan', 'telah', 'sudah', 'bisa', 'dapat', 'tidak', 'nggak', 'enggak',
GROUP BY topic 'gak', 'gk', 'ga', 'aku', 'saya', 'kamu', 'dia', 'kami', 'kita', 'mereka',
'iya', 'ya', 'yah', 'oh', 'ah', 'eh', 'lah', 'pun', 'juga', 'masih',
'saja', 'hanya', 'sama', 'atau', 'tapi', 'namun', 'sedang', 'sangat',
'begitu', 'karena', 'sebab', 'kalau', 'jika', 'maka', 'lalu', 'setelah',
'seperti', 'antara', 'oleh', 'sebagai', 'secara', 'melalui', 'dalam',
'the', 'and', 'for', 'are', 'but', 'not', 'you', 'all', 'can', 'has',
'was', 'were', 'been', 'like', 'just', 'that', 'this', 'with', 'your',
'from', 'they', 'have', 'what', 'when', 'where', 'which', 'their',
'about', 'would', 'could', 'should', 'very', 'also', 'than', 'then'
)
GROUP BY word
ORDER BY count DESC ORDER BY count DESC
LIMIT 10
`, `,
filter.params, filter.params,
); );