refactor(analytics): update topic analysis to use word frequency
Replaces the previous AI-category based topic analysis with a word-based frequency analysis using a Common Table Expression (CTE). - Implements `word_list` CTE to split message content into individual words. - Adds regex filtering to exclude URLs, Discord stickers, and emojis. - Implements a stop-word filter to remove common Indonesian and English conjunctions, pronouns, and prepositions. - Filters out words shorter than 3 characters. - Updates the aggregation to group by word and limit results to the top 10.
This commit is contained in:
@@ -317,16 +317,45 @@ export class AnalyticsRepository {
|
|||||||
|
|
||||||
const { rows } = await pool.query(
|
const { rows } = await pool.query(
|
||||||
`
|
`
|
||||||
|
WITH word_list AS (
|
||||||
SELECT
|
SELECT
|
||||||
TRIM(UNNEST(STRING_TO_ARRAY(ai_categories, ','))) AS topic,
|
LOWER(TRIM(BOTH '.,!?;:\"'()[]{}' FROM word)) AS word,
|
||||||
|
ai_moderation_score
|
||||||
|
FROM messages,
|
||||||
|
LATERAL regexp_split_to_table(content, E'\\\\s+') AS word
|
||||||
|
${filter.where}
|
||||||
|
AND content IS NOT NULL
|
||||||
|
AND content != ''
|
||||||
|
AND LENGTH(TRIM(BOTH '.,!?;:\"()[]{}' FROM word)) >= 3
|
||||||
|
-- Skip sticker & custom emoji (<:name:id> or <a:name:id>)
|
||||||
|
AND word !~ '^<a?:.+:\\d+>$'
|
||||||
|
-- Skip URLs
|
||||||
|
AND word !~ '^https?://'
|
||||||
|
AND word !~ '^discord\\.(gg|app|com)'
|
||||||
|
-- Skip common Discord embed artifacts
|
||||||
|
AND word !~ '^cdn\\.discord'
|
||||||
|
)
|
||||||
|
SELECT
|
||||||
|
word AS topic,
|
||||||
COUNT(*)::int AS count,
|
COUNT(*)::int AS count,
|
||||||
COALESCE(AVG(ai_moderation_score), 0)::real AS score
|
COALESCE(AVG(ai_moderation_score), 0)::real AS score
|
||||||
FROM messages
|
FROM word_list
|
||||||
${filter.where}
|
WHERE word NOT IN (
|
||||||
AND ai_categories IS NOT NULL
|
'yang', 'dan', 'di', 'ke', 'dari', 'dengan', 'untuk', 'pada', 'ini', 'itu',
|
||||||
AND ai_categories != ''
|
'ada', 'akan', 'telah', 'sudah', 'bisa', 'dapat', 'tidak', 'nggak', 'enggak',
|
||||||
GROUP BY topic
|
'gak', 'gk', 'ga', 'aku', 'saya', 'kamu', 'dia', 'kami', 'kita', 'mereka',
|
||||||
|
'iya', 'ya', 'yah', 'oh', 'ah', 'eh', 'lah', 'pun', 'juga', 'masih',
|
||||||
|
'saja', 'hanya', 'sama', 'atau', 'tapi', 'namun', 'sedang', 'sangat',
|
||||||
|
'begitu', 'karena', 'sebab', 'kalau', 'jika', 'maka', 'lalu', 'setelah',
|
||||||
|
'seperti', 'antara', 'oleh', 'sebagai', 'secara', 'melalui', 'dalam',
|
||||||
|
'the', 'and', 'for', 'are', 'but', 'not', 'you', 'all', 'can', 'has',
|
||||||
|
'was', 'were', 'been', 'like', 'just', 'that', 'this', 'with', 'your',
|
||||||
|
'from', 'they', 'have', 'what', 'when', 'where', 'which', 'their',
|
||||||
|
'about', 'would', 'could', 'should', 'very', 'also', 'than', 'then'
|
||||||
|
)
|
||||||
|
GROUP BY word
|
||||||
ORDER BY count DESC
|
ORDER BY count DESC
|
||||||
|
LIMIT 10
|
||||||
`,
|
`,
|
||||||
filter.params,
|
filter.params,
|
||||||
);
|
);
|
||||||
|
|||||||
Reference in New Issue
Block a user