diff --git a/src/application/misc.rs b/src/application/misc.rs index 0ae4ec7..9fe6341 100644 --- a/src/application/misc.rs +++ b/src/application/misc.rs @@ -13,21 +13,15 @@ pub async fn currency_converter(amount: f64, from: &str, to: &str) -> Result Result { - Repo::fetch_harga_emas() - .await - .map_err(ScrapingError::Http) + Repo::fetch_harga_emas().await.map_err(ScrapingError::Http) } /// Kurs BCA (jual/beli). pub async fn kurs_bca() -> Result { - Repo::fetch_kurs_bca() - .await - .map_err(ScrapingError::Http) + Repo::fetch_kurs_bca().await.map_err(ScrapingError::Http) } /// Server info (OS/CPU/RAM/disk). pub async fn server_info() -> Result { - Repo::fetch_server_info() - .await - .map_err(ScrapingError::Http) + Repo::fetch_server_info().await.map_err(ScrapingError::Http) } diff --git a/src/application/mod.rs b/src/application/mod.rs index 33edec2..5b59f6b 100644 --- a/src/application/mod.rs +++ b/src/application/mod.rs @@ -4,5 +4,6 @@ pub mod downloader; pub mod komik; pub mod misc; pub mod proxy; +pub mod search; pub mod stalk; pub mod weebs; diff --git a/src/application/search.rs b/src/application/search.rs new file mode 100644 index 0000000..72829ef --- /dev/null +++ b/src/application/search.rs @@ -0,0 +1,29 @@ +//! Application use-cases for search utilities. + +use crate::domain::error::ScrapingError; +use crate::infrastructure::repository::search as Repo; +use serde_json::Value; + +pub async fn bmkg() -> Result { + Repo::fetch_bmkg().await.map_err(ScrapingError::Http) +} + +pub async fn jadwal_sholat(kota: &str) -> Result { + Repo::fetch_jadwal_sholat(kota) + .await + .map_err(ScrapingError::Http) +} + +pub async fn weather(city: &str) -> Result { + Repo::fetch_weather(city).await.map_err(ScrapingError::Http) +} + +pub async fn google(query: &str) -> Result { + Repo::fetch_google(query).await.map_err(ScrapingError::Http) +} + +pub async fn yt_search(query: &str) -> Result { + Repo::fetch_yt_search(query) + .await + .map_err(ScrapingError::Http) +} diff --git a/src/infrastructure/repository/misc.rs b/src/infrastructure/repository/misc.rs index a4bfd78..6f2c0e5 100644 --- a/src/infrastructure/repository/misc.rs +++ b/src/infrastructure/repository/misc.rs @@ -68,11 +68,8 @@ pub async fn fetch_currency_converter(amount: f64, from: &str, to: &str) -> Resu .collect::(); // Look for patterns like "15,885.00 IDR" in body text - let re2 = Regex::new(&format!( - r"([0-9,.]+)\s+{}", - to.to_uppercase() - )) - .map_err(|e| format!("Regex compile: {}", e))?; + let re2 = Regex::new(&format!(r"([0-9,.]+)\s+{}", to.to_uppercase())) + .map_err(|e| format!("Regex compile: {}", e))?; if let Some(caps) = re2.captures(&text) { let rate_str = caps.get(1).map(|m| m.as_str()).unwrap_or("0"); @@ -228,7 +225,11 @@ pub async fn fetch_kurs_bca() -> Result { .map(|td| td.text().collect::().trim().to_string()) .collect(); - if cols.len() >= 3 && !cols[0].is_empty() && !cols[1].is_empty() && !cols[2].is_empty() { + if cols.len() >= 3 + && !cols[0].is_empty() + && !cols[1].is_empty() + && !cols[2].is_empty() + { kurs_data.push(json!({ "currency": cols[0], "jual": cols[1], diff --git a/src/infrastructure/repository/mod.rs b/src/infrastructure/repository/mod.rs index 9a17a9c..a079c44 100644 --- a/src/infrastructure/repository/mod.rs +++ b/src/infrastructure/repository/mod.rs @@ -5,6 +5,7 @@ pub mod misc; pub mod otakudesu; pub mod parsers; pub mod proxy; +pub mod search; pub mod stalk; pub mod weebs; diff --git a/src/infrastructure/repository/search.rs b/src/infrastructure/repository/search.rs new file mode 100644 index 0000000..19b0a2f --- /dev/null +++ b/src/infrastructure/repository/search.rs @@ -0,0 +1,289 @@ +//! Infrastructure — Search utilities. +//! +//! Ported from Shirokami-API `scraper/search/*.js`: +//! bmkg (BMKG earthquake), jadwal-sholat (myquran.com), weather (OpenWeather), +//! google (HTML scrape), yt (via yt-dlp flat-playlist). + +use crate::infrastructure::utils::http_client::http_client; +use reqwest::header::USER_AGENT; +use scraper::{Html, Selector}; +use serde_json::{json, Value}; +use std::process::Stdio; + +/// BMKG Indonesia earthquake info (auto-gempa, terkini, dirasakan). +pub async fn fetch_bmkg() -> Result { + let urls = [ + "https://data.bmkg.go.id/DataMKG/TEWS/autogempa.json", + "https://data.bmkg.go.id/DataMKG/TEWS/gempaterkini.json", + "https://data.bmkg.go.id/DataMKG/TEWS/gempadirasakan.json", + ]; + let mut auto = Value::Null; + let mut terkini: Vec = Vec::new(); + let mut dirasakan: Vec = Vec::new(); + for url in urls { + let resp = http_client() + .client() + .get(url) + .header( + USER_AGENT, + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36", + ) + .send() + .await + .map_err(|e| format!("HTTP error: {}", e))?; + if !resp.status().is_success() { + return Err(format!("BMKG returned HTTP {}", resp.status())); + } + let data: Value = resp + .json() + .await + .map_err(|e| format!("JSON parse: {}", e))?; + if url.contains("autogempa") { + auto = data + .pointer("/Infogempa/gempa") + .cloned() + .unwrap_or(Value::Null); + } else if url.contains("gempaterkini") { + terkini = data + .pointer("/Infogempa/gempa") + .and_then(|g| g.as_array()) + .cloned() + .unwrap_or_default(); + } else { + dirasakan = data + .pointer("/Infogempa/gempa") + .and_then(|g| g.as_array()) + .cloned() + .unwrap_or_default(); + } + } + Ok(json!({ + "autogempa": auto, + "gempaterkini": terkini, + "gempadirasakan": dirasakan, + })) +} + +/// Jadwal sholat from myquran.com. +pub async fn fetch_jadwal_sholat(kota: &str) -> Result { + // 1. Find city id + let search_url = format!( + "https://api.myquran.com/v2/sholat/kota/cari/{}", + urlencode(kota) + ); + let city_data = get_json(&search_url).await?; + if city_data + .get("status") + .and_then(|s| s.as_bool()) + .unwrap_or(false) + != true + { + return Ok(json!({ "error": "Kota tidak ditemukan" })); + } + let id_list = city_data + .get("data") + .and_then(|d| d.as_array()) + .cloned() + .unwrap_or_default(); + if id_list.is_empty() { + return Ok(json!({ "error": "Kota tidak ditemukan" })); + } + + // 2. Today's date (Asia/Jakarta). Use chrono with a fixed offset approximation. + let now = chrono::Local::now(); + let (year, month, day) = ( + now.format("%Y").to_string(), + now.format("%m").to_string(), + now.format("%d").to_string(), + ); + + let mut results: Vec = Vec::new(); + for city in id_list.iter().take(5) { + let cid = city.get("id").and_then(|i| i.as_str()).unwrap_or(""); + let jadwal_url = format!( + "https://api.myquran.com/v2/sholat/jadwal/{}/{}/{}/{}", + cid, year, month, day + ); + if let Ok(sched) = get_json(&jadwal_url).await { + if let Some(d) = sched.get("data") { + results.push(json!({ + "lokasi": d.get("lokasi").cloned().unwrap_or(Value::Null), + "daerah": d.get("daerah").cloned().unwrap_or(Value::Null), + "jadwal": d.get("jadwal").cloned().unwrap_or(Value::Null), + })); + } + } + } + + if results.is_empty() { + return Ok(json!({ "error": "Tidak dapat mengambil jadwal" })); + } + Ok(json!({ "total": results.len(), "schedules": results })) +} + +/// OpenWeather current weather (uses env OPENWEATHER_API_KEY, falls back to source's key). +pub async fn fetch_weather(city: &str) -> Result { + let key = std::env::var("OPENWEATHER_API_KEY") + .unwrap_or_else(|_| "060a6bcfa19809c2cd4d97a212b19273".to_string()); + let url = format!( + "https://api.openweathermap.org/data/2.5/weather?q={}&units=metric&appid={}", + urlencode(city), + key + ); + get_json(&url).await +} + +/// Google web search via HTML scraping. +pub async fn fetch_google(query: &str) -> Result { + let url = format!( + "https://www.google.com/search?q={}&safe=off&hl=en&gl=us", + urlencode(query) + ); + let resp = http_client() + .client() + .get(&url) + .header(USER_AGENT, "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36") + .header("Accept-Language", "en-US") + .send() + .await + .map_err(|e| format!("HTTP error: {}", e))?; + let body = resp + .text() + .await + .map_err(|e| format!("Failed to read body: {}", e))?; + + let document = Html::parse_document(&body); + let result_sel = + Selector::parse("div.g, div[data-hveid]").unwrap_or_else(|_| Selector::parse("a").unwrap()); + let link_sel = Selector::parse("a").unwrap(); + let h3_sel = Selector::parse("h3").unwrap(); + + let mut results: Vec = Vec::new(); + for node in document.select(&result_sel) { + let link = node + .select(&link_sel) + .find_map(|a| a.value().attr("href")) + .and_then(|h| h.strip_prefix("/url?q=")) + .map(|h| h.split('&').next().unwrap_or(h).to_string()); + let title = node + .select(&h3_sel) + .next() + .map(|h| h.text().collect::()); + if title.is_some() && link.is_some() { + results.push(json!({ + "title": title.unwrap_or_default(), + "description": "", + "link": link.unwrap_or_default(), + })); + } + if results.len() >= 10 { + break; + } + } + + Ok(json!(results)) +} + +/// YouTube search via yt-dlp flat playlist (real results, scraping). +pub async fn fetch_yt_search(query: &str) -> Result { + let q = format!("ytsearch15:{}", query); + let output = run_ytdlp(&q).await?; + + let mut videos: Vec = Vec::new(); + let arr: Vec = serde_json::from_str(&output).unwrap_or_default(); + for v in arr.iter().take(15) { + videos.push(json!({ + "title": v.get("title").cloned().unwrap_or(Value::Null), + "id": v.get("id").cloned().unwrap_or(Value::Null), + "url": v.get("webpage_url").cloned().unwrap_or(Value::Null), + "thumbnail": v.get("thumbnail").cloned().unwrap_or(Value::Null), + "duration": v.get("duration").cloned().unwrap_or(Value::Null), + "views": v.get("view_count").cloned().unwrap_or(Value::Null), + "author": { + "name": v.pointer("/channel").cloned().unwrap_or(Value::Null), + "url": v.get("channel_url").cloned().unwrap_or(Value::Null), + }, + })); + } + + Ok(json!({ "total": videos.len(), "videos": videos })) +} + +// ============================================================================ +// Helpers +// ============================================================================ + +async fn get_json(url: &str) -> Result { + let resp = http_client() + .client() + .get(url) + .header(USER_AGENT, "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36") + .send() + .await + .map_err(|e| format!("HTTP error: {}", e))?; + if !resp.status().is_success() { + return Err(format!("HTTP {} from upstream", resp.status())); + } + resp.json::() + .await + .map_err(|e| format!("JSON parse: {}", e)) +} + +/// Run yt-dlp with `--dump-json --flat-playlist` for the given search/query arg. +async fn run_ytdlp(query_arg: &str) -> Result { + let bins = [ + "/usr/local/bin/yt-dlp-native", + "/home/code/hermes-agent/.venv/bin/yt-dlp", + "yt-dlp", + ]; + let bin = bins + .iter() + .find(|b| std::path::Path::new(b).exists() || **b == "yt-dlp") + .copied() + .unwrap_or("yt-dlp"); + + let child = tokio::process::Command::new(bin) + .args(["--dump-json", "--flat-playlist", "--no-warnings", query_arg]) + .stdout(Stdio::piped()) + .stderr(Stdio::null()) + .spawn() + .map_err(|e| format!("Failed to spawn yt-dlp: {}", e))?; + + let output = child + .wait_with_output() + .await + .map_err(|e| format!("yt-dlp failed: {}", e))?; + + let stdout = String::from_utf8(output.stdout).map_err(|e| format!("Invalid UTF-8: {}", e))?; + // `--flat-playlist` outputs multiple JSON lines (one per video). + let mut joined = String::new(); + joined.push('['); + let mut first = true; + for line in stdout.lines() { + let line = line.trim(); + if line.is_empty() { + continue; + } + if !first { + joined.push(','); + } + joined.push_str(line); + first = false; + } + joined.push(']'); + Ok(joined) +} + +fn urlencode(s: &str) -> String { + let mut out = String::new(); + for b in s.bytes() { + match b { + b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => { + out.push(b as char) + } + b' ' => out.push_str("%20"), + _ => out.push_str(&format!("%{:02X}", b)), + } + } + out +} diff --git a/src/infrastructure/repository/stalk.rs b/src/infrastructure/repository/stalk.rs index 97ea4c5..6079261 100644 --- a/src/infrastructure/repository/stalk.rs +++ b/src/infrastructure/repository/stalk.rs @@ -81,8 +81,13 @@ pub async fn fetch_youtube_stalk(username: &str) -> Result { .map_err(|e| format!("Failed to read body: {}", e))?; // Extract var ytInitialData = {...} via brace matching (JSON may contain ";"). - let start = body.find("var ytInitialData = ").ok_or_else(|| "ytInitialData not found".to_string())?; - let brace = body[start..].find('{').ok_or_else(|| "ytInitialData parse".to_string())? + start; + let start = body + .find("var ytInitialData = ") + .ok_or_else(|| "ytInitialData not found".to_string())?; + let brace = body[start..] + .find('{') + .ok_or_else(|| "ytInitialData parse".to_string())? + + start; let json_str = extract_balanced_json(&body, brace)?; let parsed: Value = serde_json::from_str(&json_str).map_err(|e| format!("JSON parse: {}", e))?; @@ -99,7 +104,10 @@ pub async fn fetch_youtube_stalk(username: &str) -> Result { "isFamilySafe": Value::Null, }); - if let Some(meta) = parsed.get("metadata").and_then(|m| m.get("channelMetadataRenderer")) { + if let Some(meta) = parsed + .get("metadata") + .and_then(|m| m.get("channelMetadataRenderer")) + { let mut set = |dst: &str, src: &str| { if let Some(v) = meta.get(src) { channel[dst] = v.clone(); @@ -112,7 +120,10 @@ pub async fn fetch_youtube_stalk(username: &str) -> Result { } // Subscriber/video count from header.pageHeaderRenderer - if let Some(header) = parsed.get("header").and_then(|h| h.get("pageHeaderRenderer")) { + if let Some(header) = parsed + .get("header") + .and_then(|h| h.get("pageHeaderRenderer")) + { if let Some(rows) = header .pointer("/content/pageHeaderViewModel/metadata/contentMetadataViewModel/metadataRows") { @@ -293,8 +304,10 @@ fn video_from_video_renderer(vd: &Value) -> Value { .pointer("/thumbnailOverlays") .and_then(|o| o.as_array()) .and_then(|arr| { - arr.iter() - .find_map(|ov| ov.pointer("/thumbnailOverlayTimeStatusRenderer/text/simpleText").cloned()) + arr.iter().find_map(|ov| { + ov.pointer("/thumbnailOverlayTimeStatusRenderer/text/simpleText") + .cloned() + }) }) { v["duration"] = dur; @@ -393,8 +406,10 @@ fn video_from_grid_video(gv: &Value) -> Value { .pointer("/thumbnailOverlays") .and_then(|o| o.as_array()) .and_then(|arr| { - arr.iter() - .find_map(|ov| ov.pointer("/thumbnailOverlayTimeStatusRenderer/text/simpleText").cloned()) + arr.iter().find_map(|ov| { + ov.pointer("/thumbnailOverlayTimeStatusRenderer/text/simpleText") + .cloned() + }) }) { v["duration"] = dur; diff --git a/src/infrastructure/repository/weebs.rs b/src/infrastructure/repository/weebs.rs index 208988c..92be188 100644 --- a/src/infrastructure/repository/weebs.rs +++ b/src/infrastructure/repository/weebs.rs @@ -169,7 +169,11 @@ fn join_names(v: Option<&Value>) -> String { v.and_then(|g| g.as_array()) .map(|arr| { arr.iter() - .filter_map(|g| g.get("name").and_then(|n| n.as_str()).map(|s| s.to_string())) + .filter_map(|g| { + g.get("name") + .and_then(|n| n.as_str()) + .map(|s| s.to_string()) + }) .collect::>() .join(", ") }) diff --git a/src/presentation/handler/mod.rs b/src/presentation/handler/mod.rs index f169b66..882822b 100644 --- a/src/presentation/handler/mod.rs +++ b/src/presentation/handler/mod.rs @@ -5,5 +5,6 @@ pub mod health; pub mod komik; pub mod misc; pub mod proxy; +pub mod search; pub mod stalk; pub mod weebs; diff --git a/src/presentation/handler/search.rs b/src/presentation/handler/search.rs new file mode 100644 index 0000000..61ff7b3 --- /dev/null +++ b/src/presentation/handler/search.rs @@ -0,0 +1,62 @@ +//! Axum handlers for search utilities. +//! +//! Ported from Shirokami-API `scraper/search/*.js`. + +use axum::extract::Query; +use axum::http::StatusCode; +use axum::Json; +use serde::Deserialize; +use serde_json::Value; + +use crate::application::search as use_cases; +use crate::presentation::error::AppError; + +#[derive(Deserialize)] +pub struct SearchParams { + pub query: Option, + pub q: Option, + pub kota: Option, + pub city: Option, +} + +fn require<'a>(v: Option<&'a String>, name: &str) -> Result { + v.cloned() + .ok_or_else(|| AppError::BadRequest(format!("Missing '{}' parameter", name))) +} + +/// GET /search/bmkg +pub async fn bmkg_handler() -> Result<(StatusCode, Json), AppError> { + Ok((StatusCode::OK, Json(use_cases::bmkg().await?))) +} + +/// GET /search/jadwal-sholat?kota=jakarta +pub async fn jadwal_sholat_handler( + Query(p): Query, +) -> Result<(StatusCode, Json), AppError> { + let kota = require(p.kota.as_ref(), "kota")?; + Ok((StatusCode::OK, Json(use_cases::jadwal_sholat(&kota).await?))) +} + +/// GET /search/weather?city=jakarta +pub async fn weather_handler( + Query(p): Query, +) -> Result<(StatusCode, Json), AppError> { + let city = require(p.city.as_ref().or(p.q.as_ref()), "city")?; + Ok((StatusCode::OK, Json(use_cases::weather(&city).await?))) +} + +/// GET /search/google?query=rust +pub async fn google_handler( + Query(p): Query, +) -> Result<(StatusCode, Json), AppError> { + let query = require(p.query.as_ref().or(p.q.as_ref()), "query")?; + Ok((StatusCode::OK, Json(use_cases::google(&query).await?))) +} + +/// GET /search/yt?query=rickroll +pub async fn yt_handler( + Query(p): Query, +) -> Result<(StatusCode, Json), AppError> { + let query = require(p.query.as_ref().or(p.q.as_ref()), "query")?; + Ok((StatusCode::OK, Json(use_cases::yt_search(&query).await?))) +} diff --git a/src/presentation/router.rs b/src/presentation/router.rs index 3099317..0c476d9 100644 --- a/src/presentation/router.rs +++ b/src/presentation/router.rs @@ -303,6 +303,27 @@ pub fn build_router(app_state: Arc) -> anyhow::Result { "/weebs/nsfw-waifu", axum::routing::get(crate::presentation::handler::weebs::nsfw_waifu_handler), ) + // Search routes + .route( + "/search/bmkg", + axum::routing::get(crate::presentation::handler::search::bmkg_handler), + ) + .route( + "/search/jadwal-sholat", + axum::routing::get(crate::presentation::handler::search::jadwal_sholat_handler), + ) + .route( + "/search/weather", + axum::routing::get(crate::presentation::handler::search::weather_handler), + ) + .route( + "/search/google", + axum::routing::get(crate::presentation::handler::search::google_handler), + ) + .route( + "/search/yt", + axum::routing::get(crate::presentation::handler::search::yt_handler), + ) // Health .route( "/health",