diff --git a/flake.nix b/flake.nix index 79d3a79..9382603 100644 --- a/flake.nix +++ b/flake.nix @@ -15,7 +15,7 @@ }; scraper = pkgs.stdenv.mkDerivation { - name = "scraper-0.1.0"; + name = "scraper-0.1.1"; src = ./.; nativeBuildInputs = with pkgs; [ diff --git a/src/infrastructure/repository/downloader.rs b/src/infrastructure/repository/downloader.rs index 60c4e3f..f3eb4c0 100644 --- a/src/infrastructure/repository/downloader.rs +++ b/src/infrastructure/repository/downloader.rs @@ -854,12 +854,25 @@ async fn detect_media_type(url: &str) -> MediaType { /// SnapSave parser — extracts Instagram/Facebook media via snapsave.app. /// Uses downr.org as fallback for robustness, then Playwright browser scraping. pub(crate) async fn fetch_snapsave(url: &str) -> Result { - // Validate Instagram/Facebook URL - let valid_fb = regex::Regex::new(r"https?://(web\.|www\.|m\.)(facebook|fb)\.(com|watch)\S+") + // Validate Instagram/Facebook URL. + // + // Facebook link shapes (all must pass): + // https://www.facebook.com/watch/?v=123 + // https://fb.watch/abc123/ + // https://m.facebook.com/video.php?v=123 + // https://www.facebook.com/100012345678901/videos/1234567890 + // https://www.facebook.com/reel/123 + // https://www.facebook.com/story.php?story_fbid=123 + // https://web.facebook.com/... + // NOTE: the old regex required a `www.|m.|web.` subdomain AND ended the + // host capture with `(facebook|fb)\.(com|watch)` — that rejected + // `facebook.com/watch...` (no subdomain) and `fb.watch/...` entirely. + let fb_host = r"(?:[a-z0-9-]*\.)?(?:facebook|fb)\.(?:com|watch)"; + let valid_fb = regex::Regex::new(&format!(r"https?://{}(?:/|$)", fb_host)) .unwrap() .is_match(url); - let valid_ig = - regex::Regex::new(r"https?://(www\.)?instagram\.com/(p|reel|reels|tv|stories)/\S+") + let valid_ig = url.contains("instagram.com") + || regex::Regex::new(r"https?://(www\.)?instagram\.com/[^\s]+") .unwrap() .is_match(url); @@ -869,7 +882,24 @@ pub(crate) async fn fetch_snapsave(url: &str) -> Result return Ok(result), Ok(_) => {} @@ -881,7 +911,7 @@ pub(crate) async fn fetch_snapsave(url: &str) -> Result Result Some(MediaType::Video), - "mp3" | "m4a" => Some(MediaType::Audio), - _ => Some(MediaType::Video), + file_type: m.get("ext").and_then(|v| v.as_str()).map(|e| match e { + "mp4" | "m3u8" => MediaType::Video, + "mp3" | "m4a" => MediaType::Audio, + _ => MediaType::Video, }), extension: m.get("ext").and_then(|v| v.as_str()).map(|s| s.to_string()), thumbnail: None, @@ -923,6 +961,13 @@ pub(crate) async fn fetch_snapsave(url: &str) -> Result Result Option { + let title = data + .get("title") + .and_then(|v| v.as_str()) + .map(|s| s.to_string()); + let mut result = DownloadResult::success(title); + result.provider = Some("yt-dlp".to_string()); + result.thumbnail = data + .get("thumbnail") + .and_then(|v| v.as_str()) + .map(|s| s.to_string()); + result.duration = data + .get("duration") + .and_then(|v| v.as_u64()) + .map(|d| format!("{}s", d)); + + let formats = data.get("formats").and_then(|v| v.as_array()); + let requested_formats = data.get("requested_formats").and_then(|v| v.as_array()); + let mut seen = std::collections::HashSet::new(); + + // `-f bestvideo+bestaudio/best` puts the merged URLs in requested_formats. + // `best` single format lands in top-level url + formats[0]. + for list in [requested_formats, formats].into_iter().flatten() { + for f in list { + let Some(furl) = f.get("url").and_then(|v| v.as_str()) else { + continue; + }; + if !furl.starts_with("http") || !seen.insert(furl.to_string()) { + continue; + } + let proto = f.get("protocol").and_then(|v| v.as_str()).unwrap_or(""); + if proto.contains("m3u8") + || f.get("vcodec").and_then(|v| v.as_str()) == Some("none") + && f.get("acodec").and_then(|v| v.as_str()) != Some("none") + { + // Skip HLS + audio-only; Facebook's progressive mp4s are direct. + continue; + } + let ext: String = f + .get("ext") + .and_then(|v| v.as_str()) + .unwrap_or("mp4") + .to_string(); + result.media.push(MediaItem { + url: furl.to_string(), + quality: f + .get("height") + .and_then(|v| v.as_u64()) + .map(|h| format!("{}p", h)) + .or_else(|| { + f.get("format_note") + .and_then(|v| v.as_str()) + .map(|s| s.to_string()) + }), + file_type: Some(if ext == "mp4" || ext == "webm" || ext == "mkv" { + MediaType::Video + } else { + MediaType::File + }), + extension: Some(ext), + thumbnail: result.thumbnail.clone(), + file_size: f + .get("filesize") + .and_then(|v| v.as_u64()) + .map(format_filesize), + size_bytes: f.get("filesize").and_then(|v| v.as_u64()), + frame_width: f + .get("width") + .and_then(|v| v.as_u64()) + .map(|w| w.to_string()), + frame_height: f + .get("height") + .and_then(|v| v.as_u64()) + .map(|h| h.to_string()), + note: None, + }); + } + } + + // Fall back to the merged top-level URL (bestvideo+bestaudio sets it). + if result.media.is_empty() { + if let Some(dl_url) = data.get("url").and_then(|v| v.as_str()) { + if dl_url.starts_with("http") { + result.media.push(MediaItem { + url: dl_url.to_string(), + quality: None, + file_type: Some(MediaType::Video), + extension: data + .get("ext") + .and_then(|v| v.as_str()) + .map(|s| s.to_string()) + .or_else(|| Some("mp4".to_string())), + thumbnail: None, + file_size: data + .get("filesize") + .and_then(|v| v.as_u64()) + .map(format_filesize), + size_bytes: data.get("filesize").and_then(|v| v.as_u64()), + frame_width: None, + frame_height: None, + note: None, + }); + } + } + } + + if result.media.is_empty() { + None + } else { + Some(result) + } +} /// TikTok via embed-page scraping (primary method). /// Scrapes `https://www.tiktok.com/embed/v2/{video_id}` HTML and extracts the /// direct `v16m.tiktokcdn.com` MP4 URL from the `