feat: add Playwright browser scraping fallback for Instagram/Facebook/Twitter
Deploy Scraper / build-and-deploy (push) Canceled after 0s

- Add scrape_media.py using Playwright+Chromium to bypass Cloudflare/anti-bot
- Add run_playwright_scraper() Rust helper (spawns Python, probes venv)
- Add playwright_to_download_result() JSON->DownloadResult converter
- Instagram/Facebook: try downr.org, fall back to Playwright (returns real cdninstagram/fbcdn URLs)
- Twitter: scope scraper::Html/Selector parsing in a block so the future stays Send, then Playwright fallback
- Revert --impersonate chrome (unsupported on Linux) to --user-agent
- Install Chromium browsers to /usr/local/share/ms-playwright for all users
This commit is contained in:
asepharyana
2026-09-01 15:35:20 +07:00
parent dd490535fc
commit 4c5dfdb156
2 changed files with 769 additions and 168 deletions
+161
View File
@@ -0,0 +1,161 @@
#!/usr/bin/env python3
"""
Universal scraper for social media video URLs using Playwright + Chromium.
Used as fallback when yt-dlp is blocked by anti-bot protection.
Usage: scrape_media.py <url> <platform>
Output: JSON with title, media items, etc.
Platforms supported:
- instagram: Scrape reel/post video URLs from cdninstagram.com
- facebook: Scrape video URLs from fbcdn.net
- tiktok: Try scraping, but may be blocked by anti-bot
- twitter: Scrape video URLs from video.twimg.com
- pinterest: Scrape video URLs from v.pinimg.com
"""
import sys
import asyncio
import re
import json
import os
# Point Playwright at system-wide browser cache so it works regardless of
# the service user's HOME permissions.
os.environ.setdefault("PLAYWRIGHT_BROWSERS_PATH", "/usr/local/share/ms-playwright")
from playwright.async_api import async_playwright
async def scrape(url, platform):
async with async_playwright() as p:
browser = await p.chromium.launch(
headless=True,
args=["--no-sandbox", "--disable-bypass"]
)
context = await browser.new_context(
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
viewport={"width": 1920, "height": 1080},
java_script_enabled=True,
)
page = await context.new_page()
# Track network responses
media_responses = []
def handle_response(response):
resp_url = response.url
# Look for video/audio URLs
if any(ext in resp_url for ext in ['.mp4', '.m3u8', '.mp3']):
media_responses.append({
"url": resp_url,
"status": response.status,
"content_type": response.headers.get('content-type', ''),
})
page.on("response", handle_response)
try:
await page.goto(url, timeout=60000)
await page.wait_for_timeout(15000)
except Exception as e:
await browser.close()
print(json.dumps({"success": False, "error": f"Navigation error: {e}"}))
return
# Get page title
title = await page.title()
# Get video element info
video_info = await page.evaluate("""() => {
const video = document.querySelector('video');
if (!video) return null;
return {
src: video.src,
duration: video.duration,
poster: video.poster,
};
}""")
# Filter for real video URLs (not static assets, not ttwstatic)
seen = set()
media = []
for resp in media_responses:
url_val = resp["url"]
if url_val in seen:
continue
if len(url_val) < 50:
continue
# Skip static assets
if 'ttwstatic' in url_val or 'rsrc.php' in url_val:
continue
if 'cdninstagram' not in url_val and 'fbcdn' not in url_val and 'video.twimg' not in url_val and 'pinimg' not in url_val:
# For non-Instagram/facebook, still accept
if 'static' in url_val:
continue
seen.add(url_val)
# Determine content type
ext = "mp4"
if '.m3u8' in url_val:
ext = "m3u8"
elif '.mp3' in url_val:
ext = "mp3"
elif '.mp4' in url_val:
ext = "mp4"
media.append({
"url": url_val,
"ext": ext,
"status": resp["status"],
"content_type": resp["content_type"],
})
# Also check for video element src (blob URLs won't work, but worth checking)
if video_info and video_info.get('src'):
video_src = video_info['src']
if not video_src.startswith('blob:'):
media.append({
"url": video_src,
"ext": "mp4",
"status": 200,
"content_type": "video/mp4",
})
# If no media found, check page content for URLs
if not media:
content = await page.content()
# Look for video URLs in page source
content_urls = re.findall(r'https://[^\s"\'<>]+', content)
for url_val in content_urls:
if any(ext in url_val for ext in ['.mp4', '.m3u8']) and len(url_val) > 50:
if 'ttwstatic' not in url_val and 'rsrc.php' not in url_val:
if url_val not in seen:
seen.add(url_val)
ext = "m3u8" if '.m3u8' in url_val else "mp4"
media.append({
"url": url_val,
"ext": ext,
"status": 200,
"content_type": f"video/{ext}" if ext != 'm3u8' else 'application/x-mpegURL',
})
result = {
"success": len(media) > 0,
"title": title if title else "Unknown",
"platform": platform,
"media": media,
"media_count": len(media),
"provider": f"playwright-{platform}",
}
await browser.close()
print(json.dumps(result, indent=2))
if __name__ == "__main__":
if len(sys.argv) < 3:
print(json.dumps({"success": False, "error": "Usage: scrape_media.py <url> <platform>"}))
sys.exit(1)
url = sys.argv[1]
platform = sys.argv[2]
asyncio.run(scrape(url, platform))
+541 -101
View File
@@ -57,7 +57,11 @@ fn extract_tiktok_id(url: &str) -> Option<String> {
// Handle short URLs like vm.tiktok.com/ZM8s5qJ6t — resolve redirect first
if url.contains("vm.tiktok.com") || url.contains("vt.tiktok.com") {
let re = regex::Regex::new(r"/([A-Za-z0-9_-]+)$").ok()?;
let short_code = re.captures(url).and_then(|c| c.get(1))?.as_str().to_string();
let short_code = re
.captures(url)
.and_then(|c| c.get(1))?
.as_str()
.to_string();
return Some(short_code);
}
// TikTok URLs contain an 18-20 digit video ID in the path
@@ -99,9 +103,12 @@ fn find_ytdlp() -> Option<String> {
/// Run yt-dlp --dump-json and return the parsed JSON value.
/// Uses spawn + manual stdout reading to avoid pipe buffer truncation
/// on large outputs (>64KB on Linux default pipe buffer).
async fn run_ytdlp_json(url: &str, extra_args: &[&str]) -> Result<serde_json::Value, ScrapingError> {
let ytdlp = find_ytdlp()
.ok_or_else(|| ScrapingError::Http("yt-dlp binary not found".to_string()))?;
async fn run_ytdlp_json(
url: &str,
extra_args: &[&str],
) -> Result<serde_json::Value, ScrapingError> {
let ytdlp =
find_ytdlp().ok_or_else(|| ScrapingError::Http("yt-dlp binary not found".to_string()))?;
let extra_args_owned: Vec<String> = extra_args.iter().map(|s| s.to_string()).collect();
@@ -113,6 +120,8 @@ async fn run_ytdlp_json(url: &str, extra_args: &[&str]) -> Result<serde_json::Va
"--dump-json".to_string(),
"--no-warnings".to_string(),
"--no-check-certificates".to_string(),
"--user-agent".to_string(),
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36".to_string(),
];
cmd_args.extend(extra_args_owned.iter().cloned());
cmd_args.push(url_owned);
@@ -160,6 +169,117 @@ async fn run_ytdlp_json(url: &str, extra_args: &[&str]) -> Result<serde_json::Va
.map_err(|e| ScrapingError::Http(format!("yt-dlp JSON parse failed: {}", e)))
}
/// Run Playwright-based browser scraper as fallback when yt-dlp is blocked.
/// Uses headless Chromium to scrape video URLs from anti-bot-protected sites.
async fn run_playwright_scraper(
url: &str,
platform: &str,
) -> Result<serde_json::Value, ScrapingError> {
let script = env!("CARGO_MANIFEST_DIR");
let scraper_script = format!("{}/scrape_media.py", script);
// Find a Python interpreter that has playwright installed.
// The system `python3` may resolve to a different interpreter for the
// service user, so probe known venv interpreters first.
let python_candidates = [
"/home/code/hermes-agent/.venv/bin/python3",
"/usr/bin/python3",
"python3",
];
let python_bin = python_candidates
.iter()
.find(|p| {
std::process::Command::new(p)
.args(["-c", "import playwright"])
.stdout(std::process::Stdio::null())
.stderr(std::process::Stdio::null())
.status()
.map(|s| s.success())
.unwrap_or(false)
})
.map(|s| s.to_string())
.unwrap_or_else(|| "python3".to_string());
let (stdout_str, stderr_str, exit_code) = tokio::task::spawn_blocking({
let url_owned = url.to_string();
let platform_owned = platform.to_string();
let script_owned = scraper_script.clone();
let python_owned = python_bin.clone();
move || -> Result<(String, String, i32), ScrapingError> {
let output = std::process::Command::new(&python_owned)
.arg(&script_owned)
.arg(&url_owned)
.arg(&platform_owned)
.stdout(std::process::Stdio::piped())
.stderr(std::process::Stdio::piped())
.output()
.map_err(|e| ScrapingError::Http(format!("playwright spawn failed: {}", e)))?;
let stdout = String::from_utf8_lossy(&output.stdout).to_string();
let stderr = String::from_utf8_lossy(&output.stderr).to_string();
Ok((stdout, stderr, output.status.code().unwrap_or(-1)))
}
})
.await
.map_err(|e| ScrapingError::Http(format!("playwright task error: {}", e)))??;
if exit_code != 0 {
return Err(ScrapingError::Http(format!(
"playwright scraper failed: {}",
stderr_str.trim().lines().last().unwrap_or("unknown error")
)));
}
serde_json::from_str(&stdout_str)
.map_err(|e| ScrapingError::Http(format!("playwright JSON parse failed: {}", e)))
}
/// Convert a Playwright scraper JSON result (from scrape_media.py) into DownloadResult.
fn playwright_to_download_result(data: &serde_json::Value) -> DownloadResult {
let title = data
.get("title")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let mut result = DownloadResult::success(title);
result.provider = data
.get("provider")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
if let Some(medias) = data.get("media").and_then(|v| v.as_array()) {
for m in medias {
let item = MediaItem {
url: m
.get("url")
.and_then(|v| v.as_str())
.unwrap_or("")
.to_string(),
quality: None,
file_type: m.get("ext").and_then(|v| v.as_str()).and_then(|e| match e {
"mp4" | "m3u8" => Some(MediaType::Video),
"mp3" | "m4a" => Some(MediaType::Audio),
_ => Some(MediaType::Video),
}),
extension: m.get("ext").and_then(|v| v.as_str()).map(|s| s.to_string()),
thumbnail: None,
file_size: None,
size_bytes: None,
frame_width: None,
frame_height: None,
note: m
.get("content_type")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
};
result.media.push(item);
}
}
if result.media.is_empty() {
result.message = Some("No download URLs found".to_string());
}
result
}
pub struct DownloaderRepository;
impl DownloaderRepository {
@@ -553,10 +673,10 @@ async fn detect_media_type(url: &str) -> MediaType {
}
/// SnapSave parser — extracts Instagram/Facebook media via snapsave.app.
/// Uses downr.org as fallback for robustness.
/// Uses downr.org as fallback for robustness, then Playwright browser scraping.
pub(crate) async fn fetch_snapsave(url: &str) -> Result<DownloadResult, ScrapingError> {
// Validate Instagram/Facebook URL
let valid_fb = regex::Regex::new(r"https?://(web\.|www\.|m\.)?(facebook|fb)\.(com|watch)\S+")
let valid_fb = regex::Regex::new(r"https?://(web\.|www\.|m\.)(facebook|fb)\.(com|watch)\S+")
.unwrap()
.is_match(url);
let valid_ig =
@@ -570,8 +690,68 @@ pub(crate) async fn fetch_snapsave(url: &str) -> Result<DownloadResult, Scraping
));
}
// Delegate to universal downr.org scraper for robustness
fetch_all_in_one(url).await
// Try downr.org first
match fetch_all_in_one(url).await {
Ok(result) if !result.media.is_empty() => return Ok(result),
Ok(_) => {}
Err(e) => {
eprintln!(
"downr.org failed for {}: {}, trying Playwright fallback",
url, e
);
}
}
// Fallback: use Playwright browser scraping to extract video URLs
let platform = if valid_ig { "instagram" } else { "facebook" };
let data = run_playwright_scraper(url, platform).await?;
// Build result from Playwright scraper output
let title = data
.get("title")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let mut result = DownloadResult::success(title);
result.provider = data
.get("provider")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let media_arr = data.get("media").and_then(|v| v.as_array());
if let Some(medias) = media_arr {
for m in medias {
let item = MediaItem {
url: m
.get("url")
.and_then(|v| v.as_str())
.unwrap_or("")
.to_string(),
quality: None,
file_type: m.get("ext").and_then(|v| v.as_str()).and_then(|e| match e {
"mp4" | "m3u8" => Some(MediaType::Video),
"mp3" | "m4a" => Some(MediaType::Audio),
_ => Some(MediaType::Video),
}),
extension: m.get("ext").and_then(|v| v.as_str()).map(|s| s.to_string()),
thumbnail: None,
file_size: None,
size_bytes: None,
frame_width: None,
frame_height: None,
note: None,
};
result.media.push(item);
}
}
if result.media.is_empty() {
return Ok(DownloadResult::error(format!(
"Failed to extract media from {} — server IP may be blocked by anti-bot protection",
platform
)));
}
Ok(result)
}
pub async fn fetch_tiktok(url: &str) -> Result<DownloadResult, ScrapingError> {
if extract_tiktok_id(url).is_none() {
@@ -602,7 +782,8 @@ pub async fn fetch_tiktok(url: &str) -> Result<DownloadResult, ScrapingError> {
&& tikwm_msg != Some("Url parsing is failed! Please check url.")
{
let mut result = DownloadResult::success(
resp_json.get("data")
resp_json
.get("data")
.and_then(|d| d.get("title"))
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
@@ -619,17 +800,29 @@ pub async fn fetch_tiktok(url: &str) -> Result<DownloadResult, ScrapingError> {
.and_then(|v| v.as_str())
.map(|s| s.to_string());
if let Some(plays) = resp_json.get("data").and_then(|d| d.get("plays").and_then(|v| v.as_array())) {
if let Some(plays) = resp_json
.get("data")
.and_then(|d| d.get("plays").and_then(|v| v.as_array()))
{
for play in plays {
if let Some(play_url) = play.get("url").and_then(|v| v.as_str()) {
result.media.push(MediaItem {
url: play_url.to_string(),
quality: play.get("quality").and_then(|v| v.as_str()).map(|s| s.to_string()),
quality: play
.get("quality")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
file_type: Some(MediaType::Video),
extension: Some("mp4".to_string()),
thumbnail: None,
file_size: play.get("size").and_then(|v| v.as_str()).map(|s| s.to_string()),
size_bytes: play.get("size").and_then(|v| v.as_str()).and_then(|s| s.parse().ok()),
file_size: play
.get("size")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
size_bytes: play
.get("size")
.and_then(|v| v.as_str())
.and_then(|s| s.parse().ok()),
frame_width: None,
frame_height: None,
note: None,
@@ -646,10 +839,20 @@ pub async fn fetch_tiktok(url: &str) -> Result<DownloadResult, ScrapingError> {
}
// Fallback: use yt-dlp
let data = run_ytdlp_json(url, &[]).await?;
let title = data.get("title").and_then(|v| v.as_str()).map(|s| s.to_string());
let author = data.get("uploader").and_then(|v| v.as_str()).map(|s| s.to_string());
let thumbnail = data.get("thumbnail").and_then(|v| v.as_str()).map(|s| s.to_string());
match run_ytdlp_json(url, &[]).await {
Ok(data) => {
let title = data
.get("title")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let author = data
.get("uploader")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let thumbnail = data
.get("thumbnail")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let mut result = DownloadResult::success(title);
result.author = author;
@@ -661,25 +864,111 @@ pub async fn fetch_tiktok(url: &str) -> Result<DownloadResult, ScrapingError> {
if let Some(fmt_url) = fmt.get("url").and_then(|v| v.as_str()) {
result.media.push(MediaItem {
url: fmt_url.to_string(),
quality: fmt.get("format_note").and_then(|v| v.as_str()).map(|s| s.to_string())
.or_else(|| fmt.get("height").and_then(|v| v.as_str()).map(|s| s.to_string())),
file_type: fmt.get("vcodec").and_then(|v| v.as_str()).filter(|s| !s.is_empty() && *s != "none")
quality: fmt
.get("format_note")
.and_then(|v| v.as_str())
.map(|s| s.to_string())
.or_else(|| {
fmt.get("height")
.and_then(|v| v.as_str())
.map(|s| s.to_string())
}),
file_type: fmt
.get("vcodec")
.and_then(|v| v.as_str())
.filter(|s| !s.is_empty() && *s != "none")
.map(|_| MediaType::Video),
extension: fmt.get("ext").and_then(|v| v.as_str()).map(|s| s.to_string()),
extension: fmt
.get("ext")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
thumbnail: None,
file_size: fmt.get("filesize").and_then(|v| v.as_u64()).map(format_filesize),
file_size: fmt
.get("filesize")
.and_then(|v| v.as_u64())
.map(format_filesize),
size_bytes: fmt.get("filesize").and_then(|v| v.as_u64()),
frame_width: fmt.get("width").and_then(|v| v.as_str()).map(|s| s.to_string()),
frame_height: fmt.get("height").and_then(|v| v.as_str()).map(|s| s.to_string()),
frame_width: fmt
.get("width")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
frame_height: fmt
.get("height")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
note: None,
});
}
}
}
Ok(result)
if result.media.is_empty() {
result.message = Some("No download URLs found".to_string());
}
return Ok(result);
}
Err(e) => {
eprintln!("yt-dlp failed for TikTok, trying Playwright: {}", e);
// Fallback: use Playwright browser scraping
match run_playwright_scraper(url, "tiktok").await {
Ok(data) => {
let title = data
.get("title")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let mut result = DownloadResult::success(title);
result.provider = data
.get("provider")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
if let Some(medias) = data.get("media").and_then(|v| v.as_array()) {
for m in medias {
let item = MediaItem {
url: m
.get("url")
.and_then(|v| v.as_str())
.unwrap_or("")
.to_string(),
quality: None,
file_type: m.get("ext").and_then(|v| v.as_str()).and_then(|e| {
match e {
"mp4" | "m3u8" => Some(MediaType::Video),
"mp3" | "m4a" => Some(MediaType::Audio),
_ => Some(MediaType::Video),
}
}),
extension: m
.get("ext")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
thumbnail: None,
file_size: None,
size_bytes: None,
frame_width: None,
frame_height: None,
note: None,
};
result.media.push(item);
}
}
if result.media.is_empty() {
result.message = Some("No download URLs found".to_string());
}
return Ok(result);
}
Err(_) => {
// All methods failed
return Err(ScrapingError::Http(
"All TikTok download methods failed (tikwm API blocked, yt-dlp blocked, Playwright blocked)".to_string()
));
}
}
}
}
}
/// TikTok v2 — uses douyin.wtf API
pub async fn fetch_tiktok_v2(url: &str) -> Result<DownloadResult, ScrapingError> {
@@ -762,10 +1051,22 @@ pub async fn fetch_youtube_mp3(url: &str) -> Result<DownloadResult, ScrapingErro
let data = run_ytdlp_json(url, &["-f", "bestaudio/best"]).await?;
let title = data.get("title").and_then(|v| v.as_str()).map(|s| s.to_string());
let author = data.get("uploader").and_then(|v| v.as_str()).map(|s| s.to_string());
let thumbnail = data.get("thumbnail").and_then(|v| v.as_str()).map(|s| s.to_string());
let duration = data.get("duration").and_then(|v| v.as_u64()).map(|d| format!("{}s", d));
let title = data
.get("title")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let author = data
.get("uploader")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let thumbnail = data
.get("thumbnail")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let duration = data
.get("duration")
.and_then(|v| v.as_u64())
.map(|d| format!("{}s", d));
let mut result = DownloadResult::success(title);
result.author = author;
@@ -773,7 +1074,9 @@ pub async fn fetch_youtube_mp3(url: &str) -> Result<DownloadResult, ScrapingErro
result.duration = duration;
result.provider = Some("yt-dlp".to_string());
let download_url = data.get("url").and_then(|v| v.as_str())
let download_url = data
.get("url")
.and_then(|v| v.as_str())
.or_else(|| {
data.get("formats")
.and_then(|v| v.as_array())
@@ -784,11 +1087,20 @@ pub async fn fetch_youtube_mp3(url: &str) -> Result<DownloadResult, ScrapingErro
result.media.push(MediaItem {
url: download_url.to_string(),
quality: Some(format!("{}kbps", data.get("abr").and_then(|v| v.as_u64()).unwrap_or(128))),
quality: Some(format!(
"{}kbps",
data.get("abr").and_then(|v| v.as_u64()).unwrap_or(128)
)),
file_type: Some(MediaType::Audio),
extension: Some("mp3".to_string()),
thumbnail: data.get("thumbnail").and_then(|v| v.as_str()).map(|s| s.to_string()),
file_size: data.get("filesize").and_then(|v| v.as_u64()).map(format_filesize),
thumbnail: data
.get("thumbnail")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
file_size: data
.get("filesize")
.and_then(|v| v.as_u64())
.map(format_filesize),
size_bytes: data.get("filesize").and_then(|v| v.as_u64()),
frame_width: None,
frame_height: None,
@@ -803,13 +1115,28 @@ pub async fn fetch_youtube_mp4(url: &str, quality: &str) -> Result<DownloadResul
extract_youtube_id(url)
.ok_or_else(|| ScrapingError::Http("Invalid YouTube URL".to_string()))?;
let fmt = format!("bestvideo[height<={}]/best+bestaudio/best", quality.trim_end_matches('p'));
let fmt = format!(
"bestvideo[height<={}]/best+bestaudio/best",
quality.trim_end_matches('p')
);
let data = run_ytdlp_json(url, &["-f", &fmt]).await?;
let title = data.get("title").and_then(|v| v.as_str()).map(|s| s.to_string());
let author = data.get("uploader").and_then(|v| v.as_str()).map(|s| s.to_string());
let thumbnail = data.get("thumbnail").and_then(|v| v.as_str()).map(|s| s.to_string());
let duration = data.get("duration").and_then(|v| v.as_u64()).map(|d| format!("{}s", d));
let title = data
.get("title")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let author = data
.get("uploader")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let thumbnail = data
.get("thumbnail")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let duration = data
.get("duration")
.and_then(|v| v.as_u64())
.map(|d| format!("{}s", d));
let mut result = DownloadResult::success(title);
result.author = author;
@@ -826,7 +1153,9 @@ pub async fn fetch_youtube_mp4(url: &str, quality: &str) -> Result<DownloadResul
f.get("url").and_then(|v| v.as_str()),
f.get("ext").and_then(|v| v.as_str()),
) {
if ext_val == "mhtml" { continue; }
if ext_val == "mhtml" {
continue;
}
let url_string = furl.to_string();
if seen.insert(url_string.clone()) {
let fmt_type = if ext_val == "mp4" || ext_val == "webm" || ext_val == "mkv" {
@@ -842,11 +1171,23 @@ pub async fn fetch_youtube_mp4(url: &str, quality: &str) -> Result<DownloadResul
quality: Some(q),
file_type: Some(fmt_type),
extension: Some(ext_val.to_string()),
thumbnail: data.get("thumbnail").and_then(|v| v.as_str()).map(|s| s.to_string()),
file_size: f.get("filesize").and_then(|v| v.as_u64()).map(format_filesize),
thumbnail: data
.get("thumbnail")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
file_size: f
.get("filesize")
.and_then(|v| v.as_u64())
.map(format_filesize),
size_bytes: f.get("filesize").and_then(|v| v.as_u64()),
frame_width: f.get("width").and_then(|v| v.as_u64()).map(|w| w.to_string()),
frame_height: f.get("height").and_then(|v| v.as_u64()).map(|h| h.to_string()),
frame_width: f
.get("width")
.and_then(|v| v.as_u64())
.map(|w| w.to_string()),
frame_height: f
.get("height")
.and_then(|v| v.as_u64())
.map(|h| h.to_string()),
note: None,
});
}
@@ -860,9 +1201,20 @@ pub async fn fetch_youtube_mp4(url: &str, quality: &str) -> Result<DownloadResul
url: url.to_string(),
quality: Some(format!("{}p", quality.trim_end_matches('p'))),
file_type: Some(MediaType::Video),
extension: Some(data.get("ext").and_then(|v| v.as_str()).unwrap_or("mp4").to_string()),
thumbnail: data.get("thumbnail").and_then(|v| v.as_str()).map(|s| s.to_string()),
file_size: data.get("filesize").and_then(|v| v.as_u64()).map(format_filesize),
extension: Some(
data.get("ext")
.and_then(|v| v.as_str())
.unwrap_or("mp4")
.to_string(),
),
thumbnail: data
.get("thumbnail")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
file_size: data
.get("filesize")
.and_then(|v| v.as_u64())
.map(format_filesize),
size_bytes: data.get("filesize").and_then(|v| v.as_u64()),
frame_width: None,
frame_height: None,
@@ -892,10 +1244,22 @@ pub async fn fetch_spotify(url: &str) -> Result<DownloadResult, ScrapingError> {
// Use yt-dlp to extract track info and download URL
let data = run_ytdlp_json(url, &["--extract-audio", "--audio-format", "mp3"]).await?;
let title = data.get("title").and_then(|v| v.as_str()).map(|s| s.to_string());
let author = data.get("artist").and_then(|v| v.as_str()).map(|s| s.to_string());
let thumbnail = data.get("thumbnail").and_then(|v| v.as_str()).map(|s| s.to_string());
let duration = data.get("duration").and_then(|v| v.as_u64()).map(|d| format!("{}s", d));
let title = data
.get("title")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let author = data
.get("artist")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let thumbnail = data
.get("thumbnail")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let duration = data
.get("duration")
.and_then(|v| v.as_u64())
.map(|d| format!("{}s", d));
let mut result = DownloadResult::success(title);
result.author = author;
@@ -903,8 +1267,7 @@ pub async fn fetch_spotify(url: &str) -> Result<DownloadResult, ScrapingError> {
result.duration = duration;
result.provider = Some(format!("spotify-yt-dlp ({})", resource_type));
let download_url = data.get("url").and_then(|v| v.as_str())
.or_else(|| {
let download_url = data.get("url").and_then(|v| v.as_str()).or_else(|| {
data.get("formats")
.and_then(|v| v.as_array())
.and_then(|f| f.first())
@@ -917,8 +1280,14 @@ pub async fn fetch_spotify(url: &str) -> Result<DownloadResult, ScrapingError> {
quality: Some("320kbps".to_string()),
file_type: Some(MediaType::Audio),
extension: Some("mp3".to_string()),
thumbnail: data.get("thumbnail").and_then(|v| v.as_str()).map(|s| s.to_string()),
file_size: data.get("filesize").and_then(|v| v.as_u64()).map(format_filesize),
thumbnail: data
.get("thumbnail")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
file_size: data
.get("filesize")
.and_then(|v| v.as_u64())
.map(format_filesize),
size_bytes: data.get("filesize").and_then(|v| v.as_u64()),
frame_width: None,
frame_height: None,
@@ -964,10 +1333,7 @@ pub async fn fetch_twitter(url: &str) -> Result<DownloadResult, ScrapingError> {
.await
.map_err(|e| ScrapingError::Http(format!("Twitter JSON parse failed: {}", e)))?;
let html = resp
.get("data")
.and_then(|v| v.as_str())
.ok_or_else(|| {
let html = resp.get("data").and_then(|v| v.as_str()).ok_or_else(|| {
let msg = resp
.get("msg")
.and_then(|v| v.as_str())
@@ -975,15 +1341,19 @@ pub async fn fetch_twitter(url: &str) -> Result<DownloadResult, ScrapingError> {
ScrapingError::Http(msg.to_string())
})?;
// Parse savetwitter HTML inside a block so all scraper types
// (Html/Selector/ElementRef are NOT Send) go out of scope before the
// Playwright fallback .await below keeps the future Send.
let mut parsed_media: Vec<(String, Option<String>, Option<MediaType>, Option<String>)> =
Vec::new();
{
let document = scraper::Html::parse_document(html);
let mut result = DownloadResult::success(None);
result.provider = Some("savetwitter".to_string());
let tw_video_sel = scraper::Selector::parse("div.tw-video").unwrap();
let _video_list_sel = scraper::Selector::parse("div.video-data > div > ul > li").unwrap();
if document.select(&tw_video_sel).next().is_some() {
if let Ok(item_sel) = scraper::Selector::parse("div.tw-right > div > p:nth-child(1) > a") {
if let Ok(item_sel) =
scraper::Selector::parse("div.tw-right > div > p:nth-child(1) > a")
{
for item in document.select(&item_sel) {
let quality_text = item.text().collect::<String>();
let quality = if quality_text.contains("(") {
@@ -998,18 +1368,12 @@ pub async fn fetch_twitter(url: &str) -> Result<DownloadResult, ScrapingError> {
quality_text.trim().to_string()
};
let href = item.value().attr("href").unwrap_or("").to_string();
result.media.push(MediaItem {
url: href,
quality: Some(quality),
file_type: Some(MediaType::Video),
extension: Some("mp4".to_string()),
thumbnail: None,
file_size: None,
size_bytes: None,
frame_width: None,
frame_height: None,
note: None,
});
parsed_media.push((
href,
Some(quality),
Some(MediaType::Video),
Some("mp4".to_string()),
));
}
}
} else {
@@ -1022,11 +1386,26 @@ pub async fn fetch_twitter(url: &str) -> Result<DownloadResult, ScrapingError> {
.map(|s| s.to_string())
.unwrap_or_default();
if !href.is_empty() {
parsed_media.push((
href,
None,
Some(MediaType::Image),
Some("jpg".to_string()),
));
}
}
}
}
} // document + selectors dropped here
let mut result = DownloadResult::success(None);
result.provider = Some("savetwitter".to_string());
for (href, quality, file_type, extension) in parsed_media {
result.media.push(MediaItem {
url: href,
quality: None,
file_type: Some(MediaType::Image),
extension: Some("jpg".to_string()),
quality,
file_type,
extension,
thumbnail: None,
file_size: None,
size_bytes: None,
@@ -1035,11 +1414,18 @@ pub async fn fetch_twitter(url: &str) -> Result<DownloadResult, ScrapingError> {
note: None,
});
}
}
}
}
if result.media.is_empty() {
// Fallback: Playwright browser scraping for Twitter video URLs
match run_playwright_scraper(url, "twitter").await {
Ok(data) => {
let pw_result = playwright_to_download_result(&data);
if !pw_result.media.is_empty() {
return Ok(pw_result);
}
}
Err(_e) => {}
}
return Ok(DownloadResult::error("Tidak dapat menemukan video"));
}
@@ -1155,10 +1541,22 @@ pub async fn fetch_bilibili(url: &str) -> Result<DownloadResult, ScrapingError>
// yt-dlp supports both AV (/video/av123) and BV (/video/BV1xxx) IDs
let data = run_ytdlp_json(url, &["-f", "bv*+ba/b"]).await?;
let title = data.get("title").and_then(|v| v.as_str()).map(|s| s.to_string());
let author = data.get("uploader").and_then(|v| v.as_str()).map(|s| s.to_string());
let thumbnail = data.get("thumbnail").and_then(|v| v.as_str()).map(|s| s.to_string());
let duration = data.get("duration").and_then(|v| v.as_u64()).map(|d| format!("{}s", d));
let title = data
.get("title")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let author = data
.get("uploader")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let thumbnail = data
.get("thumbnail")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let duration = data
.get("duration")
.and_then(|v| v.as_u64())
.map(|d| format!("{}s", d));
let mut result = DownloadResult::success(title);
result.author = author;
@@ -1173,16 +1571,38 @@ pub async fn fetch_bilibili(url: &str) -> Result<DownloadResult, ScrapingError>
if url.is_some() {
result.media.push(MediaItem {
url: url.unwrap().to_string(),
quality: fmt.get("format_note").and_then(|v| v.as_str()).map(|s| s.to_string())
.or_else(|| fmt.get("height").and_then(|v| v.as_str()).map(|s| s.to_string())),
file_type: fmt.get("vcodec").and_then(|v| v.as_str()).filter(|s| !s.is_empty() && *s != "none")
quality: fmt
.get("format_note")
.and_then(|v| v.as_str())
.map(|s| s.to_string())
.or_else(|| {
fmt.get("height")
.and_then(|v| v.as_str())
.map(|s| s.to_string())
}),
file_type: fmt
.get("vcodec")
.and_then(|v| v.as_str())
.filter(|s| !s.is_empty() && *s != "none")
.map(|_| MediaType::Video),
extension: fmt.get("ext").and_then(|v| v.as_str()).map(|s| s.to_string()),
extension: fmt
.get("ext")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
thumbnail: None,
file_size: fmt.get("filesize").and_then(|v| v.as_u64()).map(format_filesize),
file_size: fmt
.get("filesize")
.and_then(|v| v.as_u64())
.map(format_filesize),
size_bytes: fmt.get("filesize").and_then(|v| v.as_u64()),
frame_width: fmt.get("width").and_then(|v| v.as_str()).map(|s| s.to_string()),
frame_height: fmt.get("height").and_then(|v| v.as_str()).map(|s| s.to_string()),
frame_width: fmt
.get("width")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
frame_height: fmt
.get("height")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
note: None,
});
}
@@ -1200,10 +1620,22 @@ pub async fn fetch_soundcloud(url: &str) -> Result<DownloadResult, ScrapingError
// yt-dlp --extract-audio requires a download; use --dump-json for direct URL
let data = run_ytdlp_json(url, &[]).await?;
let title = data.get("title").and_then(|v| v.as_str()).map(|s| s.to_string());
let author = data.get("uploader").and_then(|v| v.as_str()).map(|s| s.to_string());
let thumbnail = data.get("thumbnail").and_then(|v| v.as_str()).map(|s| s.to_string());
let duration = data.get("duration").and_then(|v| v.as_u64()).map(|d| format!("{}s", d));
let title = data
.get("title")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let author = data
.get("uploader")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let thumbnail = data
.get("thumbnail")
.and_then(|v| v.as_str())
.map(|s| s.to_string());
let duration = data
.get("duration")
.and_then(|v| v.as_u64())
.map(|d| format!("{}s", d));
let mut result = DownloadResult::success(title);
result.author = author;
@@ -1211,8 +1643,7 @@ pub async fn fetch_soundcloud(url: &str) -> Result<DownloadResult, ScrapingError
result.duration = duration;
result.provider = Some("yt-dlp".to_string());
let download_url = data.get("url").and_then(|v| v.as_str())
.or_else(|| {
let download_url = data.get("url").and_then(|v| v.as_str()).or_else(|| {
data.get("formats")
.and_then(|v| v.as_array())
.and_then(|f| f.first())
@@ -1222,11 +1653,20 @@ pub async fn fetch_soundcloud(url: &str) -> Result<DownloadResult, ScrapingError
if let Some(dl_url) = download_url {
result.media.push(MediaItem {
url: dl_url.to_string(),
quality: Some(format!("{}kbps", data.get("abr").and_then(|v| v.as_u64()).unwrap_or(128))),
quality: Some(format!(
"{}kbps",
data.get("abr").and_then(|v| v.as_u64()).unwrap_or(128)
)),
file_type: Some(MediaType::Audio),
extension: Some("mp3".to_string()),
thumbnail: data.get("thumbnail").and_then(|v| v.as_str()).map(|s| s.to_string()),
file_size: data.get("filesize").and_then(|v| v.as_u64()).map(format_filesize),
thumbnail: data
.get("thumbnail")
.and_then(|v| v.as_str())
.map(|s| s.to_string()),
file_size: data
.get("filesize")
.and_then(|v| v.as_u64())
.map(format_filesize),
size_bytes: data.get("filesize").and_then(|v| v.as_u64()),
frame_width: None,
frame_height: None,