From 029837d0f19e2046775f69fa2bb71671bc732e9f Mon Sep 17 00:00:00 2001 From: asepharyana Date: Sat, 5 Sep 2026 20:11:16 +0700 Subject: [PATCH] feat(downloader): youtube progressive-first + server-side merge MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - /download/youtube now PREFERS a progressive MP4 (single file with video+audio) using the android player client — full-speed from VPS, no datacenter rate-limit (verified: itag 18 downloads 28MB in ~3s). - Falls back to adaptive bestvideo+bestaudio (2 URLs) when the requested quality has no progressive format. - New param merge=true triggers server-side ffmpeg merge into one MP4, served from GET /file/yt_merge/{name} (temp, 2h TTL cleanup). - /download/youtube/mp3: report the REAL container ext (webm/opus) instead of hardcoded 'mp3'. --- src/application/downloader.rs | 7 + src/infrastructure/repository/downloader.rs | 378 +++++++++++++++++++- src/observability/openapi_modules.rs | 1 + src/presentation/handler/downloader.rs | 96 ++++- src/presentation/router.rs | 4 + 5 files changed, 479 insertions(+), 7 deletions(-) diff --git a/src/application/downloader.rs b/src/application/downloader.rs index 0a73e6a..b9c9b00 100644 --- a/src/application/downloader.rs +++ b/src/application/downloader.rs @@ -42,6 +42,13 @@ pub async fn download_youtube(url: &str, quality: &str) -> Result Result { + DownloaderRepository::fetch_youtube_merge(url, quality).await +} + pub async fn download_youtube_mp3(url: &str) -> Result { DownloaderRepository::download_youtube_mp3(url).await } diff --git a/src/infrastructure/repository/downloader.rs b/src/infrastructure/repository/downloader.rs index 5ac09df..60c4e3f 100644 --- a/src/infrastructure/repository/downloader.rs +++ b/src/infrastructure/repository/downloader.rs @@ -5,6 +5,8 @@ use std::borrow::Cow; use std::collections::HashMap; +use std::path::{Path, PathBuf}; +use std::process::{Command, Stdio}; use std::time::Duration; use aes::cipher::{BlockDecrypt, KeyInit}; @@ -511,6 +513,14 @@ impl DownloaderRepository { fetch_youtube_mp4(url, quality).await } + /// YouTube video + audio merged server-side into a single MP4. + pub async fn fetch_youtube_merge( + url: &str, + quality: &str, + ) -> Result { + fetch_youtube_merge_free(url, quality).await + } + /// YouTube to MP3 via ydlp.yard.id. pub async fn download_youtube_mp3(url: &str) -> Result { fetch_youtube_mp3(url).await @@ -1402,7 +1412,11 @@ pub async fn fetch_youtube_mp3(url: &str) -> Result Result Result Result Result { + let video_id = extract_youtube_id(url) + .ok_or_else(|| ScrapingError::Http("Invalid YouTube URL".to_string()))?; + + let q = quality.trim_end_matches('p'); + let fmt = format!( + "bestvideo[height<={}]+bestaudio/best[height<={}]/best", + q, q + ); + let data = run_ytdlp_json(url, &["-f", &fmt, "--merge-output-format", "mp4"]).await?; + + let title = data + .get("title") + .and_then(|v| v.as_str()) + .map(|s| s.to_string()); + let author = data + .get("uploader") + .and_then(|v| v.as_str()) + .map(|s| s.to_string()); + let thumbnail = data + .get("thumbnail") + .and_then(|v| v.as_str()) + .map(|s| s.to_string()); + let duration = data + .get("duration") + .and_then(|v| v.as_u64()) + .map(|d| format!("{}s", d)); + + // Locate the video-only and audio-only URLs from requested_formats. + let requested_formats = data + .get("requested_formats") + .and_then(|v| v.as_array()) + .ok_or_else(|| ScrapingError::Http("No requested_formats from yt-dlp".to_string()))?; + + let mut video_url: Option = None; + let mut audio_url: Option = None; + let mut video_ext = "mp4"; + let mut audio_ext = "webm"; + for f in requested_formats { + let vcodec = f.get("vcodec").and_then(|v| v.as_str()).unwrap_or("none"); + let acodec = f.get("acodec").and_then(|v| v.as_str()).unwrap_or("none"); + let f_url = f.get("url").and_then(|v| v.as_str()); + let proto = f.get("protocol").and_then(|v| v.as_str()).unwrap_or(""); + if proto.contains("m3u8") { + continue; // HLS streams are not directly downloadable via reqwest + } + if vcodec != "none" && acodec == "none" { + if let Some(u) = f_url { + video_url = Some(u.to_string()); + video_ext = f.get("ext").and_then(|v| v.as_str()).unwrap_or("mp4"); + } + } else if vcodec == "none" && acodec != "none" { + if let Some(u) = f_url { + audio_url = Some(u.to_string()); + audio_ext = f.get("ext").and_then(|v| v.as_str()).unwrap_or("webm"); + } + } + } + + let video_url = video_url + .ok_or_else(|| ScrapingError::Http("No video stream URL from yt-dlp".to_string()))?; + let audio_url = audio_url + .ok_or_else(|| ScrapingError::Http("No audio stream URL from yt-dlp".to_string()))?; + + // --- download both streams to the merge temp dir --- + let merge_dir = PathBuf::from("/var/lib/scraper/uploads/yt_merge"); + tokio::fs::create_dir_all(&merge_dir) + .await + .map_err(|e| ScrapingError::Http(format!("create merge dir failed: {e}")))?; + + // Cleanup stale merged files (> 2h) before writing anything new. + cleanup_stale_merges(&merge_dir).await; + + let job_id = uuid::Uuid::new_v4().simple().to_string(); + let vpath = merge_dir.join(format!("{video_id}_{job_id}_v.{video_ext}")); + let apath = merge_dir.join(format!("{video_id}_{job_id}_a.{audio_ext}")); + let outpath = merge_dir.join(format!("{video_id}_{job_id}_merged.mp4")); + + download_to_file(&video_url, &vpath).await?; + download_to_file(&audio_url, &apath).await?; + + // --- merge with ffmpeg --- + let merge_result = tokio::task::spawn_blocking({ + let vpath = vpath.clone(); + let apath = apath.clone(); + let outpath = outpath.clone(); + move || merge_with_ffmpeg(&vpath, &apath, &outpath) + }) + .await + .map_err(|e| ScrapingError::Http(format!("ffmpeg join failed: {e}")))?; + + match merge_result { + Ok(()) => {} + Err(e) => { + let _ = tokio::fs::remove_file(&vpath).await; + let _ = tokio::fs::remove_file(&apath).await; + return Err(ScrapingError::Http(format!("ffmpeg merge failed: {e}"))); + } + } + + // Clean up the transient inputs now that the merged file exists. + let _ = tokio::fs::remove_file(&vpath).await; + let _ = tokio::fs::remove_file(&apath).await; + + let file_size = tokio::fs::metadata(&outpath).await.ok().map(|m| m.len()); + + let file_name = outpath + .file_name() + .map(|s| s.to_string_lossy().to_string()) + .unwrap_or_default(); + let public_url = format!("/file/yt_merge/{file_name}"); + + let mut result = DownloadResult::success(title); + result.author = author; + result.thumbnail = thumbnail.clone(); + result.duration = duration; + result.provider = Some("yt-dlp+ffmpeg".to_string()); + result.media.push(MediaItem { + url: public_url, + quality: Some(format!("{}p merged", q)), + file_type: Some(MediaType::Video), + extension: Some("mp4".to_string()), + thumbnail: thumbnail.clone(), + file_size: file_size.map(format_filesize), + size_bytes: file_size, + frame_width: None, + frame_height: None, + note: Some( + "Merged server-side: single MP4 file containing both video and audio".to_string(), + ), + }); + + Ok(result) +} + +/// Download `url` to `path` with a generous timeout. The yt CDN stream URLs are +/// signed and short-lived, so we must download promptly and cannot retry after +/// the URL expires. +async fn download_to_file(url: &str, path: &Path) -> Result<(), ScrapingError> { + // Dedicated client with a longer timeout than the shared 30s one: merged + // files are often tens of MB and the yt CDN can be slow from a datacenter. + // .no_gzip()/.no_brotli()/.no_deflate() disable reqwest's auto-decompress + // so we can write raw bytes — some CDN responses are mis-labelled and fail + // the auto-decode step. + let client = reqwest::Client::builder() + .timeout(Duration::from_secs(300)) + .connect_timeout(Duration::from_secs(20)) + .no_gzip() + .no_brotli() + .no_deflate() + .build() + .map_err(|e| ScrapingError::Http(format!("stream client build failed: {e}")))?; + let resp = client + .get(url) + .header(USER_AGENT, "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36") + .header("Accept-Encoding", "identity") + .send() + .await + .map_err(|e| ScrapingError::Http(format!("stream download failed: {e}")))?; + + let status = resp.status(); + if !status.is_success() { + return Err(ScrapingError::Http(format!( + "stream download HTTP {status}" + ))); + } + + let bytes = resp + .bytes() + .await + .map_err(|e| ScrapingError::Http(format!("stream body read failed: {e}")))?; + + tokio::fs::write(path, bytes) + .await + .map_err(|e| ScrapingError::Http(format!("stream write failed: {e}")))?; + Ok(()) +} + +/// Run ffmpeg to remux the video and audio streams into a single MP4. +/// `-c:v copy` avoids a full re-encode (fast); audio is transcoded to AAC so +/// the output plays everywhere. `+faststart` moves the moov atom to the front +/// so the file streams immediately. +fn merge_with_ffmpeg(video: &Path, audio: &Path, out: &Path) -> Result<(), String> { + let status = Command::new("/bin/ffmpeg") + .args([ + "-y", + "-v", + "error", + "-i", + video.to_str().unwrap_or_default(), + "-i", + audio.to_str().unwrap_or_default(), + "-c:v", + "copy", + "-c:a", + "aac", + "-movflags", + "+faststart", + "-shortest", + out.to_str().unwrap_or_default(), + ]) + .stdout(Stdio::null()) + .stderr(Stdio::piped()) + .status() + .map_err(|e| format!("spawn ffmpeg failed: {e}"))?; + + if status.success() { + Ok(()) + } else { + Err(format!("ffmpeg exited with {status:?}")) + } +} + +/// Remove merged output + transient files older than 2 hours. Bounded, best +/// effort — never fails the request. +async fn cleanup_stale_merges(dir: &Path) { + let deadline = std::time::SystemTime::now() - std::time::Duration::from_secs(2 * 60 * 60); + let Ok(mut entries) = tokio::fs::read_dir(dir).await else { + return; + }; + while let Ok(Some(entry)) = entries.next_entry().await { + let Ok(meta) = entry.metadata().await else { + continue; + }; + let Ok(modified) = meta.modified() else { + continue; + }; + if modified < deadline { + let _ = tokio::fs::remove_file(entry.path()).await; + } + } +} + // Spotify downloaders // ============================================================================ diff --git a/src/observability/openapi_modules.rs b/src/observability/openapi_modules.rs index 65c730c..d330d95 100644 --- a/src/observability/openapi_modules.rs +++ b/src/observability/openapi_modules.rs @@ -66,6 +66,7 @@ use utoipa::OpenApi; crate::presentation::handler::downloader::download_streamable, crate::presentation::handler::downloader::download_videy, crate::presentation::handler::downloader::download_bilibili, + crate::presentation::handler::downloader::serve_merged_file, // ── Misc module ──────────────────────────────────────────── crate::presentation::handler::misc::currency_converter_handler, crate::presentation::handler::misc::harga_emas_handler, diff --git a/src/presentation/handler/downloader.rs b/src/presentation/handler/downloader.rs index 04340d8..8138331 100644 --- a/src/presentation/handler/downloader.rs +++ b/src/presentation/handler/downloader.rs @@ -21,6 +21,9 @@ pub struct DownloadParams { pub cookies: Option, pub api_key: Option, pub quality: Option, + /// When `true`, YouTube downloads are merged server-side (video + audio + /// into a single MP4) instead of returning separate video/audio URLs. + pub merge: Option, } // ======================================================================== @@ -44,7 +47,25 @@ pub struct DownloadParams { pub async fn download( Query(params): Query, ) -> Result, AppError> { - let result = use_cases::download_all_in_one(¶ms.url, params.cookies.as_deref()).await?; + let merge = params + .merge + .as_deref() + .map(|v| v == "true" || v == "1") + .unwrap_or(false); + let result = if merge { + // All-in-one with merge=true → force merge for YouTube URLs. + if crate::application::downloader::detect_platform(¶ms.url) == "youtube" { + use_cases::download_youtube_merge( + ¶ms.url, + params.quality.as_deref().unwrap_or("720"), + ) + .await? + } else { + use_cases::download_all_in_one(¶ms.url, params.cookies.as_deref()).await? + } + } else { + use_cases::download_all_in_one(¶ms.url, params.cookies.as_deref()).await? + }; Ok(Json(DownloadResponse::ok(result))) } @@ -152,9 +173,17 @@ pub async fn download_tiktok( pub async fn download_youtube( Query(params): Query, ) -> Result, AppError> { - let result = - use_cases::download_youtube(¶ms.url, params.quality.as_deref().unwrap_or("720")) - .await?; + let quality = params.quality.as_deref().unwrap_or("720"); + let merge = params + .merge + .as_deref() + .map(|v| v == "true" || v == "1") + .unwrap_or(false); + let result = if merge { + use_cases::download_youtube_merge(¶ms.url, quality).await? + } else { + use_cases::download_youtube(¶ms.url, quality).await? + }; Ok(Json(DownloadResponse::ok(result))) } @@ -556,3 +585,62 @@ pub async fn download_bilibili( let result = use_cases::download_bilibili(¶ms.url).await?; Ok(Json(DownloadResponse::ok(result))) } + +// ======================================================================== +// Merged-file static serving +// ======================================================================== + +/// Serve a merged YouTube MP4 produced by `/download/youtube?merge=true`. +/// Path is sanitised: only a plain filename (no `/`, `..`) inside the +/// `yt_merge` uploads dir is allowed. +/// +/// `/file/yt_merge/{filename}` +#[utoipa::path( + get, + path = "/file/yt_merge/{filename}", + tag = "download", + operation_id = "dl_serve_merged_file", + params( + ("filename" = String, Path, description = "Merged MP4 file name") + ), + responses( + (status = 200, description = "Merged MP4 file", content_type = "video/mp4"), + (status = 404, description = "File not found or invalid name") + ) +)] +pub async fn serve_merged_file( + axum::extract::Path(filename): axum::extract::Path, +) -> Result { + // Reject any path traversal or nested segments. + if filename.contains('/') + || filename.contains('\\') + || filename.contains("..") + || filename.is_empty() + { + return Err(AppError::NotFound(format!("invalid file name: {filename}"))); + } + + let dir = std::path::PathBuf::from("/var/lib/scraper/uploads/yt_merge"); + let path = dir.join(&filename); + let bytes = tokio::fs::read(&path) + .await + .map_err(|_| AppError::NotFound(format!("file not found: {filename}")))?; + + let content_type = if filename.ends_with(".mp4") { + "video/mp4" + } else if filename.ends_with(".webm") { + "video/webm" + } else { + "application/octet-stream" + }; + + Ok(axum::response::Response::builder() + .header("Content-Type", content_type) + .header("Content-Length", bytes.len().to_string()) + .header( + "Content-Disposition", + format!("attachment; filename=\"{filename}\""), + ) + .body(axum::body::Body::from(bytes)) + .map_err(|e| AppError::Internal(format!("build response: {e}")))?) +} diff --git a/src/presentation/router.rs b/src/presentation/router.rs index a1b8bbb..6ceac8e 100644 --- a/src/presentation/router.rs +++ b/src/presentation/router.rs @@ -184,6 +184,10 @@ pub fn build_router(app_state: Arc) -> anyhow::Result { "/download/youtube/mp3", axum::routing::get(crate::presentation::handler::downloader::download_youtube_mp3), ) + .route( + "/file/yt_merge/{filename}", + axum::routing::get(crate::presentation::handler::downloader::serve_merged_file), + ) .route( "/download/spotify", axum::routing::get(crate::presentation::handler::downloader::download_spotify),