import os import sys import re import asyncio import argparse from pathlib import Path from urllib.parse import urlparse from tqdm import tqdm import scraper_core def _is_filtered(url: str) -> bool: """Apply the shared master filter list (lazy import avoids circular deps).""" try: from downloader import is_filtered except Exception: return False return is_filtered(url) async def get_video_info(scraper, page, url, video_selector, uploader_eval_js=None): """Resolve page, extract video source URL, and uploader name.""" await scraper.resolve_page(page, url) # Run user uploader evaluation if provided, else generic fallback if uploader_eval_js: uploader_raw = await page.evaluate(uploader_eval_js) else: uploader_raw = await page.evaluate("""() => { const selectors = [ '.uploader', '.author', '.user-name', 'a[href*="/members/"]', 'a[href*="/user/"]', '.video-metadata a[href*="/profile/"]' ]; for (const s of selectors) { const el = document.querySelector(s); if (el && el.innerText.trim()) return el.innerText.trim(); } return 'unknown'; }""") uploader = scraper_core.clean_filename(uploader_raw) if uploader_raw else "unknown" # Try core video src extraction first (intercepts player network requests) src = await scraper.extract_media_source(page, url, video_selector) return src, uploader async def scrape_playlist(page, playlist_url: str, is_video_link_fn, next_page_selector): """Scrape all video URLs from a playlist/user profile, following pagination.""" base = playlist_url.rstrip("/") all_video_urls = [] page_num = 1 seen = set() while True: # Navigate or click to pagination if page_num > 1: clicked = False try: # Find the button/link matching the target page_num exactly btn_info = await page.evaluate(f"""(pageNum) => {{ const btn = Array.from(document.querySelectorAll('button, a')) .find(b => b.innerText.trim() === String(pageNum)); if (btn) {{ return {{ found: true, href: btn.href || null }}; }} return {{ found: false }}; }}""", page_num) if btn_info.get("found"): href = btn_info.get("href") if href and href.startswith("http"): tqdm.write(f"Scraping page {page_num} via direct href navigation: {href}") await page.goto(href, wait_until="domcontentloaded", timeout=30000) await page.wait_for_timeout(3000) clicked = True else: clicked = await page.evaluate(f"""(pageNum) => {{ const btn = Array.from(document.querySelectorAll('button, a')) .find(b => b.innerText.trim() === String(pageNum)); if (btn) {{ btn.click(); return true; }} return false; }}""", page_num) if clicked: tqdm.write(f"Scraping page {page_num} via client-side pagination click...") await page.wait_for_timeout(3000) except Exception as e: tqdm.write(f" Failed client-side click/navigation: {e}") if not clicked: # Fall back to URL navigation if "?" in base: url = f"{base}&page={page_num}" else: url = f"{base}/page/{page_num}/" if "/" in base else f"{base}?page={page_num}" tqdm.write(f"Scraping page {page_num} via URL navigation: {url}") try: await page.goto(url, wait_until="domcontentloaded", timeout=30000) await page.wait_for_timeout(2000) except Exception as e: tqdm.write(f" Error loading page {page_num}: {e}") break else: tqdm.write(f"Scraping page {page_num}: {base}") try: await page.goto(base, wait_until="domcontentloaded", timeout=30000) await page.wait_for_timeout(2000) except Exception as e: tqdm.write(f" Error loading page {page_num}: {e}") break # Extract all link tags and JSON-LD URLs links = await page.evaluate("""() => { const urls = Array.from(document.querySelectorAll('a')).map(el => el.href); const scripts = document.querySelectorAll('script[type="application/ld+json"]'); for (const s of scripts) { try { const data = JSON.parse(s.innerText); if (data['@type'] === 'ItemList' && data.itemListElement) { for (const element of data.itemListElement) { const item = element.item; if (item) { if (item.embedUrl) urls.push(item.embedUrl); if (item.url) urls.push(item.url); } } } } catch (e) {} } return urls; }""") unique_page_urls = [] for l in links: if l in seen or not is_video_link_fn(l): continue if _is_filtered(l): tqdm.write(f" x Filtered: {l}") continue seen.add(l) unique_page_urls.append(l) if not unique_page_urls: tqdm.write(" No new video links found on page.") break tqdm.write(f" Found {len(unique_page_urls)} video links") all_video_urls.extend(unique_page_urls) # Check for next page availability has_next = False if next_page_selector: next_button = await page.query_selector(next_page_selector) if next_button: has_next = True if not has_next: # Check for client-side pagination numerical button matching page_num + 1 has_next = await page.evaluate(f"""(nextPage) => {{ return Array.from(document.querySelectorAll('button, a')) .some(b => b.innerText.trim() === String(nextPage)); }}""", page_num + 1) if not has_next: # Check if there is any pagination link for next page in extracted links next_page_str = f"page={page_num + 1}" next_page_str_alt = f"/page/{page_num + 1}" has_next = any((next_page_str in l or next_page_str_alt in l) for l in links) if not has_next: tqdm.write(" No next page button or link found, done.") break page_num += 1 return all_video_urls async def worker(queue, scraper, download_dir, skip_existing, bar_pool, overall_bar, video_selector, uploader_eval_js): """Worker task that handles resolving video URL and downloading.""" while True: url = await queue.get() if url is None: queue.task_done() break try: if _is_filtered(url): tqdm.write(f" x Filtered: {url}") continue page = await scraper.new_page() src, uploader = await get_video_info(scraper, page, url, video_selector, uploader_eval_js) await page.close() if not src: tqdm.write(f" [ERROR] Skipping {url} - no video source found") continue # Create a clean filename from URL path parsed = urlparse(url) stem = parsed.path.rstrip("/").split("/")[-1] if not stem or stem == "video" or stem.isdigit(): stem = parsed.path.rstrip("/").split("/")[-2] + "_" + stem filename = scraper_core.clean_filename(stem) + ".mp4" dest_path = download_dir / uploader / filename downloaded_bytes = 0 if skip_existing: from downloader import check_existing_file is_complete, existing_bytes = await check_existing_file(dest_path, src, {"Referer": url}) if is_complete: if hasattr(overall_bar, "record_skip"): overall_bar.record_skip(existing_bytes) else: overall_bar.update(1) continue pos = bar_pool.acquire() or 4 # Download concurrently in a thread success = await asyncio.to_thread( scraper_core.download_file, src, dest_path, None, pos, url ) bar_pool.release(pos) if success: if dest_path.is_file(): downloaded_bytes = dest_path.stat().st_size label = f"{dest_path.parent.name}/{dest_path.name}" if hasattr(overall_bar, "record_download"): overall_bar.record_download(downloaded_bytes, name=label) else: overall_bar.update(1) scraper_core.append_log(download_dir / "urls.txt", url) scraper_core.append_log(download_dir / uploader / "urls.txt", url) else: overall_bar.update(1) except Exception as e: tqdm.write(f" Error processing {url}: {e}") overall_bar.update(1) finally: queue.task_done() async def process_urls(site_name, urls, download_dir, is_video_link_fn, next_page_selector, video_selector, uploader_eval_js, concurrency, skip_existing): scraper = scraper_core.PlaywrightScraper() await scraper.start() # Resolve playlists first if any target is a playlist playlist_page = await scraper.new_page() all_urls = [] for url in urls: if is_video_link_fn(url): if _is_filtered(url): tqdm.write(f" x Filtered (not queued): {url}") continue all_urls.append(url) else: playlist_urls = await scrape_playlist(playlist_page, url, is_video_link_fn, next_page_selector) all_urls.extend(playlist_urls) await playlist_page.close() if not all_urls: tqdm.write("No video URLs found to process.") await scraper.close() return tqdm.write(f"[{site_name}] Processing {len(all_urls)} videos with concurrency {concurrency} ...") # Queue setup queue = asyncio.Queue() for url in all_urls: await queue.put(url) for _ in range(concurrency): await queue.put(None) bar_pool = scraper_core.BarPositionPool(concurrency) bar_pool.available = [p + 3 for p in bar_pool.available] try: from downloader import OverallProgressTracker overall_bar = OverallProgressTracker( total=len(all_urls), desc=f"[{site_name}] Overall", ) except Exception: overall_bar = tqdm( total=len(all_urls), desc=f"[{site_name}] Overall", position=2, leave=True, ncols=80, ) workers = [ asyncio.create_task(worker(queue, scraper, download_dir, skip_existing, bar_pool, overall_bar, video_selector, uploader_eval_js)) for _ in range(concurrency) ] await asyncio.gather(*workers) overall_bar.close() # Clear terminal lines sys.stdout.write("\n" * (concurrency + 4)) sys.stdout.flush() await scraper.close() def run(site_name, is_video_link_fn, next_page_selector, video_selector="video", uploader_eval_js=None, default_playlist=""): parser = argparse.ArgumentParser( description=f"Download videos from {site_name} (headless, concurrent, multithreaded)." ) parser.add_argument( "urls", nargs="*", help="Playlist or individual video URLs.", ) parser.add_argument( "--concurrency", type=int, default=3, help="Number of concurrent downloads (default: 3).", ) parser.add_argument( "--skip-existing", action="store_true", default=True, help="Skip already-downloaded files (default: true).", ) parser.add_argument( "--no-skip-existing", action="store_false", dest="skip_existing", help="Re-download existing files.", ) args = parser.parse_args() targets = args.urls if not targets: if default_playlist: targets = [default_playlist] else: sys.exit("Error: No URLs provided.") download_dir = Path(sys.argv[0]).resolve().parent / "videos" asyncio.run(process_urls( site_name, targets, download_dir, is_video_link_fn, next_page_selector, video_selector, uploader_eval_js, args.concurrency, args.skip_existing ))