import os import re import sys import time import asyncio import logging import threading import random from pathlib import Path from urllib.parse import urlparse import requests from tqdm import tqdm logging.basicConfig( level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s", datefmt="%H:%M:%S", ) log = logging.getLogger(__name__) USER_AGENT = ( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " "AppleWebKit/537.36 (KHTML, like Gecko) " "Chrome/125.0.0.0 Safari/537.36" ) HEADERS = { "User-Agent": USER_AGENT, "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8", "Accept-Language": "en-US,en;q=0.5", "Connection": "keep-alive", } STEALTH_JS = """() => { Object.defineProperty(navigator, 'webdriver', { get: () => false }); Object.defineProperty(navigator, 'plugins', { get: () => [1, 2, 3] }); Object.defineProperty(navigator, 'languages', { get: () => ['en-US', 'en'] }); }""" class BarPositionPool: """Manages console line offsets for multiple concurrent tqdm progress bars.""" def __init__(self, size): self.lock = threading.Lock() self.available = list(range(1, size + 1)) def acquire(self): with self.lock: if self.available: return self.available.pop(0) return None def release(self, pos): with self.lock: if pos is not None and pos not in self.available: self.available.append(pos) self.available.sort() def clean_filename(name: str) -> str: return re.sub(r'[^a-zA-Z0-9_.-]', '_', name) def download_file(url: str, dest_path: Path, headers: dict = None, bar_position: int = 1, referer: str = None, cookies: dict = None) -> bool: """Download a file with requests showing a progress bar at a specific terminal line.""" dest_path.parent.mkdir(parents=True, exist_ok=True) label = f"{dest_path.parent.name}/{dest_path.name}" req_headers = dict(HEADERS) if headers: req_headers.update(headers) if referer: req_headers["Referer"] = referer try: if ".m3u8" in url or "manifest" in url: import yt_dlp pbar = tqdm( desc=label[:25], unit="B", unit_scale=True, unit_divisor=1024, position=bar_position, leave=False, ncols=80, ) class YtdlpTqdmHook: def __init__(self, pbar): self.pbar = pbar self.last_bytes = 0 def __call__(self, d): if d['status'] == 'downloading': downloaded = d.get('downloaded_bytes', 0) total = d.get('total_bytes') or d.get('total_bytes_estimate') or 0 if total and self.pbar.total is None: self.pbar.total = total delta = downloaded - self.last_bytes if delta > 0: self.pbar.update(delta) self.last_bytes = downloaded elif d['status'] == 'finished': if self.pbar.total and self.last_bytes < self.pbar.total: self.pbar.update(self.pbar.total - self.last_bytes) ydl_headers = dict(req_headers) if ydl_headers.get("User-Agent") == "stash-scraper/v1": ydl_headers.pop("User-Agent", None) ydl_opts = { 'outtmpl': str(dest_path), 'progress_hooks': [YtdlpTqdmHook(pbar)], 'quiet': True, 'noprogress': True, 'http_headers': ydl_headers, 'downloader': 'ffmpeg', 'hls_use_mpegts': True, } try: with yt_dlp.YoutubeDL(ydl_opts) as ydl: result = ydl.download([url]) if result != 0: raise Exception("yt-dlp download failed with non-zero exit code") finally: pbar.close() return True if cookies: try: from curl_cffi import requests as curl_req resp = curl_req.get( url, impersonate="chrome", headers=req_headers, cookies=cookies, stream=True, timeout=(15, 300), ) resp.raise_for_status() total = int(resp.headers.get("content-length", 0)) with tqdm( total=total, unit="B", unit_scale=True, unit_divisor=1024, desc=label[:25], position=bar_position, leave=False, ncols=80, ) as pbar: with open(dest_path, "wb") as f: for chunk in resp.iter_content(chunk_size=65536): if chunk: f.write(chunk) pbar.update(len(chunk)) return True except ImportError: pass resp = requests.get(url, headers=req_headers, stream=True, timeout=(15, 300), cookies=cookies) resp.raise_for_status() if resp.raw._connection: sock = resp.raw._connection.sock if sock: sock.settimeout(30.0) total = int(resp.headers.get("content-length", 0)) with tqdm( total=total, unit="B", unit_scale=True, unit_divisor=1024, desc=label[:25], position=bar_position, leave=False, ncols=80, ) as pbar: with open(dest_path, "wb") as f: for chunk in resp.iter_content(chunk_size=65536): if chunk: f.write(chunk) pbar.update(len(chunk)) return True except Exception as e: tqdm.write(f" [ERROR] Download failed ({label}): {e}") if dest_path.exists(): try: dest_path.unlink() except Exception: pass return False def curl_fetch(url: str, cookies: dict = None, referer: str = None, retries: int = 5) -> str: """Fetch a page using curl_cffi with browser TLS impersonation. Returns HTML or empty string.""" try: from curl_cffi import requests as curl_req except ImportError: return "" headers = dict(HEADERS) if referer: headers["Referer"] = referer for attempt in range(retries): try: resp = curl_req.get( url, impersonate="chrome", headers=headers, cookies=cookies, timeout=30, ) if resp.status_code == 200: return resp.text if resp.status_code == 403: tqdm.write(f" [curl_fetch] 403 Forbidden (attempt {attempt+1}/{retries}): {url}") elif resp.status_code == 429: wait = (2 ** attempt) * 8 + random.uniform(2.0, 6.0) tqdm.write(f" [curl_fetch] 429 Rate limited, waiting {wait:.1f}s (attempt {attempt+1}/{retries}): {url}") time.sleep(wait) else: tqdm.write(f" [curl_fetch] HTTP {resp.status_code} (attempt {attempt+1}/{retries}): {url}") except Exception as e: tqdm.write(f" [curl_fetch] Error (attempt {attempt+1}/{retries}): {e}") if attempt < retries - 1: time.sleep(2 * (attempt + 1) + random.uniform(0.5, 2.0)) return "" def extract_video_info_from_html(html: str) -> tuple: """Extract (video_src, uploader) from raw video page HTML.""" src = "" uploader = "" source_match = re.search(r']+src="([^"]+)"', html) if video_match: src = video_match.group(1) if not src: ld_match = re.search(r'"contentUrl"\s*:\s*"([^"]+)"', html) if ld_match: src = ld_match.group(1) # Strip comments section to prevent extracting commenter profiles as the uploader main_html = html for marker in ['class="ul-comments"', 'class="videobas"', 'id="ul-comments"', 'class="comments"']: if marker in main_html: main_html = main_html.split(marker, 1)[0] break uploader_match = re.search(r'Uploader:\s*([^<\s]+)', main_html) if uploader_match: uploader = uploader_match.group(1) if not uploader: profile_match = re.search(r']*href="[^"]*profiles/[^"]*"[^>]*>\s*]*alt="([^"]+)"', main_html) if profile_match: uploader = profile_match.group(1) return src.strip(), clean_filename(uploader) if uploader else "unknown" def extract_video_links_from_html(html: str) -> list: """Extract unique video page URLs from a playlist page HTML.""" links = re.findall(r'href="(https?://[^"]*?/videos/[^"]*?\.html)"', html) seen = set() unique = [] for link in links: if link not in seen: seen.add(link) unique.append(link) return unique def has_next_page(html: str) -> bool: """Check if a playlist page has a 'Next' pagination link.""" return bool(re.search(r']*>[^<]*[Nn]ext[^<]*', html)) class PlaywrightScraper: """Manages browser instance and page operations headlessly.""" def __init__(self): self.playwright = None self.browser = None self.context = None async def start(self, headless=True): from playwright.async_api import async_playwright self.playwright = await async_playwright().start() launch_args = [ "--disable-blink-features=AutomationControlled", "--no-sandbox", "--disable-dev-shm-usage", ] if headless: launch_args.append("--headless=new") self.browser = await self.playwright.chromium.launch( headless=headless, channel="chrome", args=launch_args ) self.context = await self.browser.new_context( user_agent=USER_AGENT, viewport={"width": 1920, "height": 1080}, ) async def new_page(self): page = await self.context.new_page() await page.add_init_script(STEALTH_JS) return page async def resolve_page(self, page, url: str, retries=3): for attempt in range(retries): try: await page.goto(url, wait_until="domcontentloaded", timeout=45000) await page.wait_for_timeout(4000) # Automatically handle age-gate overlays if present try: overlay = await page.query_selector(".age-verification-overlay") if overlay: confirm_btn = await overlay.query_selector("button.btn-confirm, button:has-text('Enter'), button:has-text('Yes'), button:has-text('Agree'), button:has-text('18')") if confirm_btn: await confirm_btn.click() # Wait for the overlay to disappear await page.wait_for_selector(".age-verification-overlay", state="hidden", timeout=5000) except Exception: pass title = await page.title() if "Just a moment" not in title: return tqdm.write(" Cloudflare challenge detected, waiting ...") # Check for Cloudflare Turnstile challenge container for loop in range(12): # Wait up to 60 seconds (12 * 5s) per attempt await page.wait_for_timeout(5000) title = await page.title() if "Just a moment" not in title: return # Try to locate the challenge iframe and click near the center of the checkbox try: iframe = page.frame_locator('iframe[src*="challenges.cloudflare.com"]') # Look for checkbox inside iframe if it's there checkbox = iframe.locator('input[type="checkbox"], #challenge-stage, .ctp-checkbox-label') if await checkbox.count() > 0: # Try to click/focus to trigger validation await checkbox.first.click(timeout=1000) tqdm.write(" Attempted interaction with Turnstile checkbox...") except Exception: pass except Exception as e: tqdm.write(f" Attempt {attempt + 1}/{retries} failed for {url}: {e}") if attempt < retries - 1: await page.wait_for_timeout(5000) raise RuntimeError(f"Could not resolve {url} after {retries} attempts") async def extract_media_source(self, page, url: str, video_selector="video"): """Navigates to a page, intercepts media requests, and extracts the direct video source.""" media_urls = [] # Intercept network requests for media files def handle_response(response): req = response.request content_type = response.headers.get("content-type", "").lower() url_lower = response.url.lower() # Skip static assets containing '.ts' or other extensions if any(x in content_type for x in ["text/css", "javascript", "image/", "font/"]): return if any(url_lower.endswith(ext) or f"{ext}?" in url_lower for ext in [".css", ".js", ".png", ".jpg", ".jpeg", ".gif", ".woff", ".svg", ".ico"]): return is_media = ( req.resource_type in ["media", "video"] or "video/" in content_type or "application/x-mpegurl" in content_type or any(url_lower.endswith(ext) or f"{ext}?" in url_lower for ext in [".mp4", ".m3u8", ".webm", ".ts"]) ) if is_media and response.url not in media_urls: media_urls.append(response.url) page.on("response", handle_response) try: await self.resolve_page(page, url) # Try to trigger playback if the video player exists to force network requests try: play_buttons = [ "button.vjs-big-play-button", ".play-button", ".video-player", "video", ".fp-play" ] for selector in play_buttons: el = await page.query_selector(selector) if el: await el.click(timeout=1000) await page.wait_for_timeout(2000) break except Exception: pass # Allow some time for media requests to fire await page.wait_for_timeout(3000) if media_urls: # Return the largest/best URL found (prefer .mp4 or .m3u8 over small chunks) non_ts = [u for u in media_urls if not u.endswith(".ts")] if non_ts: return non_ts[0] return media_urls[0] # Fallback to DOM inspection src = await page.evaluate(f"""(sel) => {{ // 1. Check video tag const v = document.querySelector(sel); if (v && v.currentSrc) return v.currentSrc; if (v && v.src) return v.src; // 2. Check source tags const sources = document.querySelectorAll('video source'); for (const s of sources) {{ if (s.src) return s.src; }} // 3. JSON-LD fallback const scripts = document.querySelectorAll('script[type="application/ld+json"]'); for (const s of scripts) {{ try {{ const data = JSON.parse(s.innerText); if (data.contentUrl) return data.contentUrl; if (data.embedUrl) return data.embedUrl; }} catch (e) {{}} }} return ''; }}""", video_selector) return src.strip() if src else "" finally: page.remove_listener("response", handle_response) async def close(self): if self.context: await self.context.close() if self.browser: await self.browser.close() if self.playwright: await self.playwright.stop() def append_log(log_path: Path, line: str): log_path.parent.mkdir(parents=True, exist_ok=True) with open(log_path, "a", encoding="utf-8") as f: f.write(line + "\n") def is_already_downloaded(dest_path: Path) -> bool: return dest_path.exists() and dest_path.stat().st_size > 0