import os import sys import re import asyncio import argparse import random from pathlib import Path from tqdm import tqdm sys.path.append(str(Path(__file__).resolve().parents[1])) import scraper_core BASE_DIR = Path(__file__).resolve().parent DOWNLOAD_DIR = BASE_DIR / "videos" FILTERED_WORDS = {"shit", "scat", "shitty", "poop", "horseshit", "bullshit", "cowshit", "shitting", "crap", "feces", "dung", "pungpile", "pungheap", "pooping", "crapping", "manure"} def is_filtered(url: str) -> bool: lowered = url.lower() return any(w in lowered for w in FILTERED_WORDS) async def get_cloudflare_cookies(scraper, url): """Use Playwright to solve one Cloudflare challenge and return browser cookies.""" page = await scraper.new_page() try: try: await scraper.resolve_page(page, url) except Exception as e: tqdm.write(f" Warning: Cloudflare resolve_page failed: {e}. Trying to proceed with currently gathered cookies.") raw_cookies = await scraper.context.cookies() return {c["name"]: c["value"] for c in raw_cookies} finally: await page.close() def extract_pornzoo_video_links(html: str) -> list: """Extract video links matching player or video page patterns on PornZoo.Love.""" # Matches URLs like: /video/slug or /en/video/slug patterns = [ r'href="(/video/[^"/#?]+)"', r'href="(/[^/]+/video/[^"/#?]+)"', r'href="(https?://pornzoo\.love/video/[^"/#?]+)"', r'href="(https?://pornzoo\.love/[^/]+/video/[^"/#?]+)"' ] links = [] for pattern in patterns: for match in re.findall(pattern, html): if match.startswith("/"): url = f"https://pornzoo.love{match}" else: url = match if url not in links and "/embed/" not in url: links.append(url) return links async def scrape_playlist(playlist_url: str, cookies: dict, queue, counter, overall_bar, scraper): """Scrape all video URLs from a playlist/user page.""" page_num = 1 while True: url = playlist_url if page_num > 1: # Detect pagination pattern if any (e.g. ?page=X or /page/X) if "?" in playlist_url: url = f"{playlist_url}&page={page_num}" else: url = f"{playlist_url}?page={page_num}" tqdm.write(f" Scraping page {page_num}: {url}") html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://pornzoo.love/") if not html: tqdm.write(f" Static fetch failed. Trying Playwright for playlist page...") page = await scraper.new_page() try: await scraper.resolve_page(page, url) html = await page.content() except Exception as pe: tqdm.write(f" x Playwright failed to fetch playlist page: {pe}") finally: await page.close() if not html: tqdm.write(f" Failed to fetch page {page_num}") break video_urls = extract_pornzoo_video_links(html) if not video_urls: tqdm.write(f" No video links found on page.") break tqdm.write(f" Found {len(video_urls)} video links") new_links = 0 for video_url in video_urls: if is_filtered(video_url): tqdm.write(f" x Filtered: {video_url}") continue await queue.put(video_url) counter["total"] += 1 overall_bar.total = counter["total"] overall_bar.refresh() new_links += 1 # Check for next page indicator if "Next" not in html and "next" not in html.lower(): break if new_links == 0: break page_num += 1 await asyncio.sleep(random.uniform(1.0, 2.5)) def extract_video_info(html: str, url: str) -> tuple: """Extract video embed ID, direct source, and uploader name from page HTML.""" embed_id = None uploader = "unknown_creator" # Find embed URL / ID embed_match = re.search(r'/embed/(\d+)', html) if embed_match: embed_id = embed_match.group(1) # Find uploader name user_match = re.search(r'href="[^"]*/user/([^"/]+)"', html) if user_match: uploader = user_match.group(1) uploader = scraper_core.clean_filename(uploader) return embed_id, uploader async def worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper): """Worker: fetch video page, resolve source link, and download.""" while True: url = await queue.get() if url is None: queue.task_done() break try: if is_filtered(url): tqdm.write(f" x Filtered: {url}") continue await asyncio.sleep(random.uniform(0.3, 0.8)) html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://pornzoo.love/") if not html: tqdm.write(f" Static fetch failed for {url}. Fetching page via Playwright...") page = await scraper.new_page() try: await scraper.resolve_page(page, url) html = await page.content() except Exception as pe: tqdm.write(f" x Playwright failed to fetch HTML for {url}: {pe}") finally: await page.close() if not html: tqdm.write(f" x Failed to fetch HTML for {url}") continue embed_id, uploader = extract_video_info(html, url) if not embed_id: tqdm.write(f" x Skipping {url} - could not find embed ID") continue # Standardize filename based on URL slug and embed ID stem = url.rstrip("/").rsplit("/", 1)[-1] filename = f"{scraper_core.clean_filename(stem)}-{embed_id}.mp4" dest_path = DOWNLOAD_DIR / uploader / filename if skip_existing and scraper_core.is_already_downloaded(dest_path): tqdm.write(f" > Already downloaded: {dest_path.parent.name}/{dest_path.name}") continue # Resolve direct video source via embed page embed_url = f"https://pornzoo.love/embed/{embed_id}" embed_html = await asyncio.to_thread(scraper_core.curl_fetch, embed_url, cookies=cookies, referer=url) if not embed_html: page = await scraper.new_page() try: await scraper.resolve_page(page, embed_url) embed_html = await page.content() except Exception as pe: tqdm.write(f" x Playwright failed to load embed for {url}: {pe}") finally: await page.close() src = None if embed_html: src_match = re.search(r']*src="([^"]+)"', embed_html) if src_match: src = src_match.group(1) else: src_match = re.search(r']*src="([^"]+)"', embed_html) if src_match: src = src_match.group(1) if not src: # Direct fallback guess link src = f"https://pornzoo.love/media/videos/iphone/{embed_id}.mp4" pos = bar_pool.acquire() if pos is None: pos = 1 success = await asyncio.to_thread( scraper_core.download_file, src, dest_path, {"Referer": embed_url}, pos, cookies=cookies ) bar_pool.release(pos) if success: scraper_core.append_log(DOWNLOAD_DIR / "urls.txt", url) scraper_core.append_log(DOWNLOAD_DIR / uploader / "urls.txt", url) except Exception as e: tqdm.write(f" Error processing {url}: {e}") finally: overall_bar.update(1) queue.task_done() async def producer(urls, cookies, queue, counter, overall_bar, scraper): """Feed input URLs (video pages or playlists) into the queue.""" for url in urls: if "/video/" in url: if is_filtered(url): tqdm.write(f" x Filtered (not queued): {url}") continue tqdm.write(f"Queuing direct video URL: {url}") await queue.put(url) counter["total"] += 1 overall_bar.total = counter["total"] overall_bar.refresh() else: tqdm.write(f"Scraping user/playlist: {url}") try: await scrape_playlist(url, cookies, queue, counter, overall_bar, scraper) except Exception as e: tqdm.write(f" Error scraping playlist {url}: {e}") await asyncio.sleep(random.uniform(1.0, 2.0)) async def process_urls(urls, concurrency, skip_existing): scraper = scraper_core.PlaywrightScraper() await scraper.start(headless=True) tqdm.write("Solving Cloudflare cookies...") first_url = urls[0] cookies = await get_cloudflare_cookies(scraper, first_url) tqdm.write(f"Got cookies: {list(cookies.keys())}") queue = asyncio.Queue() counter = {"total": 0} overall_bar = tqdm( total=0, desc="Overall Progress", position=0, leave=True, ncols=80, bar_format="{desc}: {n_fmt}/{total_fmt} |{bar}| {percentage:.0f}%", ) bar_pool = scraper_core.BarPositionPool(concurrency) workers = [ asyncio.create_task(worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper)) for _ in range(concurrency) ] await producer(urls, cookies, queue, counter, overall_bar, scraper) for _ in range(concurrency): await queue.put(None) await asyncio.gather(*workers) overall_bar.close() await scraper.close() sys.stdout.write("\n" * (concurrency + 1)) sys.stdout.flush() def resolve_urls(args_or_files): urls = [] for arg in args_or_files: if os.path.isfile(arg): with open(arg, "r") as f: for line in f: for token in line.strip().split(): if token and not token.startswith("#"): urls.append(token) else: urls.append(arg) return urls def main(): parser = argparse.ArgumentParser( description="Download videos from PornZoo.Love (concurrent, metadata-aware)." ) parser.add_argument( "urls", nargs="*", help="Profile URLs, direct video URLs, or .txt files containing them.", ) parser.add_argument( "--concurrency", type=int, default=2, help="Number of concurrent downloads (default: 2).", ) parser.add_argument( "--skip-existing", action="store_true", default=True, help="Skip downloading files that already exist (default: True).", ) args = parser.parse_args() if not args.urls: parser.print_help() sys.exit(1) targets = resolve_urls(args.urls) if not targets: print("No URLs found to process.") sys.exit(0) # Ensure output directory exists DOWNLOAD_DIR.mkdir(parents=True, exist_ok=True) asyncio.run(process_urls(targets, args.concurrency, args.skip_existing)) if __name__ == "__main__": main()