import os import sys import re import asyncio import argparse import random from pathlib import Path from tqdm import tqdm sys.path.append(str(Path(__file__).resolve().parents[1])) import scraper_core BASE_DIR = Path(__file__).resolve().parent DOWNLOAD_DIR = BASE_DIR / "videos" FILTERED_WORDS = {"shit", "scat", "shitty", "poop", "horseshit", "bullshit", "cowshit", "shitting", "crap", "feces", "dung", "pungpile", "pungheap", "pooping", "crapping", "manure"} def is_filtered(url: str) -> bool: lowered = url.lower() return any(w in lowered for w in FILTERED_WORDS) async def get_cloudflare_cookies(scraper, url): """Use Playwright to solve one Cloudflare challenge and return browser cookies.""" page = await scraper.new_page() try: await scraper.resolve_page(page, url) raw_cookies = await scraper.context.cookies() return {c["name"]: c["value"] for c in raw_cookies} finally: await page.close() def extract_chezcathy_video_links(html: str) -> list: """Extract video links matching player or adult page patterns on chezcathy.""" # Matches URLs like: /extreme-sex/player/35738/...html or /adult/769439/...html patterns = [ r'href="(/[^"]*?/player/\d+/[^"]*?\.html)"', r'href="(/adult/\d+/[^"]*?\.html)"', r'href="(https?://en\.chezcathy\.com/[^"]*?/player/\d+/[^"]*?\.html)"', r'href="(https?://en\.chezcathy\.com/adult/\d+/[^"]*?\.html)"' ] links = [] for pattern in patterns: for match in re.findall(pattern, html): if match.startswith("/"): url = f"https://en.chezcathy.com{match}" else: url = match if url not in links: links.append(url) return links async def scrape_playlist(playlist_url: str, cookies: dict, queue, counter, overall_bar, scraper): """Scrape all video URLs from a playlist/user page using curl_cffi with browser cookies.""" base_check = playlist_url.lower() page_num = 1 while True: # Construct pagination URL if page > 1 if page_num > 1: base_no_slash = playlist_url.rstrip("/") if "?" in playlist_url: url = f"{base_no_slash}&page={page_num}" else: url = f"{base_no_slash}?page={page_num}" else: url = playlist_url tqdm.write(f" Scraping page {page_num}: {url}") html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.chezcathy.com/") if not html: tqdm.write(f" Static fetch failed. Trying Playwright for playlist page...") page = await scraper.new_page() try: await scraper.resolve_page(page, url) html = await page.content() except Exception as pe: tqdm.write(f" x Playwright failed to fetch playlist page: {pe}") finally: await page.close() if not html: tqdm.write(f" Failed to fetch page {page_num}") break video_urls = extract_chezcathy_video_links(html) if not video_urls: tqdm.write(f" No video links found on page.") break tqdm.write(f" Found {len(video_urls)} video links") for video_url in video_urls: if is_filtered(video_url): tqdm.write(f" x Filtered: {video_url}") continue await queue.put(video_url) counter["total"] += 1 overall_bar.total = counter["total"] overall_bar.refresh() # Check for pagination next page if not scraper_core.has_next_page(html) and f"page={page_num}" not in html: # Simple check if there is a next page marker or if we found no new links break page_num += 1 await asyncio.sleep(random.uniform(1.0, 2.5)) def extract_video_info(html: str, url: str) -> tuple: """Extract video source and uploader/creator name from ChezCathy HTML.""" # Attempt to extract video source using standard scraper_core extractors src, uploader = scraper_core.extract_video_info_from_html(html) # ChezCathy specific creator/uploader extraction # Look for user profile links like: href="/user/7115/ryan_thomas" or /profile/... user_match = re.search(r'href="/user/\d+/([^"/]+)"', html) if user_match: uploader = user_match.group(1) if not uploader or uploader == "unknown": # Fallback to domain name or generic subfolder if uploader is still unknown uploader = "unknown_creator" uploader = scraper_core.clean_filename(uploader) return src, uploader async def worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper): """Worker: fetch video page HTML, extract source, and download.""" while True: url = await queue.get() if url is None: queue.task_done() break try: if is_filtered(url): tqdm.write(f" x Filtered: {url}") continue await asyncio.sleep(random.uniform(0.3, 0.8)) html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.chezcathy.com/") # Fallback to Playwright if static fetch fails if not html: tqdm.write(f" Static fetch failed for {url}. Fetching page via Playwright...") page = await scraper.new_page() try: await scraper.resolve_page(page, url) html = await page.content() except Exception as pe: tqdm.write(f" x Playwright failed to fetch HTML for {url}: {pe}") finally: await page.close() if not html: tqdm.write(f" x Failed to fetch HTML for {url}") continue src, uploader = extract_video_info(html, url) # If dynamic rendering is required to capture the media source if not src: tqdm.write(f" * Static extraction failed for {url}. Trying Playwright...") page = await scraper.new_page() try: src = await scraper.extract_media_source(page, url) except Exception as pe: tqdm.write(f" x Playwright extraction failed: {pe}") finally: await page.close() if not src: tqdm.write(f" x Skipping {url} - no video source found") continue # Standardize filename based on URL slug and ID match = re.search(r'/(\d+)/([^/]+)\.html', url) if match: video_id = match.group(1) slug = match.group(2) filename = f"{scraper_core.clean_filename(slug)}-{video_id}.mp4" else: stem = url.rstrip(".html").rstrip("/").rsplit("/", 1)[-1] filename = scraper_core.clean_filename(stem) + ".mp4" dest_path = DOWNLOAD_DIR / uploader / filename if skip_existing and scraper_core.is_already_downloaded(dest_path): tqdm.write(f" > Already downloaded: {dest_path.parent.name}/{dest_path.name}") continue pos = bar_pool.acquire() if pos is None: pos = 1 success = await asyncio.to_thread( scraper_core.download_file, src, dest_path, {"Referer": url}, pos, cookies=cookies ) bar_pool.release(pos) if success: scraper_core.append_log(DOWNLOAD_DIR / "urls.txt", url) scraper_core.append_log(DOWNLOAD_DIR / uploader / "urls.txt", url) except Exception as e: tqdm.write(f" Error processing {url}: {e}") finally: overall_bar.update(1) queue.task_done() async def producer(urls, cookies, queue, counter, overall_bar, scraper): """Feed input URLs (video pages or playlists/profile pages) into the queue.""" for url in urls: # Check if URL is a direct video page if ("/player/" in url or "/adult/" in url) and url.endswith(".html"): if is_filtered(url): tqdm.write(f" x Filtered (not queued): {url}") continue tqdm.write(f"Queuing direct video URL: {url}") await queue.put(url) counter["total"] += 1 overall_bar.total = counter["total"] overall_bar.refresh() else: tqdm.write(f"Scraping user/playlist: {url}") try: await scrape_playlist(url, cookies, queue, counter, overall_bar, scraper) except Exception as e: tqdm.write(f" Error scraping playlist {url}: {e}") await asyncio.sleep(random.uniform(1.0, 2.0)) async def process_urls(urls, concurrency, skip_existing): scraper = scraper_core.PlaywrightScraper() await scraper.start() tqdm.write("Solving Cloudflare cookies...") first_url = urls[0] cookies = await get_cloudflare_cookies(scraper, first_url) tqdm.write(f"Got cookies: {list(cookies.keys())}") queue = asyncio.Queue() counter = {"total": 0} overall_bar = tqdm( total=0, desc="Overall Progress", position=0, leave=True, ncols=80, bar_format="{desc}: {n_fmt}/{total_fmt} |{bar}| {percentage:.0f}%", ) bar_pool = scraper_core.BarPositionPool(concurrency) workers = [ asyncio.create_task(worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper)) for _ in range(concurrency) ] await producer(urls, cookies, queue, counter, overall_bar, scraper) for _ in range(concurrency): await queue.put(None) await asyncio.gather(*workers) overall_bar.close() await scraper.close() sys.stdout.write("\n" * (concurrency + 1)) sys.stdout.flush() def resolve_urls(args_or_files): urls = [] for arg in args_or_files: if os.path.isfile(arg): with open(arg, "r") as f: for line in f: for token in line.strip().split(): if token and not token.startswith("#"): urls.append(token) else: urls.append(arg) return urls def main(): parser = argparse.ArgumentParser( description="Download videos from ChezCathy (concurrent, metadata-aware)." ) parser.add_argument( "urls", nargs="*", help="Profile URLs, direct video URLs, or .txt files containing them.", ) parser.add_argument( "--concurrency", type=int, default=3, help="Number of concurrent downloads (default: 3).", ) parser.add_argument( "--skip-existing", action="store_true", default=True, help="Skip already-downloaded files (default: true).", ) args = parser.parse_args() if not args.urls: sys.exit("Please specify at least one URL or file containing URLs.") targets = resolve_urls(args.urls) asyncio.run(process_urls(targets, args.concurrency, args.skip_existing)) if __name__ == "__main__": main()