import os import sys import re import asyncio import argparse import random from pathlib import Path from tqdm import tqdm sys.path.append(str(Path(__file__).resolve().parents[1])) import scraper_core BASE_DIR = Path(__file__).resolve().parent DOWNLOAD_DIR = BASE_DIR / "videos" FILTERED_WORDS = {"shit", "scat", "shitty", "poop", "horseshit", "bullshit", "cowshit", "shitting", "crap", "feces", "dung", "pungpile", "pungheap", "pooping", "crapping", "manure"} def is_filtered(url: str) -> bool: lowered = url.lower() return any(w in lowered for w in FILTERED_WORDS) async def get_cloudflare_cookies(scraper, url): """Use Playwright to solve one Cloudflare challenge and return browser cookies.""" page = await scraper.new_page() try: await scraper.resolve_page(page, url) raw_cookies = await scraper.context.cookies() return {c["name"]: c["value"] for c in raw_cookies} finally: await page.close() async def scrape_playlist(playlist_url: str, cookies: dict, queue, counter, overall_bar): """Scrape all video URLs from a playlist using curl_cffi with browser cookies.""" base_check = playlist_url.lower() page_num = 1 while True: if page_num > 1: base_no_slash = playlist_url.rstrip("/") if "?" in playlist_url: url = f"{base_no_slash}&page={page_num}" elif "uploads-by-user" in base_check or "playlist-by-user" in base_check or "videos-commented-by-user" in base_check or "/channels/" in base_check: url = f"{base_no_slash}/page{page_num}.html" else: url = f"{base_no_slash}?page={page_num}" else: url = playlist_url tqdm.write(f" Scraping playlist page {page_num}: {url}") html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.luxuretv.com/") if not html: tqdm.write(f" Failed to fetch page {page_num}") break video_urls = scraper_core.extract_video_links_from_html(html) if not video_urls: tqdm.write(f" No video links found on page.") break tqdm.write(f" Found {len(video_urls)} unique video links") for video_url in video_urls: if is_filtered(video_url): tqdm.write(f" x Filtered: {video_url}") continue await queue.put(video_url) counter["total"] += 1 overall_bar.total = counter["total"] overall_bar.refresh() if not scraper_core.has_next_page(html): tqdm.write(f" No 'Next' button found, done.") break page_num += 1 await asyncio.sleep(random.uniform(1.0, 2.5)) def get_remote_file_size(url: str, headers: dict = None, cookies: dict = None) -> int: """Fetch the content-length of the remote video file.""" req_headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36", "Accept": "*/*", "Connection": "keep-alive", } if headers: req_headers.update(headers) if cookies: try: from curl_cffi import requests as curl_req with curl_req.get( url, impersonate="chrome", headers=req_headers, cookies=cookies, stream=True, timeout=15, ) as resp: if resp.status_code == 200: return int(resp.headers.get("content-length", 0)) except Exception: pass try: import requests with requests.get(url, headers=req_headers, stream=True, timeout=15, cookies=cookies) as resp: if resp.status_code == 200: return int(resp.headers.get("content-length", 0)) except Exception: pass return 0 async def worker(queue, cookies, skip_existing, bar_pool, overall_bar): """Worker: fetch video page HTML via curl_cffi, extract source, download.""" while True: url = await queue.get() if url is None: queue.task_done() break try: if is_filtered(url): tqdm.write(f" x Filtered: {url}") continue # Add a short polite delay to avoid rate limits during bulk scraping await asyncio.sleep(random.uniform(0.3, 0.8)) html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.luxuretv.com/") if not html: tqdm.write(f" x Failed to fetch {url}") continue src, uploader = scraper_core.extract_video_info_from_html(html) if not src: tqdm.write(f" x Skipping {url} - no video source found") continue stem = url.rstrip(".html").rstrip("/").rsplit("/", 1)[-1] filename = scraper_core.clean_filename(stem) + ".mp4" dest_path = DOWNLOAD_DIR / uploader / filename file_exists = dest_path.exists() is_incomplete = False if file_exists: local_size = dest_path.stat().st_size if local_size == 0: is_incomplete = True else: remote_size = await asyncio.to_thread( get_remote_file_size, src, {"Referer": url}, cookies ) if remote_size > 0 and local_size < remote_size: is_incomplete = True tqdm.write(f" ! Incomplete download detected for {dest_path.parent.name}/{dest_path.name} (local: {local_size} B, remote: {remote_size} B). Redownloading...") if skip_existing and file_exists and not is_incomplete: tqdm.write(f" > Already downloaded: {dest_path.parent.name}/{dest_path.name}") continue pos = bar_pool.acquire() if pos is None: pos = 1 success = await asyncio.to_thread( scraper_core.download_file, src, dest_path, {"Referer": url}, pos, cookies=cookies ) bar_pool.release(pos) if success: scraper_core.append_log(DOWNLOAD_DIR / "urls.txt", url) scraper_core.append_log(DOWNLOAD_DIR / uploader / "urls.txt", url) except Exception as e: tqdm.write(f" Error processing {url}: {e}") finally: overall_bar.update(1) queue.task_done() async def producer(urls, cookies, queue, counter, overall_bar): """Scrape playlists and feed discovered video URLs into the download queue immediately.""" for url in urls: if "/videos/" in url and ".html" in url: if is_filtered(url): tqdm.write(f" x Filtered (not queued): {url}") continue tqdm.write(f"Queuing direct video URL: {url}") await queue.put(url) counter["total"] += 1 overall_bar.total = counter["total"] overall_bar.refresh() else: tqdm.write(f"Scraping playlist: {url}") try: await scrape_playlist(url, cookies, queue, counter, overall_bar) except Exception as e: tqdm.write(f" Error scraping {url}: {e}") await asyncio.sleep(random.uniform(1.0, 2.0)) async def process_urls(urls, concurrency, skip_existing): scraper = scraper_core.PlaywrightScraper() await scraper.start() tqdm.write("Solving initial Cloudflare challenge for cookies...") first_url = urls[0] cookies = await get_cloudflare_cookies(scraper, first_url) tqdm.write(f"Got {len(cookies)} cookies: {list(cookies.keys())}") await scraper.close() queue = asyncio.Queue() counter = {"total": 0} overall_bar = tqdm( total=0, desc="Overall Progress", position=0, leave=True, ncols=80, bar_format="{desc}: {n_fmt}/{total_fmt} |{bar}| {percentage:.0f}%", ) bar_pool = scraper_core.BarPositionPool(concurrency) workers = [ asyncio.create_task(worker(queue, cookies, skip_existing, bar_pool, overall_bar)) for _ in range(concurrency) ] await producer(urls, cookies, queue, counter, overall_bar) for _ in range(concurrency): await queue.put(None) await asyncio.gather(*workers) overall_bar.close() sys.stdout.write("\n" * (concurrency + 1)) sys.stdout.flush() def resolve_urls(args_or_files): """Expand file arguments into URL lists.""" urls = [] for arg in args_or_files: if os.path.isfile(arg): with open(arg, "r") as f: for line in f: for token in line.strip().split(): if token and not token.startswith("#"): urls.append(token) else: urls.append(arg) return urls def main(): parser = argparse.ArgumentParser( description="Download videos from LuxureTV (headless, concurrent, multithreaded)." ) parser.add_argument( "urls", nargs="*", help="Playlist URLs, direct video URLs, or .txt files containing them. Default: user playlist.", ) parser.add_argument( "--concurrency", type=int, default=3, help="Number of concurrent downloads (default: 3).", ) parser.add_argument( "--skip-existing", action="store_true", default=True, help="Skip already-downloaded files (default: true).", ) parser.add_argument( "--no-skip-existing", action="store_false", dest="skip_existing", help="Re-download existing files.", ) parser.add_argument( "--user-playlist", default="https://en.luxuretv.com/playlist-by-user/267285/", help="Default user playlist URL (used when no URLs or files given).", ) args = parser.parse_args() raw = args.urls if args.urls else [args.user_playlist] targets = resolve_urls(raw) if not targets: sys.exit("No URLs found in arguments or file.") asyncio.run(process_urls(targets, args.concurrency, args.skip_existing)) if __name__ == "__main__": main()