Files
niggers/downloader.py

3364 lines
127 KiB
Python

#!/usr/bin/env python3
"""
Master downloader for every website in this workspace.
Replaces the per-website downloader scripts with one entry point. Given a
mix of website URLs, RedGIFs creator usernames, and/or .txt files containing
either (one URL or creator per line, blank lines and '#' comments allowed),
it:
* identifies the target website for every input
* downloads / names / places files EXACTLY as each individual script did:
<website_folder>/videos/<creator_or_uploader>/<file>.mp4
(RedGIFs additionally writes <creator>/links.txt and appends all_links.txt)
Examples:
python downloader.py https://en.chezcathy.com/user/7115/whatever \
https://en.luxuretv.com/videos/12345/foo.html \
https://www.thisvid.com/video/9876/ \
some_redgifs_creator another_creator urls.txt
python downloader.py --update-all-creators # refresh redgifs creators.txt
python downloader.py --list-sites # show supported sites
Adding a new site:
1. Add a registry entry in SITES (folder name, domains, handler, defaults).
2. If the site is a standard HTML <video> site, add a GENERIC_SITES entry.
3. Otherwise add a handler function under "Custom site handlers".
"""
import os
import sys
import re
import json
import asyncio
import argparse
import random
import time
from pathlib import Path
from urllib.parse import urlparse, unquote
from html import unescape
from concurrent.futures import ThreadPoolExecutor, as_completed
import threading
import queue
import io
if sys.platform == 'win32':
sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding='utf-8', errors='replace')
sys.stderr = io.TextIOWrapper(sys.stderr.buffer, encoding='utf-8', errors='replace')
import requests
from tqdm import tqdm
import scraper_core
import generic_downloader
ROOT = Path(__file__).resolve().parent
# ---------------------------------------------------------------------------
# Shared helpers
# ---------------------------------------------------------------------------
def format_size(bytes_count):
if bytes_count < 1024:
return f"{bytes_count} B"
elif bytes_count < 1024 * 1024:
return f"{bytes_count / 1024:.2f} KB"
elif bytes_count < 1024 * 1024 * 1024:
return f"{bytes_count / (1024 * 1024):.2f} MB"
else:
return f"{bytes_count / (1024 * 1024 * 1024):.2f} GB"
class MasterProgressTracker:
"""Run-wide master progress tracker that aggregates stats across all sources and creators."""
def __init__(self):
self.total_files = 0
self.processed_files = 0
self.skipped_files = 0
self.skipped_bytes = 0
self.downloaded_files = 0
self.downloaded_bytes = 0
self.start_time = time.time()
self.lock = threading.Lock()
self.bar = tqdm(
total=0,
desc="Master Progress",
position=0,
leave=True,
ncols=100,
bar_format="{desc}: {n_fmt}/{total_fmt} |{bar:54}| {percentage:.2f}%",
)
self.text_bar = tqdm(
total=0,
position=1,
leave=True,
ncols=120,
desc="Skipped: 0 files equaling 0 B | Downloaded: 0 files equaling 0 B (Avg: 0.0 B/s)",
bar_format="{desc}",
)
def add_total(self, n: int = 1):
with self.lock:
self.total_files += n
self.bar.total = self.total_files
self.bar.refresh()
def record_skip(self, size: int = 0):
with self.lock:
self.skipped_files += 1
self.skipped_bytes += size
self.processed_files = self.skipped_files + self.downloaded_files
self._refresh_bars()
def record_download(self, size: int = 0):
with self.lock:
self.downloaded_files += 1
self.downloaded_bytes += size
self.processed_files = self.skipped_files + self.downloaded_files
self._refresh_bars()
def _refresh_bars(self):
self.bar.total = self.total_files
self.bar.n = self.processed_files
self.bar.refresh()
elapsed = time.time() - self.start_time
avg_speed = self.downloaded_bytes / elapsed if elapsed > 0 else 0
text = (
f"Skipped: {self.skipped_files} files equaling {format_size(self.skipped_bytes)} | "
f"Downloaded: {self.downloaded_files} files equaling {format_size(self.downloaded_bytes)} "
f"(Avg: {format_size(avg_speed)}/s)"
)
self.text_bar.set_description_str(text)
self.text_bar.refresh()
def close(self):
with self.lock:
self.bar.close()
self.text_bar.close()
def format_status(self) -> str:
with self.lock:
elapsed = time.time() - self.start_time
avg_speed = self.downloaded_bytes / elapsed if elapsed > 0 else 0
pct = (self.processed_files / self.total_files * 100) if self.total_files > 0 else 0.0
line1 = f"Master Progress: {self.processed_files}/{self.total_files} ({pct:.2f}%)"
line2 = (
f"Skipped: {self.skipped_files} files equaling {format_size(self.skipped_bytes)} | "
f"Downloaded: {self.downloaded_files} files equaling {format_size(self.downloaded_bytes)} "
f"(Avg: {format_size(avg_speed)}/s)"
)
return f"{line1}\n{line2}"
def print_summary(self):
with self.lock:
elapsed = time.time() - self.start_time
avg_speed = self.downloaded_bytes / elapsed if elapsed > 0 else 0
mins, secs = divmod(int(elapsed), 60)
hours, mins = divmod(mins, 60)
time_str = f"{hours:02d}:{mins:02d}:{secs:02d}" if hours else f"{mins:02d}:{secs:02d}"
pct = (self.processed_files / self.total_files * 100) if self.total_files > 0 else 0.0
sep = "=" * 80
tqdm.write(f"\n{sep}")
tqdm.write(" MASTER DOWNLOAD RUN SUMMARY")
tqdm.write(sep)
tqdm.write(f"Total Targets Tracked : {self.total_files} files (Processed: {self.processed_files}, {pct:.2f}%)")
tqdm.write(
f"Skipped : {self.skipped_files} files equaling {format_size(self.skipped_bytes)}"
)
tqdm.write(
f"Downloaded : {self.downloaded_files} files equaling {format_size(self.downloaded_bytes)} "
f"(Avg: {format_size(avg_speed)}/s)"
)
tqdm.write(f"Total Run Time : {time_str}")
tqdm.write(f"{sep}\n")
MASTER_TRACKER = MasterProgressTracker()
class OverallProgressTracker:
def __init__(self, total=0, desc="Overall Progress", master: MasterProgressTracker = None):
self.bar = tqdm(
total=total,
desc=desc,
position=2,
leave=True,
ncols=100,
bar_format="{desc}: {n_fmt}/{total_fmt} |{bar:54}| {percentage:.2f}%",
)
self.text_bar = tqdm(
total=0,
position=3,
leave=True,
ncols=120,
desc="Skipped: 0 files equaling 0 B | Downloaded: 0 files equaling 0 B (Avg: 0.0 B/s)",
bar_format="{desc}",
)
self.master = master if master is not None else MASTER_TRACKER
self.skipped_files = 0
self.skipped_bytes = 0
self.downloaded_files = 0
self.downloaded_bytes = 0
self.start_time = time.time()
self.lock = threading.Lock()
if total > 0 and self.master:
self.master.add_total(total)
@property
def total(self):
return self.bar.total
@total.setter
def total(self, value):
old_val = self.bar.total or 0
self.bar.total = value
diff = value - old_val
if diff > 0 and self.master:
self.master.add_total(diff)
def refresh(self):
self.bar.refresh()
self.text_bar.refresh()
def update(self, n=1):
self.bar.update(n)
def close(self):
self.bar.close()
self.text_bar.close()
def record_skip(self, size: int = 0):
with self.lock:
self.skipped_files += 1
self.skipped_bytes += size
self._refresh_text()
self.bar.update(1)
if self.master:
self.master.record_skip(size)
def record_download(self, size: int = 0, name: str = ""):
with self.lock:
self.downloaded_files += 1
self.downloaded_bytes += size
self._refresh_text()
self.bar.update(1)
if name:
tqdm.write(f" [FINISHED] '{name}' downloaded successfully ({format_size(size)}).")
if self.master:
self.master.record_download(size)
def _refresh_text(self):
elapsed = time.time() - self.start_time
avg_speed = self.downloaded_bytes / elapsed if elapsed > 0 else 0
text = (
f"Skipped: {self.skipped_files} files equaling {format_size(self.skipped_bytes)} | "
f"Downloaded: {self.downloaded_files} files equaling {format_size(self.downloaded_bytes)} "
f"(Avg: {format_size(avg_speed)}/s)"
)
self.text_bar.set_description_str(text)
class CheckResult(int):
def __new__(cls, is_complete: bool, size: int):
val = 1 if is_complete else 0
obj = super().__new__(cls, val)
obj.is_complete = is_complete
obj.size = size
return obj
def __bool__(self):
return self.is_complete
def __iter__(self):
yield self.is_complete
yield self.size
FILTERED_WORDS = {
"shit", "scat", "shitty", "poop", "horseshit", "bullshit", "cowshit",
"shitting", "crap", "feces", "dung", "pungpile", "pungheap", "pooping",
"crapping", "manure", "turd", "turds",
}
def is_filtered(url: str) -> bool:
lowered = url.lower()
return any(w in lowered for w in FILTERED_WORDS)
def dedupe(items):
seen = set()
out = []
for it in items:
if it not in seen:
seen.add(it)
out.append(it)
return out
def safe_mkdir(path: Path, retries: int = 5, delay: float = 2.0) -> Path:
"""mkdir with retries to handle transient SMB/network share I/O errors (e.g. WinError 1117).
If the directory entry on the remote filesystem is corrupted and permanently causes
WinError 1117, falls back to an alternative directory name (e.g. name_redgifs)
so downloads can proceed without crashing the entire run.
"""
for attempt in range(retries):
try:
path.mkdir(parents=True, exist_ok=True)
return path
except OSError as e:
if attempt < retries - 1:
tqdm.write(f" [!] Disk/Network I/O error creating {path.name} (attempt {attempt+1}/{retries}): {e}. Retrying in {delay}s...")
time.sleep(delay)
else:
# Persistent I/O error on this specific directory entry
win_err = getattr(e, 'winerror', None)
if win_err == 1117 or '1117' in str(e):
for suffix in ['_redgifs', '_dir', f'_{int(time.time())}']:
fallback = path.parent / f"{path.name}{suffix}"
try:
fallback.mkdir(parents=True, exist_ok=True)
tqdm.write(
f" [!] WinError 1117 persisted for '{path.name}' (filesystem entry corrupted). "
f"Using fallback directory '{fallback.name}' instead."
)
return fallback
except Exception:
continue
raise
def safe_append_text(path: Path, text: str, retries: int = 5, delay: float = 2.0):
"""Append text to a file with retries for transient network drive I/O errors."""
for attempt in range(retries):
try:
with open(path, 'a', encoding='utf-8') as f:
f.write(text)
return
except OSError as e:
if attempt < retries - 1:
time.sleep(delay)
else:
raise
async def get_cloudflare_cookies(scraper, url):
"""Use Playwright to solve one Cloudflare challenge and return browser cookies."""
page = await scraper.new_page()
try:
try:
await scraper.resolve_page(page, url)
except Exception as e:
tqdm.write(f" Warning: Cloudflare resolve_page failed: {e}. Trying to proceed with currently gathered cookies.")
raw_cookies = await scraper.context.cookies()
return {c["name"]: c["value"] for c in raw_cookies}
finally:
await page.close()
def get_remote_file_size(url: str, headers: dict = None, cookies: dict = None) -> int:
"""Fetch the content-length of the remote video file with HEAD/GET."""
req_headers = {
"User-Agent": scraper_core.USER_AGENT,
"Accept": "*/*",
}
if headers:
req_headers.update(headers)
if cookies:
try:
from curl_cffi import requests as curl_req
# Try HEAD first
try:
resp = curl_req.head(
url, impersonate="chrome", headers=req_headers,
cookies=cookies, timeout=8,
)
cl = int(resp.headers.get("content-length", 0))
if cl > 0:
return cl
except Exception:
pass
# Stream GET fallback
with curl_req.get(
url, impersonate="chrome", headers=req_headers,
cookies=cookies, stream=True, timeout=8,
) as resp:
if resp.status_code == 200:
return int(resp.headers.get("content-length", 0))
except Exception:
pass
try:
# Try HEAD first
try:
resp = requests.head(url, headers=req_headers, timeout=8, cookies=cookies, allow_redirects=True)
cl = int(resp.headers.get("content-length", 0))
if cl > 0:
return cl
except Exception:
pass
# Stream GET fallback
with requests.get(url, headers=req_headers, stream=True, timeout=8, cookies=cookies) as resp:
if resp.status_code == 200:
return int(resp.headers.get("content-length", 0))
except Exception:
pass
return 0
def check_existing_file_sync(dest_path: Path, src: str, headers: dict = None, cookies: dict = None) -> CheckResult:
"""Check integrity of an existing file against remote file size synchronously.
Returns CheckResult(True, local_size) if file exists and is complete (skip download).
Returns CheckResult(False, 0) if file does not exist, or if it is incomplete/corrupted (trigger redownload).
Prints messages describing what is happening in both cases.
"""
if not dest_path.exists():
return CheckResult(False, 0)
label = f"{dest_path.parent.name}/{dest_path.name}"
local_size = dest_path.stat().st_size
if local_size == 0:
tqdm.write(f" [OVERWRITE] '{label}' is incomplete (0 bytes). Overwriting and redownloading...")
try:
dest_path.unlink()
except Exception:
pass
return CheckResult(False, 0)
remote_size = get_remote_file_size(src, headers, cookies)
if remote_size > 0:
if local_size < remote_size:
tqdm.write(
f" [OVERWRITE] '{label}' is incomplete "
f"(local: {format_size(local_size)}, expected: {format_size(remote_size)}). Overwriting..."
)
try:
dest_path.unlink()
except Exception:
pass
return CheckResult(False, 0)
else:
tqdm.write(
f" [SKIP] '{label}' already exists and is complete ({format_size(local_size)})."
)
return CheckResult(True, local_size)
else:
# Fallback if Content-Length header is not provided by server
tqdm.write(
f" [SKIP] '{label}' already downloaded ({format_size(local_size)}, remote size unavailable)."
)
return CheckResult(True, local_size)
async def check_existing_file(dest_path: Path, src: str, headers: dict = None, cookies: dict = None) -> CheckResult:
"""Check integrity of an existing file against remote file size asynchronously."""
return await asyncio.to_thread(check_existing_file_sync, dest_path, src, headers, cookies)
# ---------------------------------------------------------------------------
# Site registry
# ---------------------------------------------------------------------------
# handler: "generic" -> GENERIC_SITES entry drives it via generic_downloader;
# otherwise a dedicated function in "Custom site handlers" below.
SITES = {
"chezcathy": {
"folder": "chezcathy",
"domains": ["chezcathy.com"],
"handler": "chezcathy",
"default_concurrency": 3,
},
"luxuretv": {
"folder": "luxuretv",
"domains": ["luxuretv.com"],
"handler": "luxuretv",
"default_concurrency": 3,
},
"motherless": {
"folder": "motherless",
"domains": ["motherless.com"],
"handler": "generic",
"default_concurrency": 3,
},
"pmvhaven": {
"folder": "pmvhaven",
"domains": ["pmvhaven.com"],
"handler": "generic",
"default_concurrency": 3,
},
"pornzoo.love": {
"folder": "pornzoo.love",
"domains": ["pornzoo.love"],
"handler": "pornzoo",
"default_concurrency": 2,
},
"pornhub": {
"folder": "pornhub",
"domains": ["pornhub.com"],
"handler": "pornhub",
"default_concurrency": 2,
},
"redgifs": {
"folder": "redgifs",
"domains": ["redgifs.com"],
"handler": "redgifs",
"default_concurrency": 5,
},
"rule34video": {
"folder": "rule34video",
"domains": ["rule34video.com"],
"handler": "generic",
"default_concurrency": 3,
},
"thisvid": {
"folder": "thisvid",
"domains": ["thisvid.com"],
"handler": "generic",
"default_concurrency": 3,
},
"voyeurflash": {
"folder": "voyeurflash",
"domains": ["voyeurflash.com"],
"handler": "generic",
"default_concurrency": 3,
},
"xhamster": {
"folder": "xhamster",
"domains": ["xhamster.com", "xhamster.xxx", "xhamster18.com"],
"handler": "generic",
"default_concurrency": 3,
},
"xxxvideoszoo": {
"folder": "xxxvideoszoo",
"domains": ["xxxvideoszoo.com"],
"handler": "generic",
"default_concurrency": 3,
},
"zootube.vip": {
"folder": "zootube.vip",
"domains": ["zootube.vip"],
"handler": "zootubevip",
"default_concurrency": 3,
},
"zootube1": {
"folder": "zootube1",
"domains": ["zootube1.com"],
"handler": "zootube1",
"default_concurrency": 3,
},
"tickzoo.tv": {
"folder": "tickzoo.tv",
"domains": ["tickzoo.tv"],
"handler": "tickzoo",
"default_concurrency": 1,
},
"file.al": {
"folder": "file.al",
"domains": ["file.al"],
"handler": "fileal",
"default_concurrency": 1,
},
}
# ---------------------------------------------------------------------------
# Generic site configs (standard HTML <video> player sites)
# ---------------------------------------------------------------------------
def _motherless_is_video_link(url):
# Match Motherless video URL patterns like:
# https://motherless.com/AC493CD
# https://motherless.com/g/group_name/AC493CD
path = urlparse(url).path.rstrip("/")
if not path:
return False
last_part = path.split("/")[-1]
# Motherless IDs are typically 7 or 8 characters of uppercase letters and numbers
return bool(re.match(r"^[A-Z0-9]{6,9}$", last_part))
def _pmvhaven_is_video_link(url):
# Match watch/video URLs like: /video/slug_id, /videos/slug_id, or /watch/slug_id
return "/video" in url or "/watch" in url
def _rule34video_is_video_link(url):
# Match Rule34Video paths like: https://rule34video.com/video/12345/slug/
return "/video/" in url
def _thisvid_is_video_link(url):
# Match ThisVid paths like: https://thisvid.com/video/12345/
return "/video/" in url
def _voyeurflash_is_video_link(url):
# Match typical voyeurflash video paths like: https://voyeurflash.com/video/12345/xyz/
return "/video/" in url or "/watch/" in url
def _xhamster_is_video_link(url):
# Match standard xHamster video pages: contains /videos/ and doesn't point to user grids/channels
return (
"/videos/" in url
and "/users/" not in url
and "/channels/" not in url
and "/creators/" not in url
)
def _xxxvideoszoo_is_video_link(url):
return "/v/" in url or "/video/" in url or "/watch/" in url
def _zootube1_is_video_link(url):
return "/allvideos/" in url or "/video/" in url or "/watch/" in url
GENERIC_SITES = {
"motherless": {
"site_name": "Motherless",
"is_video_link": _motherless_is_video_link,
"next_page_selector": "a:has-text('Next'), a.next, a.pagination-next",
"video_selector": "video",
"uploader_eval_js": """() => {
const memberLink = document.querySelector('a[href*="/members/"]');
if (memberLink) return memberLink.innerText.trim();
const uploadText = document.querySelector('.upload-info');
if (uploadText) {
const m = uploadText.innerText.match(/by\\s+(\\S+)/i);
if (m) return m[1];
}
return 'unknown';
}""",
},
"pmvhaven": {
"site_name": "PMVHaven",
"is_video_link": _pmvhaven_is_video_link,
"next_page_selector": "a:has-text('Next'), .next, a.pagination-next",
"video_selector": "video",
"uploader_eval_js": """() => {
const el = Array.from(document.querySelectorAll('a[href*="/user/"], a[href*="/users/"], a[href*="/profile/"], .uploader-name'))
.find(a => a.innerText.trim() !== '');
if (el) {
const h3 = el.querySelector('h3');
if (h3) {
const clone = h3.cloneNode(true);
const spans = clone.querySelectorAll('span');
spans.forEach(s => s.remove());
return clone.textContent.trim();
}
return el.innerText.replace(/uploader/gi, '').trim();
}
return 'unknown';
}""",
},
"rule34video": {
"site_name": "Rule34Video",
"is_video_link": _rule34video_is_video_link,
"next_page_selector": "a.next, a:has-text('Next'), .pagination-next",
"video_selector": "video",
"uploader_eval_js": """() => {
const el = document.querySelector('a[href*="/user/"], a[href*="/profile/"], .video-uploader a');
return el ? el.innerText.trim() : 'unknown';
}""",
},
"thisvid": {
"site_name": "ThisVid",
"is_video_link": _thisvid_is_video_link,
"next_page_selector": "a:has-text('Next'), .next, a.pagination-next",
"video_selector": "video",
"uploader_eval_js": """() => {
const el = document.querySelector('a[href*="/members/"], a[href*="/profile/"], .username');
return el ? el.innerText.trim() : 'unknown';
}""",
},
"voyeurflash": {
"site_name": "VoyeurFlash",
"is_video_link": _voyeurflash_is_video_link,
"next_page_selector": "a:has-text('Next'), .next, a.pagination-next",
"video_selector": "video",
"uploader_eval_js": """() => {
const el = document.querySelector('a[href*="/members/"], a[href*="/user/"], a[href*="/profile/"]');
return el ? el.innerText.trim() : 'unknown';
}""",
},
"xhamster": {
"site_name": "xHamster",
"is_video_link": _xhamster_is_video_link,
"next_page_selector": "a:has-text('Next'), .next, a[rel='next']",
"video_selector": "video",
"uploader_eval_js": """() => {
// 1. Try channel link
const channelLink = document.querySelector('a[href*="/channels/"]');
if (channelLink && channelLink.innerText.trim()) {
return channelLink.innerText.trim().replace(/\\n/g, ' ').trim();
}
// 2. Try user link
const userLink = document.querySelector('.entity-author-container__name, a[href*="/users/"]');
if (userLink && userLink.innerText.trim()) {
return userLink.innerText.trim();
}
// 3. Try creator logo or class
const creatorLink = document.querySelector('a[href*="/creators/"] .video-uploader__name, .video-uploader__name');
if (creatorLink && creatorLink.innerText.trim()) {
return creatorLink.innerText.trim();
}
return 'unknown';
}""",
},
"xxxvideoszoo": {
"site_name": "XXXVideosZoo",
"is_video_link": _xxxvideoszoo_is_video_link,
"next_page_selector": "a:has-text('Next'), .next, a.pagination-next",
"video_selector": "video",
"uploader_eval_js": """() => {
const el = document.querySelector('a[href*="/members/"], a[href*="/user/"], a[href*="/profile/"]');
return el ? el.innerText.trim() : 'unknown';
}""",
},
}
# ---------------------------------------------------------------------------
# Input classification: URL -> site, bare token -> RedGIFs creator
# ---------------------------------------------------------------------------
def looks_like_url(token: str) -> bool:
return "://" in token or token.lower().startswith("www.")
def find_site_for_url(url: str):
"""Return the site key that owns a URL's host, or None."""
try:
parsed = urlparse(url)
host = parsed.netloc.lower()
except ValueError:
return None
if not host:
# Scheme-less input like "www.xhamster.com/xyz" - retry with a scheme
try:
host = urlparse("https://" + url).netloc.lower()
except ValueError:
return None
if not host:
return None
for key, cfg in SITES.items():
for domain in cfg["domains"]:
if host == domain or host.endswith("." + domain):
return key
return None
def classify_target(token: str):
"""Classify one CLI token -> (site_key, target).
URLs are routed by hostname; any non-URL token is treated as a RedGIFs
creator username (matching the original redgifs_creator_downloader.py
behaviour, which accepted bare creator names).
"""
if looks_like_url(token):
site = find_site_for_url(token)
if site is None:
raise ValueError(
"no supported website matched (run with --list-sites to see supported domains)"
)
if site == "redgifs":
# RedGIFs accepts /watch/<id> URLs, media URLs, and bare creator names.
return site, token
return site, token
return "redgifs", token
def resolve_targets(targets_args):
"""Expand CLI args (URLs / creator names / .txt files) into {site_key: [targets]}.
Files are read line by line and EVERY whitespace-separated token is
classified independently, so a single file may mix URLs and creators.
"""
groups = {}
warnings = []
for arg in targets_args:
p = Path(arg)
try:
is_file = p.is_file()
except OSError:
is_file = False
if is_file:
try:
with open(p, "r", encoding="utf-8") as f:
tokens = [
t for line in f if not line.lstrip().startswith("#")
for t in line.strip().split()
]
except Exception as e:
warnings.append(f"Error reading file '{arg}': {e}")
continue
else:
tokens = [arg]
for tok in tokens:
if not tok or tok.startswith("#"):
continue
try:
site, target = classify_target(tok)
except ValueError as e:
warnings.append(f"Skipping '{tok}': {e}")
continue
groups.setdefault(site, []).append(target)
return groups, warnings
# ---------------------------------------------------------------------------
# Custom site handlers
# ---------------------------------------------------------------------------
# ----- ChezCathy ------------------------------------------------------------
def _chezcathy_extract_video_links(html: str) -> list:
"""Extract video links matching player or adult page patterns on chezcathy."""
# Matches URLs like: /extreme-sex/player/35738/...html or /adult/769439/...html
patterns = [
r'href="(/[^"]*?/player/\d+/[^"]*?\.html)"',
r'href="(/adult/\d+/[^"]*?\.html)"',
r'href="(https?://en\.chezcathy\.com/[^"]*?/player/\d+/[^"]*?\.html)"',
r'href="(https?://en\.chezcathy\.com/adult/\d+/[^"]*?\.html)"'
]
links = []
for pattern in patterns:
for match in re.findall(pattern, html):
if match.startswith("/"):
url = f"https://en.chezcathy.com{match}"
else:
url = match
if url not in links:
links.append(url)
return links
async def _chezcathy_scrape_playlist(playlist_url, cookies, queue, counter, overall_bar, scraper):
"""Scrape all video URLs from a playlist/user page using curl_cffi with browser cookies."""
page_num = 1
while True:
# Construct pagination URL if page > 1
if page_num > 1:
base_no_slash = playlist_url.rstrip("/")
if "?" in playlist_url:
url = f"{base_no_slash}&page={page_num}"
else:
url = f"{base_no_slash}?page={page_num}"
else:
url = playlist_url
tqdm.write(f" Scraping page {page_num}: {url}")
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.chezcathy.com/")
if not html:
tqdm.write(f" Static fetch failed. Trying Playwright for playlist page...")
page = await scraper.new_page()
try:
await scraper.resolve_page(page, url)
html = await page.content()
except Exception as pe:
tqdm.write(f" x Playwright failed to fetch playlist page: {pe}")
finally:
await page.close()
if not html:
tqdm.write(f" Failed to fetch page {page_num}")
break
video_urls = _chezcathy_extract_video_links(html)
if not video_urls:
tqdm.write(f" No video links found on page.")
break
tqdm.write(f" Found {len(video_urls)} video links")
for video_url in video_urls:
if is_filtered(video_url):
tqdm.write(f" x Filtered: {video_url}")
continue
await queue.put(video_url)
counter["total"] += 1
overall_bar.total = counter["total"]
overall_bar.refresh()
# Check for pagination next page
if not scraper_core.has_next_page(html) and f"page={page_num}" not in html:
break
page_num += 1
await asyncio.sleep(random.uniform(1.0, 2.5))
def _chezcathy_extract_video_info(html: str, url: str):
"""Extract video source and uploader/creator name from ChezCathy HTML."""
# Attempt to extract video source using standard scraper_core extractors
src, uploader = scraper_core.extract_video_info_from_html(html)
# ChezCathy specific creator/uploader extraction
# Look for user profile links like: href="/user/7115/ryan_thomas" or /profile/...
user_match = re.search(r'href="/user/\d+/([^"/]+)"', html)
if user_match:
uploader = user_match.group(1)
if not uploader or uploader == "unknown":
uploader = "unknown_creator"
uploader = scraper_core.clean_filename(uploader)
return src, uploader
async def _chezcathy_worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper, download_dir):
"""Worker: fetch video page HTML, extract source, and download."""
while True:
url = await queue.get()
if url is None:
queue.task_done()
break
try:
if is_filtered(url):
tqdm.write(f" x Filtered: {url}")
continue
await asyncio.sleep(random.uniform(0.3, 0.8))
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.chezcathy.com/")
# Fallback to Playwright if static fetch fails
if not html:
tqdm.write(f" Static fetch failed for {url}. Fetching page via Playwright...")
page = await scraper.new_page()
try:
await scraper.resolve_page(page, url)
html = await page.content()
except Exception as pe:
tqdm.write(f" x Playwright failed to fetch HTML for {url}: {pe}")
finally:
await page.close()
if not html:
tqdm.write(f" x Failed to fetch HTML for {url}")
continue
src, uploader = _chezcathy_extract_video_info(html, url)
# If dynamic rendering is required to capture the media source
if not src:
tqdm.write(f" * Static extraction failed for {url}. Trying Playwright...")
page = await scraper.new_page()
try:
src = await scraper.extract_media_source(page, url)
except Exception as pe:
tqdm.write(f" x Playwright extraction failed: {pe}")
finally:
await page.close()
if not src:
tqdm.write(f" x Skipping {url} - no video source found")
continue
# Standardize filename based on URL slug and ID
match = re.search(r'/(\d+)/([^/]+)\.html', url)
if match:
video_id = match.group(1)
slug = match.group(2)
filename = f"{scraper_core.clean_filename(slug)}-{video_id}.mp4"
else:
stem = url.rstrip(".html").rstrip("/").rsplit("/", 1)[-1]
filename = scraper_core.clean_filename(stem) + ".mp4"
dest_path = download_dir / uploader / filename
downloaded_bytes = 0
if skip_existing:
is_complete, existing_bytes = await check_existing_file(dest_path, src, {"Referer": url}, cookies)
if is_complete:
overall_bar.record_skip(existing_bytes)
continue
pos = bar_pool.acquire()
if pos is None:
pos = 4
success = await asyncio.to_thread(
scraper_core.download_file,
src, dest_path, {"Referer": url}, pos, cookies=cookies
)
bar_pool.release(pos)
if success:
if dest_path.is_file():
downloaded_bytes = dest_path.stat().st_size
label = f"{dest_path.parent.name}/{dest_path.name}"
overall_bar.record_download(downloaded_bytes, name=label)
scraper_core.append_log(download_dir / "urls.txt", url)
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
else:
overall_bar.update(1)
except Exception as e:
tqdm.write(f" Error processing {url}: {e}")
overall_bar.update(1)
finally:
queue.task_done()
async def _chezcathy_producer(urls, cookies, queue, counter, overall_bar, scraper):
"""Feed input URLs (video pages or playlists/profile pages) into the queue."""
for url in urls:
# Check if URL is a direct video page
if ("/player/" in url or "/adult/" in url) and url.endswith(".html"):
if is_filtered(url):
tqdm.write(f" x Filtered (not queued): {url}")
continue
tqdm.write(f"Queuing direct video URL: {url}")
await queue.put(url)
counter["total"] += 1
overall_bar.total = counter["total"]
overall_bar.refresh()
else:
tqdm.write(f"Scraping user/playlist: {url}")
try:
await _chezcathy_scrape_playlist(url, cookies, queue, counter, overall_bar, scraper)
except Exception as e:
tqdm.write(f" Error scraping playlist {url}: {e}")
await asyncio.sleep(random.uniform(1.0, 2.0))
async def chezcathy_process_urls(download_dir, urls, concurrency, skip_existing):
scraper = scraper_core.PlaywrightScraper()
await scraper.start()
tqdm.write("Solving Cloudflare cookies...")
first_url = urls[0]
cookies = await get_cloudflare_cookies(scraper, first_url)
tqdm.write(f"Got cookies: {list(cookies.keys())}")
queue = asyncio.Queue()
counter = {"total": 0}
overall_bar = OverallProgressTracker(total=0, desc="Overall Progress")
bar_pool = scraper_core.BarPositionPool(concurrency)
bar_pool.available = [p + 3 for p in bar_pool.available]
workers = [
asyncio.create_task(_chezcathy_worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper, download_dir))
for _ in range(concurrency)
]
await _chezcathy_producer(urls, cookies, queue, counter, overall_bar, scraper)
for _ in range(concurrency):
await queue.put(None)
await asyncio.gather(*workers)
overall_bar.close()
await scraper.close()
sys.stdout.write("\n" * (concurrency + 4))
sys.stdout.flush()
# ----- LuxureTV -------------------------------------------------------------
async def _luxuretv_scrape_playlist(playlist_url, cookies, queue, counter, overall_bar):
"""Scrape all video URLs from a playlist using curl_cffi with browser cookies."""
base_check = playlist_url.lower()
page_num = 1
while True:
if page_num > 1:
base_no_slash = playlist_url.rstrip("/")
if "?" in playlist_url:
url = f"{base_no_slash}&page={page_num}"
elif "uploads-by-user" in base_check or "playlist-by-user" in base_check or "videos-commented-by-user" in base_check or "/channels/" in base_check:
url = f"{base_no_slash}/page{page_num}.html"
else:
url = f"{base_no_slash}?page={page_num}"
else:
url = playlist_url
tqdm.write(f" Scraping playlist page {page_num}: {url}")
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.luxuretv.com/")
if not html:
tqdm.write(f" Failed to fetch page {page_num}")
break
video_urls = scraper_core.extract_video_links_from_html(html)
if not video_urls:
tqdm.write(f" No video links found on page.")
break
tqdm.write(f" Found {len(video_urls)} unique video links")
for video_url in video_urls:
if is_filtered(video_url):
tqdm.write(f" x Filtered: {video_url}")
continue
await queue.put(video_url)
counter["total"] += 1
overall_bar.total = counter["total"]
overall_bar.refresh()
if not scraper_core.has_next_page(html):
tqdm.write(f" No 'Next' button found, done.")
break
page_num += 1
await asyncio.sleep(random.uniform(1.0, 2.5))
async def _luxuretv_worker(queue, cookies, skip_existing, bar_pool, overall_bar, download_dir):
"""Worker: fetch video page HTML via curl_cffi, extract source, download."""
while True:
url = await queue.get()
if url is None:
queue.task_done()
break
try:
if is_filtered(url):
tqdm.write(f" x Filtered: {url}")
continue
# Add a short polite delay to avoid rate limits during bulk scraping
await asyncio.sleep(random.uniform(0.3, 0.8))
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.luxuretv.com/")
if not html:
tqdm.write(f" x Failed to fetch {url}")
continue
src, uploader = scraper_core.extract_video_info_from_html(html)
if not src:
tqdm.write(f" x Skipping {url} - no video source found")
continue
stem = url.rstrip(".html").rstrip("/").rsplit("/", 1)[-1]
filename = scraper_core.clean_filename(stem) + ".mp4"
dest_path = download_dir / uploader / filename
downloaded_bytes = 0
if skip_existing:
is_complete, existing_bytes = await check_existing_file(dest_path, src, {"Referer": url}, cookies)
if is_complete:
overall_bar.record_skip(existing_bytes)
continue
pos = bar_pool.acquire()
if pos is None:
pos = 4
success = await asyncio.to_thread(
scraper_core.download_file,
src, dest_path, {"Referer": url}, pos, cookies=cookies
)
bar_pool.release(pos)
if success:
if dest_path.is_file():
downloaded_bytes = dest_path.stat().st_size
label = f"{dest_path.parent.name}/{dest_path.name}"
overall_bar.record_download(downloaded_bytes, name=label)
scraper_core.append_log(download_dir / "urls.txt", url)
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
else:
overall_bar.update(1)
except Exception as e:
tqdm.write(f" Error processing {url}: {e}")
overall_bar.update(1)
finally:
queue.task_done()
async def _luxuretv_producer(urls, cookies, queue, counter, overall_bar):
"""Scrape playlists and feed discovered video URLs into the download queue immediately."""
for url in urls:
if "/videos/" in url and ".html" in url:
if is_filtered(url):
tqdm.write(f" x Filtered (not queued): {url}")
continue
tqdm.write(f"Queuing direct video URL: {url}")
await queue.put(url)
counter["total"] += 1
overall_bar.total = counter["total"]
overall_bar.refresh()
else:
tqdm.write(f"Scraping playlist: {url}")
try:
await _luxuretv_scrape_playlist(url, cookies, queue, counter, overall_bar)
except Exception as e:
tqdm.write(f" Error scraping {url}: {e}")
await asyncio.sleep(random.uniform(1.0, 2.0))
async def luxuretv_process_urls(download_dir, urls, concurrency, skip_existing):
scraper = scraper_core.PlaywrightScraper()
await scraper.start()
tqdm.write("Solving initial Cloudflare challenge for cookies...")
first_url = urls[0]
cookies = await get_cloudflare_cookies(scraper, first_url)
tqdm.write(f"Got {len(cookies)} cookies: {list(cookies.keys())}")
await scraper.close()
queue = asyncio.Queue()
counter = {"total": 0}
overall_bar = OverallProgressTracker(total=0, desc="Overall Progress")
bar_pool = scraper_core.BarPositionPool(concurrency)
bar_pool.available = [p + 3 for p in bar_pool.available]
workers = [
asyncio.create_task(_luxuretv_worker(queue, cookies, skip_existing, bar_pool, overall_bar, download_dir))
for _ in range(concurrency)
]
await _luxuretv_producer(urls, cookies, queue, counter, overall_bar)
for _ in range(concurrency):
await queue.put(None)
await asyncio.gather(*workers)
overall_bar.close()
sys.stdout.write("\n" * (concurrency + 4))
sys.stdout.flush()
# ----- PornZoo.Love ---------------------------------------------------------
def _pornzoo_extract_video_links(html: str) -> list:
"""Extract video links matching player or video page patterns on PornZoo.Love."""
# Matches URLs like: /video/slug or /en/video/slug
patterns = [
r'href="(/video/[^"/#?]+)"',
r'href="(/[^/]+/video/[^"/#?]+)"',
r'href="(https?://pornzoo\.love/video/[^"/#?]+)"',
r'href="(https?://pornzoo\.love/[^/]+/video/[^"/#?]+)"'
]
links = []
for pattern in patterns:
for match in re.findall(pattern, html):
if match.startswith("/"):
url = f"https://pornzoo.love{match}"
else:
url = match
if url not in links and "/embed/" not in url:
links.append(url)
return links
async def _pornzoo_scrape_playlist(playlist_url, cookies, queue, counter, overall_bar, scraper):
"""Scrape all video URLs from a playlist/user page."""
page_num = 1
while True:
url = playlist_url
if page_num > 1:
# Detect pagination pattern if any (e.g. ?page=X or /page/X)
if "?" in playlist_url:
url = f"{playlist_url}&page={page_num}"
else:
url = f"{playlist_url}?page={page_num}"
tqdm.write(f" Scraping page {page_num}: {url}")
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://pornzoo.love/")
if not html:
tqdm.write(f" Static fetch failed. Trying Playwright for playlist page...")
page = await scraper.new_page()
try:
await scraper.resolve_page(page, url)
html = await page.content()
except Exception as pe:
tqdm.write(f" x Playwright failed to fetch playlist page: {pe}")
finally:
await page.close()
if not html:
tqdm.write(f" Failed to fetch page {page_num}")
break
video_urls = _pornzoo_extract_video_links(html)
if not video_urls:
tqdm.write(f" No video links found on page.")
break
tqdm.write(f" Found {len(video_urls)} video links")
new_links = 0
for video_url in video_urls:
if is_filtered(video_url):
tqdm.write(f" x Filtered: {video_url}")
continue
await queue.put(video_url)
counter["total"] += 1
overall_bar.total = counter["total"]
overall_bar.refresh()
new_links += 1
# Check for next page indicator
if "Next" not in html and "next" not in html.lower():
break
if new_links == 0:
break
page_num += 1
await asyncio.sleep(random.uniform(1.0, 2.5))
def _zhvid_extract_video_info(html: str, url: str, base: str = "https://pornzoo.love"):
"""Extract direct video source, video ID, and uploader name from page HTML.
Shared by the /zhvid.php CMS sites (pornzoo.love, zootube.vip); `base` is the
site origin used to resolve root-relative sources.
Current site structure serves the source directly on the page:
<source type="video/mp4" src="/zhvid.php?vid=123456&type=MP4">
which 302-redirects to the actual MP4 on greymedia.org. Older pages used
/embed/<id> - kept as a fallback.
"""
source = None
video_id = None
uploader = "unknown_creator"
src_match = re.search(r'<source[^>]*src="([^"]+)"', html, re.I)
if src_match:
source = src_match.group(1).replace("&amp;", "&")
if not source:
video_match = re.search(r'<video[^>]*src="([^"]+)"', html, re.I)
if video_match:
source = video_match.group(1).replace("&amp;", "&")
if source:
vid_match = re.search(r'vid=(\d+)', source)
if vid_match:
video_id = vid_match.group(1)
if not video_id:
embed_match = re.search(r'/embed/(\d+)', html)
if embed_match:
video_id = embed_match.group(1)
if not video_id and source:
m = re.search(r'/(\d+)\.mp4', source)
if m:
video_id = m.group(1)
if not video_id:
video_id = "na"
if source:
if source.startswith("//"):
source = "https:" + source
elif source.startswith("/"):
source = base + source
elif video_id != "na":
source = f"{base}/zhvid.php?vid={video_id}&type=MP4"
for pat in (
r'href="[^"]*/user/([^"/]+)"',
r'href="[^"]*/members/([^"/]+)"',
r'href="[^"]*/profile/([^"/]+)"',
):
m = re.search(pat, html)
if m:
uploader = m.group(1)
break
uploader = scraper_core.clean_filename(uploader)
return video_id, source, uploader
async def _pornzoo_worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper, download_dir):
"""Worker: fetch video page, extract direct source URL, and download."""
while True:
url = await queue.get()
if url is None:
queue.task_done()
break
try:
if is_filtered(url):
tqdm.write(f" x Filtered: {url}")
continue
await asyncio.sleep(random.uniform(0.3, 0.8))
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://pornzoo.love/")
if not html:
tqdm.write(f" Static fetch failed for {url}. Fetching page via Playwright...")
page = await scraper.new_page()
try:
await scraper.resolve_page(page, url)
html = await page.content()
except Exception as pe:
tqdm.write(f" x Playwright failed to fetch HTML for {url}: {pe}")
finally:
await page.close()
if not html:
tqdm.write(f" x Failed to fetch HTML for {url}")
continue
video_id, src, uploader = _zhvid_extract_video_info(html, url)
if not src:
tqdm.write(f" x Skipping {url} - could not find video source")
continue
stem = url.rstrip("/").rsplit("/", 1)[-1]
filename = f"{scraper_core.clean_filename(stem)}-{video_id}.mp4"
dest_path = download_dir / uploader / filename
downloaded_bytes = 0
if skip_existing:
is_complete, existing_bytes = await check_existing_file(dest_path, src, {"Referer": url}, cookies)
if is_complete:
overall_bar.record_skip(existing_bytes)
continue
pos = bar_pool.acquire()
if pos is None:
pos = 4
success = await asyncio.to_thread(
scraper_core.download_file,
src, dest_path, {"Referer": url}, pos, cookies=cookies
)
bar_pool.release(pos)
if success:
if dest_path.is_file():
downloaded_bytes = dest_path.stat().st_size
label = f"{dest_path.parent.name}/{dest_path.name}"
overall_bar.record_download(downloaded_bytes, name=label)
scraper_core.append_log(download_dir / "urls.txt", url)
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
else:
overall_bar.update(1)
except Exception as e:
tqdm.write(f" Error processing {url}: {e}")
overall_bar.update(1)
finally:
queue.task_done()
async def _pornzoo_producer(urls, cookies, queue, counter, overall_bar, scraper):
"""Feed input URLs (video pages or playlists) into the queue."""
for url in urls:
if "/video/" in url:
if is_filtered(url):
tqdm.write(f" x Filtered (not queued): {url}")
continue
tqdm.write(f"Queuing direct video URL: {url}")
await queue.put(url)
counter["total"] += 1
overall_bar.total = counter["total"]
overall_bar.refresh()
else:
tqdm.write(f"Scraping user/playlist: {url}")
try:
await _pornzoo_scrape_playlist(url, cookies, queue, counter, overall_bar, scraper)
except Exception as e:
tqdm.write(f" Error scraping playlist {url}: {e}")
await asyncio.sleep(random.uniform(1.0, 2.0))
async def pornzoo_process_urls(download_dir, urls, concurrency, skip_existing):
scraper = scraper_core.PlaywrightScraper()
await scraper.start(headless=True)
tqdm.write("Solving Cloudflare cookies...")
first_url = urls[0]
cookies = await get_cloudflare_cookies(scraper, first_url)
tqdm.write(f"Got cookies: {list(cookies.keys())}")
queue = asyncio.Queue()
counter = {"total": 0}
overall_bar = OverallProgressTracker(total=0, desc="Overall Progress")
bar_pool = scraper_core.BarPositionPool(concurrency)
bar_pool.available = [p + 3 for p in bar_pool.available]
workers = [
asyncio.create_task(_pornzoo_worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper, download_dir))
for _ in range(concurrency)
]
await _pornzoo_producer(urls, cookies, queue, counter, overall_bar, scraper)
for _ in range(concurrency):
await queue.put(None)
await asyncio.gather(*workers)
overall_bar.close()
await scraper.close()
sys.stdout.write("\n" * (concurrency + 4))
sys.stdout.flush()
# ----- Zootube.vip ----------------------------------------------------------
def _zootubevip_is_video_link(url):
return "/video/" in url
async def _zootubevip_worker(queue, skip_existing, bar_pool, overall_bar, download_dir, base):
"""Worker: fetch the static video page, extract the /zhvid.php source, download it."""
while True:
url = await queue.get()
if url is None:
queue.task_done()
break
try:
if is_filtered(url):
tqdm.write(f" x Filtered: {url}")
continue
await asyncio.sleep(random.uniform(0.3, 0.8))
html = await asyncio.to_thread(scraper_core.curl_fetch, url, referer=f"{base}/")
if not html:
tqdm.write(f" x Failed to fetch HTML for {url}")
continue
video_id, src, uploader = _zhvid_extract_video_info(html, url, base)
if not src:
tqdm.write(f" x Skipping {url} - could not find video source")
continue
stem = url.rstrip("/").rsplit("/", 1)[-1]
filename = f"{scraper_core.clean_filename(stem)}-{video_id}.mp4"
dest_path = download_dir / uploader / filename
headers = {"Referer": url}
downloaded_bytes = 0
if skip_existing:
is_complete, existing_bytes = await check_existing_file(dest_path, src, headers, None)
if is_complete:
overall_bar.record_skip(existing_bytes)
continue
pos = bar_pool.acquire()
if pos is None:
pos = 4
success = await asyncio.to_thread(
scraper_core.download_file, src, dest_path, headers, pos, cookies=None
)
bar_pool.release(pos)
if success:
if dest_path.is_file():
downloaded_bytes = dest_path.stat().st_size
label = f"{dest_path.parent.name}/{dest_path.name}"
overall_bar.record_download(downloaded_bytes, name=label)
scraper_core.append_log(download_dir / "urls.txt", url)
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
else:
overall_bar.update(1)
except Exception as e:
tqdm.write(f" Error processing {url}: {e}")
overall_bar.update(1)
finally:
queue.task_done()
async def zootubevip_process_urls(download_dir, urls, concurrency, skip_existing):
base = "https://zootube.vip"
queue = asyncio.Queue()
overall_bar = OverallProgressTracker(total=0, desc="Overall Progress")
bar_pool = scraper_core.BarPositionPool(concurrency)
bar_pool.available = [p + 3 for p in bar_pool.available]
workers = [
asyncio.create_task(
_zootubevip_worker(queue, skip_existing, bar_pool, overall_bar, download_dir, base)
)
for _ in range(concurrency)
]
for url in urls:
if not _zootubevip_is_video_link(url):
tqdm.write(f" x Not a video URL: {url}")
continue
if is_filtered(url):
tqdm.write(f" x Filtered (not queued): {url}")
continue
tqdm.write(f"Queuing direct video URL: {url}")
await queue.put(url)
overall_bar.total += 1
overall_bar.refresh()
for _ in range(concurrency):
await queue.put(None)
await asyncio.gather(*workers)
overall_bar.close()
sys.stdout.write("\n" * (concurrency + 4))
sys.stdout.flush()
# ----- tickzoo.tv ------------------------------------------------------------
# Video pages have no <video> tag - they embed a 3rd-party player
# (veev.to / firestream.to / hgcloud.to / rubyvidhub.com) via an "Embed Code"
# <textarea id="iframe">, so each embed is resolved through headless Chrome.
_TICKZOO_AD_HOSTS = (
"dtscout", "lijit", "doubleclick", "googlesyndication", "adservice",
"pubmatic", "criteo", "taboola", "outbrain", "adsystem",
)
def _tickzoo_is_video_link(url):
# tickzoo.tv video pages look like https://tickzoo.tv/<slug>/
path = urlparse(url).path.strip("/")
return bool(path) and not path.split("/")[-1].endswith(
(".html", ".htm", ".php", ".txt", ".xml", ".json", ".css")
)
def _tickzoo_extract_embed(html: str) -> str:
"""Pull the video iframe src out of the <textarea id="iframe"> embed box."""
m = re.search(r"<textarea[^>]*id=[\"']iframe[\"'][^>]*>(.*?)</textarea>", html, re.S | re.I)
if not m:
return ""
inner = m.group(1)
m = re.search(r"<iframe\b[^>]*\bsrc\s*=\s*[\"']([^\"']+)[\"']", inner, re.I)
if not m:
m = re.search(r"<iframe\b[^>]*\bsrc\s*=\s*([^\"'>\s]+)", inner, re.I)
if not m:
return ""
src = unescape(m.group(1).strip())
if any(ad in src.lower() for ad in _TICKZOO_AD_HOSTS):
return ""
return src
def _tickzoo_extract_title(html: str) -> str:
m = re.search(r"<div[^>]*class=[\"']title[\"'][^>]*>(.*?)</div>", html, re.S)
if m:
txt = re.sub(r"<[^>]+>", "", m.group(1))
return unescape(txt).strip()
m = re.search(r"<meta[^>]*property=[\"']og:title[\"'][^>]*content=[\"']([^\"']*)", html, re.I)
return unescape(m.group(1)).strip() if m else ""
def _tickzoo_extract_uploader(html: str) -> str:
"""Studio tag ('folder X') if the page carries one, else a stable default."""
for m in re.finditer(r"folder\s*[:\-]?\s*([A-Za-z0-9][A-Za-z0-9_. \-\']{0,40})", html, re.I):
name = re.sub(r"[<>\"]", "", m.group(1)).strip().strip(".")
if name:
return scraper_core.clean_filename(name)
return "unknown"
def _tickzoo_score_media_url(url: str) -> int:
"""Rank candidate media URLs; higher wins. Prefer HLS masters, then HLS, then mp4."""
ul = url.lower()
if "blank.mp4" in ul:
return -1
if "master.m3u8" in ul or "/master" in ul:
return 100
if ".m3u8" in ul or "manifest" in ul:
return 90
if "veevcdn" in ul or "veev.to" in ul:
return 80
if "/stream/" in ul:
return 70
if ".mp4" in ul or ".webm" in ul:
return 50
return 60
async def _tickzoo_resolve_media(scraper, embed_url: str, referer_url: str) -> str:
"""Resolve the direct media URL from a tickzoo embed host via headless Chrome.
Handles every embed host seen on tickzoo.tv:
* firestream.to -> POST /api/videos/<id>/resolve returns JSON
{signedVideoUrl: "<m3u8>", ...}
* hgcloud.to -> HLS master manifest fires as a network response (jwplayer)
* rubyvidhub.com -> HLS master manifest fires as a network response
* veev.to -> the mp4 URL lives in <video>.currentSrc after muted
play(); its CDN is TLS/network gated, so the download
itself may fail from some networks.
Returns the best scoring URL, or "" if nothing resolvable.
"""
page = await scraper.new_page()
candidates = {}
def note(url, score):
if not url or "blob:" in url or "data:" in url:
return
if score > candidates.get(url, -1):
candidates[url] = score
async def on_response(resp):
try:
rurl = resp.url
ct = resp.headers.get("content-type", "").lower()
base = rurl.split("?")[0]
if ("video/" in ct or "mpegurl" in ct or "octet-stream" in ct
or base.endswith((".m3u8", ".mp4", ".webm"))):
note(rurl, _tickzoo_score_media_url(rurl))
if "/api/" in rurl.lower() or "resolve" in rurl.lower():
try:
body = await resp.text()
except Exception:
return
try:
payload = json.loads(body)
except Exception:
return
for key in ("signedVideoUrl", "signedVideoSdUrl", "videoUrl", "url", "src"):
val = payload.get(key)
if isinstance(val, str) and val.startswith("http"):
note(val, 90)
break
except Exception:
pass
page.on("response", lambda r: asyncio.create_task(on_response(r)))
try:
tqdm.write(f" resolving embed {embed_url}")
await page.goto(embed_url, wait_until="domcontentloaded", timeout=45000, referer=referer_url)
await page.wait_for_timeout(4000)
# Trigger playback so the player actually requests the media.
for selector in ("button.vjs-big-play-button", ".play-button", "[class*='play']", "video"):
try:
el = await page.query_selector(selector)
if el:
await el.click(timeout=1200)
await page.wait_for_timeout(1500)
break
except Exception:
continue
try:
await page.evaluate(
"""() => { const v = document.querySelector('video');
if (v) { v.muted = true; v.play().catch(()=>{}); } }"""
)
except Exception:
pass
# Some players (firestream) answer late / intermittently - poll a while.
dom_js = """() => { const v = document.querySelector('video');
if (!v) return '';
return (v.currentSrc || v.src || ''); }"""
for _ in range(8):
await page.wait_for_timeout(4000)
try:
src = await page.evaluate(dom_js)
if src:
note(src, _tickzoo_score_media_url(src))
except Exception:
pass
if candidates:
break
good = {u: s for u, s in candidates.items() if s > 0}
if not good:
tqdm.write(f" x no media found for embed {embed_url}")
return ""
best = max(good.items(), key=lambda kv: (kv[1], kv[0]))[0]
tqdm.write(f" -> {best}")
return best
finally:
try:
await page.close()
except Exception:
pass
async def _tickzoo_process_one(scraper, url, download_dir, skip_existing, overall_bar, bar_pool):
"""Fetch the tickzoo page, resolve its embed, download the video."""
if is_filtered(url):
tqdm.write(f" x Filtered: {url}")
overall_bar.update(1)
return
html = await asyncio.to_thread(scraper_core.curl_fetch, url, referer="https://tickzoo.tv/")
if not html:
tqdm.write(f" x Failed to fetch page: {url}")
overall_bar.update(1)
return
embed_url = _tickzoo_extract_embed(html)
if not embed_url:
tqdm.write(f" x No embed iframe found on {url}")
overall_bar.update(1)
return
title = _tickzoo_extract_title(html) or url.rstrip("/").rsplit("/", 1)[-1]
uploader = _tickzoo_extract_uploader(html)
tqdm.write(f" [{title[:70]}]")
tqdm.write(f" embed: {embed_url}")
media_url = await _tickzoo_resolve_media(scraper, embed_url, url)
if not media_url:
overall_bar.update(1)
return
stem = scraper_core.clean_filename(url.rstrip("/").rsplit("/", 1)[-1])
dest_path = download_dir / uploader / f"{stem}.mp4"
is_hls = ".m3u8" in media_url.lower() or "manifest" in media_url.lower()
if skip_existing:
if is_hls:
if scraper_core.is_already_downloaded(dest_path):
tqdm.write(f" [SKIP] '{uploader}/{stem}.mp4' already downloaded (HLS).")
overall_bar.record_skip(dest_path.stat().st_size)
return
else:
is_complete, existing_bytes = await check_existing_file(
dest_path, media_url, {"Referer": embed_url}, None
)
if is_complete:
overall_bar.record_skip(existing_bytes)
return
pos = bar_pool.acquire()
if pos is None:
pos = 4
headers = None if is_hls else {"Referer": embed_url}
success = await asyncio.to_thread(
scraper_core.download_file, media_url, dest_path, headers, pos, None, None
)
bar_pool.release(pos)
if success and dest_path.is_file() and dest_path.stat().st_size > 0:
label = f"{dest_path.parent.name}/{dest_path.name}"
overall_bar.record_download(dest_path.stat().st_size, name=label)
scraper_core.append_log(download_dir / "urls.txt", url)
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
else:
overall_bar.update(1)
async def tickzoo_process_urls(download_dir, urls, concurrency, skip_existing):
"""Download tickzoo.tv video pages.
Each page embeds a 3rd-party player (veev.to / firestream.to / hgcloud.to /
rubyvidhub.com). Embeds are resolved through one shared headless Chrome
session, then handed to the shared downloader (yt-dlp for HLS streams).
Sequential on purpose: embed hosts are flaky and a browser session is reused.
"""
unique = [u for u in dedupe(urls) if _tickzoo_is_video_link(u)]
if not unique:
tqdm.write(" No tickzoo.tv video URLs provided.")
return
safe_mkdir(download_dir)
overall_bar = OverallProgressTracker(total=0, desc="tickzoo.tv")
bar_pool = scraper_core.BarPositionPool(concurrency)
bar_pool.available = [p + 3 for p in bar_pool.available]
for url in unique:
tqdm.write(f"Queuing video URL: {url}")
overall_bar.total += 1
overall_bar.refresh()
scraper = scraper_core.PlaywrightScraper()
await scraper.start()
try:
for url in unique:
try:
await _tickzoo_process_one(scraper, url, download_dir, skip_existing, overall_bar, bar_pool)
except Exception as e:
tqdm.write(f" Error processing {url}: {e}")
overall_bar.update(1)
finally:
try:
await scraper.close()
except Exception:
pass
overall_bar.close()
sys.stdout.write("\n" * (concurrency + 3))
sys.stdout.flush()
# ----- Pornhub ---------------------------------------------------------------
def _pornhub_is_video_link(url):
return "/view_video.php?viewkey=" in url
def _pornhub_extract_get_media(page_html):
m = re.search(r'"videoUrl":"((?:[^"\\]|\\.)*?get_media(?:[^"\\]|\\.)*?)"', page_html)
if not m:
return None
return m.group(1).replace("\\/", "/").replace("\\u0026", "&").replace("\\u002F", "/")
def _pornhub_extract_folder(page_html):
m = re.search(r'<a[^>]*data-label\s*=\s*["\']channel["\'][^>]*>(.*?)</a>', page_html, re.S)
if m:
name = re.sub(r"<[^>]+>", "", m.group(1)).strip()
if name:
return scraper_core.clean_filename(name)
m = re.search(r'class\s*=\s*["\']from["\'][\s\S]*?usernameBadgesWrapper[^>]*>\s*<a[^>]*>([^<]+)', page_html)
if m:
return scraper_core.clean_filename(m.group(1).strip())
m = re.search(r'"username":"([^"]+)"', page_html)
if m:
return scraper_core.clean_filename(m.group(1))
return "unknown"
def _pornhub_extract_title(page_html):
m = re.search(r"<title>(.*?)</title>", page_html, re.S)
if not m:
return ""
return re.sub(r"\s*-\s*Pornhub\.com\s*$", "", unescape(m.group(1)), flags=re.I).strip()
def _pornhub_fetch_video_source(url):
from curl_cffi import requests as cr
session = cr.Session(impersonate="chrome")
resp = session.get(url, timeout=60)
if resp.status_code != 200:
return None
page_html = resp.text
gurl = _pornhub_extract_get_media(page_html)
if not gurl:
return None
g = session.get(gurl, headers={"Referer": url, "X-Requested-With": "XMLHttpRequest"}, timeout=60)
if g.status_code != 200:
return None
data = g.json()
if not isinstance(data, list):
return None
media = [e for e in data if e.get("format") == "mp4"]
if not media:
return None
best = max(media, key=lambda e: e.get("height") or 0)
return {
"src": best["videoUrl"],
"cookies": dict(session.cookies),
"folder": _pornhub_extract_folder(page_html),
"title": _pornhub_extract_title(page_html),
}
async def _pornhub_worker(queue, skip_existing, bar_pool, overall_bar, download_dir):
while True:
url = await queue.get()
if url is None:
queue.task_done()
break
try:
if is_filtered(url):
tqdm.write(f" x Filtered: {url}")
continue
await asyncio.sleep(random.uniform(0.5, 1.5))
info = await asyncio.to_thread(_pornhub_fetch_video_source, url)
if not info:
tqdm.write(f" x Skipping {url} - could not resolve video source")
continue
vk = re.search(r"viewkey=([a-f0-9]{13})", url)
viewkey = vk.group(1) if vk else "na"
stem = info["title"] if info["title"] else viewkey
filename = f"{scraper_core.clean_filename(stem)}-{viewkey}.mp4"
dest_path = download_dir / info["folder"] / filename
headers = {"Referer": url}
downloaded_bytes = 0
if skip_existing:
is_complete, existing_bytes = await check_existing_file(dest_path, info["src"], headers, info["cookies"])
if is_complete:
overall_bar.record_skip(existing_bytes)
continue
pos = bar_pool.acquire() or 4
success = await asyncio.to_thread(
scraper_core.download_file, info["src"], dest_path, headers, pos, url, info["cookies"]
)
bar_pool.release(pos)
if success:
if dest_path.is_file():
downloaded_bytes = dest_path.stat().st_size
label = f"{dest_path.parent.name}/{dest_path.name}"
overall_bar.record_download(downloaded_bytes, name=label)
scraper_core.append_log(download_dir / "urls.txt", url)
scraper_core.append_log(download_dir / info["folder"] / "urls.txt", url)
else:
overall_bar.update(1)
except Exception as e:
tqdm.write(f" Error processing {url}: {e}")
overall_bar.update(1)
finally:
queue.task_done()
async def _pornhub_scrape_channel(channel_url, queue, counter, overall_bar):
base = channel_url.rstrip("/")
if not base.endswith("/videos"):
base += "/videos"
page_num = 1
seen = set()
while True:
url = base if page_num == 1 else f"{base}?page={page_num}"
tqdm.write(f" Scraping page {page_num}: {url}")
html = await asyncio.to_thread(scraper_core.curl_fetch, url, referer="https://www.pornhub.com/")
if not html:
tqdm.write(" Failed to fetch channel page")
break
links = []
for vk in dict.fromkeys(re.findall(r"/view_video\.php\?viewkey=([a-f0-9]{13})", html)):
u = f"https://www.pornhub.com/view_video.php?viewkey={vk}"
if u not in seen and not is_filtered(u):
seen.add(u)
links.append(u)
if not links:
tqdm.write(" No new video links found on page")
break
for u in links:
await queue.put(u)
counter["total"] += 1
overall_bar.total = counter["total"]
overall_bar.refresh()
tqdm.write(f" Found {len(links)} video links")
if f"page={page_num + 1}" not in html:
tqdm.write(" No next page link, done")
break
page_num += 1
await asyncio.sleep(random.uniform(1.0, 2.0))
async def pornhub_process_urls(download_dir, urls, concurrency, skip_existing):
queue = asyncio.Queue()
counter = {"total": 0}
overall_bar = OverallProgressTracker(total=0, desc="Overall Progress")
bar_pool = scraper_core.BarPositionPool(concurrency)
bar_pool.available = [p + 3 for p in bar_pool.available]
workers = [
asyncio.create_task(_pornhub_worker(queue, skip_existing, bar_pool, overall_bar, download_dir))
for _ in range(concurrency)
]
for url in urls:
if is_filtered(url):
tqdm.write(f" x Filtered (not queued): {url}")
continue
if _pornhub_is_video_link(url):
tqdm.write(f"Queuing direct video URL: {url}")
await queue.put(url)
counter["total"] += 1
overall_bar.total = counter["total"]
overall_bar.refresh()
else:
tqdm.write(f"Scraping channel/playlist: {url}")
try:
await _pornhub_scrape_channel(url, queue, counter, overall_bar)
except Exception as e:
tqdm.write(f" Error scraping channel {url}: {e}")
await asyncio.sleep(random.uniform(1.0, 2.0))
for _ in range(concurrency):
await queue.put(None)
await asyncio.gather(*workers)
overall_bar.close()
sys.stdout.write("\n" * (concurrency + 4))
sys.stdout.flush()
# ----- Zootube1 -------------------------------------------------------------
def _zootube1_extract_video_links(html: str) -> list:
raw = re.findall(r'hd=(https://zootube1\.com/allvideos/[^"&\s]+)', html)
raw += re.findall(r'href="(https://zootube1\.com/allvideos/[^"]+)"', html)
links = []
for url in raw:
url = url.replace("&amp;", "&").split("#")[0]
if url not in links:
links.append(url)
return links
async def _zootube1_scrape_playlist(playlist_url, cookies, queue, counter, overall_bar, scraper):
"""Scrape a /users/<id>/ page following the real ?from_videos=NN pagination.
The site ignores ?page=N (it always returns page 1) and paginates through
?from_videos=NN, where every batch yields ~28 never-before-seen videos.
"""
seen = set()
page_num = 1
while True:
if page_num == 1:
url = playlist_url
else:
url = f"{playlist_url}?from_videos={page_num:02d}"
tqdm.write(f" Scraping page {page_num}: {url}")
html = await asyncio.to_thread(
scraper_core.curl_fetch, url, cookies=cookies, referer="https://zootube1.com/"
)
if not html:
tqdm.write(f" Static fetch failed. Trying Playwright for playlist page...")
page = await scraper.new_page()
try:
await scraper.resolve_page(page, url)
html = await page.content()
except Exception as pe:
tqdm.write(f" x Playwright failed to fetch playlist page: {pe}")
finally:
await page.close()
if not html:
tqdm.write(f" Failed to fetch page {page_num}")
break
links = _zootube1_extract_video_links(html)
new_links = 0
for link in links:
if link in seen:
continue
seen.add(link)
if is_filtered(link):
tqdm.write(f" x Filtered: {link}")
continue
await queue.put(link)
counter["total"] += 1
overall_bar.total = counter["total"]
overall_bar.refresh()
new_links += 1
tqdm.write(f" Found {len(links)} links ({new_links} new)")
if new_links == 0:
break
page_num += 1
await asyncio.sleep(random.uniform(1.0, 2.0))
async def _zootube1_extract_video_info(scraper, url: str):
"""Load a video page in the browser to read the runtime player source.
The player (kt_player.js) rewrites flashvars.video_url at runtime; the
static HTML hash 404s, so a real page load is required. Returns
(video_id, source, uploader, cookies).
"""
page = await scraper.new_page()
try:
await scraper.resolve_page(page, url)
js = """() => {
const out = {source: '', uploader: '', video_id: ''};
if (typeof flashvars !== 'undefined' && flashvars) {
out.source = flashvars.video_url || '';
out.video_id = String(flashvars.video_id || '');
}
const pu = document.querySelector('.player-uploader');
if (pu) {
out.uploader = pu.innerText || pu.textContent || '';
}
return out;
}"""
info = await page.evaluate(js)
# Wait until the player finishes rewriting video_url (it drops the
# "function/N/" placeholder prefix).
for _ in range(10):
src = info.get("source") or ""
if src and not src.startswith("function/"):
break
await page.wait_for_timeout(1000)
info = await page.evaluate(js)
finally:
await page.close()
cookies = {
c["name"]: c["value"]
for c in await scraper.context.cookies()
if "zootube1.com" in c.get("domain", "")
}
source = re.sub(r"^function/\d+/", "", info.get("source") or "")
return info.get("video_id", ""), source, info.get("uploader", "unknown"), cookies
async def _zootube1_worker(queue, skip_existing, bar_pool, overall_bar, scraper, download_dir):
while True:
url = await queue.get()
if url is None:
queue.task_done()
break
try:
if is_filtered(url):
tqdm.write(f" x Filtered: {url}")
continue
await asyncio.sleep(random.uniform(0.3, 0.8))
video_id, src, uploader, cookies = await _zootube1_extract_video_info(scraper, url)
if not src:
tqdm.write(f" x Skipping {url} - could not find video source")
continue
uploader = re.sub(r"\s+", " ", uploader or "").strip()
uploader = re.sub(r"^by\s+", "", uploader, flags=re.I).strip()
uploader = scraper_core.clean_filename(uploader) or "unknown"
stem = url.rstrip("/").rsplit("/", 1)[-1]
stem = re.sub(r"\.html?$", "", stem)
suffix = f"-{video_id}" if video_id else ""
filename = f"{scraper_core.clean_filename(stem)}{suffix}.mp4"
dest_path = download_dir / uploader / filename
downloaded_bytes = 0
if skip_existing:
is_complete, existing_bytes = await check_existing_file(dest_path, src, {"Referer": url}, cookies)
if is_complete:
overall_bar.record_skip(existing_bytes)
continue
pos = bar_pool.acquire()
if pos is None:
pos = 4
success = await asyncio.to_thread(
scraper_core.download_file,
src, dest_path, {"Referer": url}, pos, cookies=cookies
)
bar_pool.release(pos)
if success:
if dest_path.is_file():
downloaded_bytes = dest_path.stat().st_size
label = f"{dest_path.parent.name}/{dest_path.name}"
overall_bar.record_download(downloaded_bytes, name=label)
scraper_core.append_log(download_dir / "urls.txt", url)
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
else:
overall_bar.update(1)
except Exception as e:
tqdm.write(f" Error processing {url}: {e}")
overall_bar.update(1)
finally:
queue.task_done()
async def _zootube1_producer(urls, cookies, queue, counter, overall_bar, scraper):
for url in urls:
if _zootube1_is_video_link(url):
if is_filtered(url):
tqdm.write(f" x Filtered (not queued): {url}")
continue
tqdm.write(f"Queuing direct video URL: {url}")
await queue.put(url)
counter["total"] += 1
overall_bar.total = counter["total"]
overall_bar.refresh()
else:
tqdm.write(f"Scraping user/playlist: {url}")
try:
await _zootube1_scrape_playlist(url, cookies, queue, counter, overall_bar, scraper)
except Exception as e:
tqdm.write(f" Error scraping playlist {url}: {e}")
await asyncio.sleep(random.uniform(1.0, 2.0))
async def zootube1_process_urls(download_dir, urls, concurrency, skip_existing):
scraper = scraper_core.PlaywrightScraper()
await scraper.start(headless=True)
cookies = {}
warmup_url = urls[0] if _zootube1_is_video_link(urls[0]) else "https://zootube1.com/"
try:
page = await scraper.new_page()
try:
await scraper.resolve_page(page, warmup_url)
cookies = {
c["name"]: c["value"]
for c in await scraper.context.cookies()
if "zootube1.com" in c.get("domain", "")
}
finally:
await page.close()
except Exception as e:
tqdm.write(f" Warning: warm-up failed: {e}")
tqdm.write(f"Got cookies: {list(cookies.keys())}")
queue = asyncio.Queue()
counter = {"total": 0}
overall_bar = OverallProgressTracker(total=0, desc="Overall Progress")
bar_pool = scraper_core.BarPositionPool(concurrency)
bar_pool.available = [p + 3 for p in bar_pool.available]
workers = [
asyncio.create_task(_zootube1_worker(queue, skip_existing, bar_pool, overall_bar, scraper, download_dir))
for _ in range(concurrency)
]
await _zootube1_producer(urls, cookies, queue, counter, overall_bar, scraper)
for _ in range(concurrency):
await queue.put(None)
await asyncio.gather(*workers)
overall_bar.close()
await scraper.close()
sys.stdout.write("\n" * (concurrency + 4))
sys.stdout.flush()
# ----- RedGIFs --------------------------------------------------------------
REDGIFS_HEADERS = {
'User-Agent': scraper_core.USER_AGENT,
'Accept': 'application/json, text/plain, */*',
'Accept-Language': 'en-US,en;q=0.9',
'Referer': 'https://www.redgifs.com/',
'Origin': 'https://www.redgifs.com',
'DNT': '1',
'Connection': 'keep-alive',
}
def _wait_cooldown(wait_seconds: float, desc: str = "Waiting") -> bool:
"""
Waits for wait_seconds while displaying a progress bar.
Accepts keyboard input (pressing Enter or any key on Windows / Enter on Unix)
to immediately interrupt and skip the cooldown wait.
Returns True if interrupted by user, False if completed normally.
"""
if wait_seconds <= 0:
return False
tqdm.write(" [i] Press ENTER at any time to skip cooldown and retry immediately (e.g. after changing IP).")
total_steps = max(1, int(wait_seconds * 10)) # 100ms intervals
bar = tqdm(total=int(wait_seconds), desc=f" {desc}", unit='s', ncols=80, leave=False)
interrupted = False
# Check if Windows msvcrt is available for instant non-blocking keypress
has_msvcrt = False
try:
import msvcrt
has_msvcrt = True
except ImportError:
pass
if has_msvcrt:
# Flush any previously buffered keys
try:
while msvcrt.kbhit():
msvcrt.getch()
except Exception:
pass
for step in range(total_steps):
try:
if msvcrt.kbhit():
ch = msvcrt.getch()
interrupted = True
break
except Exception:
pass
time.sleep(0.1)
if step % 10 == 9:
bar.update(1)
else:
# Fallback using a background thread reading sys.stdin
stop_event = threading.Event()
input_received = threading.Event()
def _reader():
try:
import select
while not stop_event.is_set():
r, _, _ = select.select([sys.stdin], [], [], 0.2)
if r:
sys.stdin.readline()
input_received.set()
break
except Exception:
pass
t = threading.Thread(target=_reader, daemon=True)
t.start()
for step in range(total_steps):
if input_received.is_set():
interrupted = True
break
time.sleep(0.1)
if step % 10 == 9:
bar.update(1)
stop_event.set()
bar.close()
if interrupted:
tqdm.write(" [!] Cooldown interrupted by user. Resuming immediately...")
return True
return False
class RedGIFsAPI:
"""Headless browser wrapper for the RedGIFs API (token-bound to browser)."""
def __init__(self, skip_cooldown: bool = False):
self._pw_cm = None
self._browser = None
self._page = None
self._token = None
self.skip_cooldown = skip_cooldown
def start(self):
tqdm.write(' Obtaining RedGIFs API token ...')
from playwright.sync_api import sync_playwright
self._pw_cm = sync_playwright()
pw = self._pw_cm.__enter__()
self._browser = pw.chromium.launch(headless=True)
context = self._browser.new_context(
user_agent=REDGIFS_HEADERS['User-Agent'],
viewport={'width': 1920, 'height': 1080},
)
self._page = context.new_page()
self._page.goto(
'https://www.redgifs.com/',
wait_until='domcontentloaded',
timeout=30000,
)
self.obtain_token()
def obtain_token(self, retries=5):
"""Fetches or refreshes the temporary token from the RedGIFs API."""
last_err = None
for attempt in range(retries):
try:
# Ensure we're on a valid redgifs page (not a Cloudflare challenge redirect)
cur = self._page.url
if 'redgifs' not in cur or 'challenge' in cur:
self._page.goto('https://www.redgifs.com/', wait_until='networkidle', timeout=30000)
elif cur != 'https://www.redgifs.com/':
self._page.goto('https://www.redgifs.com/', wait_until='domcontentloaded', timeout=30000)
result = self._page.evaluate('''async () => {
try {
const r = await fetch('https://api.redgifs.com/v2/auth/temporary');
if (!r.ok) {
return {error: "HTTP " + r.status + ": " + await r.text()};
}
const d = await r.json();
return {token: d.token};
} catch(e) {
return {error: e.message || String(e)};
}
}''')
if isinstance(result, dict) and 'error' in result:
raise RuntimeError(result['error'])
self._token = result if isinstance(result, str) else result.get('token') if isinstance(result, dict) else result
return
except Exception as e:
last_err = e
err_str = str(e)
if attempt < retries - 1:
if '429' in err_str:
if self.skip_cooldown:
tqdm.write(' Token fetch rate-limited (HTTP 429). Skipping cooldown wait (--skip-redgifs-cooldown enabled)...')
wait = 1
else:
m = re.search(r'"delay"\s*:\s*(\d+)', err_str)
wait = int(m.group(1)) + 2 if m else (15 * (attempt + 1))
tqdm.write(f' Token fetch rate-limited (HTTP 429). Waiting {wait}s...')
else:
wait = 5 * (attempt + 1)
tqdm.write(f' Token fetch failed (attempt {attempt+1}/{retries}): {e}. Retrying in {wait}s...')
_wait_cooldown(wait, desc='Waiting')
try:
self._page.goto('https://www.redgifs.com/', wait_until='domcontentloaded', timeout=30000)
except Exception:
pass
raise RuntimeError(f'Failed to obtain token after {retries} attempts: {last_err}')
def stop(self):
if self._browser:
try:
self._browser.close()
except Exception:
pass
if self._pw_cm:
try:
self._pw_cm.__exit__(None, None, None)
except Exception:
pass
def get_all_creator_videos(self, username, fresh_token=True):
"""Fetch all videos for a creator. Automatically refreshes tokens if expired."""
if fresh_token or not self._token:
self.obtain_token()
raw = self._page.evaluate('''async ({username, token}) => {
let allGifs = [];
let page = 1;
let pages = 1;
let total = 0;
let currentToken = token;
let errorMsg = '';
const sleep = ms => new Promise(r => setTimeout(r, ms));
try {
while (page <= pages) {
let r;
while (true) {
try {
r = await fetch(
'https://api.redgifs.com/v2/users/'
+ encodeURIComponent(username)
+ '/search?page=' + page + '&count=80&order=latest&type=g',
{headers: {'Authorization': 'Bearer ' + currentToken}}
);
} catch (networkErr) {
errorMsg = 'Network error: ' + (networkErr.message || String(networkErr));
r = null;
break;
}
if (r.status === 401 || r.status === 403) {
try {
const tokRes = await fetch('https://api.redgifs.com/v2/auth/temporary');
if (tokRes.status === 200) {
const tokData = await tokRes.json();
currentToken = tokData.token;
continue;
}
} catch (e) {}
}
if (r.status === 429) {
let rawBody = '';
try { rawBody = await r.text(); } catch (e) {}
errorMsg = 'API status 429: ' + rawBody;
break;
}
break;
}
if (errorMsg || !r || r.status !== 200) {
if (!errorMsg && r) {
errorMsg = 'API status ' + r.status + ': ' + await r.text();
}
break;
}
const d = await r.json();
if (page === 1) {
pages = Math.min(d.pages || 1, 100);
total = d.total || 0;
}
for (const g of (d.gifs || [])) {
allGifs.push({urls: g.urls, id: g.id});
}
page++;
await sleep(1500); // 1.5s delay between pages
}
} catch (outerErr) {
errorMsg = 'Unexpected error: ' + (outerErr.message || String(outerErr));
}
return JSON.stringify({gifs: allGifs, total: total, error: errorMsg});
}''', {'username': username, 'token': self._token})
data = json.loads(raw)
if data.get('error'):
raise RuntimeError(data['error'])
return data.get('gifs', []), int(data.get('total', 0))
def get_gif_info(self, gif_id, fresh_token=False):
if fresh_token or not self._token:
self.obtain_token()
raw = self._page.evaluate('''async ({gifId, token}) => {
let currentToken = token;
for (let attempt = 0; attempt < 2; attempt++) {
let r;
try {
r = await fetch(
'https://api.redgifs.com/v2/gifs/' + encodeURIComponent(gifId),
{headers: {'Authorization': 'Bearer ' + currentToken}}
);
} catch (networkErr) {
return JSON.stringify({error: 'Network error: ' + (networkErr.message || String(networkErr))});
}
if (r.status === 401 || r.status === 403) {
try {
const tokRes = await fetch('https://api.redgifs.com/v2/auth/temporary');
if (tokRes.status === 200) {
currentToken = (await tokRes.json()).token;
continue;
}
} catch (e) {}
}
if (r.status === 429) {
let errBody = '';
try { errBody = await r.text(); } catch (e) {}
return JSON.stringify({error: 'API status 429: ' + errBody});
}
if (!r.ok) {
return JSON.stringify({error: 'API status ' + r.status});
}
const d = await r.json();
const g = d.gif || d;
return JSON.stringify({gif: {
id: g.id || gifId,
userName: g.userName || '',
title: g.title || '',
urls: g.urls || {},
}});
}
return JSON.stringify({error: 'token refresh failed'});
}''', {'gifId': gif_id, 'token': self._token})
data = json.loads(raw)
if data.get('error'):
raise RuntimeError(data['error'])
return data.get('gif') or {}
def _redgifs_extract_filename(url: str) -> str:
return urlparse(url).path.split('/')[-1]
def _redgifs_extract_gif_id(target: str):
"""Extract a gif id from a RedGIFs watch URL or media URL; None if not one."""
try:
parsed = urlparse(target)
host = parsed.netloc.lower()
path = parsed.path
except ValueError:
return None
if not host:
host = urlparse("https://" + target).netloc.lower()
path = urlparse("https://" + target).path
path = path.rstrip("/")
if host.startswith("media."):
name = path.split("/")[-1]
if name.lower().endswith(".mp4"):
name = name[:-4]
return name or None
m = re.match(r"/watch/([A-Za-z0-9_]+)$", path)
if m:
return m.group(1)
return None
def redgifs_download_creator_videos(username, output_dir, gifs, total, concurrency, skip_existing):
"""Download one creator's videos into output_dir/<username>/ (exact original layout)."""
creator_dir = safe_mkdir(output_dir / username)
links_file = creator_dir / 'links.txt'
master_links_file = output_dir / 'all_links.txt'
download_urls = []
link_lines = []
for gif in gifs:
urls = gif['urls']
dl_url = urls.get('hd') or urls.get('sd') or next(iter(urls.values()))
filename = _redgifs_extract_filename(dl_url)
filepath = creator_dir / filename
download_urls.append((dl_url, filepath))
link_lines.append(f'{dl_url}\n')
# Save url lists
safe_append_text(links_file, ''.join(link_lines))
safe_append_text(master_links_file, ''.join(link_lines))
if len(gifs) < total:
tqdm.write(f"Downloading {len(gifs)} of {total} video(s) (capped by API page limit) for {username} with concurrency {concurrency} ...")
else:
tqdm.write(f"Downloading {total} video(s) for {username} with concurrency {concurrency} ...")
bar_pool = scraper_core.BarPositionPool(concurrency)
bar_pool.available = [p + 3 for p in bar_pool.available]
overall_bar = OverallProgressTracker(
total=len(gifs),
desc=f"Creator: {username}",
)
def download_one(dl_url, filepath):
processed = False
try:
if skip_existing:
is_complete, existing_bytes = check_existing_file_sync(filepath, dl_url, headers=REDGIFS_HEADERS)
if is_complete:
overall_bar.record_skip(existing_bytes)
processed = True
return
pos = bar_pool.acquire() or 4
try:
success = scraper_core.download_file(
dl_url, filepath, REDGIFS_HEADERS, pos
)
if success:
dl_size = filepath.stat().st_size if filepath.is_file() else 0
label = f"{filepath.parent.name}/{filepath.name}"
overall_bar.record_download(dl_size, name=label)
processed = True
finally:
bar_pool.release(pos)
except Exception as e:
tqdm.write(f" [ERROR] Failed to download:\n {dl_url}\n Error: {e}")
finally:
if not processed:
overall_bar.update(1)
with ThreadPoolExecutor(max_workers=concurrency) as executor:
futures = [executor.submit(download_one, url, path) for url, path in download_urls]
for f in as_completed(futures):
f.result()
overall_bar.close()
# Clean up console lines
sys.stdout.write("\n" * (concurrency + 4))
sys.stdout.flush()
def redgifs_update_creator_list(output_dir, creators_file):
"""--update-all-creators: merge folder names + creators.txt, write back, return merged list."""
found_folders = []
if output_dir.exists():
for entry in output_dir.iterdir():
if entry.is_dir():
found_folders.append(entry.name)
existing_creators = []
if creators_file.is_file():
try:
with open(creators_file, 'r', encoding='utf-8') as f:
for line in f:
stripped = line.strip()
if not stripped or stripped.startswith('#'):
continue
for part in re.split(r'[,\s]+', stripped):
if part:
existing_creators.append(part)
except Exception as e:
tqdm.write(f"Error reading creators.txt: {e}")
all_creators = sorted(list(set(existing_creators + found_folders)))
try:
with open(creators_file, 'w', encoding='utf-8') as f:
f.write(' '.join(all_creators))
tqdm.write(f"Updated creators.txt with {len(all_creators)} unique creators.")
except Exception as e:
tqdm.write(f"Error writing to creators.txt: {e}")
return all_creators
def redgifs_download_creators(creators, output_dir, concurrency, skip_existing, skip_cooldown: bool = False):
"""Download all videos for the given creator username(s) into output_dir/<creator>/."""
creators = dedupe(creators)
if not creators:
sys.exit('No creators provided.')
output_dir = Path(output_dir).resolve()
output_dir.mkdir(parents=True, exist_ok=True)
api = RedGIFsAPI(skip_cooldown=skip_cooldown)
try:
api.start()
except Exception as e:
sys.exit(f'Failed to initialise Playwright browser: {e}')
try:
total_creators = len(creators)
for idx, username in enumerate(creators, 1):
sep = '-' * 3
tqdm.write(f'\n{sep} [{idx}/{total_creators}] {username} {sep}')
tqdm.write(f' Fetching video URLs for {username} ...')
gifs, total = [], 0
success = False
# Retry loop with browser reset on failure
for attempt in range(1, 4):
try:
gifs, total = api.get_all_creator_videos(username, fresh_token=(attempt == 1))
success = True
break
except Exception as e:
tqdm.write(f' [ERROR] Attempt {attempt}/3 failed for {username}: {e}')
if attempt < 3:
err_str = str(e)
if '429' in err_str:
if skip_cooldown:
tqdm.write(' Rate limited. Skipping cooldown wait (--skip-redgifs-cooldown enabled)...')
wait_seconds = 1
else:
m = re.search(r'"delay"\s*:\s*(\d+)', err_str)
wait_seconds = int(m.group(1)) + 1 if m else 30
tqdm.write(f' Rate limited. Waiting {wait_seconds:.1f}s...')
skipped = _wait_cooldown(wait_seconds, desc='Waiting')
if skipped:
# User changed IP and skipped cooldown; force token refresh
try:
api.obtain_token()
except Exception:
pass
else:
wait_reset = 1 if skip_cooldown else 30
tqdm.write(f' Resetting browser context and waiting {wait_reset}s before retry...')
try:
api.stop()
except Exception:
pass
_wait_cooldown(wait_reset, desc='Reset Wait')
try:
api.start()
except Exception as restart_err:
tqdm.write(f' Failed to restart browser: {restart_err}')
if success:
tqdm.write(f' Found {total} video(s)')
if total > 0:
redgifs_download_creator_videos(
username, output_dir, gifs, total,
concurrency, skip_existing
)
else:
tqdm.write(f' [ERROR] Skipping {username} after 3 failed attempts.')
# Introduce a small pacing delay to avoid aggressive rate limits
time.sleep(2)
finally:
api.stop()
def redgifs_download_gif_urls(gif_urls, output_dir, concurrency, skip_existing, skip_cooldown: bool = False):
"""Download individual RedGIFs gifs from watch/media URLs into output_dir/<creator>/.
Downloads concurrently as URLs are resolved rather than waiting for all URLs to be parsed.
"""
unique_urls = dedupe(gif_urls)
total_targets = len(unique_urls)
if total_targets == 0:
tqdm.write(' No RedGIFs gif URLs provided.')
return
api = RedGIFsAPI(skip_cooldown=skip_cooldown)
try:
api.start()
except Exception as e:
sys.exit(f'Failed to initialise Playwright browser: {e}')
try:
tqdm.write(f"Processing and downloading {total_targets} RedGIFs gif(s) with concurrency {concurrency} ...")
bar_pool = scraper_core.BarPositionPool(concurrency)
bar_pool.available = [p + 3 for p in bar_pool.available]
overall_bar = OverallProgressTracker(total=total_targets, desc='RedGIFs gifs')
item_queue = queue.Queue(maxsize=max(50, concurrency * 4))
def download_worker():
while True:
item = item_queue.get()
if item is None:
item_queue.task_done()
break
dl_url, filepath = item
downloaded = False
skipped = False
try:
if skip_existing:
is_complete, existing_bytes = check_existing_file_sync(filepath, dl_url, headers=REDGIFS_HEADERS)
if is_complete:
overall_bar.record_skip(existing_bytes)
skipped = True
if not skipped:
pos = bar_pool.acquire() or 4
try:
success = scraper_core.download_file(dl_url, filepath, REDGIFS_HEADERS, pos)
if success:
dl_size = filepath.stat().st_size if filepath.is_file() else 0
label = f"{filepath.parent.name}/{filepath.name}"
overall_bar.record_download(dl_size, name=label)
downloaded = True
finally:
bar_pool.release(pos)
except Exception as e:
tqdm.write(f' [ERROR] Failed to download:\n {dl_url}\n Error: {e}')
finally:
if not downloaded and not skipped:
overall_bar.update(1)
item_queue.task_done()
# Start background downloader threads
download_threads = []
for _ in range(concurrency):
t = threading.Thread(target=download_worker, daemon=True)
t.start()
download_threads.append(t)
try:
for raw_url in unique_urls:
gid = _redgifs_extract_gif_id(raw_url)
if not gid:
tqdm.write(f' [!] Could not parse gif id from: {raw_url}')
overall_bar.update(1)
continue
info = None
candidates = []
lowered = gid.lower()
candidates.append(lowered)
if gid != lowered:
candidates.append(gid)
for cand in candidates:
for attempt in range(1, 4):
try:
info = api.get_gif_info(cand, fresh_token=False)
break
except Exception as e:
err_msg = str(e)
# If the API returned 404 or 410, do not burn retries
if '404' in err_msg or '410' in err_msg:
if '410' in err_msg:
# 410 Gone means the resource is permanently gone across any case variation
candidates.clear()
break
if '429' in err_msg:
if skip_cooldown:
tqdm.write(f' Rate limited (429) resolving {cand}. Skipping cooldown wait (--skip-redgifs-cooldown enabled)...')
wait_seconds = 1
else:
m = re.search(r'"delay"\s*:\s*(\d+)', err_msg)
wait_seconds = int(m.group(1)) + 2 if m else (20 * attempt)
tqdm.write(f' Rate limited (429) resolving {cand}. Waiting {wait_seconds}s...')
skipped = _wait_cooldown(wait_seconds, desc='Waiting')
if skipped:
try:
api.obtain_token()
except Exception:
pass
continue
tqdm.write(f' [ERROR] Attempt {attempt}/3 resolving:\n {raw_url}\n as {cand}: {e}')
if attempt < 3:
err_wait = 1 if skip_cooldown else (5 * attempt)
_wait_cooldown(err_wait, desc='Retry Wait')
if info or not candidates:
break
if not info:
tqdm.write(f' [ERROR] Skipping (could not resolve gif info):\n {raw_url}')
overall_bar.update(1)
continue
urls = info.get('urls') or {}
dl_url = urls.get('hd') or urls.get('sd')
if not dl_url and urls:
dl_url = next(iter(urls.values()))
if not dl_url:
tqdm.write(f' [!] No usable media URL for {gid}')
overall_bar.update(1)
continue
username = (info.get('userName') or '').strip() or 'misc'
creator_dir = safe_mkdir(output_dir / scraper_core.clean_filename(username))
filepath = creator_dir / _redgifs_extract_filename(dl_url)
safe_append_text(creator_dir / 'links.txt', dl_url + '\n')
safe_append_text(output_dir / 'all_links.txt', dl_url + '\n')
tqdm.write(f' {raw_url}\n -> {creator_dir.name}/{filepath.name}')
item_queue.put((dl_url, filepath))
time.sleep(0.1)
finally:
# Signal download workers to stop once all items in queue are processed
for _ in range(concurrency):
item_queue.put(None)
for t in download_threads:
t.join()
overall_bar.close()
sys.stdout.write("\n" * (concurrency + 4))
sys.stdout.flush()
finally:
api.stop()
# ----- File.al (XFS hoster) --------------------------------------------------
def _fileal_load_cookies(cookie_path: str) -> dict:
"""Load file.al cookies from a Netscape cookies.txt or EditThisCookie JSON export.
Login on file.al is reCAPTCHA-gated, so cookies must be exported from a
browser session. Netscape lines are tab-separated:
domain, includeSubdomains, path, secure, expiry, name, value.
"""
path = Path(cookie_path)
if not path.is_file():
tqdm.write(f' [!] Cookie file not found: {cookie_path}')
return {}
try:
text = path.read_text(encoding="utf-8", errors="replace")
except Exception as e:
tqdm.write(f' [!] Cannot read cookie file: {e}')
return {}
cookies = {}
if text.lstrip().startswith("["):
try:
for c in json.loads(text):
domain = c.get("domain") or ""
if c.get("name") and c.get("value") and "file.al" in domain:
cookies[c["name"]] = c["value"]
except Exception as e:
tqdm.write(f' [!] Failed to parse JSON cookies: {e}')
cookies = {}
else:
for line in text.splitlines():
if not line or line.startswith("#"):
continue
parts = line.split("\t")
if len(parts) >= 7 and "file.al" in parts[0]:
cookies[parts[5]] = parts[6]
return cookies
def _fileal_extract_form_payload(html):
"""Return the hidden-field payload of the download1/download2 form, or None.
XFS offers a free flow (submit method_free) and a subscribed flow (already
includes method_premium=1, which must be posted verbatim).
"""
for m in re.finditer(r"<[Ff]orm[^>]*>.*?</[Ff]orm>", html, re.S):
fields = dict(re.findall(r'<input type="hidden" name="([^"]+)" value="([^"]*)">', m.group(0)))
if fields.get("op") in ("download1", "download2"):
if "method_premium" not in fields and re.search(r'name="method_free"', m.group(0)):
fields["method_free"] = "Free Download"
return fields
return None
def _fileal_extract_direct_link(html):
"""Find the generated direct file URL on the download page.
Prefers the anchor labelled "Click here to download"; falls back to the
first off-site link that ends in a media/archive extension (the XFS CDN
pattern, excluding ad links).
"""
m = re.search(r'<a[^>]*href="([^"]+)"[^>]*>\s*Click here to download', html, re.S)
if m:
return m.group(1)
for m in re.finditer(r'<a[^>]*href="(https?://[^"]+)"', html):
host = urlparse(m.group(1)).netloc.lower()
path = urlparse(m.group(1)).path.lower()
if host != "file.al" and not host.endswith(".file.al") and path.endswith(
(".avi", ".mp4", ".mkv", ".webm", ".mov", ".flv", ".wmv", ".mpg", ".mpeg", ".zip", ".rar", ".7z")
):
return m.group(1)
return None
def _fileal_extract_filename(html, direct_url):
m = re.search(r"Filename:\s*([^<]+)", html)
if m:
name = m.group(1).strip()
if name:
return name
seg = urlparse(direct_url).path.rstrip("/").split("/")[-1]
return unquote(seg) if seg else None
def _fileal_download_one(url, output_dir, skip_existing, cookies, overall_bar, log_path):
from curl_cffi import requests as curl_req
session = curl_req.Session(impersonate="chrome")
if cookies:
session.cookies.update(cookies)
resp = session.get(url, timeout=60, allow_redirects=True)
resp.raise_for_status()
if "Premium Users only" in resp.text:
tqdm.write(f' [!] Paywalled (uploader subscription required): {url}')
overall_bar.update(1)
return
fallback_name = Path(urlparse(url).path).name
if fallback_name.lower().endswith(".html"):
fallback_name = fallback_name[:-5]
payload = _fileal_extract_form_payload(resp.text)
if not payload:
tqdm.write(f' [!] No download form on page: {url}')
overall_bar.update(1)
return
current = session.post("https://file.al/", data=payload, timeout=120, allow_redirects=True)
direct_url = filename = None
for _ in range(4):
if "Premium Users only" in current.text:
tqdm.write(f' [!] Paywalled (uploader subscription required): {url}')
overall_bar.update(1)
return
if not (current.headers.get("content-type") or "").startswith("text/html"):
direct_url = current.url
break
direct_url = _fileal_extract_direct_link(current.text)
if direct_url:
filename = _fileal_extract_filename(current.text, direct_url)
break
payload = _fileal_extract_form_payload(current.text)
if payload:
current = session.post("https://file.al/", data=payload, timeout=120, allow_redirects=True)
continue
break
if not direct_url:
tqdm.write(f' [!] No direct link resolved for: {url}')
overall_bar.update(1)
return
fname = scraper_core.clean_filename(filename or fallback_name or "file")
dest_path = output_dir / fname
dl_cookies = cookies or None
if skip_existing:
is_complete, existing_bytes = check_existing_file_sync(dest_path, direct_url, cookies=dl_cookies)
if is_complete:
overall_bar.record_skip(existing_bytes)
return
safe_append_text(log_path, direct_url + "\n")
success = scraper_core.download_file(direct_url, dest_path, cookies=dl_cookies, referer="https://file.al/")
if success:
size = dest_path.stat().st_size if dest_path.is_file() else 0
overall_bar.record_download(size, name=f"{dest_path.parent.name}/{dest_path.name}")
else:
overall_bar.update(1)
def fileal_download_urls(urls, output_dir, concurrency, skip_existing, cookies=None):
"""Download file.al links into output_dir/.
Sequential on purpose: XFS hosters are brittle under parallel requests, so
the concurrency argument is accepted only for CLI symmetry.
"""
unique = dedupe(urls)
if not unique:
tqdm.write(" No file.al URLs provided.")
return
safe_mkdir(output_dir)
overall_bar = OverallProgressTracker(total=len(unique), desc="file.al")
log_path = output_dir / "all_links.txt"
try:
for url in unique:
try:
_fileal_download_one(url, output_dir, skip_existing, cookies, overall_bar, log_path)
except Exception as e:
tqdm.write(f" [ERROR] {url}\n {e}")
overall_bar.update(1)
finally:
overall_bar.close()
sys.stdout.write("\n" * 2)
sys.stdout.flush()
# ---------------------------------------------------------------------------
# CLI
# ---------------------------------------------------------------------------
def print_site_list():
print("Supported sites (folder -> handler):")
print(f" {'Site':<16} {'Folder':<18} {'Domains':<40} {'Concurrency':>11}")
print("-" * 90)
for key, cfg in SITES.items():
print(f" {key:<16} {cfg['folder']:<18} {', '.join(cfg['domains']):<40} {cfg['default_concurrency']:>11}")
print()
print("Inputs are auto-routed: URLs by hostname, any bare token is a RedGIFs creator name.")
print(".txt files may list URLs and/or creator names (one per line, '#' = comment).")
def main():
parser = argparse.ArgumentParser(
prog="downloader",
description="Unified downloader: routes URLs/creator names to the right site and "
"places files in <website_folder>/videos/<creator>/ exactly like the "
"individual downloaders did.",
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog="""examples:
python downloader.py https://en.chezcathy.com/user/7115/foo https://www.thisvid.com/video/9876/ creator1 creator2
python downloader.py urls.txt # file may mix URLs and creator names
python downloader.py --update-all-creators
python downloader.py --list-sites
""",
)
parser.add_argument(
"targets", nargs="*",
help="URLs from any supported site, RedGIFs creator usernames, or .txt files "
"containing them (one per line, '#' = comment).",
)
parser.add_argument(
"--concurrency", type=int, default=None,
help="Number of concurrent downloads. Defaults per site: "
"chezcathy/luxuretv/generic=3, pornzoo.love=2, redgifs=5.",
)
parser.add_argument(
"--skip-existing", action="store_true", dest="skip_existing", default=True,
help="Skip already-downloaded files (default: true).",
)
parser.add_argument(
"--no-skip-existing", action="store_false", dest="skip_existing",
help="Re-download existing files.",
)
parser.add_argument(
"-o", "--output", default=str(ROOT / "redgifs" / "videos"),
help="RedGIFs output directory (default: redgifs/videos).",
)
parser.add_argument(
"--update-all-creators", action="store_true",
help="RedGIFs: gets the name of each creator folder in --output, merges them into "
"redgifs/creators.txt (deduplicated), then uses that list as the download list.",
)
parser.add_argument(
"--user-playlist", default=None,
help="LuxureTV default user playlist (used when no targets are given).",
)
parser.add_argument(
"--skip-redgifs-cooldown", "--skip-cooldown", action="store_true", dest="skip_redgifs_cooldown",
help="Skip RedGIFs API rate limit / HTTP 429 cooldown wait period (useful if IP was changed).",
)
parser.add_argument(
"--list-sites", action="store_true",
help="List supported websites and exit.",
)
parser.add_argument(
"--cookies", default=None,
help="Path to a browser cookie export (Netscape cookies.txt or EditThisCookie "
"JSON) for sites that need an authenticated session, e.g. file.al.",
)
args = parser.parse_args()
if args.list_sites:
print_site_list()
return
creators_file = ROOT / "redgifs" / "creators.txt"
groups, warnings = resolve_targets(args.targets)
for w in warnings:
print(f" [!] {w}", file=sys.stderr)
if args.update_all_creators:
names = redgifs_update_creator_list(Path(args.output).resolve(), creators_file)
if names:
groups["redgifs"] = dedupe(groups.get("redgifs", []) + names)
if not groups:
if args.user_playlist:
groups["luxuretv"] = [args.user_playlist]
else:
raw = input('RedGIFs creator username(s) (comma/space separated): ').strip()
creators = [c.strip() for c in re.split(r'[,\s]+', raw) if c.strip()]
if not creators:
sys.exit('No creators provided.')
groups["redgifs"] = creators
if not groups:
sys.exit("No targets to process. Pass URLs, creator names, or .txt files.")
for site_key, targets in groups.items():
cfg = SITES[site_key]
concurrency = args.concurrency if args.concurrency is not None else cfg["default_concurrency"]
download_dir = ROOT / cfg["folder"] / "videos"
handler = cfg["handler"]
tqdm.write(f"\n=== [{site_key}] processing {len(targets)} target(s) into {download_dir} ===")
if handler == "generic":
g = GENERIC_SITES[site_key]
asyncio.run(generic_downloader.process_urls(
g["site_name"], targets, download_dir,
g["is_video_link"], g["next_page_selector"],
g["video_selector"], g["uploader_eval_js"],
concurrency, args.skip_existing,
))
elif handler == "chezcathy":
asyncio.run(chezcathy_process_urls(download_dir, targets, concurrency, args.skip_existing))
elif handler == "luxuretv":
asyncio.run(luxuretv_process_urls(download_dir, targets, concurrency, args.skip_existing))
elif handler == "pornzoo":
asyncio.run(pornzoo_process_urls(download_dir, targets, concurrency, args.skip_existing))
elif handler == "pornhub":
asyncio.run(pornhub_process_urls(download_dir, targets, concurrency, args.skip_existing))
elif handler == "zootubevip":
asyncio.run(zootubevip_process_urls(download_dir, targets, concurrency, args.skip_existing))
elif handler == "zootube1":
asyncio.run(zootube1_process_urls(download_dir, targets, concurrency, args.skip_existing))
elif handler == "tickzoo":
asyncio.run(tickzoo_process_urls(download_dir, targets, concurrency, args.skip_existing))
elif handler == "redgifs":
output_dir = Path(args.output).resolve()
creator_names = [t for t in targets if not looks_like_url(t)]
gif_urls = [t for t in targets if looks_like_url(t)]
if creator_names:
redgifs_download_creators(creator_names, output_dir, concurrency, args.skip_existing, skip_cooldown=args.skip_redgifs_cooldown)
if gif_urls:
redgifs_download_gif_urls(gif_urls, output_dir, concurrency, args.skip_existing, skip_cooldown=args.skip_redgifs_cooldown)
elif handler == "fileal":
file_cookies = _fileal_load_cookies(args.cookies) if args.cookies else {}
if args.cookies and not file_cookies:
print(f" [!] No file.al cookies loaded from '{args.cookies}' - uploader paywalls will be skipped.", file=sys.stderr)
fileal_download_urls(targets, download_dir, concurrency, args.skip_existing, file_cookies)
else:
print(f" [!] Unknown handler '{handler}' for site '{site_key}' - skipped.", file=sys.stderr)
# Print final master summary for the entire run
MASTER_TRACKER.close()
MASTER_TRACKER.print_summary()
if __name__ == "__main__":
main()