2853 lines
104 KiB
Python
2853 lines
104 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Master downloader for every website in this workspace.
|
|
|
|
Replaces the per-website downloader scripts with one entry point. Given a
|
|
mix of website URLs, RedGIFs creator usernames, and/or .txt files containing
|
|
either (one URL or creator per line, blank lines and '#' comments allowed),
|
|
it:
|
|
|
|
* identifies the target website for every input
|
|
* downloads / names / places files EXACTLY as each individual script did:
|
|
<website_folder>/videos/<creator_or_uploader>/<file>.mp4
|
|
(RedGIFs additionally writes <creator>/links.txt and appends all_links.txt)
|
|
|
|
Examples:
|
|
python downloader.py https://en.chezcathy.com/user/7115/whatever \
|
|
https://en.luxuretv.com/videos/12345/foo.html \
|
|
https://www.thisvid.com/video/9876/ \
|
|
some_redgifs_creator another_creator urls.txt
|
|
|
|
python downloader.py --update-all-creators # refresh redgifs creators.txt
|
|
python downloader.py --list-sites # show supported sites
|
|
|
|
Adding a new site:
|
|
1. Add a registry entry in SITES (folder name, domains, handler, defaults).
|
|
2. If the site is a standard HTML <video> site, add a GENERIC_SITES entry.
|
|
3. Otherwise add a handler function under "Custom site handlers".
|
|
"""
|
|
|
|
import os
|
|
import sys
|
|
import re
|
|
import json
|
|
import asyncio
|
|
import argparse
|
|
import random
|
|
import time
|
|
from pathlib import Path
|
|
from urllib.parse import urlparse
|
|
from html import unescape
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
import threading
|
|
|
|
import requests
|
|
from tqdm import tqdm
|
|
|
|
import scraper_core
|
|
import generic_downloader
|
|
|
|
ROOT = Path(__file__).resolve().parent
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Shared helpers
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def format_size(bytes_count):
|
|
if bytes_count < 1024:
|
|
return f"{bytes_count} B"
|
|
elif bytes_count < 1024 * 1024:
|
|
return f"{bytes_count / 1024:.2f} KB"
|
|
elif bytes_count < 1024 * 1024 * 1024:
|
|
return f"{bytes_count / (1024 * 1024):.2f} MB"
|
|
else:
|
|
return f"{bytes_count / (1024 * 1024 * 1024):.2f} GB"
|
|
|
|
|
|
class MasterProgressTracker:
|
|
"""Run-wide master progress tracker that aggregates stats across all sources and creators."""
|
|
def __init__(self):
|
|
self.total_files = 0
|
|
self.processed_files = 0
|
|
self.skipped_files = 0
|
|
self.skipped_bytes = 0
|
|
self.downloaded_files = 0
|
|
self.downloaded_bytes = 0
|
|
self.start_time = time.time()
|
|
self.lock = threading.Lock()
|
|
self.bar = tqdm(
|
|
total=0,
|
|
desc="Master Progress",
|
|
position=0,
|
|
leave=True,
|
|
ncols=100,
|
|
bar_format="{desc}: {n_fmt}/{total_fmt} |{bar:54}| {percentage:.2f}%",
|
|
)
|
|
self.text_bar = tqdm(
|
|
total=0,
|
|
position=1,
|
|
leave=True,
|
|
ncols=120,
|
|
desc="Skipped: 0 files equaling 0 B | Downloaded: 0 files equaling 0 B (Avg: 0.0 B/s)",
|
|
bar_format="{desc}",
|
|
)
|
|
|
|
def add_total(self, n: int = 1):
|
|
with self.lock:
|
|
self.total_files += n
|
|
self.bar.total = self.total_files
|
|
self.bar.refresh()
|
|
|
|
def record_skip(self, size: int = 0):
|
|
with self.lock:
|
|
self.skipped_files += 1
|
|
self.skipped_bytes += size
|
|
self.processed_files = self.skipped_files + self.downloaded_files
|
|
self._refresh_bars()
|
|
|
|
def record_download(self, size: int = 0):
|
|
with self.lock:
|
|
self.downloaded_files += 1
|
|
self.downloaded_bytes += size
|
|
self.processed_files = self.skipped_files + self.downloaded_files
|
|
self._refresh_bars()
|
|
|
|
def _refresh_bars(self):
|
|
self.bar.total = self.total_files
|
|
self.bar.n = self.processed_files
|
|
self.bar.refresh()
|
|
elapsed = time.time() - self.start_time
|
|
avg_speed = self.downloaded_bytes / elapsed if elapsed > 0 else 0
|
|
text = (
|
|
f"Skipped: {self.skipped_files} files equaling {format_size(self.skipped_bytes)} | "
|
|
f"Downloaded: {self.downloaded_files} files equaling {format_size(self.downloaded_bytes)} "
|
|
f"(Avg: {format_size(avg_speed)}/s)"
|
|
)
|
|
self.text_bar.set_description_str(text)
|
|
self.text_bar.refresh()
|
|
|
|
def close(self):
|
|
with self.lock:
|
|
self.bar.close()
|
|
self.text_bar.close()
|
|
|
|
def format_status(self) -> str:
|
|
with self.lock:
|
|
elapsed = time.time() - self.start_time
|
|
avg_speed = self.downloaded_bytes / elapsed if elapsed > 0 else 0
|
|
pct = (self.processed_files / self.total_files * 100) if self.total_files > 0 else 0.0
|
|
line1 = f"Master Progress: {self.processed_files}/{self.total_files} ({pct:.2f}%)"
|
|
line2 = (
|
|
f"Skipped: {self.skipped_files} files equaling {format_size(self.skipped_bytes)} | "
|
|
f"Downloaded: {self.downloaded_files} files equaling {format_size(self.downloaded_bytes)} "
|
|
f"(Avg: {format_size(avg_speed)}/s)"
|
|
)
|
|
return f"{line1}\n{line2}"
|
|
|
|
def print_summary(self):
|
|
with self.lock:
|
|
elapsed = time.time() - self.start_time
|
|
avg_speed = self.downloaded_bytes / elapsed if elapsed > 0 else 0
|
|
mins, secs = divmod(int(elapsed), 60)
|
|
hours, mins = divmod(mins, 60)
|
|
time_str = f"{hours:02d}:{mins:02d}:{secs:02d}" if hours else f"{mins:02d}:{secs:02d}"
|
|
pct = (self.processed_files / self.total_files * 100) if self.total_files > 0 else 0.0
|
|
|
|
sep = "=" * 80
|
|
tqdm.write(f"\n{sep}")
|
|
tqdm.write(" MASTER DOWNLOAD RUN SUMMARY")
|
|
tqdm.write(sep)
|
|
tqdm.write(f"Total Targets Tracked : {self.total_files} files (Processed: {self.processed_files}, {pct:.2f}%)")
|
|
tqdm.write(
|
|
f"Skipped : {self.skipped_files} files equaling {format_size(self.skipped_bytes)}"
|
|
)
|
|
tqdm.write(
|
|
f"Downloaded : {self.downloaded_files} files equaling {format_size(self.downloaded_bytes)} "
|
|
f"(Avg: {format_size(avg_speed)}/s)"
|
|
)
|
|
tqdm.write(f"Total Run Time : {time_str}")
|
|
tqdm.write(f"{sep}\n")
|
|
|
|
|
|
MASTER_TRACKER = MasterProgressTracker()
|
|
|
|
|
|
class OverallProgressTracker:
|
|
def __init__(self, total=0, desc="Overall Progress", master: MasterProgressTracker = None):
|
|
self.bar = tqdm(
|
|
total=total,
|
|
desc=desc,
|
|
position=2,
|
|
leave=True,
|
|
ncols=100,
|
|
bar_format="{desc}: {n_fmt}/{total_fmt} |{bar:54}| {percentage:.2f}%",
|
|
)
|
|
self.text_bar = tqdm(
|
|
total=0,
|
|
position=3,
|
|
leave=True,
|
|
ncols=120,
|
|
desc="Skipped: 0 files equaling 0 B | Downloaded: 0 files equaling 0 B (Avg: 0.0 B/s)",
|
|
bar_format="{desc}",
|
|
)
|
|
self.master = master if master is not None else MASTER_TRACKER
|
|
self.skipped_files = 0
|
|
self.skipped_bytes = 0
|
|
self.downloaded_files = 0
|
|
self.downloaded_bytes = 0
|
|
self.start_time = time.time()
|
|
self.lock = threading.Lock()
|
|
if total > 0 and self.master:
|
|
self.master.add_total(total)
|
|
|
|
@property
|
|
def total(self):
|
|
return self.bar.total
|
|
|
|
@total.setter
|
|
def total(self, value):
|
|
old_val = self.bar.total or 0
|
|
self.bar.total = value
|
|
diff = value - old_val
|
|
if diff > 0 and self.master:
|
|
self.master.add_total(diff)
|
|
|
|
def refresh(self):
|
|
self.bar.refresh()
|
|
self.text_bar.refresh()
|
|
|
|
def update(self, n=1):
|
|
self.bar.update(n)
|
|
|
|
def close(self):
|
|
self.bar.close()
|
|
self.text_bar.close()
|
|
|
|
def record_skip(self, size: int = 0):
|
|
with self.lock:
|
|
self.skipped_files += 1
|
|
self.skipped_bytes += size
|
|
self._refresh_text()
|
|
self.bar.update(1)
|
|
if self.master:
|
|
self.master.record_skip(size)
|
|
|
|
def record_download(self, size: int = 0, name: str = ""):
|
|
with self.lock:
|
|
self.downloaded_files += 1
|
|
self.downloaded_bytes += size
|
|
self._refresh_text()
|
|
self.bar.update(1)
|
|
if name:
|
|
tqdm.write(f" [FINISHED] '{name}' downloaded successfully ({format_size(size)}).")
|
|
if self.master:
|
|
self.master.record_download(size)
|
|
|
|
def _refresh_text(self):
|
|
elapsed = time.time() - self.start_time
|
|
avg_speed = self.downloaded_bytes / elapsed if elapsed > 0 else 0
|
|
text = (
|
|
f"Skipped: {self.skipped_files} files equaling {format_size(self.skipped_bytes)} | "
|
|
f"Downloaded: {self.downloaded_files} files equaling {format_size(self.downloaded_bytes)} "
|
|
f"(Avg: {format_size(avg_speed)}/s)"
|
|
)
|
|
self.text_bar.set_description_str(text)
|
|
|
|
|
|
class CheckResult(int):
|
|
def __new__(cls, is_complete: bool, size: int):
|
|
val = 1 if is_complete else 0
|
|
obj = super().__new__(cls, val)
|
|
obj.is_complete = is_complete
|
|
obj.size = size
|
|
return obj
|
|
|
|
def __bool__(self):
|
|
return self.is_complete
|
|
|
|
def __iter__(self):
|
|
yield self.is_complete
|
|
yield self.size
|
|
|
|
FILTERED_WORDS = {
|
|
"shit", "scat", "shitty", "poop", "horseshit", "bullshit", "cowshit",
|
|
"shitting", "crap", "feces", "dung", "pungpile", "pungheap", "pooping",
|
|
"crapping", "manure", "turd", "turds",
|
|
}
|
|
|
|
|
|
def is_filtered(url: str) -> bool:
|
|
lowered = url.lower()
|
|
return any(w in lowered for w in FILTERED_WORDS)
|
|
|
|
|
|
def dedupe(items):
|
|
seen = set()
|
|
out = []
|
|
for it in items:
|
|
if it not in seen:
|
|
seen.add(it)
|
|
out.append(it)
|
|
return out
|
|
|
|
|
|
def safe_mkdir(path: Path, retries: int = 5, delay: float = 2.0) -> Path:
|
|
"""mkdir with retries to handle transient SMB/network share I/O errors (e.g. WinError 1117).
|
|
|
|
If the directory entry on the remote filesystem is corrupted and permanently causes
|
|
WinError 1117, falls back to an alternative directory name (e.g. name_redgifs)
|
|
so downloads can proceed without crashing the entire run.
|
|
"""
|
|
for attempt in range(retries):
|
|
try:
|
|
path.mkdir(parents=True, exist_ok=True)
|
|
return path
|
|
except OSError as e:
|
|
if attempt < retries - 1:
|
|
tqdm.write(f" [!] Disk/Network I/O error creating {path.name} (attempt {attempt+1}/{retries}): {e}. Retrying in {delay}s...")
|
|
time.sleep(delay)
|
|
else:
|
|
# Persistent I/O error on this specific directory entry
|
|
win_err = getattr(e, 'winerror', None)
|
|
if win_err == 1117 or '1117' in str(e):
|
|
for suffix in ['_redgifs', '_dir', f'_{int(time.time())}']:
|
|
fallback = path.parent / f"{path.name}{suffix}"
|
|
try:
|
|
fallback.mkdir(parents=True, exist_ok=True)
|
|
tqdm.write(
|
|
f" [!] WinError 1117 persisted for '{path.name}' (filesystem entry corrupted). "
|
|
f"Using fallback directory '{fallback.name}' instead."
|
|
)
|
|
return fallback
|
|
except Exception:
|
|
continue
|
|
raise
|
|
|
|
|
|
def safe_append_text(path: Path, text: str, retries: int = 5, delay: float = 2.0):
|
|
"""Append text to a file with retries for transient network drive I/O errors."""
|
|
for attempt in range(retries):
|
|
try:
|
|
with open(path, 'a', encoding='utf-8') as f:
|
|
f.write(text)
|
|
return
|
|
except OSError as e:
|
|
if attempt < retries - 1:
|
|
time.sleep(delay)
|
|
else:
|
|
raise
|
|
|
|
|
|
async def get_cloudflare_cookies(scraper, url):
|
|
"""Use Playwright to solve one Cloudflare challenge and return browser cookies."""
|
|
page = await scraper.new_page()
|
|
try:
|
|
try:
|
|
await scraper.resolve_page(page, url)
|
|
except Exception as e:
|
|
tqdm.write(f" Warning: Cloudflare resolve_page failed: {e}. Trying to proceed with currently gathered cookies.")
|
|
raw_cookies = await scraper.context.cookies()
|
|
return {c["name"]: c["value"] for c in raw_cookies}
|
|
finally:
|
|
await page.close()
|
|
|
|
|
|
def get_remote_file_size(url: str, headers: dict = None, cookies: dict = None) -> int:
|
|
"""Fetch the content-length of the remote video file with HEAD/GET."""
|
|
req_headers = {
|
|
"User-Agent": scraper_core.USER_AGENT,
|
|
"Accept": "*/*",
|
|
}
|
|
if headers:
|
|
req_headers.update(headers)
|
|
|
|
if cookies:
|
|
try:
|
|
from curl_cffi import requests as curl_req
|
|
# Try HEAD first
|
|
try:
|
|
resp = curl_req.head(
|
|
url, impersonate="chrome", headers=req_headers,
|
|
cookies=cookies, timeout=8,
|
|
)
|
|
cl = int(resp.headers.get("content-length", 0))
|
|
if cl > 0:
|
|
return cl
|
|
except Exception:
|
|
pass
|
|
|
|
# Stream GET fallback
|
|
with curl_req.get(
|
|
url, impersonate="chrome", headers=req_headers,
|
|
cookies=cookies, stream=True, timeout=8,
|
|
) as resp:
|
|
if resp.status_code == 200:
|
|
return int(resp.headers.get("content-length", 0))
|
|
except Exception:
|
|
pass
|
|
|
|
try:
|
|
# Try HEAD first
|
|
try:
|
|
resp = requests.head(url, headers=req_headers, timeout=8, cookies=cookies, allow_redirects=True)
|
|
cl = int(resp.headers.get("content-length", 0))
|
|
if cl > 0:
|
|
return cl
|
|
except Exception:
|
|
pass
|
|
|
|
# Stream GET fallback
|
|
with requests.get(url, headers=req_headers, stream=True, timeout=8, cookies=cookies) as resp:
|
|
if resp.status_code == 200:
|
|
return int(resp.headers.get("content-length", 0))
|
|
except Exception:
|
|
pass
|
|
|
|
return 0
|
|
|
|
|
|
def check_existing_file_sync(dest_path: Path, src: str, headers: dict = None, cookies: dict = None) -> CheckResult:
|
|
"""Check integrity of an existing file against remote file size synchronously.
|
|
|
|
Returns CheckResult(True, local_size) if file exists and is complete (skip download).
|
|
Returns CheckResult(False, 0) if file does not exist, or if it is incomplete/corrupted (trigger redownload).
|
|
Prints messages describing what is happening in both cases.
|
|
"""
|
|
if not dest_path.exists():
|
|
return CheckResult(False, 0)
|
|
|
|
label = f"{dest_path.parent.name}/{dest_path.name}"
|
|
local_size = dest_path.stat().st_size
|
|
|
|
if local_size == 0:
|
|
tqdm.write(f" [OVERWRITE] '{label}' is incomplete (0 bytes). Overwriting and redownloading...")
|
|
try:
|
|
dest_path.unlink()
|
|
except Exception:
|
|
pass
|
|
return CheckResult(False, 0)
|
|
|
|
remote_size = get_remote_file_size(src, headers, cookies)
|
|
|
|
if remote_size > 0:
|
|
if local_size < remote_size:
|
|
tqdm.write(
|
|
f" [OVERWRITE] '{label}' is incomplete "
|
|
f"(local: {format_size(local_size)}, expected: {format_size(remote_size)}). Overwriting..."
|
|
)
|
|
try:
|
|
dest_path.unlink()
|
|
except Exception:
|
|
pass
|
|
return CheckResult(False, 0)
|
|
else:
|
|
tqdm.write(
|
|
f" [SKIP] '{label}' already exists and is complete ({format_size(local_size)})."
|
|
)
|
|
return CheckResult(True, local_size)
|
|
else:
|
|
# Fallback if Content-Length header is not provided by server
|
|
tqdm.write(
|
|
f" [SKIP] '{label}' already downloaded ({format_size(local_size)}, remote size unavailable)."
|
|
)
|
|
return CheckResult(True, local_size)
|
|
|
|
|
|
async def check_existing_file(dest_path: Path, src: str, headers: dict = None, cookies: dict = None) -> CheckResult:
|
|
"""Check integrity of an existing file against remote file size asynchronously."""
|
|
return await asyncio.to_thread(check_existing_file_sync, dest_path, src, headers, cookies)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Site registry
|
|
# ---------------------------------------------------------------------------
|
|
# handler: "generic" -> GENERIC_SITES entry drives it via generic_downloader;
|
|
# otherwise a dedicated function in "Custom site handlers" below.
|
|
|
|
SITES = {
|
|
"chezcathy": {
|
|
"folder": "chezcathy",
|
|
"domains": ["chezcathy.com"],
|
|
"handler": "chezcathy",
|
|
"default_concurrency": 3,
|
|
},
|
|
"luxuretv": {
|
|
"folder": "luxuretv",
|
|
"domains": ["luxuretv.com"],
|
|
"handler": "luxuretv",
|
|
"default_concurrency": 3,
|
|
},
|
|
"motherless": {
|
|
"folder": "motherless",
|
|
"domains": ["motherless.com"],
|
|
"handler": "generic",
|
|
"default_concurrency": 3,
|
|
},
|
|
"pmvhaven": {
|
|
"folder": "pmvhaven",
|
|
"domains": ["pmvhaven.com"],
|
|
"handler": "generic",
|
|
"default_concurrency": 3,
|
|
},
|
|
"pornzoo.love": {
|
|
"folder": "pornzoo.love",
|
|
"domains": ["pornzoo.love"],
|
|
"handler": "pornzoo",
|
|
"default_concurrency": 2,
|
|
},
|
|
"pornhub": {
|
|
"folder": "pornhub",
|
|
"domains": ["pornhub.com"],
|
|
"handler": "pornhub",
|
|
"default_concurrency": 2,
|
|
},
|
|
"redgifs": {
|
|
"folder": "redgifs",
|
|
"domains": ["redgifs.com"],
|
|
"handler": "redgifs",
|
|
"default_concurrency": 5,
|
|
},
|
|
"rule34video": {
|
|
"folder": "rule34video",
|
|
"domains": ["rule34video.com"],
|
|
"handler": "generic",
|
|
"default_concurrency": 3,
|
|
},
|
|
"thisvid": {
|
|
"folder": "thisvid",
|
|
"domains": ["thisvid.com"],
|
|
"handler": "generic",
|
|
"default_concurrency": 3,
|
|
},
|
|
"voyeurflash": {
|
|
"folder": "voyeurflash",
|
|
"domains": ["voyeurflash.com"],
|
|
"handler": "generic",
|
|
"default_concurrency": 3,
|
|
},
|
|
"xhamster": {
|
|
"folder": "xhamster",
|
|
"domains": ["xhamster.com", "xhamster.xxx", "xhamster18.com"],
|
|
"handler": "generic",
|
|
"default_concurrency": 3,
|
|
},
|
|
"xxxvideoszoo": {
|
|
"folder": "xxxvideoszoo",
|
|
"domains": ["xxxvideoszoo.com"],
|
|
"handler": "generic",
|
|
"default_concurrency": 3,
|
|
},
|
|
"zootube.vip": {
|
|
"folder": "zootube.vip",
|
|
"domains": ["zootube.vip"],
|
|
"handler": "zootubevip",
|
|
"default_concurrency": 3,
|
|
},
|
|
"zootube1": {
|
|
"folder": "zootube1",
|
|
"domains": ["zootube1.com"],
|
|
"handler": "zootube1",
|
|
"default_concurrency": 3,
|
|
},
|
|
}
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Generic site configs (standard HTML <video> player sites)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _motherless_is_video_link(url):
|
|
# Match Motherless video URL patterns like:
|
|
# https://motherless.com/AC493CD
|
|
# https://motherless.com/g/group_name/AC493CD
|
|
path = urlparse(url).path.rstrip("/")
|
|
if not path:
|
|
return False
|
|
last_part = path.split("/")[-1]
|
|
# Motherless IDs are typically 7 or 8 characters of uppercase letters and numbers
|
|
return bool(re.match(r"^[A-Z0-9]{6,9}$", last_part))
|
|
|
|
|
|
def _pmvhaven_is_video_link(url):
|
|
# Match watch/video URLs like: /video/slug_id, /videos/slug_id, or /watch/slug_id
|
|
return "/video" in url or "/watch" in url
|
|
|
|
|
|
def _rule34video_is_video_link(url):
|
|
# Match Rule34Video paths like: https://rule34video.com/video/12345/slug/
|
|
return "/video/" in url
|
|
|
|
|
|
def _thisvid_is_video_link(url):
|
|
# Match ThisVid paths like: https://thisvid.com/video/12345/
|
|
return "/video/" in url
|
|
|
|
|
|
def _voyeurflash_is_video_link(url):
|
|
# Match typical voyeurflash video paths like: https://voyeurflash.com/video/12345/xyz/
|
|
return "/video/" in url or "/watch/" in url
|
|
|
|
|
|
def _xhamster_is_video_link(url):
|
|
# Match standard xHamster video pages: contains /videos/ and doesn't point to user grids/channels
|
|
return (
|
|
"/videos/" in url
|
|
and "/users/" not in url
|
|
and "/channels/" not in url
|
|
and "/creators/" not in url
|
|
)
|
|
|
|
|
|
def _xxxvideoszoo_is_video_link(url):
|
|
return "/v/" in url or "/video/" in url or "/watch/" in url
|
|
|
|
|
|
def _zootube1_is_video_link(url):
|
|
return "/allvideos/" in url or "/video/" in url or "/watch/" in url
|
|
|
|
|
|
GENERIC_SITES = {
|
|
"motherless": {
|
|
"site_name": "Motherless",
|
|
"is_video_link": _motherless_is_video_link,
|
|
"next_page_selector": "a:has-text('Next'), a.next, a.pagination-next",
|
|
"video_selector": "video",
|
|
"uploader_eval_js": """() => {
|
|
const memberLink = document.querySelector('a[href*="/members/"]');
|
|
if (memberLink) return memberLink.innerText.trim();
|
|
const uploadText = document.querySelector('.upload-info');
|
|
if (uploadText) {
|
|
const m = uploadText.innerText.match(/by\\s+(\\S+)/i);
|
|
if (m) return m[1];
|
|
}
|
|
return 'unknown';
|
|
}""",
|
|
},
|
|
"pmvhaven": {
|
|
"site_name": "PMVHaven",
|
|
"is_video_link": _pmvhaven_is_video_link,
|
|
"next_page_selector": "a:has-text('Next'), .next, a.pagination-next",
|
|
"video_selector": "video",
|
|
"uploader_eval_js": """() => {
|
|
const el = Array.from(document.querySelectorAll('a[href*="/user/"], a[href*="/users/"], a[href*="/profile/"], .uploader-name'))
|
|
.find(a => a.innerText.trim() !== '');
|
|
if (el) {
|
|
const h3 = el.querySelector('h3');
|
|
if (h3) {
|
|
const clone = h3.cloneNode(true);
|
|
const spans = clone.querySelectorAll('span');
|
|
spans.forEach(s => s.remove());
|
|
return clone.textContent.trim();
|
|
}
|
|
return el.innerText.replace(/uploader/gi, '').trim();
|
|
}
|
|
return 'unknown';
|
|
}""",
|
|
},
|
|
"rule34video": {
|
|
"site_name": "Rule34Video",
|
|
"is_video_link": _rule34video_is_video_link,
|
|
"next_page_selector": "a.next, a:has-text('Next'), .pagination-next",
|
|
"video_selector": "video",
|
|
"uploader_eval_js": """() => {
|
|
const el = document.querySelector('a[href*="/user/"], a[href*="/profile/"], .video-uploader a');
|
|
return el ? el.innerText.trim() : 'unknown';
|
|
}""",
|
|
},
|
|
"thisvid": {
|
|
"site_name": "ThisVid",
|
|
"is_video_link": _thisvid_is_video_link,
|
|
"next_page_selector": "a:has-text('Next'), .next, a.pagination-next",
|
|
"video_selector": "video",
|
|
"uploader_eval_js": """() => {
|
|
const el = document.querySelector('a[href*="/members/"], a[href*="/profile/"], .username');
|
|
return el ? el.innerText.trim() : 'unknown';
|
|
}""",
|
|
},
|
|
"voyeurflash": {
|
|
"site_name": "VoyeurFlash",
|
|
"is_video_link": _voyeurflash_is_video_link,
|
|
"next_page_selector": "a:has-text('Next'), .next, a.pagination-next",
|
|
"video_selector": "video",
|
|
"uploader_eval_js": """() => {
|
|
const el = document.querySelector('a[href*="/members/"], a[href*="/user/"], a[href*="/profile/"]');
|
|
return el ? el.innerText.trim() : 'unknown';
|
|
}""",
|
|
},
|
|
"xhamster": {
|
|
"site_name": "xHamster",
|
|
"is_video_link": _xhamster_is_video_link,
|
|
"next_page_selector": "a:has-text('Next'), .next, a[rel='next']",
|
|
"video_selector": "video",
|
|
"uploader_eval_js": """() => {
|
|
// 1. Try channel link
|
|
const channelLink = document.querySelector('a[href*="/channels/"]');
|
|
if (channelLink && channelLink.innerText.trim()) {
|
|
return channelLink.innerText.trim().replace(/\\n/g, ' ').trim();
|
|
}
|
|
|
|
// 2. Try user link
|
|
const userLink = document.querySelector('.entity-author-container__name, a[href*="/users/"]');
|
|
if (userLink && userLink.innerText.trim()) {
|
|
return userLink.innerText.trim();
|
|
}
|
|
|
|
// 3. Try creator logo or class
|
|
const creatorLink = document.querySelector('a[href*="/creators/"] .video-uploader__name, .video-uploader__name');
|
|
if (creatorLink && creatorLink.innerText.trim()) {
|
|
return creatorLink.innerText.trim();
|
|
}
|
|
|
|
return 'unknown';
|
|
}""",
|
|
},
|
|
"xxxvideoszoo": {
|
|
"site_name": "XXXVideosZoo",
|
|
"is_video_link": _xxxvideoszoo_is_video_link,
|
|
"next_page_selector": "a:has-text('Next'), .next, a.pagination-next",
|
|
"video_selector": "video",
|
|
"uploader_eval_js": """() => {
|
|
const el = document.querySelector('a[href*="/members/"], a[href*="/user/"], a[href*="/profile/"]');
|
|
return el ? el.innerText.trim() : 'unknown';
|
|
}""",
|
|
},
|
|
}
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Input classification: URL -> site, bare token -> RedGIFs creator
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def looks_like_url(token: str) -> bool:
|
|
return "://" in token or token.lower().startswith("www.")
|
|
|
|
|
|
def find_site_for_url(url: str):
|
|
"""Return the site key that owns a URL's host, or None."""
|
|
try:
|
|
parsed = urlparse(url)
|
|
host = parsed.netloc.lower()
|
|
except ValueError:
|
|
return None
|
|
if not host:
|
|
# Scheme-less input like "www.xhamster.com/xyz" - retry with a scheme
|
|
try:
|
|
host = urlparse("https://" + url).netloc.lower()
|
|
except ValueError:
|
|
return None
|
|
if not host:
|
|
return None
|
|
for key, cfg in SITES.items():
|
|
for domain in cfg["domains"]:
|
|
if host == domain or host.endswith("." + domain):
|
|
return key
|
|
return None
|
|
|
|
|
|
def classify_target(token: str):
|
|
"""Classify one CLI token -> (site_key, target).
|
|
|
|
URLs are routed by hostname; any non-URL token is treated as a RedGIFs
|
|
creator username (matching the original redgifs_creator_downloader.py
|
|
behaviour, which accepted bare creator names).
|
|
"""
|
|
if looks_like_url(token):
|
|
site = find_site_for_url(token)
|
|
if site is None:
|
|
raise ValueError(
|
|
"no supported website matched (run with --list-sites to see supported domains)"
|
|
)
|
|
if site == "redgifs":
|
|
# RedGIFs accepts /watch/<id> URLs, media URLs, and bare creator names.
|
|
return site, token
|
|
return site, token
|
|
return "redgifs", token
|
|
|
|
|
|
def resolve_targets(targets_args):
|
|
"""Expand CLI args (URLs / creator names / .txt files) into {site_key: [targets]}.
|
|
|
|
Files are read line by line and EVERY whitespace-separated token is
|
|
classified independently, so a single file may mix URLs and creators.
|
|
"""
|
|
groups = {}
|
|
warnings = []
|
|
for arg in targets_args:
|
|
p = Path(arg)
|
|
try:
|
|
is_file = p.is_file()
|
|
except OSError:
|
|
is_file = False
|
|
if is_file:
|
|
try:
|
|
with open(p, "r", encoding="utf-8") as f:
|
|
tokens = [
|
|
t for line in f if not line.lstrip().startswith("#")
|
|
for t in line.strip().split()
|
|
]
|
|
except Exception as e:
|
|
warnings.append(f"Error reading file '{arg}': {e}")
|
|
continue
|
|
else:
|
|
tokens = [arg]
|
|
for tok in tokens:
|
|
if not tok or tok.startswith("#"):
|
|
continue
|
|
try:
|
|
site, target = classify_target(tok)
|
|
except ValueError as e:
|
|
warnings.append(f"Skipping '{tok}': {e}")
|
|
continue
|
|
groups.setdefault(site, []).append(target)
|
|
return groups, warnings
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Custom site handlers
|
|
# ---------------------------------------------------------------------------
|
|
|
|
# ----- ChezCathy ------------------------------------------------------------
|
|
|
|
|
|
def _chezcathy_extract_video_links(html: str) -> list:
|
|
"""Extract video links matching player or adult page patterns on chezcathy."""
|
|
# Matches URLs like: /extreme-sex/player/35738/...html or /adult/769439/...html
|
|
patterns = [
|
|
r'href="(/[^"]*?/player/\d+/[^"]*?\.html)"',
|
|
r'href="(/adult/\d+/[^"]*?\.html)"',
|
|
r'href="(https?://en\.chezcathy\.com/[^"]*?/player/\d+/[^"]*?\.html)"',
|
|
r'href="(https?://en\.chezcathy\.com/adult/\d+/[^"]*?\.html)"'
|
|
]
|
|
links = []
|
|
for pattern in patterns:
|
|
for match in re.findall(pattern, html):
|
|
if match.startswith("/"):
|
|
url = f"https://en.chezcathy.com{match}"
|
|
else:
|
|
url = match
|
|
if url not in links:
|
|
links.append(url)
|
|
return links
|
|
|
|
|
|
async def _chezcathy_scrape_playlist(playlist_url, cookies, queue, counter, overall_bar, scraper):
|
|
"""Scrape all video URLs from a playlist/user page using curl_cffi with browser cookies."""
|
|
page_num = 1
|
|
|
|
while True:
|
|
# Construct pagination URL if page > 1
|
|
if page_num > 1:
|
|
base_no_slash = playlist_url.rstrip("/")
|
|
if "?" in playlist_url:
|
|
url = f"{base_no_slash}&page={page_num}"
|
|
else:
|
|
url = f"{base_no_slash}?page={page_num}"
|
|
else:
|
|
url = playlist_url
|
|
|
|
tqdm.write(f" Scraping page {page_num}: {url}")
|
|
|
|
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.chezcathy.com/")
|
|
if not html:
|
|
tqdm.write(f" Static fetch failed. Trying Playwright for playlist page...")
|
|
page = await scraper.new_page()
|
|
try:
|
|
await scraper.resolve_page(page, url)
|
|
html = await page.content()
|
|
except Exception as pe:
|
|
tqdm.write(f" x Playwright failed to fetch playlist page: {pe}")
|
|
finally:
|
|
await page.close()
|
|
|
|
if not html:
|
|
tqdm.write(f" Failed to fetch page {page_num}")
|
|
break
|
|
|
|
video_urls = _chezcathy_extract_video_links(html)
|
|
|
|
if not video_urls:
|
|
tqdm.write(f" No video links found on page.")
|
|
break
|
|
|
|
tqdm.write(f" Found {len(video_urls)} video links")
|
|
for video_url in video_urls:
|
|
if is_filtered(video_url):
|
|
tqdm.write(f" x Filtered: {video_url}")
|
|
continue
|
|
await queue.put(video_url)
|
|
counter["total"] += 1
|
|
overall_bar.total = counter["total"]
|
|
overall_bar.refresh()
|
|
|
|
# Check for pagination next page
|
|
if not scraper_core.has_next_page(html) and f"page={page_num}" not in html:
|
|
break
|
|
|
|
page_num += 1
|
|
await asyncio.sleep(random.uniform(1.0, 2.5))
|
|
|
|
|
|
def _chezcathy_extract_video_info(html: str, url: str):
|
|
"""Extract video source and uploader/creator name from ChezCathy HTML."""
|
|
# Attempt to extract video source using standard scraper_core extractors
|
|
src, uploader = scraper_core.extract_video_info_from_html(html)
|
|
|
|
# ChezCathy specific creator/uploader extraction
|
|
# Look for user profile links like: href="/user/7115/ryan_thomas" or /profile/...
|
|
user_match = re.search(r'href="/user/\d+/([^"/]+)"', html)
|
|
if user_match:
|
|
uploader = user_match.group(1)
|
|
|
|
if not uploader or uploader == "unknown":
|
|
uploader = "unknown_creator"
|
|
|
|
uploader = scraper_core.clean_filename(uploader)
|
|
return src, uploader
|
|
|
|
|
|
async def _chezcathy_worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper, download_dir):
|
|
"""Worker: fetch video page HTML, extract source, and download."""
|
|
while True:
|
|
url = await queue.get()
|
|
if url is None:
|
|
queue.task_done()
|
|
break
|
|
|
|
try:
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered: {url}")
|
|
continue
|
|
|
|
await asyncio.sleep(random.uniform(0.3, 0.8))
|
|
|
|
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.chezcathy.com/")
|
|
|
|
# Fallback to Playwright if static fetch fails
|
|
if not html:
|
|
tqdm.write(f" Static fetch failed for {url}. Fetching page via Playwright...")
|
|
page = await scraper.new_page()
|
|
try:
|
|
await scraper.resolve_page(page, url)
|
|
html = await page.content()
|
|
except Exception as pe:
|
|
tqdm.write(f" x Playwright failed to fetch HTML for {url}: {pe}")
|
|
finally:
|
|
await page.close()
|
|
|
|
if not html:
|
|
tqdm.write(f" x Failed to fetch HTML for {url}")
|
|
continue
|
|
|
|
src, uploader = _chezcathy_extract_video_info(html, url)
|
|
|
|
# If dynamic rendering is required to capture the media source
|
|
if not src:
|
|
tqdm.write(f" * Static extraction failed for {url}. Trying Playwright...")
|
|
page = await scraper.new_page()
|
|
try:
|
|
src = await scraper.extract_media_source(page, url)
|
|
except Exception as pe:
|
|
tqdm.write(f" x Playwright extraction failed: {pe}")
|
|
finally:
|
|
await page.close()
|
|
|
|
if not src:
|
|
tqdm.write(f" x Skipping {url} - no video source found")
|
|
continue
|
|
|
|
# Standardize filename based on URL slug and ID
|
|
match = re.search(r'/(\d+)/([^/]+)\.html', url)
|
|
if match:
|
|
video_id = match.group(1)
|
|
slug = match.group(2)
|
|
filename = f"{scraper_core.clean_filename(slug)}-{video_id}.mp4"
|
|
else:
|
|
stem = url.rstrip(".html").rstrip("/").rsplit("/", 1)[-1]
|
|
filename = scraper_core.clean_filename(stem) + ".mp4"
|
|
dest_path = download_dir / uploader / filename
|
|
|
|
downloaded_bytes = 0
|
|
if skip_existing:
|
|
is_complete, existing_bytes = await check_existing_file(dest_path, src, {"Referer": url}, cookies)
|
|
if is_complete:
|
|
overall_bar.record_skip(existing_bytes)
|
|
continue
|
|
|
|
pos = bar_pool.acquire()
|
|
if pos is None:
|
|
pos = 4
|
|
|
|
success = await asyncio.to_thread(
|
|
scraper_core.download_file,
|
|
src, dest_path, {"Referer": url}, pos, cookies=cookies
|
|
)
|
|
|
|
bar_pool.release(pos)
|
|
|
|
if success:
|
|
if dest_path.is_file():
|
|
downloaded_bytes = dest_path.stat().st_size
|
|
label = f"{dest_path.parent.name}/{dest_path.name}"
|
|
overall_bar.record_download(downloaded_bytes, name=label)
|
|
scraper_core.append_log(download_dir / "urls.txt", url)
|
|
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
|
|
else:
|
|
overall_bar.update(1)
|
|
|
|
except Exception as e:
|
|
tqdm.write(f" Error processing {url}: {e}")
|
|
overall_bar.update(1)
|
|
finally:
|
|
queue.task_done()
|
|
|
|
|
|
async def _chezcathy_producer(urls, cookies, queue, counter, overall_bar, scraper):
|
|
"""Feed input URLs (video pages or playlists/profile pages) into the queue."""
|
|
for url in urls:
|
|
# Check if URL is a direct video page
|
|
if ("/player/" in url or "/adult/" in url) and url.endswith(".html"):
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered (not queued): {url}")
|
|
continue
|
|
tqdm.write(f"Queuing direct video URL: {url}")
|
|
await queue.put(url)
|
|
counter["total"] += 1
|
|
overall_bar.total = counter["total"]
|
|
overall_bar.refresh()
|
|
else:
|
|
tqdm.write(f"Scraping user/playlist: {url}")
|
|
try:
|
|
await _chezcathy_scrape_playlist(url, cookies, queue, counter, overall_bar, scraper)
|
|
except Exception as e:
|
|
tqdm.write(f" Error scraping playlist {url}: {e}")
|
|
|
|
await asyncio.sleep(random.uniform(1.0, 2.0))
|
|
|
|
|
|
async def chezcathy_process_urls(download_dir, urls, concurrency, skip_existing):
|
|
scraper = scraper_core.PlaywrightScraper()
|
|
await scraper.start()
|
|
|
|
tqdm.write("Solving Cloudflare cookies...")
|
|
first_url = urls[0]
|
|
cookies = await get_cloudflare_cookies(scraper, first_url)
|
|
tqdm.write(f"Got cookies: {list(cookies.keys())}")
|
|
|
|
queue = asyncio.Queue()
|
|
counter = {"total": 0}
|
|
overall_bar = OverallProgressTracker(total=0, desc="Overall Progress")
|
|
bar_pool = scraper_core.BarPositionPool(concurrency)
|
|
bar_pool.available = [p + 3 for p in bar_pool.available]
|
|
|
|
workers = [
|
|
asyncio.create_task(_chezcathy_worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper, download_dir))
|
|
for _ in range(concurrency)
|
|
]
|
|
|
|
await _chezcathy_producer(urls, cookies, queue, counter, overall_bar, scraper)
|
|
|
|
for _ in range(concurrency):
|
|
await queue.put(None)
|
|
|
|
await asyncio.gather(*workers)
|
|
overall_bar.close()
|
|
await scraper.close()
|
|
|
|
sys.stdout.write("\n" * (concurrency + 4))
|
|
sys.stdout.flush()
|
|
|
|
# ----- LuxureTV -------------------------------------------------------------
|
|
|
|
|
|
async def _luxuretv_scrape_playlist(playlist_url, cookies, queue, counter, overall_bar):
|
|
"""Scrape all video URLs from a playlist using curl_cffi with browser cookies."""
|
|
base_check = playlist_url.lower()
|
|
page_num = 1
|
|
|
|
while True:
|
|
if page_num > 1:
|
|
base_no_slash = playlist_url.rstrip("/")
|
|
if "?" in playlist_url:
|
|
url = f"{base_no_slash}&page={page_num}"
|
|
elif "uploads-by-user" in base_check or "playlist-by-user" in base_check or "videos-commented-by-user" in base_check or "/channels/" in base_check:
|
|
url = f"{base_no_slash}/page{page_num}.html"
|
|
else:
|
|
url = f"{base_no_slash}?page={page_num}"
|
|
else:
|
|
url = playlist_url
|
|
|
|
tqdm.write(f" Scraping playlist page {page_num}: {url}")
|
|
|
|
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.luxuretv.com/")
|
|
if not html:
|
|
tqdm.write(f" Failed to fetch page {page_num}")
|
|
break
|
|
|
|
video_urls = scraper_core.extract_video_links_from_html(html)
|
|
|
|
if not video_urls:
|
|
tqdm.write(f" No video links found on page.")
|
|
break
|
|
|
|
tqdm.write(f" Found {len(video_urls)} unique video links")
|
|
for video_url in video_urls:
|
|
if is_filtered(video_url):
|
|
tqdm.write(f" x Filtered: {video_url}")
|
|
continue
|
|
await queue.put(video_url)
|
|
counter["total"] += 1
|
|
overall_bar.total = counter["total"]
|
|
overall_bar.refresh()
|
|
|
|
if not scraper_core.has_next_page(html):
|
|
tqdm.write(f" No 'Next' button found, done.")
|
|
break
|
|
|
|
page_num += 1
|
|
await asyncio.sleep(random.uniform(1.0, 2.5))
|
|
|
|
|
|
async def _luxuretv_worker(queue, cookies, skip_existing, bar_pool, overall_bar, download_dir):
|
|
"""Worker: fetch video page HTML via curl_cffi, extract source, download."""
|
|
while True:
|
|
url = await queue.get()
|
|
if url is None:
|
|
queue.task_done()
|
|
break
|
|
|
|
try:
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered: {url}")
|
|
continue
|
|
|
|
# Add a short polite delay to avoid rate limits during bulk scraping
|
|
await asyncio.sleep(random.uniform(0.3, 0.8))
|
|
|
|
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.luxuretv.com/")
|
|
|
|
if not html:
|
|
tqdm.write(f" x Failed to fetch {url}")
|
|
continue
|
|
|
|
src, uploader = scraper_core.extract_video_info_from_html(html)
|
|
|
|
if not src:
|
|
tqdm.write(f" x Skipping {url} - no video source found")
|
|
continue
|
|
|
|
stem = url.rstrip(".html").rstrip("/").rsplit("/", 1)[-1]
|
|
filename = scraper_core.clean_filename(stem) + ".mp4"
|
|
dest_path = download_dir / uploader / filename
|
|
|
|
downloaded_bytes = 0
|
|
if skip_existing:
|
|
is_complete, existing_bytes = await check_existing_file(dest_path, src, {"Referer": url}, cookies)
|
|
if is_complete:
|
|
overall_bar.record_skip(existing_bytes)
|
|
continue
|
|
|
|
pos = bar_pool.acquire()
|
|
if pos is None:
|
|
pos = 4
|
|
|
|
success = await asyncio.to_thread(
|
|
scraper_core.download_file,
|
|
src, dest_path, {"Referer": url}, pos, cookies=cookies
|
|
)
|
|
|
|
bar_pool.release(pos)
|
|
|
|
if success:
|
|
if dest_path.is_file():
|
|
downloaded_bytes = dest_path.stat().st_size
|
|
label = f"{dest_path.parent.name}/{dest_path.name}"
|
|
overall_bar.record_download(downloaded_bytes, name=label)
|
|
scraper_core.append_log(download_dir / "urls.txt", url)
|
|
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
|
|
else:
|
|
overall_bar.update(1)
|
|
|
|
except Exception as e:
|
|
tqdm.write(f" Error processing {url}: {e}")
|
|
overall_bar.update(1)
|
|
finally:
|
|
queue.task_done()
|
|
|
|
|
|
async def _luxuretv_producer(urls, cookies, queue, counter, overall_bar):
|
|
"""Scrape playlists and feed discovered video URLs into the download queue immediately."""
|
|
for url in urls:
|
|
if "/videos/" in url and ".html" in url:
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered (not queued): {url}")
|
|
continue
|
|
tqdm.write(f"Queuing direct video URL: {url}")
|
|
await queue.put(url)
|
|
counter["total"] += 1
|
|
overall_bar.total = counter["total"]
|
|
overall_bar.refresh()
|
|
else:
|
|
tqdm.write(f"Scraping playlist: {url}")
|
|
try:
|
|
await _luxuretv_scrape_playlist(url, cookies, queue, counter, overall_bar)
|
|
except Exception as e:
|
|
tqdm.write(f" Error scraping {url}: {e}")
|
|
|
|
await asyncio.sleep(random.uniform(1.0, 2.0))
|
|
|
|
|
|
async def luxuretv_process_urls(download_dir, urls, concurrency, skip_existing):
|
|
scraper = scraper_core.PlaywrightScraper()
|
|
await scraper.start()
|
|
|
|
tqdm.write("Solving initial Cloudflare challenge for cookies...")
|
|
first_url = urls[0]
|
|
cookies = await get_cloudflare_cookies(scraper, first_url)
|
|
tqdm.write(f"Got {len(cookies)} cookies: {list(cookies.keys())}")
|
|
await scraper.close()
|
|
|
|
queue = asyncio.Queue()
|
|
counter = {"total": 0}
|
|
overall_bar = OverallProgressTracker(total=0, desc="Overall Progress")
|
|
bar_pool = scraper_core.BarPositionPool(concurrency)
|
|
bar_pool.available = [p + 3 for p in bar_pool.available]
|
|
|
|
workers = [
|
|
asyncio.create_task(_luxuretv_worker(queue, cookies, skip_existing, bar_pool, overall_bar, download_dir))
|
|
for _ in range(concurrency)
|
|
]
|
|
|
|
await _luxuretv_producer(urls, cookies, queue, counter, overall_bar)
|
|
|
|
for _ in range(concurrency):
|
|
await queue.put(None)
|
|
|
|
await asyncio.gather(*workers)
|
|
overall_bar.close()
|
|
|
|
sys.stdout.write("\n" * (concurrency + 4))
|
|
sys.stdout.flush()
|
|
|
|
# ----- PornZoo.Love ---------------------------------------------------------
|
|
|
|
|
|
def _pornzoo_extract_video_links(html: str) -> list:
|
|
"""Extract video links matching player or video page patterns on PornZoo.Love."""
|
|
# Matches URLs like: /video/slug or /en/video/slug
|
|
patterns = [
|
|
r'href="(/video/[^"/#?]+)"',
|
|
r'href="(/[^/]+/video/[^"/#?]+)"',
|
|
r'href="(https?://pornzoo\.love/video/[^"/#?]+)"',
|
|
r'href="(https?://pornzoo\.love/[^/]+/video/[^"/#?]+)"'
|
|
]
|
|
links = []
|
|
for pattern in patterns:
|
|
for match in re.findall(pattern, html):
|
|
if match.startswith("/"):
|
|
url = f"https://pornzoo.love{match}"
|
|
else:
|
|
url = match
|
|
if url not in links and "/embed/" not in url:
|
|
links.append(url)
|
|
return links
|
|
|
|
|
|
async def _pornzoo_scrape_playlist(playlist_url, cookies, queue, counter, overall_bar, scraper):
|
|
"""Scrape all video URLs from a playlist/user page."""
|
|
page_num = 1
|
|
|
|
while True:
|
|
url = playlist_url
|
|
if page_num > 1:
|
|
# Detect pagination pattern if any (e.g. ?page=X or /page/X)
|
|
if "?" in playlist_url:
|
|
url = f"{playlist_url}&page={page_num}"
|
|
else:
|
|
url = f"{playlist_url}?page={page_num}"
|
|
|
|
tqdm.write(f" Scraping page {page_num}: {url}")
|
|
|
|
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://pornzoo.love/")
|
|
if not html:
|
|
tqdm.write(f" Static fetch failed. Trying Playwright for playlist page...")
|
|
page = await scraper.new_page()
|
|
try:
|
|
await scraper.resolve_page(page, url)
|
|
html = await page.content()
|
|
except Exception as pe:
|
|
tqdm.write(f" x Playwright failed to fetch playlist page: {pe}")
|
|
finally:
|
|
await page.close()
|
|
|
|
if not html:
|
|
tqdm.write(f" Failed to fetch page {page_num}")
|
|
break
|
|
|
|
video_urls = _pornzoo_extract_video_links(html)
|
|
|
|
if not video_urls:
|
|
tqdm.write(f" No video links found on page.")
|
|
break
|
|
|
|
tqdm.write(f" Found {len(video_urls)} video links")
|
|
new_links = 0
|
|
for video_url in video_urls:
|
|
if is_filtered(video_url):
|
|
tqdm.write(f" x Filtered: {video_url}")
|
|
continue
|
|
await queue.put(video_url)
|
|
counter["total"] += 1
|
|
overall_bar.total = counter["total"]
|
|
overall_bar.refresh()
|
|
new_links += 1
|
|
|
|
# Check for next page indicator
|
|
if "Next" not in html and "next" not in html.lower():
|
|
break
|
|
if new_links == 0:
|
|
break
|
|
|
|
page_num += 1
|
|
await asyncio.sleep(random.uniform(1.0, 2.5))
|
|
|
|
|
|
def _zhvid_extract_video_info(html: str, url: str, base: str = "https://pornzoo.love"):
|
|
"""Extract direct video source, video ID, and uploader name from page HTML.
|
|
|
|
Shared by the /zhvid.php CMS sites (pornzoo.love, zootube.vip); `base` is the
|
|
site origin used to resolve root-relative sources.
|
|
|
|
Current site structure serves the source directly on the page:
|
|
<source type="video/mp4" src="/zhvid.php?vid=123456&type=MP4">
|
|
which 302-redirects to the actual MP4 on greymedia.org. Older pages used
|
|
/embed/<id> - kept as a fallback.
|
|
"""
|
|
source = None
|
|
video_id = None
|
|
uploader = "unknown_creator"
|
|
|
|
src_match = re.search(r'<source[^>]*src="([^"]+)"', html, re.I)
|
|
if src_match:
|
|
source = src_match.group(1).replace("&", "&")
|
|
|
|
if not source:
|
|
video_match = re.search(r'<video[^>]*src="([^"]+)"', html, re.I)
|
|
if video_match:
|
|
source = video_match.group(1).replace("&", "&")
|
|
|
|
if source:
|
|
vid_match = re.search(r'vid=(\d+)', source)
|
|
if vid_match:
|
|
video_id = vid_match.group(1)
|
|
|
|
if not video_id:
|
|
embed_match = re.search(r'/embed/(\d+)', html)
|
|
if embed_match:
|
|
video_id = embed_match.group(1)
|
|
|
|
if not video_id and source:
|
|
m = re.search(r'/(\d+)\.mp4', source)
|
|
if m:
|
|
video_id = m.group(1)
|
|
|
|
if not video_id:
|
|
video_id = "na"
|
|
|
|
if source:
|
|
if source.startswith("//"):
|
|
source = "https:" + source
|
|
elif source.startswith("/"):
|
|
source = base + source
|
|
elif video_id != "na":
|
|
source = f"{base}/zhvid.php?vid={video_id}&type=MP4"
|
|
|
|
for pat in (
|
|
r'href="[^"]*/user/([^"/]+)"',
|
|
r'href="[^"]*/members/([^"/]+)"',
|
|
r'href="[^"]*/profile/([^"/]+)"',
|
|
):
|
|
m = re.search(pat, html)
|
|
if m:
|
|
uploader = m.group(1)
|
|
break
|
|
|
|
uploader = scraper_core.clean_filename(uploader)
|
|
return video_id, source, uploader
|
|
|
|
|
|
async def _pornzoo_worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper, download_dir):
|
|
"""Worker: fetch video page, extract direct source URL, and download."""
|
|
while True:
|
|
url = await queue.get()
|
|
if url is None:
|
|
queue.task_done()
|
|
break
|
|
|
|
try:
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered: {url}")
|
|
continue
|
|
|
|
await asyncio.sleep(random.uniform(0.3, 0.8))
|
|
|
|
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://pornzoo.love/")
|
|
|
|
if not html:
|
|
tqdm.write(f" Static fetch failed for {url}. Fetching page via Playwright...")
|
|
page = await scraper.new_page()
|
|
try:
|
|
await scraper.resolve_page(page, url)
|
|
html = await page.content()
|
|
except Exception as pe:
|
|
tqdm.write(f" x Playwright failed to fetch HTML for {url}: {pe}")
|
|
finally:
|
|
await page.close()
|
|
|
|
if not html:
|
|
tqdm.write(f" x Failed to fetch HTML for {url}")
|
|
continue
|
|
|
|
video_id, src, uploader = _zhvid_extract_video_info(html, url)
|
|
|
|
if not src:
|
|
tqdm.write(f" x Skipping {url} - could not find video source")
|
|
continue
|
|
|
|
stem = url.rstrip("/").rsplit("/", 1)[-1]
|
|
filename = f"{scraper_core.clean_filename(stem)}-{video_id}.mp4"
|
|
dest_path = download_dir / uploader / filename
|
|
|
|
downloaded_bytes = 0
|
|
if skip_existing:
|
|
is_complete, existing_bytes = await check_existing_file(dest_path, src, {"Referer": url}, cookies)
|
|
if is_complete:
|
|
overall_bar.record_skip(existing_bytes)
|
|
continue
|
|
|
|
pos = bar_pool.acquire()
|
|
if pos is None:
|
|
pos = 4
|
|
|
|
success = await asyncio.to_thread(
|
|
scraper_core.download_file,
|
|
src, dest_path, {"Referer": url}, pos, cookies=cookies
|
|
)
|
|
|
|
bar_pool.release(pos)
|
|
|
|
if success:
|
|
if dest_path.is_file():
|
|
downloaded_bytes = dest_path.stat().st_size
|
|
label = f"{dest_path.parent.name}/{dest_path.name}"
|
|
overall_bar.record_download(downloaded_bytes, name=label)
|
|
scraper_core.append_log(download_dir / "urls.txt", url)
|
|
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
|
|
else:
|
|
overall_bar.update(1)
|
|
|
|
except Exception as e:
|
|
tqdm.write(f" Error processing {url}: {e}")
|
|
overall_bar.update(1)
|
|
finally:
|
|
queue.task_done()
|
|
|
|
|
|
async def _pornzoo_producer(urls, cookies, queue, counter, overall_bar, scraper):
|
|
"""Feed input URLs (video pages or playlists) into the queue."""
|
|
for url in urls:
|
|
if "/video/" in url:
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered (not queued): {url}")
|
|
continue
|
|
tqdm.write(f"Queuing direct video URL: {url}")
|
|
await queue.put(url)
|
|
counter["total"] += 1
|
|
overall_bar.total = counter["total"]
|
|
overall_bar.refresh()
|
|
else:
|
|
tqdm.write(f"Scraping user/playlist: {url}")
|
|
try:
|
|
await _pornzoo_scrape_playlist(url, cookies, queue, counter, overall_bar, scraper)
|
|
except Exception as e:
|
|
tqdm.write(f" Error scraping playlist {url}: {e}")
|
|
|
|
await asyncio.sleep(random.uniform(1.0, 2.0))
|
|
|
|
|
|
async def pornzoo_process_urls(download_dir, urls, concurrency, skip_existing):
|
|
scraper = scraper_core.PlaywrightScraper()
|
|
await scraper.start(headless=True)
|
|
|
|
tqdm.write("Solving Cloudflare cookies...")
|
|
first_url = urls[0]
|
|
cookies = await get_cloudflare_cookies(scraper, first_url)
|
|
tqdm.write(f"Got cookies: {list(cookies.keys())}")
|
|
|
|
queue = asyncio.Queue()
|
|
counter = {"total": 0}
|
|
overall_bar = OverallProgressTracker(total=0, desc="Overall Progress")
|
|
bar_pool = scraper_core.BarPositionPool(concurrency)
|
|
bar_pool.available = [p + 3 for p in bar_pool.available]
|
|
|
|
workers = [
|
|
asyncio.create_task(_pornzoo_worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper, download_dir))
|
|
for _ in range(concurrency)
|
|
]
|
|
|
|
await _pornzoo_producer(urls, cookies, queue, counter, overall_bar, scraper)
|
|
|
|
for _ in range(concurrency):
|
|
await queue.put(None)
|
|
|
|
await asyncio.gather(*workers)
|
|
overall_bar.close()
|
|
await scraper.close()
|
|
|
|
sys.stdout.write("\n" * (concurrency + 4))
|
|
sys.stdout.flush()
|
|
|
|
# ----- Zootube.vip ----------------------------------------------------------
|
|
|
|
|
|
def _zootubevip_is_video_link(url):
|
|
return "/video/" in url
|
|
|
|
|
|
async def _zootubevip_worker(queue, skip_existing, bar_pool, overall_bar, download_dir, base):
|
|
"""Worker: fetch the static video page, extract the /zhvid.php source, download it."""
|
|
while True:
|
|
url = await queue.get()
|
|
if url is None:
|
|
queue.task_done()
|
|
break
|
|
|
|
try:
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered: {url}")
|
|
continue
|
|
|
|
await asyncio.sleep(random.uniform(0.3, 0.8))
|
|
|
|
html = await asyncio.to_thread(scraper_core.curl_fetch, url, referer=f"{base}/")
|
|
|
|
if not html:
|
|
tqdm.write(f" x Failed to fetch HTML for {url}")
|
|
continue
|
|
|
|
video_id, src, uploader = _zhvid_extract_video_info(html, url, base)
|
|
|
|
if not src:
|
|
tqdm.write(f" x Skipping {url} - could not find video source")
|
|
continue
|
|
|
|
stem = url.rstrip("/").rsplit("/", 1)[-1]
|
|
filename = f"{scraper_core.clean_filename(stem)}-{video_id}.mp4"
|
|
dest_path = download_dir / uploader / filename
|
|
headers = {"Referer": url}
|
|
|
|
downloaded_bytes = 0
|
|
if skip_existing:
|
|
is_complete, existing_bytes = await check_existing_file(dest_path, src, headers, None)
|
|
if is_complete:
|
|
overall_bar.record_skip(existing_bytes)
|
|
continue
|
|
|
|
pos = bar_pool.acquire()
|
|
if pos is None:
|
|
pos = 4
|
|
|
|
success = await asyncio.to_thread(
|
|
scraper_core.download_file, src, dest_path, headers, pos, cookies=None
|
|
)
|
|
|
|
bar_pool.release(pos)
|
|
|
|
if success:
|
|
if dest_path.is_file():
|
|
downloaded_bytes = dest_path.stat().st_size
|
|
label = f"{dest_path.parent.name}/{dest_path.name}"
|
|
overall_bar.record_download(downloaded_bytes, name=label)
|
|
scraper_core.append_log(download_dir / "urls.txt", url)
|
|
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
|
|
else:
|
|
overall_bar.update(1)
|
|
|
|
except Exception as e:
|
|
tqdm.write(f" Error processing {url}: {e}")
|
|
overall_bar.update(1)
|
|
finally:
|
|
queue.task_done()
|
|
|
|
|
|
async def zootubevip_process_urls(download_dir, urls, concurrency, skip_existing):
|
|
base = "https://zootube.vip"
|
|
|
|
queue = asyncio.Queue()
|
|
overall_bar = OverallProgressTracker(total=0, desc="Overall Progress")
|
|
bar_pool = scraper_core.BarPositionPool(concurrency)
|
|
bar_pool.available = [p + 3 for p in bar_pool.available]
|
|
|
|
workers = [
|
|
asyncio.create_task(
|
|
_zootubevip_worker(queue, skip_existing, bar_pool, overall_bar, download_dir, base)
|
|
)
|
|
for _ in range(concurrency)
|
|
]
|
|
|
|
for url in urls:
|
|
if not _zootubevip_is_video_link(url):
|
|
tqdm.write(f" x Not a video URL: {url}")
|
|
continue
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered (not queued): {url}")
|
|
continue
|
|
tqdm.write(f"Queuing direct video URL: {url}")
|
|
await queue.put(url)
|
|
overall_bar.total += 1
|
|
overall_bar.refresh()
|
|
|
|
for _ in range(concurrency):
|
|
await queue.put(None)
|
|
|
|
await asyncio.gather(*workers)
|
|
overall_bar.close()
|
|
|
|
sys.stdout.write("\n" * (concurrency + 4))
|
|
sys.stdout.flush()
|
|
|
|
|
|
# ----- Pornhub ---------------------------------------------------------------
|
|
|
|
def _pornhub_is_video_link(url):
|
|
return "/view_video.php?viewkey=" in url
|
|
|
|
|
|
def _pornhub_extract_get_media(page_html):
|
|
m = re.search(r'"videoUrl":"((?:[^"\\]|\\.)*?get_media(?:[^"\\]|\\.)*?)"', page_html)
|
|
if not m:
|
|
return None
|
|
return m.group(1).replace("\\/", "/").replace("\\u0026", "&").replace("\\u002F", "/")
|
|
|
|
|
|
def _pornhub_extract_folder(page_html):
|
|
m = re.search(r'<a[^>]*data-label\s*=\s*["\']channel["\'][^>]*>(.*?)</a>', page_html, re.S)
|
|
if m:
|
|
name = re.sub(r"<[^>]+>", "", m.group(1)).strip()
|
|
if name:
|
|
return scraper_core.clean_filename(name)
|
|
m = re.search(r'class\s*=\s*["\']from["\'][\s\S]*?usernameBadgesWrapper[^>]*>\s*<a[^>]*>([^<]+)', page_html)
|
|
if m:
|
|
return scraper_core.clean_filename(m.group(1).strip())
|
|
m = re.search(r'"username":"([^"]+)"', page_html)
|
|
if m:
|
|
return scraper_core.clean_filename(m.group(1))
|
|
return "unknown"
|
|
|
|
|
|
def _pornhub_extract_title(page_html):
|
|
m = re.search(r"<title>(.*?)</title>", page_html, re.S)
|
|
if not m:
|
|
return ""
|
|
return re.sub(r"\s*-\s*Pornhub\.com\s*$", "", unescape(m.group(1)), flags=re.I).strip()
|
|
|
|
|
|
def _pornhub_fetch_video_source(url):
|
|
from curl_cffi import requests as cr
|
|
session = cr.Session(impersonate="chrome")
|
|
resp = session.get(url, timeout=60)
|
|
if resp.status_code != 200:
|
|
return None
|
|
page_html = resp.text
|
|
gurl = _pornhub_extract_get_media(page_html)
|
|
if not gurl:
|
|
return None
|
|
g = session.get(gurl, headers={"Referer": url, "X-Requested-With": "XMLHttpRequest"}, timeout=60)
|
|
if g.status_code != 200:
|
|
return None
|
|
data = g.json()
|
|
if not isinstance(data, list):
|
|
return None
|
|
media = [e for e in data if e.get("format") == "mp4"]
|
|
if not media:
|
|
return None
|
|
best = max(media, key=lambda e: e.get("height") or 0)
|
|
return {
|
|
"src": best["videoUrl"],
|
|
"cookies": dict(session.cookies),
|
|
"folder": _pornhub_extract_folder(page_html),
|
|
"title": _pornhub_extract_title(page_html),
|
|
}
|
|
|
|
|
|
async def _pornhub_worker(queue, skip_existing, bar_pool, overall_bar, download_dir):
|
|
while True:
|
|
url = await queue.get()
|
|
if url is None:
|
|
queue.task_done()
|
|
break
|
|
try:
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered: {url}")
|
|
continue
|
|
|
|
await asyncio.sleep(random.uniform(0.5, 1.5))
|
|
|
|
info = await asyncio.to_thread(_pornhub_fetch_video_source, url)
|
|
if not info:
|
|
tqdm.write(f" x Skipping {url} - could not resolve video source")
|
|
continue
|
|
|
|
vk = re.search(r"viewkey=([a-f0-9]{13})", url)
|
|
viewkey = vk.group(1) if vk else "na"
|
|
stem = info["title"] if info["title"] else viewkey
|
|
filename = f"{scraper_core.clean_filename(stem)}-{viewkey}.mp4"
|
|
dest_path = download_dir / info["folder"] / filename
|
|
headers = {"Referer": url}
|
|
|
|
downloaded_bytes = 0
|
|
if skip_existing:
|
|
is_complete, existing_bytes = await check_existing_file(dest_path, info["src"], headers, info["cookies"])
|
|
if is_complete:
|
|
overall_bar.record_skip(existing_bytes)
|
|
continue
|
|
|
|
pos = bar_pool.acquire() or 4
|
|
success = await asyncio.to_thread(
|
|
scraper_core.download_file, info["src"], dest_path, headers, pos, url, info["cookies"]
|
|
)
|
|
bar_pool.release(pos)
|
|
|
|
if success:
|
|
if dest_path.is_file():
|
|
downloaded_bytes = dest_path.stat().st_size
|
|
label = f"{dest_path.parent.name}/{dest_path.name}"
|
|
overall_bar.record_download(downloaded_bytes, name=label)
|
|
scraper_core.append_log(download_dir / "urls.txt", url)
|
|
scraper_core.append_log(download_dir / info["folder"] / "urls.txt", url)
|
|
else:
|
|
overall_bar.update(1)
|
|
except Exception as e:
|
|
tqdm.write(f" Error processing {url}: {e}")
|
|
overall_bar.update(1)
|
|
finally:
|
|
queue.task_done()
|
|
|
|
|
|
async def _pornhub_scrape_channel(channel_url, queue, counter, overall_bar):
|
|
base = channel_url.rstrip("/")
|
|
if not base.endswith("/videos"):
|
|
base += "/videos"
|
|
page_num = 1
|
|
seen = set()
|
|
while True:
|
|
url = base if page_num == 1 else f"{base}?page={page_num}"
|
|
tqdm.write(f" Scraping page {page_num}: {url}")
|
|
html = await asyncio.to_thread(scraper_core.curl_fetch, url, referer="https://www.pornhub.com/")
|
|
if not html:
|
|
tqdm.write(" Failed to fetch channel page")
|
|
break
|
|
links = []
|
|
for vk in dict.fromkeys(re.findall(r"/view_video\.php\?viewkey=([a-f0-9]{13})", html)):
|
|
u = f"https://www.pornhub.com/view_video.php?viewkey={vk}"
|
|
if u not in seen and not is_filtered(u):
|
|
seen.add(u)
|
|
links.append(u)
|
|
if not links:
|
|
tqdm.write(" No new video links found on page")
|
|
break
|
|
for u in links:
|
|
await queue.put(u)
|
|
counter["total"] += 1
|
|
overall_bar.total = counter["total"]
|
|
overall_bar.refresh()
|
|
tqdm.write(f" Found {len(links)} video links")
|
|
if f"page={page_num + 1}" not in html:
|
|
tqdm.write(" No next page link, done")
|
|
break
|
|
page_num += 1
|
|
await asyncio.sleep(random.uniform(1.0, 2.0))
|
|
|
|
|
|
async def pornhub_process_urls(download_dir, urls, concurrency, skip_existing):
|
|
queue = asyncio.Queue()
|
|
counter = {"total": 0}
|
|
overall_bar = OverallProgressTracker(total=0, desc="Overall Progress")
|
|
bar_pool = scraper_core.BarPositionPool(concurrency)
|
|
bar_pool.available = [p + 3 for p in bar_pool.available]
|
|
|
|
workers = [
|
|
asyncio.create_task(_pornhub_worker(queue, skip_existing, bar_pool, overall_bar, download_dir))
|
|
for _ in range(concurrency)
|
|
]
|
|
|
|
for url in urls:
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered (not queued): {url}")
|
|
continue
|
|
if _pornhub_is_video_link(url):
|
|
tqdm.write(f"Queuing direct video URL: {url}")
|
|
await queue.put(url)
|
|
counter["total"] += 1
|
|
overall_bar.total = counter["total"]
|
|
overall_bar.refresh()
|
|
else:
|
|
tqdm.write(f"Scraping channel/playlist: {url}")
|
|
try:
|
|
await _pornhub_scrape_channel(url, queue, counter, overall_bar)
|
|
except Exception as e:
|
|
tqdm.write(f" Error scraping channel {url}: {e}")
|
|
await asyncio.sleep(random.uniform(1.0, 2.0))
|
|
|
|
for _ in range(concurrency):
|
|
await queue.put(None)
|
|
|
|
await asyncio.gather(*workers)
|
|
overall_bar.close()
|
|
|
|
sys.stdout.write("\n" * (concurrency + 4))
|
|
sys.stdout.flush()
|
|
|
|
|
|
# ----- Zootube1 -------------------------------------------------------------
|
|
|
|
|
|
def _zootube1_extract_video_links(html: str) -> list:
|
|
raw = re.findall(r'hd=(https://zootube1\.com/allvideos/[^"&\s]+)', html)
|
|
raw += re.findall(r'href="(https://zootube1\.com/allvideos/[^"]+)"', html)
|
|
|
|
links = []
|
|
for url in raw:
|
|
url = url.replace("&", "&").split("#")[0]
|
|
if url not in links:
|
|
links.append(url)
|
|
return links
|
|
|
|
|
|
async def _zootube1_scrape_playlist(playlist_url, cookies, queue, counter, overall_bar, scraper):
|
|
"""Scrape a /users/<id>/ page following the real ?from_videos=NN pagination.
|
|
|
|
The site ignores ?page=N (it always returns page 1) and paginates through
|
|
?from_videos=NN, where every batch yields ~28 never-before-seen videos.
|
|
"""
|
|
seen = set()
|
|
page_num = 1
|
|
|
|
while True:
|
|
if page_num == 1:
|
|
url = playlist_url
|
|
else:
|
|
url = f"{playlist_url}?from_videos={page_num:02d}"
|
|
|
|
tqdm.write(f" Scraping page {page_num}: {url}")
|
|
|
|
html = await asyncio.to_thread(
|
|
scraper_core.curl_fetch, url, cookies=cookies, referer="https://zootube1.com/"
|
|
)
|
|
if not html:
|
|
tqdm.write(f" Static fetch failed. Trying Playwright for playlist page...")
|
|
page = await scraper.new_page()
|
|
try:
|
|
await scraper.resolve_page(page, url)
|
|
html = await page.content()
|
|
except Exception as pe:
|
|
tqdm.write(f" x Playwright failed to fetch playlist page: {pe}")
|
|
finally:
|
|
await page.close()
|
|
|
|
if not html:
|
|
tqdm.write(f" Failed to fetch page {page_num}")
|
|
break
|
|
|
|
links = _zootube1_extract_video_links(html)
|
|
new_links = 0
|
|
for link in links:
|
|
if link in seen:
|
|
continue
|
|
seen.add(link)
|
|
if is_filtered(link):
|
|
tqdm.write(f" x Filtered: {link}")
|
|
continue
|
|
await queue.put(link)
|
|
counter["total"] += 1
|
|
overall_bar.total = counter["total"]
|
|
overall_bar.refresh()
|
|
new_links += 1
|
|
|
|
tqdm.write(f" Found {len(links)} links ({new_links} new)")
|
|
|
|
if new_links == 0:
|
|
break
|
|
|
|
page_num += 1
|
|
await asyncio.sleep(random.uniform(1.0, 2.0))
|
|
|
|
|
|
async def _zootube1_extract_video_info(scraper, url: str):
|
|
"""Load a video page in the browser to read the runtime player source.
|
|
|
|
The player (kt_player.js) rewrites flashvars.video_url at runtime; the
|
|
static HTML hash 404s, so a real page load is required. Returns
|
|
(video_id, source, uploader, cookies).
|
|
"""
|
|
page = await scraper.new_page()
|
|
try:
|
|
await scraper.resolve_page(page, url)
|
|
js = """() => {
|
|
const out = {source: '', uploader: '', video_id: ''};
|
|
if (typeof flashvars !== 'undefined' && flashvars) {
|
|
out.source = flashvars.video_url || '';
|
|
out.video_id = String(flashvars.video_id || '');
|
|
}
|
|
const pu = document.querySelector('.player-uploader');
|
|
if (pu) {
|
|
out.uploader = pu.innerText || pu.textContent || '';
|
|
}
|
|
return out;
|
|
}"""
|
|
info = await page.evaluate(js)
|
|
# Wait until the player finishes rewriting video_url (it drops the
|
|
# "function/N/" placeholder prefix).
|
|
for _ in range(10):
|
|
src = info.get("source") or ""
|
|
if src and not src.startswith("function/"):
|
|
break
|
|
await page.wait_for_timeout(1000)
|
|
info = await page.evaluate(js)
|
|
finally:
|
|
await page.close()
|
|
|
|
cookies = {
|
|
c["name"]: c["value"]
|
|
for c in await scraper.context.cookies()
|
|
if "zootube1.com" in c.get("domain", "")
|
|
}
|
|
source = re.sub(r"^function/\d+/", "", info.get("source") or "")
|
|
return info.get("video_id", ""), source, info.get("uploader", "unknown"), cookies
|
|
|
|
|
|
async def _zootube1_worker(queue, skip_existing, bar_pool, overall_bar, scraper, download_dir):
|
|
while True:
|
|
url = await queue.get()
|
|
if url is None:
|
|
queue.task_done()
|
|
break
|
|
|
|
try:
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered: {url}")
|
|
continue
|
|
|
|
await asyncio.sleep(random.uniform(0.3, 0.8))
|
|
|
|
video_id, src, uploader, cookies = await _zootube1_extract_video_info(scraper, url)
|
|
|
|
if not src:
|
|
tqdm.write(f" x Skipping {url} - could not find video source")
|
|
continue
|
|
|
|
uploader = re.sub(r"\s+", " ", uploader or "").strip()
|
|
uploader = re.sub(r"^by\s+", "", uploader, flags=re.I).strip()
|
|
uploader = scraper_core.clean_filename(uploader) or "unknown"
|
|
|
|
stem = url.rstrip("/").rsplit("/", 1)[-1]
|
|
stem = re.sub(r"\.html?$", "", stem)
|
|
suffix = f"-{video_id}" if video_id else ""
|
|
filename = f"{scraper_core.clean_filename(stem)}{suffix}.mp4"
|
|
dest_path = download_dir / uploader / filename
|
|
|
|
downloaded_bytes = 0
|
|
if skip_existing:
|
|
is_complete, existing_bytes = await check_existing_file(dest_path, src, {"Referer": url}, cookies)
|
|
if is_complete:
|
|
overall_bar.record_skip(existing_bytes)
|
|
continue
|
|
|
|
pos = bar_pool.acquire()
|
|
if pos is None:
|
|
pos = 4
|
|
|
|
success = await asyncio.to_thread(
|
|
scraper_core.download_file,
|
|
src, dest_path, {"Referer": url}, pos, cookies=cookies
|
|
)
|
|
|
|
bar_pool.release(pos)
|
|
|
|
if success:
|
|
if dest_path.is_file():
|
|
downloaded_bytes = dest_path.stat().st_size
|
|
label = f"{dest_path.parent.name}/{dest_path.name}"
|
|
overall_bar.record_download(downloaded_bytes, name=label)
|
|
scraper_core.append_log(download_dir / "urls.txt", url)
|
|
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
|
|
else:
|
|
overall_bar.update(1)
|
|
|
|
except Exception as e:
|
|
tqdm.write(f" Error processing {url}: {e}")
|
|
overall_bar.update(1)
|
|
finally:
|
|
queue.task_done()
|
|
|
|
|
|
async def _zootube1_producer(urls, cookies, queue, counter, overall_bar, scraper):
|
|
for url in urls:
|
|
if _zootube1_is_video_link(url):
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered (not queued): {url}")
|
|
continue
|
|
tqdm.write(f"Queuing direct video URL: {url}")
|
|
await queue.put(url)
|
|
counter["total"] += 1
|
|
overall_bar.total = counter["total"]
|
|
overall_bar.refresh()
|
|
else:
|
|
tqdm.write(f"Scraping user/playlist: {url}")
|
|
try:
|
|
await _zootube1_scrape_playlist(url, cookies, queue, counter, overall_bar, scraper)
|
|
except Exception as e:
|
|
tqdm.write(f" Error scraping playlist {url}: {e}")
|
|
|
|
await asyncio.sleep(random.uniform(1.0, 2.0))
|
|
|
|
|
|
async def zootube1_process_urls(download_dir, urls, concurrency, skip_existing):
|
|
scraper = scraper_core.PlaywrightScraper()
|
|
await scraper.start(headless=True)
|
|
|
|
cookies = {}
|
|
warmup_url = urls[0] if _zootube1_is_video_link(urls[0]) else "https://zootube1.com/"
|
|
try:
|
|
page = await scraper.new_page()
|
|
try:
|
|
await scraper.resolve_page(page, warmup_url)
|
|
cookies = {
|
|
c["name"]: c["value"]
|
|
for c in await scraper.context.cookies()
|
|
if "zootube1.com" in c.get("domain", "")
|
|
}
|
|
finally:
|
|
await page.close()
|
|
except Exception as e:
|
|
tqdm.write(f" Warning: warm-up failed: {e}")
|
|
|
|
tqdm.write(f"Got cookies: {list(cookies.keys())}")
|
|
|
|
queue = asyncio.Queue()
|
|
counter = {"total": 0}
|
|
overall_bar = OverallProgressTracker(total=0, desc="Overall Progress")
|
|
bar_pool = scraper_core.BarPositionPool(concurrency)
|
|
bar_pool.available = [p + 3 for p in bar_pool.available]
|
|
|
|
workers = [
|
|
asyncio.create_task(_zootube1_worker(queue, skip_existing, bar_pool, overall_bar, scraper, download_dir))
|
|
for _ in range(concurrency)
|
|
]
|
|
|
|
await _zootube1_producer(urls, cookies, queue, counter, overall_bar, scraper)
|
|
|
|
for _ in range(concurrency):
|
|
await queue.put(None)
|
|
|
|
await asyncio.gather(*workers)
|
|
overall_bar.close()
|
|
await scraper.close()
|
|
|
|
sys.stdout.write("\n" * (concurrency + 4))
|
|
sys.stdout.flush()
|
|
|
|
|
|
# ----- RedGIFs --------------------------------------------------------------
|
|
|
|
REDGIFS_HEADERS = {
|
|
'User-Agent': scraper_core.USER_AGENT,
|
|
'Accept': 'application/json, text/plain, */*',
|
|
'Accept-Language': 'en-US,en;q=0.9',
|
|
'Referer': 'https://www.redgifs.com/',
|
|
'Origin': 'https://www.redgifs.com',
|
|
'DNT': '1',
|
|
'Connection': 'keep-alive',
|
|
}
|
|
|
|
|
|
def _wait_cooldown(wait_seconds: float, desc: str = "Waiting") -> bool:
|
|
"""
|
|
Waits for wait_seconds while displaying a progress bar.
|
|
Accepts keyboard input (pressing Enter or any key on Windows / Enter on Unix)
|
|
to immediately interrupt and skip the cooldown wait.
|
|
Returns True if interrupted by user, False if completed normally.
|
|
"""
|
|
if wait_seconds <= 0:
|
|
return False
|
|
|
|
tqdm.write(" [i] Press ENTER at any time to skip cooldown and retry immediately (e.g. after changing IP).")
|
|
total_steps = max(1, int(wait_seconds * 10)) # 100ms intervals
|
|
bar = tqdm(total=int(wait_seconds), desc=f" {desc}", unit='s', ncols=80, leave=False)
|
|
|
|
interrupted = False
|
|
|
|
# Check if Windows msvcrt is available for instant non-blocking keypress
|
|
has_msvcrt = False
|
|
try:
|
|
import msvcrt
|
|
has_msvcrt = True
|
|
except ImportError:
|
|
pass
|
|
|
|
if has_msvcrt:
|
|
# Flush any previously buffered keys
|
|
try:
|
|
while msvcrt.kbhit():
|
|
msvcrt.getch()
|
|
except Exception:
|
|
pass
|
|
|
|
for step in range(total_steps):
|
|
try:
|
|
if msvcrt.kbhit():
|
|
ch = msvcrt.getch()
|
|
interrupted = True
|
|
break
|
|
except Exception:
|
|
pass
|
|
time.sleep(0.1)
|
|
if step % 10 == 9:
|
|
bar.update(1)
|
|
else:
|
|
# Fallback using a background thread reading sys.stdin
|
|
stop_event = threading.Event()
|
|
input_received = threading.Event()
|
|
|
|
def _reader():
|
|
try:
|
|
import select
|
|
while not stop_event.is_set():
|
|
r, _, _ = select.select([sys.stdin], [], [], 0.2)
|
|
if r:
|
|
sys.stdin.readline()
|
|
input_received.set()
|
|
break
|
|
except Exception:
|
|
pass
|
|
|
|
t = threading.Thread(target=_reader, daemon=True)
|
|
t.start()
|
|
|
|
for step in range(total_steps):
|
|
if input_received.is_set():
|
|
interrupted = True
|
|
break
|
|
time.sleep(0.1)
|
|
if step % 10 == 9:
|
|
bar.update(1)
|
|
stop_event.set()
|
|
|
|
bar.close()
|
|
if interrupted:
|
|
tqdm.write(" [!] Cooldown interrupted by user. Resuming immediately...")
|
|
return True
|
|
return False
|
|
|
|
|
|
class RedGIFsAPI:
|
|
"""Headless browser wrapper for the RedGIFs API (token-bound to browser)."""
|
|
def __init__(self, skip_cooldown: bool = False):
|
|
self._pw_cm = None
|
|
self._browser = None
|
|
self._page = None
|
|
self._token = None
|
|
self.skip_cooldown = skip_cooldown
|
|
|
|
def start(self):
|
|
tqdm.write(' Obtaining RedGIFs API token ...')
|
|
from playwright.sync_api import sync_playwright
|
|
self._pw_cm = sync_playwright()
|
|
pw = self._pw_cm.__enter__()
|
|
self._browser = pw.chromium.launch(headless=True)
|
|
context = self._browser.new_context(
|
|
user_agent=REDGIFS_HEADERS['User-Agent'],
|
|
viewport={'width': 1920, 'height': 1080},
|
|
)
|
|
self._page = context.new_page()
|
|
self._page.goto(
|
|
'https://www.redgifs.com/',
|
|
wait_until='domcontentloaded',
|
|
timeout=30000,
|
|
)
|
|
self.obtain_token()
|
|
|
|
def obtain_token(self, retries=5):
|
|
"""Fetches or refreshes the temporary token from the RedGIFs API."""
|
|
last_err = None
|
|
for attempt in range(retries):
|
|
try:
|
|
# Ensure we're on a valid redgifs page (not a Cloudflare challenge redirect)
|
|
cur = self._page.url
|
|
if 'redgifs' not in cur or 'challenge' in cur:
|
|
self._page.goto('https://www.redgifs.com/', wait_until='networkidle', timeout=30000)
|
|
elif cur != 'https://www.redgifs.com/':
|
|
self._page.goto('https://www.redgifs.com/', wait_until='domcontentloaded', timeout=30000)
|
|
|
|
result = self._page.evaluate('''async () => {
|
|
try {
|
|
const r = await fetch('https://api.redgifs.com/v2/auth/temporary');
|
|
if (!r.ok) {
|
|
return {error: "HTTP " + r.status + ": " + await r.text()};
|
|
}
|
|
const d = await r.json();
|
|
return {token: d.token};
|
|
} catch(e) {
|
|
return {error: e.message || String(e)};
|
|
}
|
|
}''')
|
|
|
|
if isinstance(result, dict) and 'error' in result:
|
|
raise RuntimeError(result['error'])
|
|
self._token = result if isinstance(result, str) else result.get('token') if isinstance(result, dict) else result
|
|
return
|
|
except Exception as e:
|
|
last_err = e
|
|
err_str = str(e)
|
|
if attempt < retries - 1:
|
|
if '429' in err_str:
|
|
if self.skip_cooldown:
|
|
tqdm.write(' Token fetch rate-limited (HTTP 429). Skipping cooldown wait (--skip-redgifs-cooldown enabled)...')
|
|
wait = 1
|
|
else:
|
|
m = re.search(r'"delay"\s*:\s*(\d+)', err_str)
|
|
wait = int(m.group(1)) + 2 if m else (15 * (attempt + 1))
|
|
tqdm.write(f' Token fetch rate-limited (HTTP 429). Waiting {wait}s...')
|
|
else:
|
|
wait = 5 * (attempt + 1)
|
|
tqdm.write(f' Token fetch failed (attempt {attempt+1}/{retries}): {e}. Retrying in {wait}s...')
|
|
_wait_cooldown(wait, desc='Waiting')
|
|
try:
|
|
self._page.goto('https://www.redgifs.com/', wait_until='domcontentloaded', timeout=30000)
|
|
except Exception:
|
|
pass
|
|
raise RuntimeError(f'Failed to obtain token after {retries} attempts: {last_err}')
|
|
|
|
def stop(self):
|
|
if self._browser:
|
|
try:
|
|
self._browser.close()
|
|
except Exception:
|
|
pass
|
|
if self._pw_cm:
|
|
try:
|
|
self._pw_cm.__exit__(None, None, None)
|
|
except Exception:
|
|
pass
|
|
|
|
def get_all_creator_videos(self, username, fresh_token=True):
|
|
"""Fetch all videos for a creator. Automatically refreshes tokens if expired."""
|
|
if fresh_token or not self._token:
|
|
self.obtain_token()
|
|
|
|
raw = self._page.evaluate('''async ({username, token}) => {
|
|
let allGifs = [];
|
|
let page = 1;
|
|
let pages = 1;
|
|
let total = 0;
|
|
let currentToken = token;
|
|
let errorMsg = '';
|
|
|
|
const sleep = ms => new Promise(r => setTimeout(r, ms));
|
|
|
|
try {
|
|
while (page <= pages) {
|
|
let r;
|
|
|
|
while (true) {
|
|
try {
|
|
r = await fetch(
|
|
'https://api.redgifs.com/v2/users/'
|
|
+ encodeURIComponent(username)
|
|
+ '/search?page=' + page + '&count=80&order=latest&type=g',
|
|
{headers: {'Authorization': 'Bearer ' + currentToken}}
|
|
);
|
|
} catch (networkErr) {
|
|
errorMsg = 'Network error: ' + (networkErr.message || String(networkErr));
|
|
r = null;
|
|
break;
|
|
}
|
|
|
|
if (r.status === 401 || r.status === 403) {
|
|
try {
|
|
const tokRes = await fetch('https://api.redgifs.com/v2/auth/temporary');
|
|
if (tokRes.status === 200) {
|
|
const tokData = await tokRes.json();
|
|
currentToken = tokData.token;
|
|
continue;
|
|
}
|
|
} catch (e) {}
|
|
}
|
|
|
|
if (r.status === 429) {
|
|
let rawBody = '';
|
|
try { rawBody = await r.text(); } catch (e) {}
|
|
errorMsg = 'API status 429: ' + rawBody;
|
|
break;
|
|
}
|
|
|
|
break;
|
|
}
|
|
|
|
if (errorMsg || !r || r.status !== 200) {
|
|
if (!errorMsg && r) {
|
|
errorMsg = 'API status ' + r.status + ': ' + await r.text();
|
|
}
|
|
break;
|
|
}
|
|
|
|
const d = await r.json();
|
|
if (page === 1) {
|
|
pages = Math.min(d.pages || 1, 100);
|
|
total = d.total || 0;
|
|
}
|
|
for (const g of (d.gifs || [])) {
|
|
allGifs.push({urls: g.urls, id: g.id});
|
|
}
|
|
page++;
|
|
await sleep(1500); // 1.5s delay between pages
|
|
}
|
|
} catch (outerErr) {
|
|
errorMsg = 'Unexpected error: ' + (outerErr.message || String(outerErr));
|
|
}
|
|
|
|
return JSON.stringify({gifs: allGifs, total: total, error: errorMsg});
|
|
}''', {'username': username, 'token': self._token})
|
|
|
|
data = json.loads(raw)
|
|
if data.get('error'):
|
|
raise RuntimeError(data['error'])
|
|
|
|
return data.get('gifs', []), int(data.get('total', 0))
|
|
|
|
def get_gif_info(self, gif_id, fresh_token=False):
|
|
if fresh_token or not self._token:
|
|
self.obtain_token()
|
|
|
|
raw = self._page.evaluate('''async ({gifId, token}) => {
|
|
let currentToken = token;
|
|
for (let attempt = 0; attempt < 2; attempt++) {
|
|
let r;
|
|
try {
|
|
r = await fetch(
|
|
'https://api.redgifs.com/v2/gifs/' + encodeURIComponent(gifId),
|
|
{headers: {'Authorization': 'Bearer ' + currentToken}}
|
|
);
|
|
} catch (networkErr) {
|
|
return JSON.stringify({error: 'Network error: ' + (networkErr.message || String(networkErr))});
|
|
}
|
|
if (r.status === 401 || r.status === 403) {
|
|
try {
|
|
const tokRes = await fetch('https://api.redgifs.com/v2/auth/temporary');
|
|
if (tokRes.status === 200) {
|
|
currentToken = (await tokRes.json()).token;
|
|
continue;
|
|
}
|
|
} catch (e) {}
|
|
}
|
|
if (r.status === 429) {
|
|
let errBody = '';
|
|
try { errBody = await r.text(); } catch (e) {}
|
|
return JSON.stringify({error: 'API status 429: ' + errBody});
|
|
}
|
|
if (!r.ok) {
|
|
return JSON.stringify({error: 'API status ' + r.status});
|
|
}
|
|
const d = await r.json();
|
|
const g = d.gif || d;
|
|
return JSON.stringify({gif: {
|
|
id: g.id || gifId,
|
|
userName: g.userName || '',
|
|
title: g.title || '',
|
|
urls: g.urls || {},
|
|
}});
|
|
}
|
|
return JSON.stringify({error: 'token refresh failed'});
|
|
}''', {'gifId': gif_id, 'token': self._token})
|
|
|
|
data = json.loads(raw)
|
|
if data.get('error'):
|
|
raise RuntimeError(data['error'])
|
|
|
|
return data.get('gif') or {}
|
|
|
|
|
|
def _redgifs_extract_filename(url: str) -> str:
|
|
return urlparse(url).path.split('/')[-1]
|
|
|
|
|
|
def _redgifs_extract_gif_id(target: str):
|
|
"""Extract a gif id from a RedGIFs watch URL or media URL; None if not one."""
|
|
try:
|
|
parsed = urlparse(target)
|
|
host = parsed.netloc.lower()
|
|
path = parsed.path
|
|
except ValueError:
|
|
return None
|
|
if not host:
|
|
host = urlparse("https://" + target).netloc.lower()
|
|
path = urlparse("https://" + target).path
|
|
path = path.rstrip("/")
|
|
if host.startswith("media."):
|
|
name = path.split("/")[-1]
|
|
if name.lower().endswith(".mp4"):
|
|
name = name[:-4]
|
|
return name or None
|
|
m = re.match(r"/watch/([A-Za-z0-9_]+)$", path)
|
|
if m:
|
|
return m.group(1)
|
|
return None
|
|
|
|
|
|
def redgifs_download_creator_videos(username, output_dir, gifs, total, concurrency, skip_existing):
|
|
"""Download one creator's videos into output_dir/<username>/ (exact original layout)."""
|
|
creator_dir = safe_mkdir(output_dir / username)
|
|
links_file = creator_dir / 'links.txt'
|
|
master_links_file = output_dir / 'all_links.txt'
|
|
|
|
download_urls = []
|
|
link_lines = []
|
|
for gif in gifs:
|
|
urls = gif['urls']
|
|
dl_url = urls.get('hd') or urls.get('sd') or next(iter(urls.values()))
|
|
filename = _redgifs_extract_filename(dl_url)
|
|
filepath = creator_dir / filename
|
|
download_urls.append((dl_url, filepath))
|
|
link_lines.append(f'{dl_url}\n')
|
|
|
|
# Save url lists
|
|
safe_append_text(links_file, ''.join(link_lines))
|
|
safe_append_text(master_links_file, ''.join(link_lines))
|
|
|
|
if len(gifs) < total:
|
|
tqdm.write(f"Downloading {len(gifs)} of {total} video(s) (capped by API page limit) for {username} with concurrency {concurrency} ...")
|
|
else:
|
|
tqdm.write(f"Downloading {total} video(s) for {username} with concurrency {concurrency} ...")
|
|
|
|
bar_pool = scraper_core.BarPositionPool(concurrency)
|
|
bar_pool.available = [p + 3 for p in bar_pool.available]
|
|
overall_bar = OverallProgressTracker(
|
|
total=len(gifs),
|
|
desc=f"Creator: {username}",
|
|
)
|
|
|
|
def download_one(dl_url, filepath):
|
|
processed = False
|
|
try:
|
|
if skip_existing:
|
|
is_complete, existing_bytes = check_existing_file_sync(filepath, dl_url, headers=REDGIFS_HEADERS)
|
|
if is_complete:
|
|
overall_bar.record_skip(existing_bytes)
|
|
processed = True
|
|
return
|
|
|
|
pos = bar_pool.acquire() or 4
|
|
try:
|
|
success = scraper_core.download_file(
|
|
dl_url, filepath, REDGIFS_HEADERS, pos
|
|
)
|
|
if success:
|
|
dl_size = filepath.stat().st_size if filepath.is_file() else 0
|
|
label = f"{filepath.parent.name}/{filepath.name}"
|
|
overall_bar.record_download(dl_size, name=label)
|
|
processed = True
|
|
finally:
|
|
bar_pool.release(pos)
|
|
except Exception as e:
|
|
tqdm.write(f" [ERROR] Failed to download:\n {dl_url}\n Error: {e}")
|
|
finally:
|
|
if not processed:
|
|
overall_bar.update(1)
|
|
|
|
with ThreadPoolExecutor(max_workers=concurrency) as executor:
|
|
futures = [executor.submit(download_one, url, path) for url, path in download_urls]
|
|
for f in as_completed(futures):
|
|
f.result()
|
|
|
|
overall_bar.close()
|
|
|
|
# Clean up console lines
|
|
sys.stdout.write("\n" * (concurrency + 4))
|
|
sys.stdout.flush()
|
|
|
|
|
|
def redgifs_update_creator_list(output_dir, creators_file):
|
|
"""--update-all-creators: merge folder names + creators.txt, write back, return merged list."""
|
|
found_folders = []
|
|
if output_dir.exists():
|
|
for entry in output_dir.iterdir():
|
|
if entry.is_dir():
|
|
found_folders.append(entry.name)
|
|
|
|
existing_creators = []
|
|
if creators_file.is_file():
|
|
try:
|
|
with open(creators_file, 'r', encoding='utf-8') as f:
|
|
for line in f:
|
|
stripped = line.strip()
|
|
if not stripped or stripped.startswith('#'):
|
|
continue
|
|
for part in re.split(r'[,\s]+', stripped):
|
|
if part:
|
|
existing_creators.append(part)
|
|
except Exception as e:
|
|
tqdm.write(f"Error reading creators.txt: {e}")
|
|
|
|
all_creators = sorted(list(set(existing_creators + found_folders)))
|
|
try:
|
|
with open(creators_file, 'w', encoding='utf-8') as f:
|
|
f.write(' '.join(all_creators))
|
|
tqdm.write(f"Updated creators.txt with {len(all_creators)} unique creators.")
|
|
except Exception as e:
|
|
tqdm.write(f"Error writing to creators.txt: {e}")
|
|
|
|
return all_creators
|
|
|
|
|
|
def redgifs_download_creators(creators, output_dir, concurrency, skip_existing, skip_cooldown: bool = False):
|
|
"""Download all videos for the given creator username(s) into output_dir/<creator>/."""
|
|
creators = dedupe(creators)
|
|
if not creators:
|
|
sys.exit('No creators provided.')
|
|
|
|
output_dir = Path(output_dir).resolve()
|
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
api = RedGIFsAPI(skip_cooldown=skip_cooldown)
|
|
try:
|
|
api.start()
|
|
except Exception as e:
|
|
sys.exit(f'Failed to initialise Playwright browser: {e}')
|
|
|
|
try:
|
|
total_creators = len(creators)
|
|
for idx, username in enumerate(creators, 1):
|
|
sep = '-' * 3
|
|
tqdm.write(f'\n{sep} [{idx}/{total_creators}] {username} {sep}')
|
|
tqdm.write(f' Fetching video URLs for {username} ...')
|
|
|
|
gifs, total = [], 0
|
|
success = False
|
|
|
|
# Retry loop with browser reset on failure
|
|
for attempt in range(1, 4):
|
|
try:
|
|
gifs, total = api.get_all_creator_videos(username, fresh_token=(attempt == 1))
|
|
success = True
|
|
break
|
|
except Exception as e:
|
|
tqdm.write(f' [ERROR] Attempt {attempt}/3 failed for {username}: {e}')
|
|
if attempt < 3:
|
|
err_str = str(e)
|
|
if '429' in err_str:
|
|
if skip_cooldown:
|
|
tqdm.write(' Rate limited. Skipping cooldown wait (--skip-redgifs-cooldown enabled)...')
|
|
wait_seconds = 1
|
|
else:
|
|
m = re.search(r'"delay"\s*:\s*(\d+)', err_str)
|
|
wait_seconds = int(m.group(1)) + 1 if m else 30
|
|
tqdm.write(f' Rate limited. Waiting {wait_seconds:.1f}s...')
|
|
skipped = _wait_cooldown(wait_seconds, desc='Waiting')
|
|
if skipped:
|
|
# User changed IP and skipped cooldown; force token refresh
|
|
try:
|
|
api.obtain_token()
|
|
except Exception:
|
|
pass
|
|
else:
|
|
wait_reset = 1 if skip_cooldown else 30
|
|
tqdm.write(f' Resetting browser context and waiting {wait_reset}s before retry...')
|
|
try:
|
|
api.stop()
|
|
except Exception:
|
|
pass
|
|
_wait_cooldown(wait_reset, desc='Reset Wait')
|
|
try:
|
|
api.start()
|
|
except Exception as restart_err:
|
|
tqdm.write(f' Failed to restart browser: {restart_err}')
|
|
|
|
if success:
|
|
tqdm.write(f' Found {total} video(s)')
|
|
if total > 0:
|
|
redgifs_download_creator_videos(
|
|
username, output_dir, gifs, total,
|
|
concurrency, skip_existing
|
|
)
|
|
else:
|
|
tqdm.write(f' [ERROR] Skipping {username} after 3 failed attempts.')
|
|
|
|
# Introduce a small pacing delay to avoid aggressive rate limits
|
|
time.sleep(2)
|
|
|
|
finally:
|
|
api.stop()
|
|
|
|
|
|
def redgifs_download_gif_urls(gif_urls, output_dir, concurrency, skip_existing, skip_cooldown: bool = False):
|
|
"""Download individual RedGIFs gifs from watch/media URLs into output_dir/<creator>/."""
|
|
api = RedGIFsAPI(skip_cooldown=skip_cooldown)
|
|
try:
|
|
api.start()
|
|
except Exception as e:
|
|
sys.exit(f'Failed to initialise Playwright browser: {e}')
|
|
|
|
try:
|
|
items = []
|
|
for raw_url in dedupe(gif_urls):
|
|
gid = _redgifs_extract_gif_id(raw_url)
|
|
if not gid:
|
|
tqdm.write(f' [!] Could not parse gif id from: {raw_url}')
|
|
continue
|
|
info = None
|
|
candidates = []
|
|
lowered = gid.lower()
|
|
candidates.append(lowered)
|
|
if gid != lowered:
|
|
candidates.append(gid)
|
|
|
|
for cand in candidates:
|
|
for attempt in range(1, 4):
|
|
try:
|
|
info = api.get_gif_info(cand, fresh_token=False)
|
|
break
|
|
except Exception as e:
|
|
err_msg = str(e)
|
|
# If the API returned 404 or 410, do not burn retries
|
|
if '404' in err_msg or '410' in err_msg:
|
|
if '410' in err_msg:
|
|
# 410 Gone means the resource is permanently gone across any case variation
|
|
candidates.clear()
|
|
break
|
|
if '429' in err_msg:
|
|
if skip_cooldown:
|
|
tqdm.write(f' Rate limited (429) resolving {cand}. Skipping cooldown wait (--skip-redgifs-cooldown enabled)...')
|
|
wait_seconds = 1
|
|
else:
|
|
m = re.search(r'"delay"\s*:\s*(\d+)', err_msg)
|
|
wait_seconds = int(m.group(1)) + 2 if m else (20 * attempt)
|
|
tqdm.write(f' Rate limited (429) resolving {cand}. Waiting {wait_seconds}s...')
|
|
skipped = _wait_cooldown(wait_seconds, desc='Waiting')
|
|
if skipped:
|
|
try:
|
|
api.obtain_token()
|
|
except Exception:
|
|
pass
|
|
continue
|
|
|
|
tqdm.write(f' [ERROR] Attempt {attempt}/3 resolving:\n {raw_url}\n as {cand}: {e}')
|
|
if attempt < 3:
|
|
err_wait = 1 if skip_cooldown else (5 * attempt)
|
|
_wait_cooldown(err_wait, desc='Retry Wait')
|
|
if info or not candidates:
|
|
break
|
|
if not info:
|
|
tqdm.write(f' [ERROR] Skipping (could not resolve gif info):\n {raw_url}')
|
|
continue
|
|
urls = info.get('urls') or {}
|
|
dl_url = urls.get('hd') or urls.get('sd')
|
|
if not dl_url and urls:
|
|
dl_url = next(iter(urls.values()))
|
|
if not dl_url:
|
|
tqdm.write(f' [!] No usable media URL for {gid}')
|
|
continue
|
|
username = (info.get('userName') or '').strip() or 'misc'
|
|
creator_dir = safe_mkdir(output_dir / scraper_core.clean_filename(username))
|
|
filepath = creator_dir / _redgifs_extract_filename(dl_url)
|
|
items.append((dl_url, filepath))
|
|
safe_append_text(creator_dir / 'links.txt', dl_url + '\n')
|
|
safe_append_text(output_dir / 'all_links.txt', dl_url + '\n')
|
|
tqdm.write(f' {raw_url}\n -> {creator_dir.name}/{filepath.name}')
|
|
time.sleep(0.1)
|
|
|
|
if not items:
|
|
tqdm.write(' No RedGIFs gif URLs could be resolved.')
|
|
return
|
|
|
|
tqdm.write(f"Downloading {len(items)} RedGIFs gif(s) with concurrency {concurrency} ...")
|
|
bar_pool = scraper_core.BarPositionPool(concurrency)
|
|
bar_pool.available = [p + 3 for p in bar_pool.available]
|
|
overall_bar = OverallProgressTracker(total=len(items), desc='RedGIFs gifs')
|
|
|
|
def download_one(dl_url, filepath):
|
|
downloaded = False
|
|
try:
|
|
if skip_existing:
|
|
is_complete, existing_bytes = check_existing_file_sync(filepath, dl_url, headers=REDGIFS_HEADERS)
|
|
if is_complete:
|
|
overall_bar.record_skip(existing_bytes)
|
|
return
|
|
pos = bar_pool.acquire() or 4
|
|
try:
|
|
success = scraper_core.download_file(dl_url, filepath, REDGIFS_HEADERS, pos)
|
|
if success:
|
|
dl_size = filepath.stat().st_size if filepath.is_file() else 0
|
|
label = f"{filepath.parent.name}/{filepath.name}"
|
|
overall_bar.record_download(dl_size, name=label)
|
|
downloaded = True
|
|
finally:
|
|
bar_pool.release(pos)
|
|
except Exception as e:
|
|
tqdm.write(f' [ERROR] Failed to download:\n {dl_url}\n Error: {e}')
|
|
finally:
|
|
if not downloaded:
|
|
overall_bar.update(1)
|
|
|
|
with ThreadPoolExecutor(max_workers=concurrency) as executor:
|
|
futures = [executor.submit(download_one, dl, fp) for dl, fp in items]
|
|
for f in as_completed(futures):
|
|
f.result()
|
|
|
|
overall_bar.close()
|
|
sys.stdout.write("\n" * (concurrency + 4))
|
|
sys.stdout.flush()
|
|
finally:
|
|
api.stop()
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# CLI
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def print_site_list():
|
|
print("Supported sites (folder -> handler):")
|
|
print(f" {'Site':<16} {'Folder':<18} {'Domains':<40} {'Concurrency':>11}")
|
|
print("-" * 90)
|
|
for key, cfg in SITES.items():
|
|
print(f" {key:<16} {cfg['folder']:<18} {', '.join(cfg['domains']):<40} {cfg['default_concurrency']:>11}")
|
|
print()
|
|
print("Inputs are auto-routed: URLs by hostname, any bare token is a RedGIFs creator name.")
|
|
print(".txt files may list URLs and/or creator names (one per line, '#' = comment).")
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(
|
|
prog="downloader",
|
|
description="Unified downloader: routes URLs/creator names to the right site and "
|
|
"places files in <website_folder>/videos/<creator>/ exactly like the "
|
|
"individual downloaders did.",
|
|
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
epilog="""examples:
|
|
python downloader.py https://en.chezcathy.com/user/7115/foo https://www.thisvid.com/video/9876/ creator1 creator2
|
|
python downloader.py urls.txt # file may mix URLs and creator names
|
|
python downloader.py --update-all-creators
|
|
python downloader.py --list-sites
|
|
""",
|
|
)
|
|
parser.add_argument(
|
|
"targets", nargs="*",
|
|
help="URLs from any supported site, RedGIFs creator usernames, or .txt files "
|
|
"containing them (one per line, '#' = comment).",
|
|
)
|
|
parser.add_argument(
|
|
"--concurrency", type=int, default=None,
|
|
help="Number of concurrent downloads. Defaults per site: "
|
|
"chezcathy/luxuretv/generic=3, pornzoo.love=2, redgifs=5.",
|
|
)
|
|
parser.add_argument(
|
|
"--skip-existing", action="store_true", dest="skip_existing", default=True,
|
|
help="Skip already-downloaded files (default: true).",
|
|
)
|
|
parser.add_argument(
|
|
"--no-skip-existing", action="store_false", dest="skip_existing",
|
|
help="Re-download existing files.",
|
|
)
|
|
parser.add_argument(
|
|
"-o", "--output", default=str(ROOT / "redgifs" / "videos"),
|
|
help="RedGIFs output directory (default: redgifs/videos).",
|
|
)
|
|
parser.add_argument(
|
|
"--update-all-creators", action="store_true",
|
|
help="RedGIFs: gets the name of each creator folder in --output, merges them into "
|
|
"redgifs/creators.txt (deduplicated), then uses that list as the download list.",
|
|
)
|
|
parser.add_argument(
|
|
"--user-playlist", default=None,
|
|
help="LuxureTV default user playlist (used when no targets are given).",
|
|
)
|
|
parser.add_argument(
|
|
"--skip-redgifs-cooldown", "--skip-cooldown", action="store_true", dest="skip_redgifs_cooldown",
|
|
help="Skip RedGIFs API rate limit / HTTP 429 cooldown wait period (useful if IP was changed).",
|
|
)
|
|
parser.add_argument(
|
|
"--list-sites", action="store_true",
|
|
help="List supported websites and exit.",
|
|
)
|
|
args = parser.parse_args()
|
|
|
|
if args.list_sites:
|
|
print_site_list()
|
|
return
|
|
|
|
creators_file = ROOT / "redgifs" / "creators.txt"
|
|
groups, warnings = resolve_targets(args.targets)
|
|
for w in warnings:
|
|
print(f" [!] {w}", file=sys.stderr)
|
|
|
|
if args.update_all_creators:
|
|
names = redgifs_update_creator_list(Path(args.output).resolve(), creators_file)
|
|
if names:
|
|
groups["redgifs"] = dedupe(groups.get("redgifs", []) + names)
|
|
|
|
if not groups:
|
|
if args.user_playlist:
|
|
groups["luxuretv"] = [args.user_playlist]
|
|
else:
|
|
raw = input('RedGIFs creator username(s) (comma/space separated): ').strip()
|
|
creators = [c.strip() for c in re.split(r'[,\s]+', raw) if c.strip()]
|
|
if not creators:
|
|
sys.exit('No creators provided.')
|
|
groups["redgifs"] = creators
|
|
|
|
if not groups:
|
|
sys.exit("No targets to process. Pass URLs, creator names, or .txt files.")
|
|
|
|
for site_key, targets in groups.items():
|
|
cfg = SITES[site_key]
|
|
concurrency = args.concurrency if args.concurrency is not None else cfg["default_concurrency"]
|
|
download_dir = ROOT / cfg["folder"] / "videos"
|
|
handler = cfg["handler"]
|
|
|
|
tqdm.write(f"\n=== [{site_key}] processing {len(targets)} target(s) into {download_dir} ===")
|
|
|
|
if handler == "generic":
|
|
g = GENERIC_SITES[site_key]
|
|
asyncio.run(generic_downloader.process_urls(
|
|
g["site_name"], targets, download_dir,
|
|
g["is_video_link"], g["next_page_selector"],
|
|
g["video_selector"], g["uploader_eval_js"],
|
|
concurrency, args.skip_existing,
|
|
))
|
|
elif handler == "chezcathy":
|
|
asyncio.run(chezcathy_process_urls(download_dir, targets, concurrency, args.skip_existing))
|
|
elif handler == "luxuretv":
|
|
asyncio.run(luxuretv_process_urls(download_dir, targets, concurrency, args.skip_existing))
|
|
elif handler == "pornzoo":
|
|
asyncio.run(pornzoo_process_urls(download_dir, targets, concurrency, args.skip_existing))
|
|
elif handler == "pornhub":
|
|
asyncio.run(pornhub_process_urls(download_dir, targets, concurrency, args.skip_existing))
|
|
elif handler == "zootubevip":
|
|
asyncio.run(zootubevip_process_urls(download_dir, targets, concurrency, args.skip_existing))
|
|
elif handler == "zootube1":
|
|
asyncio.run(zootube1_process_urls(download_dir, targets, concurrency, args.skip_existing))
|
|
elif handler == "redgifs":
|
|
output_dir = Path(args.output).resolve()
|
|
creator_names = [t for t in targets if not looks_like_url(t)]
|
|
gif_urls = [t for t in targets if looks_like_url(t)]
|
|
if creator_names:
|
|
redgifs_download_creators(creator_names, output_dir, concurrency, args.skip_existing, skip_cooldown=args.skip_redgifs_cooldown)
|
|
if gif_urls:
|
|
redgifs_download_gif_urls(gif_urls, output_dir, concurrency, args.skip_existing, skip_cooldown=args.skip_redgifs_cooldown)
|
|
else:
|
|
print(f" [!] Unknown handler '{handler}' for site '{site_key}' - skipped.", file=sys.stderr)
|
|
|
|
# Print final master summary for the entire run
|
|
MASTER_TRACKER.close()
|
|
MASTER_TRACKER.print_summary()
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main() |