Files
niggers/pornzoo.love/downloader.py
T

338 lines
11 KiB
Python

import os
import sys
import re
import asyncio
import argparse
import random
from pathlib import Path
from tqdm import tqdm
sys.path.append(str(Path(__file__).resolve().parents[1]))
import scraper_core
BASE_DIR = Path(__file__).resolve().parent
DOWNLOAD_DIR = BASE_DIR / "videos"
FILTERED_WORDS = {"shit", "scat", "shitty", "poop", "horseshit", "bullshit", "cowshit", "shitting", "crap", "feces", "dung", "pungpile", "pungheap", "pooping", "crapping", "manure"}
def is_filtered(url: str) -> bool:
lowered = url.lower()
return any(w in lowered for w in FILTERED_WORDS)
async def get_cloudflare_cookies(scraper, url):
"""Use Playwright to solve one Cloudflare challenge and return browser cookies."""
page = await scraper.new_page()
try:
try:
await scraper.resolve_page(page, url)
except Exception as e:
tqdm.write(f" Warning: Cloudflare resolve_page failed: {e}. Trying to proceed with currently gathered cookies.")
raw_cookies = await scraper.context.cookies()
return {c["name"]: c["value"] for c in raw_cookies}
finally:
await page.close()
def extract_pornzoo_video_links(html: str) -> list:
"""Extract video links matching player or video page patterns on PornZoo.Love."""
# Matches URLs like: /video/slug or /en/video/slug
patterns = [
r'href="(/video/[^"/#?]+)"',
r'href="(/[^/]+/video/[^"/#?]+)"',
r'href="(https?://pornzoo\.love/video/[^"/#?]+)"',
r'href="(https?://pornzoo\.love/[^/]+/video/[^"/#?]+)"'
]
links = []
for pattern in patterns:
for match in re.findall(pattern, html):
if match.startswith("/"):
url = f"https://pornzoo.love{match}"
else:
url = match
if url not in links and "/embed/" not in url:
links.append(url)
return links
async def scrape_playlist(playlist_url: str, cookies: dict, queue, counter, overall_bar, scraper):
"""Scrape all video URLs from a playlist/user page."""
page_num = 1
while True:
url = playlist_url
if page_num > 1:
# Detect pagination pattern if any (e.g. ?page=X or /page/X)
if "?" in playlist_url:
url = f"{playlist_url}&page={page_num}"
else:
url = f"{playlist_url}?page={page_num}"
tqdm.write(f" Scraping page {page_num}: {url}")
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://pornzoo.love/")
if not html:
tqdm.write(f" Static fetch failed. Trying Playwright for playlist page...")
page = await scraper.new_page()
try:
await scraper.resolve_page(page, url)
html = await page.content()
except Exception as pe:
tqdm.write(f" x Playwright failed to fetch playlist page: {pe}")
finally:
await page.close()
if not html:
tqdm.write(f" Failed to fetch page {page_num}")
break
video_urls = extract_pornzoo_video_links(html)
if not video_urls:
tqdm.write(f" No video links found on page.")
break
tqdm.write(f" Found {len(video_urls)} video links")
new_links = 0
for video_url in video_urls:
if is_filtered(video_url):
tqdm.write(f" x Filtered: {video_url}")
continue
await queue.put(video_url)
counter["total"] += 1
overall_bar.total = counter["total"]
overall_bar.refresh()
new_links += 1
# Check for next page indicator
if "Next" not in html and "next" not in html.lower():
break
if new_links == 0:
break
page_num += 1
await asyncio.sleep(random.uniform(1.0, 2.5))
def extract_video_info(html: str, url: str) -> tuple:
"""Extract video embed ID, direct source, and uploader name from page HTML."""
embed_id = None
uploader = "unknown_creator"
# Find embed URL / ID
embed_match = re.search(r'/embed/(\d+)', html)
if embed_match:
embed_id = embed_match.group(1)
# Find uploader name
user_match = re.search(r'href="[^"]*/user/([^"/]+)"', html)
if user_match:
uploader = user_match.group(1)
uploader = scraper_core.clean_filename(uploader)
return embed_id, uploader
async def worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper):
"""Worker: fetch video page, resolve source link, and download."""
while True:
url = await queue.get()
if url is None:
queue.task_done()
break
try:
if is_filtered(url):
tqdm.write(f" x Filtered: {url}")
continue
await asyncio.sleep(random.uniform(0.3, 0.8))
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://pornzoo.love/")
if not html:
tqdm.write(f" Static fetch failed for {url}. Fetching page via Playwright...")
page = await scraper.new_page()
try:
await scraper.resolve_page(page, url)
html = await page.content()
except Exception as pe:
tqdm.write(f" x Playwright failed to fetch HTML for {url}: {pe}")
finally:
await page.close()
if not html:
tqdm.write(f" x Failed to fetch HTML for {url}")
continue
embed_id, uploader = extract_video_info(html, url)
if not embed_id:
tqdm.write(f" x Skipping {url} - could not find embed ID")
continue
# Standardize filename based on URL slug and embed ID
stem = url.rstrip("/").rsplit("/", 1)[-1]
filename = f"{scraper_core.clean_filename(stem)}-{embed_id}.mp4"
dest_path = DOWNLOAD_DIR / uploader / filename
if skip_existing and scraper_core.is_already_downloaded(dest_path):
tqdm.write(f" > Already downloaded: {dest_path.parent.name}/{dest_path.name}")
continue
# Resolve direct video source via embed page
embed_url = f"https://pornzoo.love/embed/{embed_id}"
embed_html = await asyncio.to_thread(scraper_core.curl_fetch, embed_url, cookies=cookies, referer=url)
if not embed_html:
page = await scraper.new_page()
try:
await scraper.resolve_page(page, embed_url)
embed_html = await page.content()
except Exception as pe:
tqdm.write(f" x Playwright failed to load embed for {url}: {pe}")
finally:
await page.close()
src = None
if embed_html:
src_match = re.search(r'<video[^>]*src="([^"]+)"', embed_html)
if src_match:
src = src_match.group(1)
else:
src_match = re.search(r'<source[^>]*src="([^"]+)"', embed_html)
if src_match:
src = src_match.group(1)
if not src:
# Direct fallback guess link
src = f"https://pornzoo.love/media/videos/iphone/{embed_id}.mp4"
pos = bar_pool.acquire()
if pos is None:
pos = 1
success = await asyncio.to_thread(
scraper_core.download_file,
src, dest_path, {"Referer": embed_url}, pos, cookies=cookies
)
bar_pool.release(pos)
if success:
scraper_core.append_log(DOWNLOAD_DIR / "urls.txt", url)
scraper_core.append_log(DOWNLOAD_DIR / uploader / "urls.txt", url)
except Exception as e:
tqdm.write(f" Error processing {url}: {e}")
finally:
overall_bar.update(1)
queue.task_done()
async def producer(urls, cookies, queue, counter, overall_bar, scraper):
"""Feed input URLs (video pages or playlists) into the queue."""
for url in urls:
if "/video/" in url:
if is_filtered(url):
tqdm.write(f" x Filtered (not queued): {url}")
continue
tqdm.write(f"Queuing direct video URL: {url}")
await queue.put(url)
counter["total"] += 1
overall_bar.total = counter["total"]
overall_bar.refresh()
else:
tqdm.write(f"Scraping user/playlist: {url}")
try:
await scrape_playlist(url, cookies, queue, counter, overall_bar, scraper)
except Exception as e:
tqdm.write(f" Error scraping playlist {url}: {e}")
await asyncio.sleep(random.uniform(1.0, 2.0))
async def process_urls(urls, concurrency, skip_existing):
scraper = scraper_core.PlaywrightScraper()
await scraper.start(headless=True)
tqdm.write("Solving Cloudflare cookies...")
first_url = urls[0]
cookies = await get_cloudflare_cookies(scraper, first_url)
tqdm.write(f"Got cookies: {list(cookies.keys())}")
queue = asyncio.Queue()
counter = {"total": 0}
overall_bar = tqdm(
total=0,
desc="Overall Progress",
position=0,
leave=True,
ncols=80,
bar_format="{desc}: {n_fmt}/{total_fmt} |{bar}| {percentage:.0f}%",
)
bar_pool = scraper_core.BarPositionPool(concurrency)
workers = [
asyncio.create_task(worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper))
for _ in range(concurrency)
]
await producer(urls, cookies, queue, counter, overall_bar, scraper)
for _ in range(concurrency):
await queue.put(None)
await asyncio.gather(*workers)
overall_bar.close()
await scraper.close()
sys.stdout.write("\n" * (concurrency + 1))
sys.stdout.flush()
def resolve_urls(args_or_files):
urls = []
for arg in args_or_files:
if os.path.isfile(arg):
with open(arg, "r") as f:
for line in f:
for token in line.strip().split():
if token and not token.startswith("#"):
urls.append(token)
else:
urls.append(arg)
return urls
def main():
parser = argparse.ArgumentParser(
description="Download videos from PornZoo.Love (concurrent, metadata-aware)."
)
parser.add_argument(
"urls",
nargs="*",
help="Profile URLs, direct video URLs, or .txt files containing them.",
)
parser.add_argument(
"--concurrency",
type=int,
default=2,
help="Number of concurrent downloads (default: 2).",
)
parser.add_argument(
"--skip-existing",
action="store_true",
default=True,
help="Skip downloading files that already exist (default: True).",
)
args = parser.parse_args()
if not args.urls:
parser.print_help()
sys.exit(1)
targets = resolve_urls(args.urls)
if not targets:
print("No URLs found to process.")
sys.exit(0)
# Ensure output directory exists
DOWNLOAD_DIR.mkdir(parents=True, exist_ok=True)
asyncio.run(process_urls(targets, args.concurrency, args.skip_existing))
if __name__ == "__main__":
main()