Initial backup: folder structure manifest, tree summary, scraper scripts, and text metadata
This commit is contained in:
commit
f164832dcc
10504 files changed
+2423594
No files matched your search
@@ -0,0 +1,337 @@
|
||||
import os
|
||||
import sys
|
||||
import re
|
||||
import asyncio
|
||||
import argparse
|
||||
import random
|
||||
from pathlib import Path
|
||||
from tqdm import tqdm
|
||||
|
||||
sys.path.append(str(Path(__file__).resolve().parents[1]))
|
||||
import scraper_core
|
||||
|
||||
BASE_DIR = Path(__file__).resolve().parent
|
||||
DOWNLOAD_DIR = BASE_DIR / "videos"
|
||||
|
||||
FILTERED_WORDS = {"shit", "scat", "shitty", "poop", "horseshit", "bullshit", "cowshit", "shitting", "crap", "feces", "dung", "pungpile", "pungheap", "pooping", "crapping", "manure"}
|
||||
|
||||
def is_filtered(url: str) -> bool:
|
||||
lowered = url.lower()
|
||||
return any(w in lowered for w in FILTERED_WORDS)
|
||||
|
||||
async def get_cloudflare_cookies(scraper, url):
|
||||
"""Use Playwright to solve one Cloudflare challenge and return browser cookies."""
|
||||
page = await scraper.new_page()
|
||||
try:
|
||||
try:
|
||||
await scraper.resolve_page(page, url)
|
||||
except Exception as e:
|
||||
tqdm.write(f" Warning: Cloudflare resolve_page failed: {e}. Trying to proceed with currently gathered cookies.")
|
||||
raw_cookies = await scraper.context.cookies()
|
||||
return {c["name"]: c["value"] for c in raw_cookies}
|
||||
finally:
|
||||
await page.close()
|
||||
|
||||
def extract_pornzoo_video_links(html: str) -> list:
|
||||
"""Extract video links matching player or video page patterns on PornZoo.Love."""
|
||||
# Matches URLs like: /video/slug or /en/video/slug
|
||||
patterns = [
|
||||
r'href="(/video/[^"/#?]+)"',
|
||||
r'href="(/[^/]+/video/[^"/#?]+)"',
|
||||
r'href="(https?://pornzoo\.love/video/[^"/#?]+)"',
|
||||
r'href="(https?://pornzoo\.love/[^/]+/video/[^"/#?]+)"'
|
||||
]
|
||||
links = []
|
||||
for pattern in patterns:
|
||||
for match in re.findall(pattern, html):
|
||||
if match.startswith("/"):
|
||||
url = f"https://pornzoo.love{match}"
|
||||
else:
|
||||
url = match
|
||||
if url not in links and "/embed/" not in url:
|
||||
links.append(url)
|
||||
return links
|
||||
|
||||
async def scrape_playlist(playlist_url: str, cookies: dict, queue, counter, overall_bar, scraper):
|
||||
"""Scrape all video URLs from a playlist/user page."""
|
||||
page_num = 1
|
||||
|
||||
while True:
|
||||
url = playlist_url
|
||||
if page_num > 1:
|
||||
# Detect pagination pattern if any (e.g. ?page=X or /page/X)
|
||||
if "?" in playlist_url:
|
||||
url = f"{playlist_url}&page={page_num}"
|
||||
else:
|
||||
url = f"{playlist_url}?page={page_num}"
|
||||
|
||||
tqdm.write(f" Scraping page {page_num}: {url}")
|
||||
|
||||
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://pornzoo.love/")
|
||||
if not html:
|
||||
tqdm.write(f" Static fetch failed. Trying Playwright for playlist page...")
|
||||
page = await scraper.new_page()
|
||||
try:
|
||||
await scraper.resolve_page(page, url)
|
||||
html = await page.content()
|
||||
except Exception as pe:
|
||||
tqdm.write(f" x Playwright failed to fetch playlist page: {pe}")
|
||||
finally:
|
||||
await page.close()
|
||||
|
||||
if not html:
|
||||
tqdm.write(f" Failed to fetch page {page_num}")
|
||||
break
|
||||
|
||||
video_urls = extract_pornzoo_video_links(html)
|
||||
|
||||
if not video_urls:
|
||||
tqdm.write(f" No video links found on page.")
|
||||
break
|
||||
|
||||
tqdm.write(f" Found {len(video_urls)} video links")
|
||||
new_links = 0
|
||||
for video_url in video_urls:
|
||||
if is_filtered(video_url):
|
||||
tqdm.write(f" x Filtered: {video_url}")
|
||||
continue
|
||||
await queue.put(video_url)
|
||||
counter["total"] += 1
|
||||
overall_bar.total = counter["total"]
|
||||
overall_bar.refresh()
|
||||
new_links += 1
|
||||
|
||||
# Check for next page indicator
|
||||
if "Next" not in html and "next" not in html.lower():
|
||||
break
|
||||
if new_links == 0:
|
||||
break
|
||||
|
||||
page_num += 1
|
||||
await asyncio.sleep(random.uniform(1.0, 2.5))
|
||||
|
||||
def extract_video_info(html: str, url: str) -> tuple:
|
||||
"""Extract video embed ID, direct source, and uploader name from page HTML."""
|
||||
embed_id = None
|
||||
uploader = "unknown_creator"
|
||||
|
||||
# Find embed URL / ID
|
||||
embed_match = re.search(r'/embed/(\d+)', html)
|
||||
if embed_match:
|
||||
embed_id = embed_match.group(1)
|
||||
|
||||
# Find uploader name
|
||||
user_match = re.search(r'href="[^"]*/user/([^"/]+)"', html)
|
||||
if user_match:
|
||||
uploader = user_match.group(1)
|
||||
|
||||
uploader = scraper_core.clean_filename(uploader)
|
||||
return embed_id, uploader
|
||||
|
||||
async def worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper):
|
||||
"""Worker: fetch video page, resolve source link, and download."""
|
||||
while True:
|
||||
url = await queue.get()
|
||||
if url is None:
|
||||
queue.task_done()
|
||||
break
|
||||
|
||||
try:
|
||||
if is_filtered(url):
|
||||
tqdm.write(f" x Filtered: {url}")
|
||||
continue
|
||||
|
||||
await asyncio.sleep(random.uniform(0.3, 0.8))
|
||||
|
||||
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://pornzoo.love/")
|
||||
|
||||
if not html:
|
||||
tqdm.write(f" Static fetch failed for {url}. Fetching page via Playwright...")
|
||||
page = await scraper.new_page()
|
||||
try:
|
||||
await scraper.resolve_page(page, url)
|
||||
html = await page.content()
|
||||
except Exception as pe:
|
||||
tqdm.write(f" x Playwright failed to fetch HTML for {url}: {pe}")
|
||||
finally:
|
||||
await page.close()
|
||||
|
||||
if not html:
|
||||
tqdm.write(f" x Failed to fetch HTML for {url}")
|
||||
continue
|
||||
|
||||
embed_id, uploader = extract_video_info(html, url)
|
||||
|
||||
if not embed_id:
|
||||
tqdm.write(f" x Skipping {url} - could not find embed ID")
|
||||
continue
|
||||
|
||||
# Standardize filename based on URL slug and embed ID
|
||||
stem = url.rstrip("/").rsplit("/", 1)[-1]
|
||||
filename = f"{scraper_core.clean_filename(stem)}-{embed_id}.mp4"
|
||||
dest_path = DOWNLOAD_DIR / uploader / filename
|
||||
|
||||
if skip_existing and scraper_core.is_already_downloaded(dest_path):
|
||||
tqdm.write(f" > Already downloaded: {dest_path.parent.name}/{dest_path.name}")
|
||||
continue
|
||||
|
||||
# Resolve direct video source via embed page
|
||||
embed_url = f"https://pornzoo.love/embed/{embed_id}"
|
||||
embed_html = await asyncio.to_thread(scraper_core.curl_fetch, embed_url, cookies=cookies, referer=url)
|
||||
|
||||
if not embed_html:
|
||||
page = await scraper.new_page()
|
||||
try:
|
||||
await scraper.resolve_page(page, embed_url)
|
||||
embed_html = await page.content()
|
||||
except Exception as pe:
|
||||
tqdm.write(f" x Playwright failed to load embed for {url}: {pe}")
|
||||
finally:
|
||||
await page.close()
|
||||
|
||||
src = None
|
||||
if embed_html:
|
||||
src_match = re.search(r'<video[^>]*src="([^"]+)"', embed_html)
|
||||
if src_match:
|
||||
src = src_match.group(1)
|
||||
else:
|
||||
src_match = re.search(r'<source[^>]*src="([^"]+)"', embed_html)
|
||||
if src_match:
|
||||
src = src_match.group(1)
|
||||
|
||||
if not src:
|
||||
# Direct fallback guess link
|
||||
src = f"https://pornzoo.love/media/videos/iphone/{embed_id}.mp4"
|
||||
|
||||
pos = bar_pool.acquire()
|
||||
if pos is None:
|
||||
pos = 1
|
||||
|
||||
success = await asyncio.to_thread(
|
||||
scraper_core.download_file,
|
||||
src, dest_path, {"Referer": embed_url}, pos, cookies=cookies
|
||||
)
|
||||
|
||||
bar_pool.release(pos)
|
||||
|
||||
if success:
|
||||
scraper_core.append_log(DOWNLOAD_DIR / "urls.txt", url)
|
||||
scraper_core.append_log(DOWNLOAD_DIR / uploader / "urls.txt", url)
|
||||
|
||||
except Exception as e:
|
||||
tqdm.write(f" Error processing {url}: {e}")
|
||||
finally:
|
||||
overall_bar.update(1)
|
||||
queue.task_done()
|
||||
|
||||
async def producer(urls, cookies, queue, counter, overall_bar, scraper):
|
||||
"""Feed input URLs (video pages or playlists) into the queue."""
|
||||
for url in urls:
|
||||
if "/video/" in url:
|
||||
if is_filtered(url):
|
||||
tqdm.write(f" x Filtered (not queued): {url}")
|
||||
continue
|
||||
tqdm.write(f"Queuing direct video URL: {url}")
|
||||
await queue.put(url)
|
||||
counter["total"] += 1
|
||||
overall_bar.total = counter["total"]
|
||||
overall_bar.refresh()
|
||||
else:
|
||||
tqdm.write(f"Scraping user/playlist: {url}")
|
||||
try:
|
||||
await scrape_playlist(url, cookies, queue, counter, overall_bar, scraper)
|
||||
except Exception as e:
|
||||
tqdm.write(f" Error scraping playlist {url}: {e}")
|
||||
|
||||
await asyncio.sleep(random.uniform(1.0, 2.0))
|
||||
|
||||
async def process_urls(urls, concurrency, skip_existing):
|
||||
scraper = scraper_core.PlaywrightScraper()
|
||||
await scraper.start(headless=True)
|
||||
|
||||
tqdm.write("Solving Cloudflare cookies...")
|
||||
first_url = urls[0]
|
||||
cookies = await get_cloudflare_cookies(scraper, first_url)
|
||||
tqdm.write(f"Got cookies: {list(cookies.keys())}")
|
||||
|
||||
queue = asyncio.Queue()
|
||||
counter = {"total": 0}
|
||||
overall_bar = tqdm(
|
||||
total=0,
|
||||
desc="Overall Progress",
|
||||
position=0,
|
||||
leave=True,
|
||||
ncols=80,
|
||||
bar_format="{desc}: {n_fmt}/{total_fmt} |{bar}| {percentage:.0f}%",
|
||||
)
|
||||
bar_pool = scraper_core.BarPositionPool(concurrency)
|
||||
|
||||
workers = [
|
||||
asyncio.create_task(worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper))
|
||||
for _ in range(concurrency)
|
||||
]
|
||||
|
||||
await producer(urls, cookies, queue, counter, overall_bar, scraper)
|
||||
|
||||
for _ in range(concurrency):
|
||||
await queue.put(None)
|
||||
|
||||
await asyncio.gather(*workers)
|
||||
overall_bar.close()
|
||||
await scraper.close()
|
||||
|
||||
sys.stdout.write("\n" * (concurrency + 1))
|
||||
sys.stdout.flush()
|
||||
|
||||
def resolve_urls(args_or_files):
|
||||
urls = []
|
||||
for arg in args_or_files:
|
||||
if os.path.isfile(arg):
|
||||
with open(arg, "r") as f:
|
||||
for line in f:
|
||||
for token in line.strip().split():
|
||||
if token and not token.startswith("#"):
|
||||
urls.append(token)
|
||||
else:
|
||||
urls.append(arg)
|
||||
return urls
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Download videos from PornZoo.Love (concurrent, metadata-aware)."
|
||||
)
|
||||
parser.add_argument(
|
||||
"urls",
|
||||
nargs="*",
|
||||
help="Profile URLs, direct video URLs, or .txt files containing them.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--concurrency",
|
||||
type=int,
|
||||
default=2,
|
||||
help="Number of concurrent downloads (default: 2).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--skip-existing",
|
||||
action="store_true",
|
||||
default=True,
|
||||
help="Skip downloading files that already exist (default: True).",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
if not args.urls:
|
||||
parser.print_help()
|
||||
sys.exit(1)
|
||||
|
||||
targets = resolve_urls(args.urls)
|
||||
if not targets:
|
||||
print("No URLs found to process.")
|
||||
sys.exit(0)
|
||||
|
||||
# Ensure output directory exists
|
||||
DOWNLOAD_DIR.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
asyncio.run(process_urls(targets, args.concurrency, args.skip_existing))
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,34 @@
|
||||
https://pornzoo.love/en/video/guy-ravages-a-creature-s-narrow-orifice-with-his-stiff-dick
|
||||
https://pornzoo.love/en/video/enthusiastic-guy-fisting-and-fingering-his-pet-s-narrow-opening
|
||||
https://pornzoo.love/en/video/guy-stretches-open-a-female-dog-s-narrow-vagina-with-his-erect-penis
|
||||
https://pornzoo.love/en/video/guy-shoves-his-stiff-dick-far-into-a-beast-s-rear-end
|
||||
https://pornzoo.love/video/guy-penetrating-a-beast-s-orifice-with-his-cock
|
||||
https://pornzoo.love/video/incredible-anthology-style-clip-featuring-steamy-bestiality
|
||||
https://pornzoo.love/video/aroused-webcam-model-gets-screwed-by-a-canine-on-her-mattress
|
||||
https://pornzoo.love/video/guy-employs-his-penis-to-satisfy-this-snug-opening
|
||||
https://pornzoo.love/video/astonishing-animal-passion-featuring-alluring-beasts
|
||||
https://pornzoo.love/video/hd-intimate-butt-sex-with-a-kinky-creature-up-close
|
||||
https://pornzoo.love/video/man-s-stunning-cock-pounds-a-narrow-animal-orifice
|
||||
https://pornzoo.love/video/webcam-model-pleasures-herself-viewing-bestiality-videos-online
|
||||
https://pornzoo.love/en/video/guy-swiftly-inserts-his-hard-dick-into-a-canine-s-orifice
|
||||
https://pornzoo.love/en/video/man-s-cock-smoothly-penetrates-deep-into-beast-s-snug-orifice
|
||||
https://pornzoo.love/en/video/narrow-anus-expanded-by-a-hot-creature
|
||||
https://pornzoo.love/en/video/narrow-beast-orifice-expanded-by-a-zoophilia-deviant
|
||||
https://pornzoo.love/en/video/guy-s-stiff-dick-engulfed-by-a-beast-s-hungry-opening
|
||||
https://pornzoo.love/en/video/guy-demolishing-a-savage-babe-s-narrow-slit-with-his-shaft
|
||||
https://pornzoo.love/en/video/guy-thrusts-his-rigid-dick-deep-into-this-wild-chick-s-snug-pussy
|
||||
https://pornzoo.love/en/video/guy-thrusting-his-thick-cock-into-a-beast-s-gullet
|
||||
https://pornzoo.love/en/video/steamy-anal-opening-being-penetrated-by-an-uncircumcised-dick-right-here
|
||||
https://pornzoo.love/en/video/blonde-in-cat-mask-has-her-snug-pussy-drilled-roughly
|
||||
https://pornzoo.love/video/married-woman-in-dark-hosiery-craves-large-cock-doggy-style
|
||||
https://pornzoo.love/video/collection-of-the-steamiest-bestiality-scenes-with-a-brunette-woman
|
||||
https://pornzoo.love/video/well-endowed-guy-boldly-penetrates-his-female-dog-s-snug-vagina
|
||||
https://pornzoo.love/video/guy-fucks-a-steamy-horse-pussy-and-cums-inside
|
||||
https://pornzoo.love/video/guy-rams-his-dick-into-a-cow-s-mouth-hardcore
|
||||
https://pornzoo.love/video/guy-pounds-mare-vigorously-in-intense-positions-and-floods-her-with-his-cum
|
||||
https://pornzoo.love/video/beast-loving-anus-penetrated-by-a-filthy-animal
|
||||
https://pornzoo.love/video/guy-penetrates-his-female-dog-s-snug-vagina-from-the-rear
|
||||
https://pornzoo.love/video/guy-pounding-a-pooch-s-obedient-gullet-on-video
|
||||
https://pornzoo.love/video/guy-aggressively-deep-throating-his-tame-pooch-in-intimate-detail
|
||||
https://pornzoo.love/video/fantastic-homosexual-zoophilia-anal-scene-featuring-an-obedient-guy
|
||||
https://pornzoo.love/video/savage-deepthroat-ravaging-on-the-brink-of-beastly-cruelty
|
||||
@@ -0,0 +1,34 @@
|
||||
https://pornzoo.love/en/video/guy-ravages-a-creature-s-narrow-orifice-with-his-stiff-dick
|
||||
https://pornzoo.love/en/video/enthusiastic-guy-fisting-and-fingering-his-pet-s-narrow-opening
|
||||
https://pornzoo.love/en/video/guy-stretches-open-a-female-dog-s-narrow-vagina-with-his-erect-penis
|
||||
https://pornzoo.love/en/video/guy-shoves-his-stiff-dick-far-into-a-beast-s-rear-end
|
||||
https://pornzoo.love/video/guy-penetrating-a-beast-s-orifice-with-his-cock
|
||||
https://pornzoo.love/video/incredible-anthology-style-clip-featuring-steamy-bestiality
|
||||
https://pornzoo.love/video/aroused-webcam-model-gets-screwed-by-a-canine-on-her-mattress
|
||||
https://pornzoo.love/video/guy-employs-his-penis-to-satisfy-this-snug-opening
|
||||
https://pornzoo.love/video/astonishing-animal-passion-featuring-alluring-beasts
|
||||
https://pornzoo.love/video/hd-intimate-butt-sex-with-a-kinky-creature-up-close
|
||||
https://pornzoo.love/video/man-s-stunning-cock-pounds-a-narrow-animal-orifice
|
||||
https://pornzoo.love/video/webcam-model-pleasures-herself-viewing-bestiality-videos-online
|
||||
https://pornzoo.love/en/video/guy-swiftly-inserts-his-hard-dick-into-a-canine-s-orifice
|
||||
https://pornzoo.love/en/video/man-s-cock-smoothly-penetrates-deep-into-beast-s-snug-orifice
|
||||
https://pornzoo.love/en/video/narrow-anus-expanded-by-a-hot-creature
|
||||
https://pornzoo.love/en/video/narrow-beast-orifice-expanded-by-a-zoophilia-deviant
|
||||
https://pornzoo.love/en/video/guy-s-stiff-dick-engulfed-by-a-beast-s-hungry-opening
|
||||
https://pornzoo.love/en/video/guy-demolishing-a-savage-babe-s-narrow-slit-with-his-shaft
|
||||
https://pornzoo.love/en/video/guy-thrusts-his-rigid-dick-deep-into-this-wild-chick-s-snug-pussy
|
||||
https://pornzoo.love/en/video/guy-thrusting-his-thick-cock-into-a-beast-s-gullet
|
||||
https://pornzoo.love/en/video/steamy-anal-opening-being-penetrated-by-an-uncircumcised-dick-right-here
|
||||
https://pornzoo.love/en/video/blonde-in-cat-mask-has-her-snug-pussy-drilled-roughly
|
||||
https://pornzoo.love/video/married-woman-in-dark-hosiery-craves-large-cock-doggy-style
|
||||
https://pornzoo.love/video/collection-of-the-steamiest-bestiality-scenes-with-a-brunette-woman
|
||||
https://pornzoo.love/video/well-endowed-guy-boldly-penetrates-his-female-dog-s-snug-vagina
|
||||
https://pornzoo.love/video/guy-fucks-a-steamy-horse-pussy-and-cums-inside
|
||||
https://pornzoo.love/video/guy-rams-his-dick-into-a-cow-s-mouth-hardcore
|
||||
https://pornzoo.love/video/guy-pounds-mare-vigorously-in-intense-positions-and-floods-her-with-his-cum
|
||||
https://pornzoo.love/video/beast-loving-anus-penetrated-by-a-filthy-animal
|
||||
https://pornzoo.love/video/guy-penetrates-his-female-dog-s-snug-vagina-from-the-rear
|
||||
https://pornzoo.love/video/guy-pounding-a-pooch-s-obedient-gullet-on-video
|
||||
https://pornzoo.love/video/guy-aggressively-deep-throating-his-tame-pooch-in-intimate-detail
|
||||
https://pornzoo.love/video/fantastic-homosexual-zoophilia-anal-scene-featuring-an-obedient-guy
|
||||
https://pornzoo.love/video/savage-deepthroat-ravaging-on-the-brink-of-beastly-cruelty
|
||||
Reference in new issue
Block a user