318 lines
11 KiB
Python
318 lines
11 KiB
Python
import os
|
|
import sys
|
|
import re
|
|
import asyncio
|
|
import argparse
|
|
import random
|
|
from pathlib import Path
|
|
from tqdm import tqdm
|
|
|
|
sys.path.append(str(Path(__file__).resolve().parents[1]))
|
|
import scraper_core
|
|
|
|
BASE_DIR = Path(__file__).resolve().parent
|
|
DOWNLOAD_DIR = BASE_DIR / "videos"
|
|
|
|
FILTERED_WORDS = {"shit", "scat", "shitty", "poop", "horseshit", "bullshit", "cowshit", "shitting", "crap", "feces", "dung", "pungpile", "pungheap", "pooping", "crapping", "manure"}
|
|
|
|
def is_filtered(url: str) -> bool:
|
|
lowered = url.lower()
|
|
return any(w in lowered for w in FILTERED_WORDS)
|
|
|
|
async def get_cloudflare_cookies(scraper, url):
|
|
"""Use Playwright to solve one Cloudflare challenge and return browser cookies."""
|
|
page = await scraper.new_page()
|
|
try:
|
|
await scraper.resolve_page(page, url)
|
|
raw_cookies = await scraper.context.cookies()
|
|
return {c["name"]: c["value"] for c in raw_cookies}
|
|
finally:
|
|
await page.close()
|
|
|
|
def extract_chezcathy_video_links(html: str) -> list:
|
|
"""Extract video links matching player or adult page patterns on chezcathy."""
|
|
# Matches URLs like: /extreme-sex/player/35738/...html or /adult/769439/...html
|
|
patterns = [
|
|
r'href="(/[^"]*?/player/\d+/[^"]*?\.html)"',
|
|
r'href="(/adult/\d+/[^"]*?\.html)"',
|
|
r'href="(https?://en\.chezcathy\.com/[^"]*?/player/\d+/[^"]*?\.html)"',
|
|
r'href="(https?://en\.chezcathy\.com/adult/\d+/[^"]*?\.html)"'
|
|
]
|
|
links = []
|
|
for pattern in patterns:
|
|
for match in re.findall(pattern, html):
|
|
if match.startswith("/"):
|
|
url = f"https://en.chezcathy.com{match}"
|
|
else:
|
|
url = match
|
|
if url not in links:
|
|
links.append(url)
|
|
return links
|
|
|
|
async def scrape_playlist(playlist_url: str, cookies: dict, queue, counter, overall_bar, scraper):
|
|
"""Scrape all video URLs from a playlist/user page using curl_cffi with browser cookies."""
|
|
base_check = playlist_url.lower()
|
|
page_num = 1
|
|
|
|
while True:
|
|
# Construct pagination URL if page > 1
|
|
if page_num > 1:
|
|
base_no_slash = playlist_url.rstrip("/")
|
|
if "?" in playlist_url:
|
|
url = f"{base_no_slash}&page={page_num}"
|
|
else:
|
|
url = f"{base_no_slash}?page={page_num}"
|
|
else:
|
|
url = playlist_url
|
|
|
|
tqdm.write(f" Scraping page {page_num}: {url}")
|
|
|
|
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.chezcathy.com/")
|
|
if not html:
|
|
tqdm.write(f" Static fetch failed. Trying Playwright for playlist page...")
|
|
page = await scraper.new_page()
|
|
try:
|
|
await scraper.resolve_page(page, url)
|
|
html = await page.content()
|
|
except Exception as pe:
|
|
tqdm.write(f" x Playwright failed to fetch playlist page: {pe}")
|
|
finally:
|
|
await page.close()
|
|
|
|
if not html:
|
|
tqdm.write(f" Failed to fetch page {page_num}")
|
|
break
|
|
|
|
video_urls = extract_chezcathy_video_links(html)
|
|
|
|
if not video_urls:
|
|
tqdm.write(f" No video links found on page.")
|
|
break
|
|
|
|
tqdm.write(f" Found {len(video_urls)} video links")
|
|
for video_url in video_urls:
|
|
if is_filtered(video_url):
|
|
tqdm.write(f" x Filtered: {video_url}")
|
|
continue
|
|
await queue.put(video_url)
|
|
counter["total"] += 1
|
|
overall_bar.total = counter["total"]
|
|
overall_bar.refresh()
|
|
|
|
# Check for pagination next page
|
|
if not scraper_core.has_next_page(html) and f"page={page_num}" not in html:
|
|
# Simple check if there is a next page marker or if we found no new links
|
|
break
|
|
|
|
page_num += 1
|
|
await asyncio.sleep(random.uniform(1.0, 2.5))
|
|
|
|
def extract_video_info(html: str, url: str) -> tuple:
|
|
"""Extract video source and uploader/creator name from ChezCathy HTML."""
|
|
# Attempt to extract video source using standard scraper_core extractors
|
|
src, uploader = scraper_core.extract_video_info_from_html(html)
|
|
|
|
# ChezCathy specific creator/uploader extraction
|
|
# Look for user profile links like: href="/user/7115/ryan_thomas" or /profile/...
|
|
user_match = re.search(r'href="/user/\d+/([^"/]+)"', html)
|
|
if user_match:
|
|
uploader = user_match.group(1)
|
|
|
|
if not uploader or uploader == "unknown":
|
|
# Fallback to domain name or generic subfolder if uploader is still unknown
|
|
uploader = "unknown_creator"
|
|
|
|
uploader = scraper_core.clean_filename(uploader)
|
|
return src, uploader
|
|
|
|
async def worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper):
|
|
"""Worker: fetch video page HTML, extract source, and download."""
|
|
while True:
|
|
url = await queue.get()
|
|
if url is None:
|
|
queue.task_done()
|
|
break
|
|
|
|
try:
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered: {url}")
|
|
continue
|
|
|
|
await asyncio.sleep(random.uniform(0.3, 0.8))
|
|
|
|
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.chezcathy.com/")
|
|
|
|
# Fallback to Playwright if static fetch fails
|
|
if not html:
|
|
tqdm.write(f" Static fetch failed for {url}. Fetching page via Playwright...")
|
|
page = await scraper.new_page()
|
|
try:
|
|
await scraper.resolve_page(page, url)
|
|
html = await page.content()
|
|
except Exception as pe:
|
|
tqdm.write(f" x Playwright failed to fetch HTML for {url}: {pe}")
|
|
finally:
|
|
await page.close()
|
|
|
|
if not html:
|
|
tqdm.write(f" x Failed to fetch HTML for {url}")
|
|
continue
|
|
|
|
src, uploader = extract_video_info(html, url)
|
|
|
|
# If dynamic rendering is required to capture the media source
|
|
if not src:
|
|
tqdm.write(f" * Static extraction failed for {url}. Trying Playwright...")
|
|
page = await scraper.new_page()
|
|
try:
|
|
src = await scraper.extract_media_source(page, url)
|
|
except Exception as pe:
|
|
tqdm.write(f" x Playwright extraction failed: {pe}")
|
|
finally:
|
|
await page.close()
|
|
|
|
if not src:
|
|
tqdm.write(f" x Skipping {url} - no video source found")
|
|
continue
|
|
|
|
# Standardize filename based on URL slug and ID
|
|
match = re.search(r'/(\d+)/([^/]+)\.html', url)
|
|
if match:
|
|
video_id = match.group(1)
|
|
slug = match.group(2)
|
|
filename = f"{scraper_core.clean_filename(slug)}-{video_id}.mp4"
|
|
else:
|
|
stem = url.rstrip(".html").rstrip("/").rsplit("/", 1)[-1]
|
|
filename = scraper_core.clean_filename(stem) + ".mp4"
|
|
dest_path = DOWNLOAD_DIR / uploader / filename
|
|
|
|
if skip_existing and scraper_core.is_already_downloaded(dest_path):
|
|
tqdm.write(f" > Already downloaded: {dest_path.parent.name}/{dest_path.name}")
|
|
continue
|
|
|
|
pos = bar_pool.acquire()
|
|
if pos is None:
|
|
pos = 1
|
|
|
|
success = await asyncio.to_thread(
|
|
scraper_core.download_file,
|
|
src, dest_path, {"Referer": url}, pos, cookies=cookies
|
|
)
|
|
|
|
bar_pool.release(pos)
|
|
|
|
if success:
|
|
scraper_core.append_log(DOWNLOAD_DIR / "urls.txt", url)
|
|
scraper_core.append_log(DOWNLOAD_DIR / uploader / "urls.txt", url)
|
|
|
|
except Exception as e:
|
|
tqdm.write(f" Error processing {url}: {e}")
|
|
finally:
|
|
overall_bar.update(1)
|
|
queue.task_done()
|
|
|
|
async def producer(urls, cookies, queue, counter, overall_bar, scraper):
|
|
"""Feed input URLs (video pages or playlists/profile pages) into the queue."""
|
|
for url in urls:
|
|
# Check if URL is a direct video page
|
|
if ("/player/" in url or "/adult/" in url) and url.endswith(".html"):
|
|
if is_filtered(url):
|
|
tqdm.write(f" x Filtered (not queued): {url}")
|
|
continue
|
|
tqdm.write(f"Queuing direct video URL: {url}")
|
|
await queue.put(url)
|
|
counter["total"] += 1
|
|
overall_bar.total = counter["total"]
|
|
overall_bar.refresh()
|
|
else:
|
|
tqdm.write(f"Scraping user/playlist: {url}")
|
|
try:
|
|
await scrape_playlist(url, cookies, queue, counter, overall_bar, scraper)
|
|
except Exception as e:
|
|
tqdm.write(f" Error scraping playlist {url}: {e}")
|
|
|
|
await asyncio.sleep(random.uniform(1.0, 2.0))
|
|
|
|
async def process_urls(urls, concurrency, skip_existing):
|
|
scraper = scraper_core.PlaywrightScraper()
|
|
await scraper.start()
|
|
|
|
tqdm.write("Solving Cloudflare cookies...")
|
|
first_url = urls[0]
|
|
cookies = await get_cloudflare_cookies(scraper, first_url)
|
|
tqdm.write(f"Got cookies: {list(cookies.keys())}")
|
|
|
|
queue = asyncio.Queue()
|
|
counter = {"total": 0}
|
|
overall_bar = tqdm(
|
|
total=0,
|
|
desc="Overall Progress",
|
|
position=0,
|
|
leave=True,
|
|
ncols=80,
|
|
bar_format="{desc}: {n_fmt}/{total_fmt} |{bar}| {percentage:.0f}%",
|
|
)
|
|
bar_pool = scraper_core.BarPositionPool(concurrency)
|
|
|
|
workers = [
|
|
asyncio.create_task(worker(queue, cookies, skip_existing, bar_pool, overall_bar, scraper))
|
|
for _ in range(concurrency)
|
|
]
|
|
|
|
await producer(urls, cookies, queue, counter, overall_bar, scraper)
|
|
|
|
for _ in range(concurrency):
|
|
await queue.put(None)
|
|
|
|
await asyncio.gather(*workers)
|
|
overall_bar.close()
|
|
await scraper.close()
|
|
|
|
sys.stdout.write("\n" * (concurrency + 1))
|
|
sys.stdout.flush()
|
|
|
|
def resolve_urls(args_or_files):
|
|
urls = []
|
|
for arg in args_or_files:
|
|
if os.path.isfile(arg):
|
|
with open(arg, "r") as f:
|
|
for line in f:
|
|
for token in line.strip().split():
|
|
if token and not token.startswith("#"):
|
|
urls.append(token)
|
|
else:
|
|
urls.append(arg)
|
|
return urls
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(
|
|
description="Download videos from ChezCathy (concurrent, metadata-aware)."
|
|
)
|
|
parser.add_argument(
|
|
"urls",
|
|
nargs="*",
|
|
help="Profile URLs, direct video URLs, or .txt files containing them.",
|
|
)
|
|
parser.add_argument(
|
|
"--concurrency",
|
|
type=int,
|
|
default=3,
|
|
help="Number of concurrent downloads (default: 3).",
|
|
)
|
|
parser.add_argument(
|
|
"--skip-existing",
|
|
action="store_true",
|
|
default=True,
|
|
help="Skip already-downloaded files (default: true).",
|
|
)
|
|
|
|
args = parser.parse_args()
|
|
if not args.urls:
|
|
sys.exit("Please specify at least one URL or file containing URLs.")
|
|
|
|
targets = resolve_urls(args.urls)
|
|
asyncio.run(process_urls(targets, args.concurrency, args.skip_existing))
|
|
|
|
if __name__ == "__main__":
|
|
main()
|