Files
niggers/generic_downloader.py
T

365 lines
13 KiB
Python

import os
import sys
import re
import asyncio
import argparse
from pathlib import Path
from urllib.parse import urlparse
from tqdm import tqdm
import scraper_core
def _is_filtered(url: str) -> bool:
"""Apply the shared master filter list (lazy import avoids circular deps)."""
try:
from downloader import is_filtered
except Exception:
return False
return is_filtered(url)
async def get_video_info(scraper, page, url, video_selector, uploader_eval_js=None):
"""Resolve page, extract video source URL, and uploader name."""
await scraper.resolve_page(page, url)
# Run user uploader evaluation if provided, else generic fallback
if uploader_eval_js:
uploader_raw = await page.evaluate(uploader_eval_js)
else:
uploader_raw = await page.evaluate("""() => {
const selectors = [
'.uploader', '.author', '.user-name',
'a[href*="/members/"]', 'a[href*="/user/"]',
'.video-metadata a[href*="/profile/"]'
];
for (const s of selectors) {
const el = document.querySelector(s);
if (el && el.innerText.trim()) return el.innerText.trim();
}
return 'unknown';
}""")
uploader = scraper_core.clean_filename(uploader_raw) if uploader_raw else "unknown"
# Try core video src extraction first (intercepts player network requests)
src = await scraper.extract_media_source(page, url, video_selector)
return src, uploader
async def scrape_playlist(page, playlist_url: str, is_video_link_fn, next_page_selector):
"""Scrape all video URLs from a playlist/user profile, following pagination."""
base = playlist_url.rstrip("/")
all_video_urls = []
page_num = 1
seen = set()
while True:
# Navigate or click to pagination
if page_num > 1:
clicked = False
try:
# Find the button/link matching the target page_num exactly
btn_info = await page.evaluate(f"""(pageNum) => {{
const btn = Array.from(document.querySelectorAll('button, a'))
.find(b => b.innerText.trim() === String(pageNum));
if (btn) {{
return {{
found: true,
href: btn.href || null
}};
}}
return {{ found: false }};
}}""", page_num)
if btn_info.get("found"):
href = btn_info.get("href")
if href and href.startswith("http"):
tqdm.write(f"Scraping page {page_num} via direct href navigation: {href}")
await page.goto(href, wait_until="domcontentloaded", timeout=30000)
await page.wait_for_timeout(3000)
clicked = True
else:
clicked = await page.evaluate(f"""(pageNum) => {{
const btn = Array.from(document.querySelectorAll('button, a'))
.find(b => b.innerText.trim() === String(pageNum));
if (btn) {{
btn.click();
return true;
}}
return false;
}}""", page_num)
if clicked:
tqdm.write(f"Scraping page {page_num} via client-side pagination click...")
await page.wait_for_timeout(3000)
except Exception as e:
tqdm.write(f" Failed client-side click/navigation: {e}")
if not clicked:
# Fall back to URL navigation
if "?" in base:
url = f"{base}&page={page_num}"
else:
url = f"{base}/page/{page_num}/" if "/" in base else f"{base}?page={page_num}"
tqdm.write(f"Scraping page {page_num} via URL navigation: {url}")
try:
await page.goto(url, wait_until="domcontentloaded", timeout=30000)
await page.wait_for_timeout(2000)
except Exception as e:
tqdm.write(f" Error loading page {page_num}: {e}")
break
else:
tqdm.write(f"Scraping page {page_num}: {base}")
try:
await page.goto(base, wait_until="domcontentloaded", timeout=30000)
await page.wait_for_timeout(2000)
except Exception as e:
tqdm.write(f" Error loading page {page_num}: {e}")
break
# Extract all link tags and JSON-LD URLs
links = await page.evaluate("""() => {
const urls = Array.from(document.querySelectorAll('a')).map(el => el.href);
const scripts = document.querySelectorAll('script[type="application/ld+json"]');
for (const s of scripts) {
try {
const data = JSON.parse(s.innerText);
if (data['@type'] === 'ItemList' && data.itemListElement) {
for (const element of data.itemListElement) {
const item = element.item;
if (item) {
if (item.embedUrl) urls.push(item.embedUrl);
if (item.url) urls.push(item.url);
}
}
}
} catch (e) {}
}
return urls;
}""")
unique_page_urls = []
for l in links:
if l in seen or not is_video_link_fn(l):
continue
if _is_filtered(l):
tqdm.write(f" x Filtered: {l}")
continue
seen.add(l)
unique_page_urls.append(l)
if not unique_page_urls:
tqdm.write(" No new video links found on page.")
break
tqdm.write(f" Found {len(unique_page_urls)} video links")
all_video_urls.extend(unique_page_urls)
# Check for next page availability
has_next = False
if next_page_selector:
next_button = await page.query_selector(next_page_selector)
if next_button:
has_next = True
if not has_next:
# Check for client-side pagination numerical button matching page_num + 1
has_next = await page.evaluate(f"""(nextPage) => {{
return Array.from(document.querySelectorAll('button, a'))
.some(b => b.innerText.trim() === String(nextPage));
}}""", page_num + 1)
if not has_next:
# Check if there is any pagination link for next page in extracted links
next_page_str = f"page={page_num + 1}"
next_page_str_alt = f"/page/{page_num + 1}"
has_next = any((next_page_str in l or next_page_str_alt in l) for l in links)
if not has_next:
tqdm.write(" No next page button or link found, done.")
break
page_num += 1
return all_video_urls
async def worker(queue, scraper, download_dir, skip_existing, bar_pool, overall_bar, video_selector, uploader_eval_js):
"""Worker task that handles resolving video URL and downloading."""
while True:
url = await queue.get()
if url is None:
queue.task_done()
break
try:
if _is_filtered(url):
tqdm.write(f" x Filtered: {url}")
continue
page = await scraper.new_page()
src, uploader = await get_video_info(scraper, page, url, video_selector, uploader_eval_js)
await page.close()
if not src:
tqdm.write(f" [ERROR] Skipping {url} - no video source found")
continue
# Create a clean filename from URL path
parsed = urlparse(url)
stem = parsed.path.rstrip("/").split("/")[-1]
if not stem or stem == "video" or stem.isdigit():
stem = parsed.path.rstrip("/").split("/")[-2] + "_" + stem
filename = scraper_core.clean_filename(stem) + ".mp4"
dest_path = download_dir / uploader / filename
downloaded_bytes = 0
if skip_existing:
from downloader import check_existing_file
is_complete, existing_bytes = await check_existing_file(dest_path, src, {"Referer": url})
if is_complete:
if hasattr(overall_bar, "record_skip"):
overall_bar.record_skip(existing_bytes)
else:
overall_bar.update(1)
continue
pos = bar_pool.acquire() or 4
# Download concurrently in a thread
success = await asyncio.to_thread(
scraper_core.download_file,
src, dest_path, None, pos, url
)
bar_pool.release(pos)
if success:
if dest_path.is_file():
downloaded_bytes = dest_path.stat().st_size
label = f"{dest_path.parent.name}/{dest_path.name}"
if hasattr(overall_bar, "record_download"):
overall_bar.record_download(downloaded_bytes, name=label)
else:
overall_bar.update(1)
scraper_core.append_log(download_dir / "urls.txt", url)
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
else:
overall_bar.update(1)
except Exception as e:
tqdm.write(f" Error processing {url}: {e}")
overall_bar.update(1)
finally:
queue.task_done()
async def process_urls(site_name, urls, download_dir, is_video_link_fn, next_page_selector, video_selector, uploader_eval_js, concurrency, skip_existing):
scraper = scraper_core.PlaywrightScraper()
await scraper.start()
# Resolve playlists first if any target is a playlist
playlist_page = await scraper.new_page()
all_urls = []
for url in urls:
if is_video_link_fn(url):
if _is_filtered(url):
tqdm.write(f" x Filtered (not queued): {url}")
continue
all_urls.append(url)
else:
playlist_urls = await scrape_playlist(playlist_page, url, is_video_link_fn, next_page_selector)
all_urls.extend(playlist_urls)
await playlist_page.close()
if not all_urls:
tqdm.write("No video URLs found to process.")
await scraper.close()
return
tqdm.write(f"[{site_name}] Processing {len(all_urls)} videos with concurrency {concurrency} ...")
# Queue setup
queue = asyncio.Queue()
for url in all_urls:
await queue.put(url)
for _ in range(concurrency):
await queue.put(None)
bar_pool = scraper_core.BarPositionPool(concurrency)
bar_pool.available = [p + 3 for p in bar_pool.available]
try:
from downloader import OverallProgressTracker
overall_bar = OverallProgressTracker(
total=len(all_urls),
desc=f"[{site_name}] Overall",
)
except Exception:
overall_bar = tqdm(
total=len(all_urls),
desc=f"[{site_name}] Overall",
position=2,
leave=True,
ncols=80,
)
workers = [
asyncio.create_task(worker(queue, scraper, download_dir, skip_existing, bar_pool, overall_bar, video_selector, uploader_eval_js))
for _ in range(concurrency)
]
await asyncio.gather(*workers)
overall_bar.close()
# Clear terminal lines
sys.stdout.write("\n" * (concurrency + 4))
sys.stdout.flush()
await scraper.close()
def run(site_name, is_video_link_fn, next_page_selector, video_selector="video", uploader_eval_js=None, default_playlist=""):
parser = argparse.ArgumentParser(
description=f"Download videos from {site_name} (headless, concurrent, multithreaded)."
)
parser.add_argument(
"urls",
nargs="*",
help="Playlist or individual video URLs.",
)
parser.add_argument(
"--concurrency",
type=int,
default=3,
help="Number of concurrent downloads (default: 3).",
)
parser.add_argument(
"--skip-existing",
action="store_true",
default=True,
help="Skip already-downloaded files (default: true).",
)
parser.add_argument(
"--no-skip-existing",
action="store_false",
dest="skip_existing",
help="Re-download existing files.",
)
args = parser.parse_args()
targets = args.urls
if not targets:
if default_playlist:
targets = [default_playlist]
else:
sys.exit("Error: No URLs provided.")
download_dir = Path(sys.argv[0]).resolve().parent / "videos"
asyncio.run(process_urls(
site_name, targets, download_dir,
is_video_link_fn, next_page_selector,
video_selector, uploader_eval_js,
args.concurrency, args.skip_existing
))