365 lines
13 KiB
Python
365 lines
13 KiB
Python
import os
|
|
import sys
|
|
import re
|
|
import asyncio
|
|
import argparse
|
|
from pathlib import Path
|
|
from urllib.parse import urlparse
|
|
from tqdm import tqdm
|
|
|
|
import scraper_core
|
|
|
|
|
|
def _is_filtered(url: str) -> bool:
|
|
"""Apply the shared master filter list (lazy import avoids circular deps)."""
|
|
try:
|
|
from downloader import is_filtered
|
|
except Exception:
|
|
return False
|
|
return is_filtered(url)
|
|
|
|
|
|
async def get_video_info(scraper, page, url, video_selector, uploader_eval_js=None):
|
|
"""Resolve page, extract video source URL, and uploader name."""
|
|
await scraper.resolve_page(page, url)
|
|
|
|
# Run user uploader evaluation if provided, else generic fallback
|
|
if uploader_eval_js:
|
|
uploader_raw = await page.evaluate(uploader_eval_js)
|
|
else:
|
|
uploader_raw = await page.evaluate("""() => {
|
|
const selectors = [
|
|
'.uploader', '.author', '.user-name',
|
|
'a[href*="/members/"]', 'a[href*="/user/"]',
|
|
'.video-metadata a[href*="/profile/"]'
|
|
];
|
|
for (const s of selectors) {
|
|
const el = document.querySelector(s);
|
|
if (el && el.innerText.trim()) return el.innerText.trim();
|
|
}
|
|
return 'unknown';
|
|
}""")
|
|
|
|
uploader = scraper_core.clean_filename(uploader_raw) if uploader_raw else "unknown"
|
|
|
|
# Try core video src extraction first (intercepts player network requests)
|
|
src = await scraper.extract_media_source(page, url, video_selector)
|
|
return src, uploader
|
|
|
|
|
|
async def scrape_playlist(page, playlist_url: str, is_video_link_fn, next_page_selector):
|
|
"""Scrape all video URLs from a playlist/user profile, following pagination."""
|
|
base = playlist_url.rstrip("/")
|
|
all_video_urls = []
|
|
page_num = 1
|
|
seen = set()
|
|
|
|
while True:
|
|
# Navigate or click to pagination
|
|
if page_num > 1:
|
|
clicked = False
|
|
try:
|
|
# Find the button/link matching the target page_num exactly
|
|
btn_info = await page.evaluate(f"""(pageNum) => {{
|
|
const btn = Array.from(document.querySelectorAll('button, a'))
|
|
.find(b => b.innerText.trim() === String(pageNum));
|
|
if (btn) {{
|
|
return {{
|
|
found: true,
|
|
href: btn.href || null
|
|
}};
|
|
}}
|
|
return {{ found: false }};
|
|
}}""", page_num)
|
|
|
|
if btn_info.get("found"):
|
|
href = btn_info.get("href")
|
|
if href and href.startswith("http"):
|
|
tqdm.write(f"Scraping page {page_num} via direct href navigation: {href}")
|
|
await page.goto(href, wait_until="domcontentloaded", timeout=30000)
|
|
await page.wait_for_timeout(3000)
|
|
clicked = True
|
|
else:
|
|
clicked = await page.evaluate(f"""(pageNum) => {{
|
|
const btn = Array.from(document.querySelectorAll('button, a'))
|
|
.find(b => b.innerText.trim() === String(pageNum));
|
|
if (btn) {{
|
|
btn.click();
|
|
return true;
|
|
}}
|
|
return false;
|
|
}}""", page_num)
|
|
if clicked:
|
|
tqdm.write(f"Scraping page {page_num} via client-side pagination click...")
|
|
await page.wait_for_timeout(3000)
|
|
except Exception as e:
|
|
tqdm.write(f" Failed client-side click/navigation: {e}")
|
|
|
|
if not clicked:
|
|
# Fall back to URL navigation
|
|
if "?" in base:
|
|
url = f"{base}&page={page_num}"
|
|
else:
|
|
url = f"{base}/page/{page_num}/" if "/" in base else f"{base}?page={page_num}"
|
|
tqdm.write(f"Scraping page {page_num} via URL navigation: {url}")
|
|
try:
|
|
await page.goto(url, wait_until="domcontentloaded", timeout=30000)
|
|
await page.wait_for_timeout(2000)
|
|
except Exception as e:
|
|
tqdm.write(f" Error loading page {page_num}: {e}")
|
|
break
|
|
else:
|
|
tqdm.write(f"Scraping page {page_num}: {base}")
|
|
try:
|
|
await page.goto(base, wait_until="domcontentloaded", timeout=30000)
|
|
await page.wait_for_timeout(2000)
|
|
except Exception as e:
|
|
tqdm.write(f" Error loading page {page_num}: {e}")
|
|
break
|
|
|
|
# Extract all link tags and JSON-LD URLs
|
|
links = await page.evaluate("""() => {
|
|
const urls = Array.from(document.querySelectorAll('a')).map(el => el.href);
|
|
const scripts = document.querySelectorAll('script[type="application/ld+json"]');
|
|
for (const s of scripts) {
|
|
try {
|
|
const data = JSON.parse(s.innerText);
|
|
if (data['@type'] === 'ItemList' && data.itemListElement) {
|
|
for (const element of data.itemListElement) {
|
|
const item = element.item;
|
|
if (item) {
|
|
if (item.embedUrl) urls.push(item.embedUrl);
|
|
if (item.url) urls.push(item.url);
|
|
}
|
|
}
|
|
}
|
|
} catch (e) {}
|
|
}
|
|
return urls;
|
|
}""")
|
|
|
|
unique_page_urls = []
|
|
for l in links:
|
|
if l in seen or not is_video_link_fn(l):
|
|
continue
|
|
if _is_filtered(l):
|
|
tqdm.write(f" x Filtered: {l}")
|
|
continue
|
|
seen.add(l)
|
|
unique_page_urls.append(l)
|
|
|
|
if not unique_page_urls:
|
|
tqdm.write(" No new video links found on page.")
|
|
break
|
|
|
|
tqdm.write(f" Found {len(unique_page_urls)} video links")
|
|
all_video_urls.extend(unique_page_urls)
|
|
|
|
# Check for next page availability
|
|
has_next = False
|
|
if next_page_selector:
|
|
next_button = await page.query_selector(next_page_selector)
|
|
if next_button:
|
|
has_next = True
|
|
|
|
if not has_next:
|
|
# Check for client-side pagination numerical button matching page_num + 1
|
|
has_next = await page.evaluate(f"""(nextPage) => {{
|
|
return Array.from(document.querySelectorAll('button, a'))
|
|
.some(b => b.innerText.trim() === String(nextPage));
|
|
}}""", page_num + 1)
|
|
|
|
if not has_next:
|
|
# Check if there is any pagination link for next page in extracted links
|
|
next_page_str = f"page={page_num + 1}"
|
|
next_page_str_alt = f"/page/{page_num + 1}"
|
|
has_next = any((next_page_str in l or next_page_str_alt in l) for l in links)
|
|
|
|
if not has_next:
|
|
tqdm.write(" No next page button or link found, done.")
|
|
break
|
|
|
|
page_num += 1
|
|
|
|
return all_video_urls
|
|
|
|
|
|
async def worker(queue, scraper, download_dir, skip_existing, bar_pool, overall_bar, video_selector, uploader_eval_js):
|
|
"""Worker task that handles resolving video URL and downloading."""
|
|
while True:
|
|
url = await queue.get()
|
|
if url is None:
|
|
queue.task_done()
|
|
break
|
|
|
|
try:
|
|
if _is_filtered(url):
|
|
tqdm.write(f" x Filtered: {url}")
|
|
continue
|
|
page = await scraper.new_page()
|
|
src, uploader = await get_video_info(scraper, page, url, video_selector, uploader_eval_js)
|
|
await page.close()
|
|
|
|
if not src:
|
|
tqdm.write(f" [ERROR] Skipping {url} - no video source found")
|
|
continue
|
|
|
|
# Create a clean filename from URL path
|
|
parsed = urlparse(url)
|
|
stem = parsed.path.rstrip("/").split("/")[-1]
|
|
if not stem or stem == "video" or stem.isdigit():
|
|
stem = parsed.path.rstrip("/").split("/")[-2] + "_" + stem
|
|
filename = scraper_core.clean_filename(stem) + ".mp4"
|
|
dest_path = download_dir / uploader / filename
|
|
|
|
downloaded_bytes = 0
|
|
if skip_existing:
|
|
from downloader import check_existing_file
|
|
is_complete, existing_bytes = await check_existing_file(dest_path, src, {"Referer": url})
|
|
if is_complete:
|
|
if hasattr(overall_bar, "record_skip"):
|
|
overall_bar.record_skip(existing_bytes)
|
|
else:
|
|
overall_bar.update(1)
|
|
continue
|
|
|
|
pos = bar_pool.acquire() or 4
|
|
|
|
# Download concurrently in a thread
|
|
success = await asyncio.to_thread(
|
|
scraper_core.download_file,
|
|
src, dest_path, None, pos, url
|
|
)
|
|
|
|
bar_pool.release(pos)
|
|
|
|
if success:
|
|
if dest_path.is_file():
|
|
downloaded_bytes = dest_path.stat().st_size
|
|
label = f"{dest_path.parent.name}/{dest_path.name}"
|
|
if hasattr(overall_bar, "record_download"):
|
|
overall_bar.record_download(downloaded_bytes, name=label)
|
|
else:
|
|
overall_bar.update(1)
|
|
scraper_core.append_log(download_dir / "urls.txt", url)
|
|
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
|
|
else:
|
|
overall_bar.update(1)
|
|
|
|
except Exception as e:
|
|
tqdm.write(f" Error processing {url}: {e}")
|
|
overall_bar.update(1)
|
|
finally:
|
|
queue.task_done()
|
|
|
|
|
|
async def process_urls(site_name, urls, download_dir, is_video_link_fn, next_page_selector, video_selector, uploader_eval_js, concurrency, skip_existing):
|
|
scraper = scraper_core.PlaywrightScraper()
|
|
await scraper.start()
|
|
|
|
# Resolve playlists first if any target is a playlist
|
|
playlist_page = await scraper.new_page()
|
|
all_urls = []
|
|
for url in urls:
|
|
if is_video_link_fn(url):
|
|
if _is_filtered(url):
|
|
tqdm.write(f" x Filtered (not queued): {url}")
|
|
continue
|
|
all_urls.append(url)
|
|
else:
|
|
playlist_urls = await scrape_playlist(playlist_page, url, is_video_link_fn, next_page_selector)
|
|
all_urls.extend(playlist_urls)
|
|
await playlist_page.close()
|
|
|
|
if not all_urls:
|
|
tqdm.write("No video URLs found to process.")
|
|
await scraper.close()
|
|
return
|
|
|
|
tqdm.write(f"[{site_name}] Processing {len(all_urls)} videos with concurrency {concurrency} ...")
|
|
|
|
# Queue setup
|
|
queue = asyncio.Queue()
|
|
for url in all_urls:
|
|
await queue.put(url)
|
|
for _ in range(concurrency):
|
|
await queue.put(None)
|
|
|
|
bar_pool = scraper_core.BarPositionPool(concurrency)
|
|
bar_pool.available = [p + 3 for p in bar_pool.available]
|
|
|
|
try:
|
|
from downloader import OverallProgressTracker
|
|
overall_bar = OverallProgressTracker(
|
|
total=len(all_urls),
|
|
desc=f"[{site_name}] Overall",
|
|
)
|
|
except Exception:
|
|
overall_bar = tqdm(
|
|
total=len(all_urls),
|
|
desc=f"[{site_name}] Overall",
|
|
position=2,
|
|
leave=True,
|
|
ncols=80,
|
|
)
|
|
|
|
workers = [
|
|
asyncio.create_task(worker(queue, scraper, download_dir, skip_existing, bar_pool, overall_bar, video_selector, uploader_eval_js))
|
|
for _ in range(concurrency)
|
|
]
|
|
|
|
await asyncio.gather(*workers)
|
|
overall_bar.close()
|
|
|
|
# Clear terminal lines
|
|
sys.stdout.write("\n" * (concurrency + 4))
|
|
sys.stdout.flush()
|
|
|
|
await scraper.close()
|
|
|
|
|
|
def run(site_name, is_video_link_fn, next_page_selector, video_selector="video", uploader_eval_js=None, default_playlist=""):
|
|
parser = argparse.ArgumentParser(
|
|
description=f"Download videos from {site_name} (headless, concurrent, multithreaded)."
|
|
)
|
|
parser.add_argument(
|
|
"urls",
|
|
nargs="*",
|
|
help="Playlist or individual video URLs.",
|
|
)
|
|
parser.add_argument(
|
|
"--concurrency",
|
|
type=int,
|
|
default=3,
|
|
help="Number of concurrent downloads (default: 3).",
|
|
)
|
|
parser.add_argument(
|
|
"--skip-existing",
|
|
action="store_true",
|
|
default=True,
|
|
help="Skip already-downloaded files (default: true).",
|
|
)
|
|
parser.add_argument(
|
|
"--no-skip-existing",
|
|
action="store_false",
|
|
dest="skip_existing",
|
|
help="Re-download existing files.",
|
|
)
|
|
|
|
args = parser.parse_args()
|
|
|
|
targets = args.urls
|
|
if not targets:
|
|
if default_playlist:
|
|
targets = [default_playlist]
|
|
else:
|
|
sys.exit("Error: No URLs provided.")
|
|
|
|
download_dir = Path(sys.argv[0]).resolve().parent / "videos"
|
|
asyncio.run(process_urls(
|
|
site_name, targets, download_dir,
|
|
is_video_link_fn, next_page_selector,
|
|
video_selector, uploader_eval_js,
|
|
args.concurrency, args.skip_existing
|
|
))
|