Initial backup: folder structure manifest, tree summary, scraper scripts, and text metadata
This commit is contained in:
commit
f164832dcc
10504 files changed
+2423594
No files matched your search
@@ -0,0 +1,364 @@
|
||||
import os
|
||||
import sys
|
||||
import re
|
||||
import asyncio
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
from urllib.parse import urlparse
|
||||
from tqdm import tqdm
|
||||
|
||||
import scraper_core
|
||||
|
||||
|
||||
def _is_filtered(url: str) -> bool:
|
||||
"""Apply the shared master filter list (lazy import avoids circular deps)."""
|
||||
try:
|
||||
from downloader import is_filtered
|
||||
except Exception:
|
||||
return False
|
||||
return is_filtered(url)
|
||||
|
||||
|
||||
async def get_video_info(scraper, page, url, video_selector, uploader_eval_js=None):
|
||||
"""Resolve page, extract video source URL, and uploader name."""
|
||||
await scraper.resolve_page(page, url)
|
||||
|
||||
# Run user uploader evaluation if provided, else generic fallback
|
||||
if uploader_eval_js:
|
||||
uploader_raw = await page.evaluate(uploader_eval_js)
|
||||
else:
|
||||
uploader_raw = await page.evaluate("""() => {
|
||||
const selectors = [
|
||||
'.uploader', '.author', '.user-name',
|
||||
'a[href*="/members/"]', 'a[href*="/user/"]',
|
||||
'.video-metadata a[href*="/profile/"]'
|
||||
];
|
||||
for (const s of selectors) {
|
||||
const el = document.querySelector(s);
|
||||
if (el && el.innerText.trim()) return el.innerText.trim();
|
||||
}
|
||||
return 'unknown';
|
||||
}""")
|
||||
|
||||
uploader = scraper_core.clean_filename(uploader_raw) if uploader_raw else "unknown"
|
||||
|
||||
# Try core video src extraction first (intercepts player network requests)
|
||||
src = await scraper.extract_media_source(page, url, video_selector)
|
||||
return src, uploader
|
||||
|
||||
|
||||
async def scrape_playlist(page, playlist_url: str, is_video_link_fn, next_page_selector):
|
||||
"""Scrape all video URLs from a playlist/user profile, following pagination."""
|
||||
base = playlist_url.rstrip("/")
|
||||
all_video_urls = []
|
||||
page_num = 1
|
||||
seen = set()
|
||||
|
||||
while True:
|
||||
# Navigate or click to pagination
|
||||
if page_num > 1:
|
||||
clicked = False
|
||||
try:
|
||||
# Find the button/link matching the target page_num exactly
|
||||
btn_info = await page.evaluate(f"""(pageNum) => {{
|
||||
const btn = Array.from(document.querySelectorAll('button, a'))
|
||||
.find(b => b.innerText.trim() === String(pageNum));
|
||||
if (btn) {{
|
||||
return {{
|
||||
found: true,
|
||||
href: btn.href || null
|
||||
}};
|
||||
}}
|
||||
return {{ found: false }};
|
||||
}}""", page_num)
|
||||
|
||||
if btn_info.get("found"):
|
||||
href = btn_info.get("href")
|
||||
if href and href.startswith("http"):
|
||||
tqdm.write(f"Scraping page {page_num} via direct href navigation: {href}")
|
||||
await page.goto(href, wait_until="domcontentloaded", timeout=30000)
|
||||
await page.wait_for_timeout(3000)
|
||||
clicked = True
|
||||
else:
|
||||
clicked = await page.evaluate(f"""(pageNum) => {{
|
||||
const btn = Array.from(document.querySelectorAll('button, a'))
|
||||
.find(b => b.innerText.trim() === String(pageNum));
|
||||
if (btn) {{
|
||||
btn.click();
|
||||
return true;
|
||||
}}
|
||||
return false;
|
||||
}}""", page_num)
|
||||
if clicked:
|
||||
tqdm.write(f"Scraping page {page_num} via client-side pagination click...")
|
||||
await page.wait_for_timeout(3000)
|
||||
except Exception as e:
|
||||
tqdm.write(f" Failed client-side click/navigation: {e}")
|
||||
|
||||
if not clicked:
|
||||
# Fall back to URL navigation
|
||||
if "?" in base:
|
||||
url = f"{base}&page={page_num}"
|
||||
else:
|
||||
url = f"{base}/page/{page_num}/" if "/" in base else f"{base}?page={page_num}"
|
||||
tqdm.write(f"Scraping page {page_num} via URL navigation: {url}")
|
||||
try:
|
||||
await page.goto(url, wait_until="domcontentloaded", timeout=30000)
|
||||
await page.wait_for_timeout(2000)
|
||||
except Exception as e:
|
||||
tqdm.write(f" Error loading page {page_num}: {e}")
|
||||
break
|
||||
else:
|
||||
tqdm.write(f"Scraping page {page_num}: {base}")
|
||||
try:
|
||||
await page.goto(base, wait_until="domcontentloaded", timeout=30000)
|
||||
await page.wait_for_timeout(2000)
|
||||
except Exception as e:
|
||||
tqdm.write(f" Error loading page {page_num}: {e}")
|
||||
break
|
||||
|
||||
# Extract all link tags and JSON-LD URLs
|
||||
links = await page.evaluate("""() => {
|
||||
const urls = Array.from(document.querySelectorAll('a')).map(el => el.href);
|
||||
const scripts = document.querySelectorAll('script[type="application/ld+json"]');
|
||||
for (const s of scripts) {
|
||||
try {
|
||||
const data = JSON.parse(s.innerText);
|
||||
if (data['@type'] === 'ItemList' && data.itemListElement) {
|
||||
for (const element of data.itemListElement) {
|
||||
const item = element.item;
|
||||
if (item) {
|
||||
if (item.embedUrl) urls.push(item.embedUrl);
|
||||
if (item.url) urls.push(item.url);
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (e) {}
|
||||
}
|
||||
return urls;
|
||||
}""")
|
||||
|
||||
unique_page_urls = []
|
||||
for l in links:
|
||||
if l in seen or not is_video_link_fn(l):
|
||||
continue
|
||||
if _is_filtered(l):
|
||||
tqdm.write(f" x Filtered: {l}")
|
||||
continue
|
||||
seen.add(l)
|
||||
unique_page_urls.append(l)
|
||||
|
||||
if not unique_page_urls:
|
||||
tqdm.write(" No new video links found on page.")
|
||||
break
|
||||
|
||||
tqdm.write(f" Found {len(unique_page_urls)} video links")
|
||||
all_video_urls.extend(unique_page_urls)
|
||||
|
||||
# Check for next page availability
|
||||
has_next = False
|
||||
if next_page_selector:
|
||||
next_button = await page.query_selector(next_page_selector)
|
||||
if next_button:
|
||||
has_next = True
|
||||
|
||||
if not has_next:
|
||||
# Check for client-side pagination numerical button matching page_num + 1
|
||||
has_next = await page.evaluate(f"""(nextPage) => {{
|
||||
return Array.from(document.querySelectorAll('button, a'))
|
||||
.some(b => b.innerText.trim() === String(nextPage));
|
||||
}}""", page_num + 1)
|
||||
|
||||
if not has_next:
|
||||
# Check if there is any pagination link for next page in extracted links
|
||||
next_page_str = f"page={page_num + 1}"
|
||||
next_page_str_alt = f"/page/{page_num + 1}"
|
||||
has_next = any((next_page_str in l or next_page_str_alt in l) for l in links)
|
||||
|
||||
if not has_next:
|
||||
tqdm.write(" No next page button or link found, done.")
|
||||
break
|
||||
|
||||
page_num += 1
|
||||
|
||||
return all_video_urls
|
||||
|
||||
|
||||
async def worker(queue, scraper, download_dir, skip_existing, bar_pool, overall_bar, video_selector, uploader_eval_js):
|
||||
"""Worker task that handles resolving video URL and downloading."""
|
||||
while True:
|
||||
url = await queue.get()
|
||||
if url is None:
|
||||
queue.task_done()
|
||||
break
|
||||
|
||||
try:
|
||||
if _is_filtered(url):
|
||||
tqdm.write(f" x Filtered: {url}")
|
||||
continue
|
||||
page = await scraper.new_page()
|
||||
src, uploader = await get_video_info(scraper, page, url, video_selector, uploader_eval_js)
|
||||
await page.close()
|
||||
|
||||
if not src:
|
||||
tqdm.write(f" [ERROR] Skipping {url} - no video source found")
|
||||
continue
|
||||
|
||||
# Create a clean filename from URL path
|
||||
parsed = urlparse(url)
|
||||
stem = parsed.path.rstrip("/").split("/")[-1]
|
||||
if not stem or stem == "video" or stem.isdigit():
|
||||
stem = parsed.path.rstrip("/").split("/")[-2] + "_" + stem
|
||||
filename = scraper_core.clean_filename(stem) + ".mp4"
|
||||
dest_path = download_dir / uploader / filename
|
||||
|
||||
downloaded_bytes = 0
|
||||
if skip_existing:
|
||||
from downloader import check_existing_file
|
||||
is_complete, existing_bytes = await check_existing_file(dest_path, src, {"Referer": url})
|
||||
if is_complete:
|
||||
if hasattr(overall_bar, "record_skip"):
|
||||
overall_bar.record_skip(existing_bytes)
|
||||
else:
|
||||
overall_bar.update(1)
|
||||
continue
|
||||
|
||||
pos = bar_pool.acquire() or 4
|
||||
|
||||
# Download concurrently in a thread
|
||||
success = await asyncio.to_thread(
|
||||
scraper_core.download_file,
|
||||
src, dest_path, None, pos, url
|
||||
)
|
||||
|
||||
bar_pool.release(pos)
|
||||
|
||||
if success:
|
||||
if dest_path.is_file():
|
||||
downloaded_bytes = dest_path.stat().st_size
|
||||
label = f"{dest_path.parent.name}/{dest_path.name}"
|
||||
if hasattr(overall_bar, "record_download"):
|
||||
overall_bar.record_download(downloaded_bytes, name=label)
|
||||
else:
|
||||
overall_bar.update(1)
|
||||
scraper_core.append_log(download_dir / "urls.txt", url)
|
||||
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
|
||||
else:
|
||||
overall_bar.update(1)
|
||||
|
||||
except Exception as e:
|
||||
tqdm.write(f" Error processing {url}: {e}")
|
||||
overall_bar.update(1)
|
||||
finally:
|
||||
queue.task_done()
|
||||
|
||||
|
||||
async def process_urls(site_name, urls, download_dir, is_video_link_fn, next_page_selector, video_selector, uploader_eval_js, concurrency, skip_existing):
|
||||
scraper = scraper_core.PlaywrightScraper()
|
||||
await scraper.start()
|
||||
|
||||
# Resolve playlists first if any target is a playlist
|
||||
playlist_page = await scraper.new_page()
|
||||
all_urls = []
|
||||
for url in urls:
|
||||
if is_video_link_fn(url):
|
||||
if _is_filtered(url):
|
||||
tqdm.write(f" x Filtered (not queued): {url}")
|
||||
continue
|
||||
all_urls.append(url)
|
||||
else:
|
||||
playlist_urls = await scrape_playlist(playlist_page, url, is_video_link_fn, next_page_selector)
|
||||
all_urls.extend(playlist_urls)
|
||||
await playlist_page.close()
|
||||
|
||||
if not all_urls:
|
||||
tqdm.write("No video URLs found to process.")
|
||||
await scraper.close()
|
||||
return
|
||||
|
||||
tqdm.write(f"[{site_name}] Processing {len(all_urls)} videos with concurrency {concurrency} ...")
|
||||
|
||||
# Queue setup
|
||||
queue = asyncio.Queue()
|
||||
for url in all_urls:
|
||||
await queue.put(url)
|
||||
for _ in range(concurrency):
|
||||
await queue.put(None)
|
||||
|
||||
bar_pool = scraper_core.BarPositionPool(concurrency)
|
||||
bar_pool.available = [p + 3 for p in bar_pool.available]
|
||||
|
||||
try:
|
||||
from downloader import OverallProgressTracker
|
||||
overall_bar = OverallProgressTracker(
|
||||
total=len(all_urls),
|
||||
desc=f"[{site_name}] Overall",
|
||||
)
|
||||
except Exception:
|
||||
overall_bar = tqdm(
|
||||
total=len(all_urls),
|
||||
desc=f"[{site_name}] Overall",
|
||||
position=2,
|
||||
leave=True,
|
||||
ncols=80,
|
||||
)
|
||||
|
||||
workers = [
|
||||
asyncio.create_task(worker(queue, scraper, download_dir, skip_existing, bar_pool, overall_bar, video_selector, uploader_eval_js))
|
||||
for _ in range(concurrency)
|
||||
]
|
||||
|
||||
await asyncio.gather(*workers)
|
||||
overall_bar.close()
|
||||
|
||||
# Clear terminal lines
|
||||
sys.stdout.write("\n" * (concurrency + 4))
|
||||
sys.stdout.flush()
|
||||
|
||||
await scraper.close()
|
||||
|
||||
|
||||
def run(site_name, is_video_link_fn, next_page_selector, video_selector="video", uploader_eval_js=None, default_playlist=""):
|
||||
parser = argparse.ArgumentParser(
|
||||
description=f"Download videos from {site_name} (headless, concurrent, multithreaded)."
|
||||
)
|
||||
parser.add_argument(
|
||||
"urls",
|
||||
nargs="*",
|
||||
help="Playlist or individual video URLs.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--concurrency",
|
||||
type=int,
|
||||
default=3,
|
||||
help="Number of concurrent downloads (default: 3).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--skip-existing",
|
||||
action="store_true",
|
||||
default=True,
|
||||
help="Skip already-downloaded files (default: true).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--no-skip-existing",
|
||||
action="store_false",
|
||||
dest="skip_existing",
|
||||
help="Re-download existing files.",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
targets = args.urls
|
||||
if not targets:
|
||||
if default_playlist:
|
||||
targets = [default_playlist]
|
||||
else:
|
||||
sys.exit("Error: No URLs provided.")
|
||||
|
||||
download_dir = Path(sys.argv[0]).resolve().parent / "videos"
|
||||
asyncio.run(process_urls(
|
||||
site_name, targets, download_dir,
|
||||
is_video_link_fn, next_page_selector,
|
||||
video_selector, uploader_eval_js,
|
||||
args.concurrency, args.skip_existing
|
||||
))
|
||||
Reference in new issue
Block a user