458 lines
17 KiB
Python
458 lines
17 KiB
Python
import os
|
|
import re
|
|
import sys
|
|
import time
|
|
import asyncio
|
|
import logging
|
|
import threading
|
|
import random
|
|
from pathlib import Path
|
|
from urllib.parse import urlparse
|
|
import requests
|
|
from tqdm import tqdm
|
|
|
|
logging.basicConfig(
|
|
level=logging.INFO,
|
|
format="%(asctime)s [%(levelname)s] %(message)s",
|
|
datefmt="%H:%M:%S",
|
|
)
|
|
log = logging.getLogger(__name__)
|
|
|
|
USER_AGENT = (
|
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
|
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
|
"Chrome/125.0.0.0 Safari/537.36"
|
|
)
|
|
|
|
HEADERS = {
|
|
"User-Agent": USER_AGENT,
|
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
|
"Accept-Language": "en-US,en;q=0.5",
|
|
"Connection": "keep-alive",
|
|
}
|
|
|
|
STEALTH_JS = """() => {
|
|
Object.defineProperty(navigator, 'webdriver', { get: () => false });
|
|
Object.defineProperty(navigator, 'plugins', { get: () => [1, 2, 3] });
|
|
Object.defineProperty(navigator, 'languages', { get: () => ['en-US', 'en'] });
|
|
}"""
|
|
|
|
class BarPositionPool:
|
|
"""Manages console line offsets for multiple concurrent tqdm progress bars."""
|
|
def __init__(self, size):
|
|
self.lock = threading.Lock()
|
|
self.available = list(range(1, size + 1))
|
|
|
|
def acquire(self):
|
|
with self.lock:
|
|
if self.available:
|
|
return self.available.pop(0)
|
|
return None
|
|
|
|
def release(self, pos):
|
|
with self.lock:
|
|
if pos is not None and pos not in self.available:
|
|
self.available.append(pos)
|
|
self.available.sort()
|
|
|
|
|
|
def clean_filename(name: str) -> str:
|
|
return re.sub(r'[^a-zA-Z0-9_.-]', '_', name)
|
|
|
|
|
|
def download_file(url: str, dest_path: Path, headers: dict = None, bar_position: int = 1, referer: str = None, cookies: dict = None) -> bool:
|
|
"""Download a file with requests showing a progress bar at a specific terminal line."""
|
|
dest_path.parent.mkdir(parents=True, exist_ok=True)
|
|
label = f"{dest_path.parent.name}/{dest_path.name}"
|
|
|
|
req_headers = dict(HEADERS)
|
|
if headers:
|
|
req_headers.update(headers)
|
|
if referer:
|
|
req_headers["Referer"] = referer
|
|
|
|
try:
|
|
if ".m3u8" in url or "manifest" in url:
|
|
import yt_dlp
|
|
|
|
pbar = tqdm(
|
|
desc=label[:25],
|
|
unit="B",
|
|
unit_scale=True,
|
|
unit_divisor=1024,
|
|
position=bar_position,
|
|
leave=False,
|
|
ncols=80,
|
|
)
|
|
|
|
class YtdlpTqdmHook:
|
|
def __init__(self, pbar):
|
|
self.pbar = pbar
|
|
self.last_bytes = 0
|
|
|
|
def __call__(self, d):
|
|
if d['status'] == 'downloading':
|
|
downloaded = d.get('downloaded_bytes', 0)
|
|
total = d.get('total_bytes') or d.get('total_bytes_estimate') or 0
|
|
if total and self.pbar.total is None:
|
|
self.pbar.total = total
|
|
delta = downloaded - self.last_bytes
|
|
if delta > 0:
|
|
self.pbar.update(delta)
|
|
self.last_bytes = downloaded
|
|
elif d['status'] == 'finished':
|
|
if self.pbar.total and self.last_bytes < self.pbar.total:
|
|
self.pbar.update(self.pbar.total - self.last_bytes)
|
|
|
|
ydl_headers = dict(req_headers)
|
|
if ydl_headers.get("User-Agent") == "stash-scraper/v1":
|
|
ydl_headers.pop("User-Agent", None)
|
|
|
|
ydl_opts = {
|
|
'outtmpl': str(dest_path),
|
|
'progress_hooks': [YtdlpTqdmHook(pbar)],
|
|
'quiet': True,
|
|
'noprogress': True,
|
|
'http_headers': ydl_headers,
|
|
'downloader': 'ffmpeg',
|
|
'hls_use_mpegts': True,
|
|
}
|
|
try:
|
|
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
|
result = ydl.download([url])
|
|
if result != 0:
|
|
raise Exception("yt-dlp download failed with non-zero exit code")
|
|
finally:
|
|
pbar.close()
|
|
return True
|
|
|
|
if cookies:
|
|
try:
|
|
from curl_cffi import requests as curl_req
|
|
resp = curl_req.get(
|
|
url, impersonate="chrome", headers=req_headers,
|
|
cookies=cookies, stream=True, timeout=(15, 300),
|
|
)
|
|
resp.raise_for_status()
|
|
total = int(resp.headers.get("content-length", 0))
|
|
|
|
with tqdm(
|
|
total=total, unit="B", unit_scale=True, unit_divisor=1024,
|
|
desc=label[:25], position=bar_position, leave=False, ncols=80,
|
|
) as pbar:
|
|
with open(dest_path, "wb") as f:
|
|
for chunk in resp.iter_content(chunk_size=65536):
|
|
if chunk:
|
|
f.write(chunk)
|
|
pbar.update(len(chunk))
|
|
return True
|
|
except ImportError:
|
|
pass
|
|
|
|
resp = requests.get(url, headers=req_headers, stream=True, timeout=(15, 300), cookies=cookies)
|
|
resp.raise_for_status()
|
|
|
|
if resp.raw._connection:
|
|
sock = resp.raw._connection.sock
|
|
if sock:
|
|
sock.settimeout(30.0)
|
|
|
|
total = int(resp.headers.get("content-length", 0))
|
|
|
|
with tqdm(
|
|
total=total, unit="B", unit_scale=True, unit_divisor=1024,
|
|
desc=label[:25], position=bar_position, leave=False, ncols=80,
|
|
) as pbar:
|
|
with open(dest_path, "wb") as f:
|
|
for chunk in resp.iter_content(chunk_size=65536):
|
|
if chunk:
|
|
f.write(chunk)
|
|
pbar.update(len(chunk))
|
|
return True
|
|
except Exception as e:
|
|
tqdm.write(f" [ERROR] Download failed ({label}): {e}")
|
|
if dest_path.exists():
|
|
try:
|
|
dest_path.unlink()
|
|
except Exception:
|
|
pass
|
|
return False
|
|
|
|
|
|
def curl_fetch(url: str, cookies: dict = None, referer: str = None, retries: int = 5) -> str:
|
|
"""Fetch a page using curl_cffi with browser TLS impersonation. Returns HTML or empty string."""
|
|
try:
|
|
from curl_cffi import requests as curl_req
|
|
except ImportError:
|
|
return ""
|
|
|
|
headers = dict(HEADERS)
|
|
if referer:
|
|
headers["Referer"] = referer
|
|
|
|
for attempt in range(retries):
|
|
try:
|
|
resp = curl_req.get(
|
|
url, impersonate="chrome", headers=headers,
|
|
cookies=cookies, timeout=30,
|
|
)
|
|
if resp.status_code == 200:
|
|
return resp.text
|
|
if resp.status_code == 403:
|
|
tqdm.write(f" [curl_fetch] 403 Forbidden (attempt {attempt+1}/{retries}): {url}")
|
|
elif resp.status_code == 429:
|
|
wait = (2 ** attempt) * 8 + random.uniform(2.0, 6.0)
|
|
tqdm.write(f" [curl_fetch] 429 Rate limited, waiting {wait:.1f}s (attempt {attempt+1}/{retries}): {url}")
|
|
time.sleep(wait)
|
|
else:
|
|
tqdm.write(f" [curl_fetch] HTTP {resp.status_code} (attempt {attempt+1}/{retries}): {url}")
|
|
except Exception as e:
|
|
tqdm.write(f" [curl_fetch] Error (attempt {attempt+1}/{retries}): {e}")
|
|
|
|
if attempt < retries - 1:
|
|
time.sleep(2 * (attempt + 1) + random.uniform(0.5, 2.0))
|
|
|
|
return ""
|
|
|
|
|
|
def extract_video_info_from_html(html: str) -> tuple:
|
|
"""Extract (video_src, uploader) from raw video page HTML."""
|
|
src = ""
|
|
uploader = ""
|
|
|
|
source_match = re.search(r'<source\s+src="([^"]+)"', html)
|
|
if source_match:
|
|
src = source_match.group(1)
|
|
|
|
if not src:
|
|
video_match = re.search(r'<video[^>]+src="([^"]+)"', html)
|
|
if video_match:
|
|
src = video_match.group(1)
|
|
|
|
if not src:
|
|
ld_match = re.search(r'"contentUrl"\s*:\s*"([^"]+)"', html)
|
|
if ld_match:
|
|
src = ld_match.group(1)
|
|
|
|
# Strip comments section to prevent extracting commenter profiles as the uploader
|
|
main_html = html
|
|
for marker in ['class="ul-comments"', 'class="videobas"', 'id="ul-comments"', 'class="comments"']:
|
|
if marker in main_html:
|
|
main_html = main_html.split(marker, 1)[0]
|
|
break
|
|
|
|
uploader_match = re.search(r'Uploader:\s*([^<\s]+)', main_html)
|
|
if uploader_match:
|
|
uploader = uploader_match.group(1)
|
|
|
|
if not uploader:
|
|
profile_match = re.search(r'<a[^>]*href="[^"]*profiles/[^"]*"[^>]*>\s*<img[^>]*alt="([^"]+)"', main_html)
|
|
if profile_match:
|
|
uploader = profile_match.group(1)
|
|
|
|
return src.strip(), clean_filename(uploader) if uploader else "unknown"
|
|
|
|
|
|
def extract_video_links_from_html(html: str) -> list:
|
|
"""Extract unique video page URLs from a playlist page HTML."""
|
|
links = re.findall(r'href="(https?://[^"]*?/videos/[^"]*?\.html)"', html)
|
|
seen = set()
|
|
unique = []
|
|
for link in links:
|
|
if link not in seen:
|
|
seen.add(link)
|
|
unique.append(link)
|
|
return unique
|
|
|
|
|
|
def has_next_page(html: str) -> bool:
|
|
"""Check if a playlist page has a 'Next' pagination link."""
|
|
return bool(re.search(r'<a[^>]*>[^<]*[Nn]ext[^<]*</a>', html))
|
|
|
|
|
|
class PlaywrightScraper:
|
|
"""Manages browser instance and page operations headlessly."""
|
|
def __init__(self):
|
|
self.playwright = None
|
|
self.browser = None
|
|
self.context = None
|
|
|
|
async def start(self, headless=True):
|
|
from playwright.async_api import async_playwright
|
|
self.playwright = await async_playwright().start()
|
|
|
|
launch_args = [
|
|
"--disable-blink-features=AutomationControlled",
|
|
"--no-sandbox",
|
|
"--disable-dev-shm-usage",
|
|
]
|
|
if headless:
|
|
launch_args.append("--headless=new")
|
|
|
|
self.browser = await self.playwright.chromium.launch(
|
|
headless=headless,
|
|
channel="chrome",
|
|
args=launch_args
|
|
)
|
|
self.context = await self.browser.new_context(
|
|
user_agent=USER_AGENT,
|
|
viewport={"width": 1920, "height": 1080},
|
|
)
|
|
|
|
async def new_page(self):
|
|
page = await self.context.new_page()
|
|
await page.add_init_script(STEALTH_JS)
|
|
return page
|
|
|
|
async def resolve_page(self, page, url: str, retries=3):
|
|
for attempt in range(retries):
|
|
try:
|
|
await page.goto(url, wait_until="domcontentloaded", timeout=45000)
|
|
await page.wait_for_timeout(4000)
|
|
|
|
# Automatically handle age-gate overlays if present
|
|
try:
|
|
overlay = await page.query_selector(".age-verification-overlay")
|
|
if overlay:
|
|
confirm_btn = await overlay.query_selector("button.btn-confirm, button:has-text('Enter'), button:has-text('Yes'), button:has-text('Agree'), button:has-text('18')")
|
|
if confirm_btn:
|
|
await confirm_btn.click()
|
|
# Wait for the overlay to disappear
|
|
await page.wait_for_selector(".age-verification-overlay", state="hidden", timeout=5000)
|
|
except Exception:
|
|
pass
|
|
|
|
title = await page.title()
|
|
if "Just a moment" not in title:
|
|
return
|
|
tqdm.write(" Cloudflare challenge detected, waiting ...")
|
|
|
|
# Check for Cloudflare Turnstile challenge container
|
|
for loop in range(12): # Wait up to 60 seconds (12 * 5s) per attempt
|
|
await page.wait_for_timeout(5000)
|
|
title = await page.title()
|
|
if "Just a moment" not in title:
|
|
return
|
|
|
|
# Try to locate the challenge iframe and click near the center of the checkbox
|
|
try:
|
|
iframe = page.frame_locator('iframe[src*="challenges.cloudflare.com"]')
|
|
# Look for checkbox inside iframe if it's there
|
|
checkbox = iframe.locator('input[type="checkbox"], #challenge-stage, .ctp-checkbox-label')
|
|
if await checkbox.count() > 0:
|
|
# Try to click/focus to trigger validation
|
|
await checkbox.first.click(timeout=1000)
|
|
tqdm.write(" Attempted interaction with Turnstile checkbox...")
|
|
except Exception:
|
|
pass
|
|
except Exception as e:
|
|
tqdm.write(f" Attempt {attempt + 1}/{retries} failed for {url}: {e}")
|
|
if attempt < retries - 1:
|
|
await page.wait_for_timeout(5000)
|
|
raise RuntimeError(f"Could not resolve {url} after {retries} attempts")
|
|
|
|
async def extract_media_source(self, page, url: str, video_selector="video"):
|
|
"""Navigates to a page, intercepts media requests, and extracts the direct video source."""
|
|
media_urls = []
|
|
|
|
# Intercept network requests for media files
|
|
def handle_response(response):
|
|
req = response.request
|
|
content_type = response.headers.get("content-type", "").lower()
|
|
url_lower = response.url.lower()
|
|
|
|
# Skip static assets containing '.ts' or other extensions
|
|
if any(x in content_type for x in ["text/css", "javascript", "image/", "font/"]):
|
|
return
|
|
if any(url_lower.endswith(ext) or f"{ext}?" in url_lower for ext in [".css", ".js", ".png", ".jpg", ".jpeg", ".gif", ".woff", ".svg", ".ico"]):
|
|
return
|
|
|
|
is_media = (
|
|
req.resource_type in ["media", "video"] or
|
|
"video/" in content_type or
|
|
"application/x-mpegurl" in content_type or
|
|
any(url_lower.endswith(ext) or f"{ext}?" in url_lower for ext in [".mp4", ".m3u8", ".webm", ".ts"])
|
|
)
|
|
if is_media and response.url not in media_urls:
|
|
media_urls.append(response.url)
|
|
|
|
page.on("response", handle_response)
|
|
|
|
try:
|
|
await self.resolve_page(page, url)
|
|
|
|
# Try to trigger playback if the video player exists to force network requests
|
|
try:
|
|
play_buttons = [
|
|
"button.vjs-big-play-button",
|
|
".play-button",
|
|
".video-player",
|
|
"video",
|
|
".fp-play"
|
|
]
|
|
for selector in play_buttons:
|
|
el = await page.query_selector(selector)
|
|
if el:
|
|
await el.click(timeout=1000)
|
|
await page.wait_for_timeout(2000)
|
|
break
|
|
except Exception:
|
|
pass
|
|
|
|
# Allow some time for media requests to fire
|
|
await page.wait_for_timeout(3000)
|
|
|
|
if media_urls:
|
|
# Return the largest/best URL found (prefer .mp4 or .m3u8 over small chunks)
|
|
non_ts = [u for u in media_urls if not u.endswith(".ts")]
|
|
if non_ts:
|
|
return non_ts[0]
|
|
return media_urls[0]
|
|
|
|
# Fallback to DOM inspection
|
|
src = await page.evaluate(f"""(sel) => {{
|
|
// 1. Check video tag
|
|
const v = document.querySelector(sel);
|
|
if (v && v.currentSrc) return v.currentSrc;
|
|
if (v && v.src) return v.src;
|
|
|
|
// 2. Check source tags
|
|
const sources = document.querySelectorAll('video source');
|
|
for (const s of sources) {{
|
|
if (s.src) return s.src;
|
|
}}
|
|
|
|
// 3. JSON-LD fallback
|
|
const scripts = document.querySelectorAll('script[type="application/ld+json"]');
|
|
for (const s of scripts) {{
|
|
try {{
|
|
const data = JSON.parse(s.innerText);
|
|
if (data.contentUrl) return data.contentUrl;
|
|
if (data.embedUrl) return data.embedUrl;
|
|
}} catch (e) {{}}
|
|
}}
|
|
return '';
|
|
}}""", video_selector)
|
|
|
|
return src.strip() if src else ""
|
|
finally:
|
|
page.remove_listener("response", handle_response)
|
|
|
|
async def close(self):
|
|
if self.context:
|
|
await self.context.close()
|
|
if self.browser:
|
|
await self.browser.close()
|
|
if self.playwright:
|
|
await self.playwright.stop()
|
|
|
|
|
|
def append_log(log_path: Path, line: str):
|
|
log_path.parent.mkdir(parents=True, exist_ok=True)
|
|
with open(log_path, "a", encoding="utf-8") as f:
|
|
f.write(line + "\n")
|
|
|
|
|
|
def is_already_downloaded(dest_path: Path) -> bool:
|
|
return dest_path.exists() and dest_path.stat().st_size > 0
|