Initial backup: folder structure manifest, tree summary, scraper scripts, and text metadata
This commit is contained in:
commit
f164832dcc
10504 files changed
+2423594
No files matched your search
+453
@@ -0,0 +1,453 @@
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
import asyncio
|
||||
import logging
|
||||
import threading
|
||||
import random
|
||||
from pathlib import Path
|
||||
from urllib.parse import urlparse
|
||||
import requests
|
||||
from tqdm import tqdm
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s [%(levelname)s] %(message)s",
|
||||
datefmt="%H:%M:%S",
|
||||
)
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
USER_AGENT = (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||
"Chrome/125.0.0.0 Safari/537.36"
|
||||
)
|
||||
|
||||
HEADERS = {
|
||||
"User-Agent": USER_AGENT,
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Language": "en-US,en;q=0.5",
|
||||
"Connection": "keep-alive",
|
||||
}
|
||||
|
||||
STEALTH_JS = """() => {
|
||||
Object.defineProperty(navigator, 'webdriver', { get: () => false });
|
||||
Object.defineProperty(navigator, 'plugins', { get: () => [1, 2, 3] });
|
||||
Object.defineProperty(navigator, 'languages', { get: () => ['en-US', 'en'] });
|
||||
}"""
|
||||
|
||||
class BarPositionPool:
|
||||
"""Manages console line offsets for multiple concurrent tqdm progress bars."""
|
||||
def __init__(self, size):
|
||||
self.lock = threading.Lock()
|
||||
self.available = list(range(1, size + 1))
|
||||
|
||||
def acquire(self):
|
||||
with self.lock:
|
||||
if self.available:
|
||||
return self.available.pop(0)
|
||||
return None
|
||||
|
||||
def release(self, pos):
|
||||
with self.lock:
|
||||
if pos is not None and pos not in self.available:
|
||||
self.available.append(pos)
|
||||
self.available.sort()
|
||||
|
||||
|
||||
def clean_filename(name: str) -> str:
|
||||
return re.sub(r'[^a-zA-Z0-9_.-]', '_', name)
|
||||
|
||||
|
||||
def download_file(url: str, dest_path: Path, headers: dict = None, bar_position: int = 1, referer: str = None, cookies: dict = None) -> bool:
|
||||
"""Download a file with requests showing a progress bar at a specific terminal line."""
|
||||
dest_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
label = f"{dest_path.parent.name}/{dest_path.name}"
|
||||
|
||||
req_headers = dict(HEADERS)
|
||||
if headers:
|
||||
req_headers.update(headers)
|
||||
if referer:
|
||||
req_headers["Referer"] = referer
|
||||
|
||||
try:
|
||||
if ".m3u8" in url or "manifest" in url:
|
||||
import yt_dlp
|
||||
|
||||
pbar = tqdm(
|
||||
desc=label[:25],
|
||||
unit="B",
|
||||
unit_scale=True,
|
||||
unit_divisor=1024,
|
||||
position=bar_position,
|
||||
leave=False,
|
||||
ncols=80,
|
||||
)
|
||||
|
||||
class YtdlpTqdmHook:
|
||||
def __init__(self, pbar):
|
||||
self.pbar = pbar
|
||||
self.last_bytes = 0
|
||||
|
||||
def __call__(self, d):
|
||||
if d['status'] == 'downloading':
|
||||
downloaded = d.get('downloaded_bytes', 0)
|
||||
total = d.get('total_bytes') or d.get('total_bytes_estimate') or 0
|
||||
if total and self.pbar.total is None:
|
||||
self.pbar.total = total
|
||||
delta = downloaded - self.last_bytes
|
||||
if delta > 0:
|
||||
self.pbar.update(delta)
|
||||
self.last_bytes = downloaded
|
||||
elif d['status'] == 'finished':
|
||||
if self.pbar.total and self.last_bytes < self.pbar.total:
|
||||
self.pbar.update(self.pbar.total - self.last_bytes)
|
||||
|
||||
ydl_headers = dict(req_headers)
|
||||
if ydl_headers.get("User-Agent") == "stash-scraper/v1":
|
||||
ydl_headers.pop("User-Agent", None)
|
||||
|
||||
ydl_opts = {
|
||||
'outtmpl': str(dest_path),
|
||||
'progress_hooks': [YtdlpTqdmHook(pbar)],
|
||||
'quiet': True,
|
||||
'noprogress': True,
|
||||
'http_headers': ydl_headers,
|
||||
}
|
||||
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
||||
result = ydl.download([url])
|
||||
if result != 0:
|
||||
raise Exception("yt-dlp download failed with non-zero exit code")
|
||||
pbar.close()
|
||||
return True
|
||||
|
||||
if cookies:
|
||||
try:
|
||||
from curl_cffi import requests as curl_req
|
||||
resp = curl_req.get(
|
||||
url, impersonate="chrome", headers=req_headers,
|
||||
cookies=cookies, stream=True, timeout=(15, 300),
|
||||
)
|
||||
resp.raise_for_status()
|
||||
total = int(resp.headers.get("content-length", 0))
|
||||
|
||||
with tqdm(
|
||||
total=total, unit="B", unit_scale=True, unit_divisor=1024,
|
||||
desc=label[:25], position=bar_position, leave=False, ncols=80,
|
||||
) as pbar:
|
||||
with open(dest_path, "wb") as f:
|
||||
for chunk in resp.iter_content(chunk_size=65536):
|
||||
if chunk:
|
||||
f.write(chunk)
|
||||
pbar.update(len(chunk))
|
||||
return True
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
resp = requests.get(url, headers=req_headers, stream=True, timeout=(15, 300), cookies=cookies)
|
||||
resp.raise_for_status()
|
||||
|
||||
if resp.raw._connection:
|
||||
sock = resp.raw._connection.sock
|
||||
if sock:
|
||||
sock.settimeout(30.0)
|
||||
|
||||
total = int(resp.headers.get("content-length", 0))
|
||||
|
||||
with tqdm(
|
||||
total=total, unit="B", unit_scale=True, unit_divisor=1024,
|
||||
desc=label[:25], position=bar_position, leave=False, ncols=80,
|
||||
) as pbar:
|
||||
with open(dest_path, "wb") as f:
|
||||
for chunk in resp.iter_content(chunk_size=65536):
|
||||
if chunk:
|
||||
f.write(chunk)
|
||||
pbar.update(len(chunk))
|
||||
return True
|
||||
except Exception as e:
|
||||
tqdm.write(f" [ERROR] Download failed ({label}): {e}")
|
||||
if dest_path.exists():
|
||||
try:
|
||||
dest_path.unlink()
|
||||
except Exception:
|
||||
pass
|
||||
return False
|
||||
|
||||
|
||||
def curl_fetch(url: str, cookies: dict = None, referer: str = None, retries: int = 5) -> str:
|
||||
"""Fetch a page using curl_cffi with browser TLS impersonation. Returns HTML or empty string."""
|
||||
try:
|
||||
from curl_cffi import requests as curl_req
|
||||
except ImportError:
|
||||
return ""
|
||||
|
||||
headers = dict(HEADERS)
|
||||
if referer:
|
||||
headers["Referer"] = referer
|
||||
|
||||
for attempt in range(retries):
|
||||
try:
|
||||
resp = curl_req.get(
|
||||
url, impersonate="chrome", headers=headers,
|
||||
cookies=cookies, timeout=30,
|
||||
)
|
||||
if resp.status_code == 200:
|
||||
return resp.text
|
||||
if resp.status_code == 403:
|
||||
tqdm.write(f" [curl_fetch] 403 Forbidden (attempt {attempt+1}/{retries}): {url}")
|
||||
elif resp.status_code == 429:
|
||||
wait = (2 ** attempt) * 8 + random.uniform(2.0, 6.0)
|
||||
tqdm.write(f" [curl_fetch] 429 Rate limited, waiting {wait:.1f}s (attempt {attempt+1}/{retries}): {url}")
|
||||
time.sleep(wait)
|
||||
else:
|
||||
tqdm.write(f" [curl_fetch] HTTP {resp.status_code} (attempt {attempt+1}/{retries}): {url}")
|
||||
except Exception as e:
|
||||
tqdm.write(f" [curl_fetch] Error (attempt {attempt+1}/{retries}): {e}")
|
||||
|
||||
if attempt < retries - 1:
|
||||
time.sleep(2 * (attempt + 1) + random.uniform(0.5, 2.0))
|
||||
|
||||
return ""
|
||||
|
||||
|
||||
def extract_video_info_from_html(html: str) -> tuple:
|
||||
"""Extract (video_src, uploader) from raw video page HTML."""
|
||||
src = ""
|
||||
uploader = ""
|
||||
|
||||
source_match = re.search(r'<source\s+src="([^"]+)"', html)
|
||||
if source_match:
|
||||
src = source_match.group(1)
|
||||
|
||||
if not src:
|
||||
video_match = re.search(r'<video[^>]+src="([^"]+)"', html)
|
||||
if video_match:
|
||||
src = video_match.group(1)
|
||||
|
||||
if not src:
|
||||
ld_match = re.search(r'"contentUrl"\s*:\s*"([^"]+)"', html)
|
||||
if ld_match:
|
||||
src = ld_match.group(1)
|
||||
|
||||
# Strip comments section to prevent extracting commenter profiles as the uploader
|
||||
main_html = html
|
||||
for marker in ['class="ul-comments"', 'class="videobas"', 'id="ul-comments"', 'class="comments"']:
|
||||
if marker in main_html:
|
||||
main_html = main_html.split(marker, 1)[0]
|
||||
break
|
||||
|
||||
uploader_match = re.search(r'Uploader:\s*([^<\s]+)', main_html)
|
||||
if uploader_match:
|
||||
uploader = uploader_match.group(1)
|
||||
|
||||
if not uploader:
|
||||
profile_match = re.search(r'<a[^>]*href="[^"]*profiles/[^"]*"[^>]*>\s*<img[^>]*alt="([^"]+)"', main_html)
|
||||
if profile_match:
|
||||
uploader = profile_match.group(1)
|
||||
|
||||
return src.strip(), clean_filename(uploader) if uploader else "unknown"
|
||||
|
||||
|
||||
def extract_video_links_from_html(html: str) -> list:
|
||||
"""Extract unique video page URLs from a playlist page HTML."""
|
||||
links = re.findall(r'href="(https?://[^"]*?/videos/[^"]*?\.html)"', html)
|
||||
seen = set()
|
||||
unique = []
|
||||
for link in links:
|
||||
if link not in seen:
|
||||
seen.add(link)
|
||||
unique.append(link)
|
||||
return unique
|
||||
|
||||
|
||||
def has_next_page(html: str) -> bool:
|
||||
"""Check if a playlist page has a 'Next' pagination link."""
|
||||
return bool(re.search(r'<a[^>]*>[^<]*[Nn]ext[^<]*</a>', html))
|
||||
|
||||
|
||||
class PlaywrightScraper:
|
||||
"""Manages browser instance and page operations headlessly."""
|
||||
def __init__(self):
|
||||
self.playwright = None
|
||||
self.browser = None
|
||||
self.context = None
|
||||
|
||||
async def start(self, headless=True):
|
||||
from playwright.async_api import async_playwright
|
||||
self.playwright = await async_playwright().start()
|
||||
|
||||
launch_args = [
|
||||
"--disable-blink-features=AutomationControlled",
|
||||
"--no-sandbox",
|
||||
"--disable-dev-shm-usage",
|
||||
]
|
||||
if headless:
|
||||
launch_args.append("--headless=new")
|
||||
|
||||
self.browser = await self.playwright.chromium.launch(
|
||||
headless=headless,
|
||||
channel="chrome",
|
||||
args=launch_args
|
||||
)
|
||||
self.context = await self.browser.new_context(
|
||||
user_agent=USER_AGENT,
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
)
|
||||
|
||||
async def new_page(self):
|
||||
page = await self.context.new_page()
|
||||
await page.add_init_script(STEALTH_JS)
|
||||
return page
|
||||
|
||||
async def resolve_page(self, page, url: str, retries=3):
|
||||
for attempt in range(retries):
|
||||
try:
|
||||
await page.goto(url, wait_until="domcontentloaded", timeout=45000)
|
||||
await page.wait_for_timeout(4000)
|
||||
|
||||
# Automatically handle age-gate overlays if present
|
||||
try:
|
||||
overlay = await page.query_selector(".age-verification-overlay")
|
||||
if overlay:
|
||||
confirm_btn = await overlay.query_selector("button.btn-confirm, button:has-text('Enter'), button:has-text('Yes'), button:has-text('Agree'), button:has-text('18')")
|
||||
if confirm_btn:
|
||||
await confirm_btn.click()
|
||||
# Wait for the overlay to disappear
|
||||
await page.wait_for_selector(".age-verification-overlay", state="hidden", timeout=5000)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
title = await page.title()
|
||||
if "Just a moment" not in title:
|
||||
return
|
||||
tqdm.write(" Cloudflare challenge detected, waiting ...")
|
||||
|
||||
# Check for Cloudflare Turnstile challenge container
|
||||
for loop in range(12): # Wait up to 60 seconds (12 * 5s) per attempt
|
||||
await page.wait_for_timeout(5000)
|
||||
title = await page.title()
|
||||
if "Just a moment" not in title:
|
||||
return
|
||||
|
||||
# Try to locate the challenge iframe and click near the center of the checkbox
|
||||
try:
|
||||
iframe = page.frame_locator('iframe[src*="challenges.cloudflare.com"]')
|
||||
# Look for checkbox inside iframe if it's there
|
||||
checkbox = iframe.locator('input[type="checkbox"], #challenge-stage, .ctp-checkbox-label')
|
||||
if await checkbox.count() > 0:
|
||||
# Try to click/focus to trigger validation
|
||||
await checkbox.first.click(timeout=1000)
|
||||
tqdm.write(" Attempted interaction with Turnstile checkbox...")
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as e:
|
||||
tqdm.write(f" Attempt {attempt + 1}/{retries} failed for {url}: {e}")
|
||||
if attempt < retries - 1:
|
||||
await page.wait_for_timeout(5000)
|
||||
raise RuntimeError(f"Could not resolve {url} after {retries} attempts")
|
||||
|
||||
async def extract_media_source(self, page, url: str, video_selector="video"):
|
||||
"""Navigates to a page, intercepts media requests, and extracts the direct video source."""
|
||||
media_urls = []
|
||||
|
||||
# Intercept network requests for media files
|
||||
def handle_response(response):
|
||||
req = response.request
|
||||
content_type = response.headers.get("content-type", "").lower()
|
||||
url_lower = response.url.lower()
|
||||
|
||||
# Skip static assets containing '.ts' or other extensions
|
||||
if any(x in content_type for x in ["text/css", "javascript", "image/", "font/"]):
|
||||
return
|
||||
if any(url_lower.endswith(ext) or f"{ext}?" in url_lower for ext in [".css", ".js", ".png", ".jpg", ".jpeg", ".gif", ".woff", ".svg", ".ico"]):
|
||||
return
|
||||
|
||||
is_media = (
|
||||
req.resource_type in ["media", "video"] or
|
||||
"video/" in content_type or
|
||||
"application/x-mpegurl" in content_type or
|
||||
any(url_lower.endswith(ext) or f"{ext}?" in url_lower for ext in [".mp4", ".m3u8", ".webm", ".ts"])
|
||||
)
|
||||
if is_media and response.url not in media_urls:
|
||||
media_urls.append(response.url)
|
||||
|
||||
page.on("response", handle_response)
|
||||
|
||||
try:
|
||||
await self.resolve_page(page, url)
|
||||
|
||||
# Try to trigger playback if the video player exists to force network requests
|
||||
try:
|
||||
play_buttons = [
|
||||
"button.vjs-big-play-button",
|
||||
".play-button",
|
||||
".video-player",
|
||||
"video",
|
||||
".fp-play"
|
||||
]
|
||||
for selector in play_buttons:
|
||||
el = await page.query_selector(selector)
|
||||
if el:
|
||||
await el.click(timeout=1000)
|
||||
await page.wait_for_timeout(2000)
|
||||
break
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Allow some time for media requests to fire
|
||||
await page.wait_for_timeout(3000)
|
||||
|
||||
if media_urls:
|
||||
# Return the largest/best URL found (prefer .mp4 or .m3u8 over small chunks)
|
||||
non_ts = [u for u in media_urls if not u.endswith(".ts")]
|
||||
if non_ts:
|
||||
return non_ts[0]
|
||||
return media_urls[0]
|
||||
|
||||
# Fallback to DOM inspection
|
||||
src = await page.evaluate(f"""(sel) => {{
|
||||
// 1. Check video tag
|
||||
const v = document.querySelector(sel);
|
||||
if (v && v.currentSrc) return v.currentSrc;
|
||||
if (v && v.src) return v.src;
|
||||
|
||||
// 2. Check source tags
|
||||
const sources = document.querySelectorAll('video source');
|
||||
for (const s of sources) {{
|
||||
if (s.src) return s.src;
|
||||
}}
|
||||
|
||||
// 3. JSON-LD fallback
|
||||
const scripts = document.querySelectorAll('script[type="application/ld+json"]');
|
||||
for (const s of scripts) {{
|
||||
try {{
|
||||
const data = JSON.parse(s.innerText);
|
||||
if (data.contentUrl) return data.contentUrl;
|
||||
if (data.embedUrl) return data.embedUrl;
|
||||
}} catch (e) {{}}
|
||||
}}
|
||||
return '';
|
||||
}}""", video_selector)
|
||||
|
||||
return src.strip() if src else ""
|
||||
finally:
|
||||
page.remove_listener("response", handle_response)
|
||||
|
||||
async def close(self):
|
||||
if self.context:
|
||||
await self.context.close()
|
||||
if self.browser:
|
||||
await self.browser.close()
|
||||
if self.playwright:
|
||||
await self.playwright.stop()
|
||||
|
||||
|
||||
def append_log(log_path: Path, line: str):
|
||||
log_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(log_path, "a", encoding="utf-8") as f:
|
||||
f.write(line + "\n")
|
||||
|
||||
|
||||
def is_already_downloaded(dest_path: Path) -> bool:
|
||||
return dest_path.exists() and dest_path.stat().st_size > 0
|
||||
Reference in new issue
Block a user