Automated daily backup: 2026-10-11 03:13

This commit is contained in:
gooner committed 2026-10-11 03:14:54 -04:00
1 parent 87811e0a16
commit 824cdedd7e
25 files changed
+11087 -491

No files matched your search

+457 -453
View File
@@ -1,453 +1,457 @@
import os
import re
import sys
import time
import asyncio
import logging
import threading
import random
from pathlib import Path
from urllib.parse import urlparse
import requests
from tqdm import tqdm
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [%(levelname)s] %(message)s",
datefmt="%H:%M:%S",
)
log = logging.getLogger(__name__)
USER_AGENT = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/125.0.0.0 Safari/537.36"
)
HEADERS = {
"User-Agent": USER_AGENT,
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
"Accept-Language": "en-US,en;q=0.5",
"Connection": "keep-alive",
}
STEALTH_JS = """() => {
Object.defineProperty(navigator, 'webdriver', { get: () => false });
Object.defineProperty(navigator, 'plugins', { get: () => [1, 2, 3] });
Object.defineProperty(navigator, 'languages', { get: () => ['en-US', 'en'] });
}"""
class BarPositionPool:
"""Manages console line offsets for multiple concurrent tqdm progress bars."""
def __init__(self, size):
self.lock = threading.Lock()
self.available = list(range(1, size + 1))
def acquire(self):
with self.lock:
if self.available:
return self.available.pop(0)
return None
def release(self, pos):
with self.lock:
if pos is not None and pos not in self.available:
self.available.append(pos)
self.available.sort()
def clean_filename(name: str) -> str:
return re.sub(r'[^a-zA-Z0-9_.-]', '_', name)
def download_file(url: str, dest_path: Path, headers: dict = None, bar_position: int = 1, referer: str = None, cookies: dict = None) -> bool:
"""Download a file with requests showing a progress bar at a specific terminal line."""
dest_path.parent.mkdir(parents=True, exist_ok=True)
label = f"{dest_path.parent.name}/{dest_path.name}"
req_headers = dict(HEADERS)
if headers:
req_headers.update(headers)
if referer:
req_headers["Referer"] = referer
try:
if ".m3u8" in url or "manifest" in url:
import yt_dlp
pbar = tqdm(
desc=label[:25],
unit="B",
unit_scale=True,
unit_divisor=1024,
position=bar_position,
leave=False,
ncols=80,
)
class YtdlpTqdmHook:
def __init__(self, pbar):
self.pbar = pbar
self.last_bytes = 0
def __call__(self, d):
if d['status'] == 'downloading':
downloaded = d.get('downloaded_bytes', 0)
total = d.get('total_bytes') or d.get('total_bytes_estimate') or 0
if total and self.pbar.total is None:
self.pbar.total = total
delta = downloaded - self.last_bytes
if delta > 0:
self.pbar.update(delta)
self.last_bytes = downloaded
elif d['status'] == 'finished':
if self.pbar.total and self.last_bytes < self.pbar.total:
self.pbar.update(self.pbar.total - self.last_bytes)
ydl_headers = dict(req_headers)
if ydl_headers.get("User-Agent") == "stash-scraper/v1":
ydl_headers.pop("User-Agent", None)
ydl_opts = {
'outtmpl': str(dest_path),
'progress_hooks': [YtdlpTqdmHook(pbar)],
'quiet': True,
'noprogress': True,
'http_headers': ydl_headers,
}
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
result = ydl.download([url])
if result != 0:
raise Exception("yt-dlp download failed with non-zero exit code")
pbar.close()
return True
if cookies:
try:
from curl_cffi import requests as curl_req
resp = curl_req.get(
url, impersonate="chrome", headers=req_headers,
cookies=cookies, stream=True, timeout=(15, 300),
)
resp.raise_for_status()
total = int(resp.headers.get("content-length", 0))
with tqdm(
total=total, unit="B", unit_scale=True, unit_divisor=1024,
desc=label[:25], position=bar_position, leave=False, ncols=80,
) as pbar:
with open(dest_path, "wb") as f:
for chunk in resp.iter_content(chunk_size=65536):
if chunk:
f.write(chunk)
pbar.update(len(chunk))
return True
except ImportError:
pass
resp = requests.get(url, headers=req_headers, stream=True, timeout=(15, 300), cookies=cookies)
resp.raise_for_status()
if resp.raw._connection:
sock = resp.raw._connection.sock
if sock:
sock.settimeout(30.0)
total = int(resp.headers.get("content-length", 0))
with tqdm(
total=total, unit="B", unit_scale=True, unit_divisor=1024,
desc=label[:25], position=bar_position, leave=False, ncols=80,
) as pbar:
with open(dest_path, "wb") as f:
for chunk in resp.iter_content(chunk_size=65536):
if chunk:
f.write(chunk)
pbar.update(len(chunk))
return True
except Exception as e:
tqdm.write(f" [ERROR] Download failed ({label}): {e}")
if dest_path.exists():
try:
dest_path.unlink()
except Exception:
pass
return False
def curl_fetch(url: str, cookies: dict = None, referer: str = None, retries: int = 5) -> str:
"""Fetch a page using curl_cffi with browser TLS impersonation. Returns HTML or empty string."""
try:
from curl_cffi import requests as curl_req
except ImportError:
return ""
headers = dict(HEADERS)
if referer:
headers["Referer"] = referer
for attempt in range(retries):
try:
resp = curl_req.get(
url, impersonate="chrome", headers=headers,
cookies=cookies, timeout=30,
)
if resp.status_code == 200:
return resp.text
if resp.status_code == 403:
tqdm.write(f" [curl_fetch] 403 Forbidden (attempt {attempt+1}/{retries}): {url}")
elif resp.status_code == 429:
wait = (2 ** attempt) * 8 + random.uniform(2.0, 6.0)
tqdm.write(f" [curl_fetch] 429 Rate limited, waiting {wait:.1f}s (attempt {attempt+1}/{retries}): {url}")
time.sleep(wait)
else:
tqdm.write(f" [curl_fetch] HTTP {resp.status_code} (attempt {attempt+1}/{retries}): {url}")
except Exception as e:
tqdm.write(f" [curl_fetch] Error (attempt {attempt+1}/{retries}): {e}")
if attempt < retries - 1:
time.sleep(2 * (attempt + 1) + random.uniform(0.5, 2.0))
return ""
def extract_video_info_from_html(html: str) -> tuple:
"""Extract (video_src, uploader) from raw video page HTML."""
src = ""
uploader = ""
source_match = re.search(r'<source\s+src="([^"]+)"', html)
if source_match:
src = source_match.group(1)
if not src:
video_match = re.search(r'<video[^>]+src="([^"]+)"', html)
if video_match:
src = video_match.group(1)
if not src:
ld_match = re.search(r'"contentUrl"\s*:\s*"([^"]+)"', html)
if ld_match:
src = ld_match.group(1)
# Strip comments section to prevent extracting commenter profiles as the uploader
main_html = html
for marker in ['class="ul-comments"', 'class="videobas"', 'id="ul-comments"', 'class="comments"']:
if marker in main_html:
main_html = main_html.split(marker, 1)[0]
break
uploader_match = re.search(r'Uploader:\s*([^<\s]+)', main_html)
if uploader_match:
uploader = uploader_match.group(1)
if not uploader:
profile_match = re.search(r'<a[^>]*href="[^"]*profiles/[^"]*"[^>]*>\s*<img[^>]*alt="([^"]+)"', main_html)
if profile_match:
uploader = profile_match.group(1)
return src.strip(), clean_filename(uploader) if uploader else "unknown"
def extract_video_links_from_html(html: str) -> list:
"""Extract unique video page URLs from a playlist page HTML."""
links = re.findall(r'href="(https?://[^"]*?/videos/[^"]*?\.html)"', html)
seen = set()
unique = []
for link in links:
if link not in seen:
seen.add(link)
unique.append(link)
return unique
def has_next_page(html: str) -> bool:
"""Check if a playlist page has a 'Next' pagination link."""
return bool(re.search(r'<a[^>]*>[^<]*[Nn]ext[^<]*</a>', html))
class PlaywrightScraper:
"""Manages browser instance and page operations headlessly."""
def __init__(self):
self.playwright = None
self.browser = None
self.context = None
async def start(self, headless=True):
from playwright.async_api import async_playwright
self.playwright = await async_playwright().start()
launch_args = [
"--disable-blink-features=AutomationControlled",
"--no-sandbox",
"--disable-dev-shm-usage",
]
if headless:
launch_args.append("--headless=new")
self.browser = await self.playwright.chromium.launch(
headless=headless,
channel="chrome",
args=launch_args
)
self.context = await self.browser.new_context(
user_agent=USER_AGENT,
viewport={"width": 1920, "height": 1080},
)
async def new_page(self):
page = await self.context.new_page()
await page.add_init_script(STEALTH_JS)
return page
async def resolve_page(self, page, url: str, retries=3):
for attempt in range(retries):
try:
await page.goto(url, wait_until="domcontentloaded", timeout=45000)
await page.wait_for_timeout(4000)
# Automatically handle age-gate overlays if present
try:
overlay = await page.query_selector(".age-verification-overlay")
if overlay:
confirm_btn = await overlay.query_selector("button.btn-confirm, button:has-text('Enter'), button:has-text('Yes'), button:has-text('Agree'), button:has-text('18')")
if confirm_btn:
await confirm_btn.click()
# Wait for the overlay to disappear
await page.wait_for_selector(".age-verification-overlay", state="hidden", timeout=5000)
except Exception:
pass
title = await page.title()
if "Just a moment" not in title:
return
tqdm.write(" Cloudflare challenge detected, waiting ...")
# Check for Cloudflare Turnstile challenge container
for loop in range(12): # Wait up to 60 seconds (12 * 5s) per attempt
await page.wait_for_timeout(5000)
title = await page.title()
if "Just a moment" not in title:
return
# Try to locate the challenge iframe and click near the center of the checkbox
try:
iframe = page.frame_locator('iframe[src*="challenges.cloudflare.com"]')
# Look for checkbox inside iframe if it's there
checkbox = iframe.locator('input[type="checkbox"], #challenge-stage, .ctp-checkbox-label')
if await checkbox.count() > 0:
# Try to click/focus to trigger validation
await checkbox.first.click(timeout=1000)
tqdm.write(" Attempted interaction with Turnstile checkbox...")
except Exception:
pass
except Exception as e:
tqdm.write(f" Attempt {attempt + 1}/{retries} failed for {url}: {e}")
if attempt < retries - 1:
await page.wait_for_timeout(5000)
raise RuntimeError(f"Could not resolve {url} after {retries} attempts")
async def extract_media_source(self, page, url: str, video_selector="video"):
"""Navigates to a page, intercepts media requests, and extracts the direct video source."""
media_urls = []
# Intercept network requests for media files
def handle_response(response):
req = response.request
content_type = response.headers.get("content-type", "").lower()
url_lower = response.url.lower()
# Skip static assets containing '.ts' or other extensions
if any(x in content_type for x in ["text/css", "javascript", "image/", "font/"]):
return
if any(url_lower.endswith(ext) or f"{ext}?" in url_lower for ext in [".css", ".js", ".png", ".jpg", ".jpeg", ".gif", ".woff", ".svg", ".ico"]):
return
is_media = (
req.resource_type in ["media", "video"] or
"video/" in content_type or
"application/x-mpegurl" in content_type or
any(url_lower.endswith(ext) or f"{ext}?" in url_lower for ext in [".mp4", ".m3u8", ".webm", ".ts"])
)
if is_media and response.url not in media_urls:
media_urls.append(response.url)
page.on("response", handle_response)
try:
await self.resolve_page(page, url)
# Try to trigger playback if the video player exists to force network requests
try:
play_buttons = [
"button.vjs-big-play-button",
".play-button",
".video-player",
"video",
".fp-play"
]
for selector in play_buttons:
el = await page.query_selector(selector)
if el:
await el.click(timeout=1000)
await page.wait_for_timeout(2000)
break
except Exception:
pass
# Allow some time for media requests to fire
await page.wait_for_timeout(3000)
if media_urls:
# Return the largest/best URL found (prefer .mp4 or .m3u8 over small chunks)
non_ts = [u for u in media_urls if not u.endswith(".ts")]
if non_ts:
return non_ts[0]
return media_urls[0]
# Fallback to DOM inspection
src = await page.evaluate(f"""(sel) => {{
// 1. Check video tag
const v = document.querySelector(sel);
if (v && v.currentSrc) return v.currentSrc;
if (v && v.src) return v.src;
// 2. Check source tags
const sources = document.querySelectorAll('video source');
for (const s of sources) {{
if (s.src) return s.src;
}}
// 3. JSON-LD fallback
const scripts = document.querySelectorAll('script[type="application/ld+json"]');
for (const s of scripts) {{
try {{
const data = JSON.parse(s.innerText);
if (data.contentUrl) return data.contentUrl;
if (data.embedUrl) return data.embedUrl;
}} catch (e) {{}}
}}
return '';
}}""", video_selector)
return src.strip() if src else ""
finally:
page.remove_listener("response", handle_response)
async def close(self):
if self.context:
await self.context.close()
if self.browser:
await self.browser.close()
if self.playwright:
await self.playwright.stop()
def append_log(log_path: Path, line: str):
log_path.parent.mkdir(parents=True, exist_ok=True)
with open(log_path, "a", encoding="utf-8") as f:
f.write(line + "\n")
def is_already_downloaded(dest_path: Path) -> bool:
return dest_path.exists() and dest_path.stat().st_size > 0
import os
import re
import sys
import time
import asyncio
import logging
import threading
import random
from pathlib import Path
from urllib.parse import urlparse
import requests
from tqdm import tqdm
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [%(levelname)s] %(message)s",
datefmt="%H:%M:%S",
)
log = logging.getLogger(__name__)
USER_AGENT = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/125.0.0.0 Safari/537.36"
)
HEADERS = {
"User-Agent": USER_AGENT,
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
"Accept-Language": "en-US,en;q=0.5",
"Connection": "keep-alive",
}
STEALTH_JS = """() => {
Object.defineProperty(navigator, 'webdriver', { get: () => false });
Object.defineProperty(navigator, 'plugins', { get: () => [1, 2, 3] });
Object.defineProperty(navigator, 'languages', { get: () => ['en-US', 'en'] });
}"""
class BarPositionPool:
"""Manages console line offsets for multiple concurrent tqdm progress bars."""
def __init__(self, size):
self.lock = threading.Lock()
self.available = list(range(1, size + 1))
def acquire(self):
with self.lock:
if self.available:
return self.available.pop(0)
return None
def release(self, pos):
with self.lock:
if pos is not None and pos not in self.available:
self.available.append(pos)
self.available.sort()
def clean_filename(name: str) -> str:
return re.sub(r'[^a-zA-Z0-9_.-]', '_', name)
def download_file(url: str, dest_path: Path, headers: dict = None, bar_position: int = 1, referer: str = None, cookies: dict = None) -> bool:
"""Download a file with requests showing a progress bar at a specific terminal line."""
dest_path.parent.mkdir(parents=True, exist_ok=True)
label = f"{dest_path.parent.name}/{dest_path.name}"
req_headers = dict(HEADERS)
if headers:
req_headers.update(headers)
if referer:
req_headers["Referer"] = referer
try:
if ".m3u8" in url or "manifest" in url:
import yt_dlp
pbar = tqdm(
desc=label[:25],
unit="B",
unit_scale=True,
unit_divisor=1024,
position=bar_position,
leave=False,
ncols=80,
)
class YtdlpTqdmHook:
def __init__(self, pbar):
self.pbar = pbar
self.last_bytes = 0
def __call__(self, d):
if d['status'] == 'downloading':
downloaded = d.get('downloaded_bytes', 0)
total = d.get('total_bytes') or d.get('total_bytes_estimate') or 0
if total and self.pbar.total is None:
self.pbar.total = total
delta = downloaded - self.last_bytes
if delta > 0:
self.pbar.update(delta)
self.last_bytes = downloaded
elif d['status'] == 'finished':
if self.pbar.total and self.last_bytes < self.pbar.total:
self.pbar.update(self.pbar.total - self.last_bytes)
ydl_headers = dict(req_headers)
if ydl_headers.get("User-Agent") == "stash-scraper/v1":
ydl_headers.pop("User-Agent", None)
ydl_opts = {
'outtmpl': str(dest_path),
'progress_hooks': [YtdlpTqdmHook(pbar)],
'quiet': True,
'noprogress': True,
'http_headers': ydl_headers,
'downloader': 'ffmpeg',
'hls_use_mpegts': True,
}
try:
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
result = ydl.download([url])
if result != 0:
raise Exception("yt-dlp download failed with non-zero exit code")
finally:
pbar.close()
return True
if cookies:
try:
from curl_cffi import requests as curl_req
resp = curl_req.get(
url, impersonate="chrome", headers=req_headers,
cookies=cookies, stream=True, timeout=(15, 300),
)
resp.raise_for_status()
total = int(resp.headers.get("content-length", 0))
with tqdm(
total=total, unit="B", unit_scale=True, unit_divisor=1024,
desc=label[:25], position=bar_position, leave=False, ncols=80,
) as pbar:
with open(dest_path, "wb") as f:
for chunk in resp.iter_content(chunk_size=65536):
if chunk:
f.write(chunk)
pbar.update(len(chunk))
return True
except ImportError:
pass
resp = requests.get(url, headers=req_headers, stream=True, timeout=(15, 300), cookies=cookies)
resp.raise_for_status()
if resp.raw._connection:
sock = resp.raw._connection.sock
if sock:
sock.settimeout(30.0)
total = int(resp.headers.get("content-length", 0))
with tqdm(
total=total, unit="B", unit_scale=True, unit_divisor=1024,
desc=label[:25], position=bar_position, leave=False, ncols=80,
) as pbar:
with open(dest_path, "wb") as f:
for chunk in resp.iter_content(chunk_size=65536):
if chunk:
f.write(chunk)
pbar.update(len(chunk))
return True
except Exception as e:
tqdm.write(f" [ERROR] Download failed ({label}): {e}")
if dest_path.exists():
try:
dest_path.unlink()
except Exception:
pass
return False
def curl_fetch(url: str, cookies: dict = None, referer: str = None, retries: int = 5) -> str:
"""Fetch a page using curl_cffi with browser TLS impersonation. Returns HTML or empty string."""
try:
from curl_cffi import requests as curl_req
except ImportError:
return ""
headers = dict(HEADERS)
if referer:
headers["Referer"] = referer
for attempt in range(retries):
try:
resp = curl_req.get(
url, impersonate="chrome", headers=headers,
cookies=cookies, timeout=30,
)
if resp.status_code == 200:
return resp.text
if resp.status_code == 403:
tqdm.write(f" [curl_fetch] 403 Forbidden (attempt {attempt+1}/{retries}): {url}")
elif resp.status_code == 429:
wait = (2 ** attempt) * 8 + random.uniform(2.0, 6.0)
tqdm.write(f" [curl_fetch] 429 Rate limited, waiting {wait:.1f}s (attempt {attempt+1}/{retries}): {url}")
time.sleep(wait)
else:
tqdm.write(f" [curl_fetch] HTTP {resp.status_code} (attempt {attempt+1}/{retries}): {url}")
except Exception as e:
tqdm.write(f" [curl_fetch] Error (attempt {attempt+1}/{retries}): {e}")
if attempt < retries - 1:
time.sleep(2 * (attempt + 1) + random.uniform(0.5, 2.0))
return ""
def extract_video_info_from_html(html: str) -> tuple:
"""Extract (video_src, uploader) from raw video page HTML."""
src = ""
uploader = ""
source_match = re.search(r'<source\s+src="([^"]+)"', html)
if source_match:
src = source_match.group(1)
if not src:
video_match = re.search(r'<video[^>]+src="([^"]+)"', html)
if video_match:
src = video_match.group(1)
if not src:
ld_match = re.search(r'"contentUrl"\s*:\s*"([^"]+)"', html)
if ld_match:
src = ld_match.group(1)
# Strip comments section to prevent extracting commenter profiles as the uploader
main_html = html
for marker in ['class="ul-comments"', 'class="videobas"', 'id="ul-comments"', 'class="comments"']:
if marker in main_html:
main_html = main_html.split(marker, 1)[0]
break
uploader_match = re.search(r'Uploader:\s*([^<\s]+)', main_html)
if uploader_match:
uploader = uploader_match.group(1)
if not uploader:
profile_match = re.search(r'<a[^>]*href="[^"]*profiles/[^"]*"[^>]*>\s*<img[^>]*alt="([^"]+)"', main_html)
if profile_match:
uploader = profile_match.group(1)
return src.strip(), clean_filename(uploader) if uploader else "unknown"
def extract_video_links_from_html(html: str) -> list:
"""Extract unique video page URLs from a playlist page HTML."""
links = re.findall(r'href="(https?://[^"]*?/videos/[^"]*?\.html)"', html)
seen = set()
unique = []
for link in links:
if link not in seen:
seen.add(link)
unique.append(link)
return unique
def has_next_page(html: str) -> bool:
"""Check if a playlist page has a 'Next' pagination link."""
return bool(re.search(r'<a[^>]*>[^<]*[Nn]ext[^<]*</a>', html))
class PlaywrightScraper:
"""Manages browser instance and page operations headlessly."""
def __init__(self):
self.playwright = None
self.browser = None
self.context = None
async def start(self, headless=True):
from playwright.async_api import async_playwright
self.playwright = await async_playwright().start()
launch_args = [
"--disable-blink-features=AutomationControlled",
"--no-sandbox",
"--disable-dev-shm-usage",
]
if headless:
launch_args.append("--headless=new")
self.browser = await self.playwright.chromium.launch(
headless=headless,
channel="chrome",
args=launch_args
)
self.context = await self.browser.new_context(
user_agent=USER_AGENT,
viewport={"width": 1920, "height": 1080},
)
async def new_page(self):
page = await self.context.new_page()
await page.add_init_script(STEALTH_JS)
return page
async def resolve_page(self, page, url: str, retries=3):
for attempt in range(retries):
try:
await page.goto(url, wait_until="domcontentloaded", timeout=45000)
await page.wait_for_timeout(4000)
# Automatically handle age-gate overlays if present
try:
overlay = await page.query_selector(".age-verification-overlay")
if overlay:
confirm_btn = await overlay.query_selector("button.btn-confirm, button:has-text('Enter'), button:has-text('Yes'), button:has-text('Agree'), button:has-text('18')")
if confirm_btn:
await confirm_btn.click()
# Wait for the overlay to disappear
await page.wait_for_selector(".age-verification-overlay", state="hidden", timeout=5000)
except Exception:
pass
title = await page.title()
if "Just a moment" not in title:
return
tqdm.write(" Cloudflare challenge detected, waiting ...")
# Check for Cloudflare Turnstile challenge container
for loop in range(12): # Wait up to 60 seconds (12 * 5s) per attempt
await page.wait_for_timeout(5000)
title = await page.title()
if "Just a moment" not in title:
return
# Try to locate the challenge iframe and click near the center of the checkbox
try:
iframe = page.frame_locator('iframe[src*="challenges.cloudflare.com"]')
# Look for checkbox inside iframe if it's there
checkbox = iframe.locator('input[type="checkbox"], #challenge-stage, .ctp-checkbox-label')
if await checkbox.count() > 0:
# Try to click/focus to trigger validation
await checkbox.first.click(timeout=1000)
tqdm.write(" Attempted interaction with Turnstile checkbox...")
except Exception:
pass
except Exception as e:
tqdm.write(f" Attempt {attempt + 1}/{retries} failed for {url}: {e}")
if attempt < retries - 1:
await page.wait_for_timeout(5000)
raise RuntimeError(f"Could not resolve {url} after {retries} attempts")
async def extract_media_source(self, page, url: str, video_selector="video"):
"""Navigates to a page, intercepts media requests, and extracts the direct video source."""
media_urls = []
# Intercept network requests for media files
def handle_response(response):
req = response.request
content_type = response.headers.get("content-type", "").lower()
url_lower = response.url.lower()
# Skip static assets containing '.ts' or other extensions
if any(x in content_type for x in ["text/css", "javascript", "image/", "font/"]):
return
if any(url_lower.endswith(ext) or f"{ext}?" in url_lower for ext in [".css", ".js", ".png", ".jpg", ".jpeg", ".gif", ".woff", ".svg", ".ico"]):
return
is_media = (
req.resource_type in ["media", "video"] or
"video/" in content_type or
"application/x-mpegurl" in content_type or
any(url_lower.endswith(ext) or f"{ext}?" in url_lower for ext in [".mp4", ".m3u8", ".webm", ".ts"])
)
if is_media and response.url not in media_urls:
media_urls.append(response.url)
page.on("response", handle_response)
try:
await self.resolve_page(page, url)
# Try to trigger playback if the video player exists to force network requests
try:
play_buttons = [
"button.vjs-big-play-button",
".play-button",
".video-player",
"video",
".fp-play"
]
for selector in play_buttons:
el = await page.query_selector(selector)
if el:
await el.click(timeout=1000)
await page.wait_for_timeout(2000)
break
except Exception:
pass
# Allow some time for media requests to fire
await page.wait_for_timeout(3000)
if media_urls:
# Return the largest/best URL found (prefer .mp4 or .m3u8 over small chunks)
non_ts = [u for u in media_urls if not u.endswith(".ts")]
if non_ts:
return non_ts[0]
return media_urls[0]
# Fallback to DOM inspection
src = await page.evaluate(f"""(sel) => {{
// 1. Check video tag
const v = document.querySelector(sel);
if (v && v.currentSrc) return v.currentSrc;
if (v && v.src) return v.src;
// 2. Check source tags
const sources = document.querySelectorAll('video source');
for (const s of sources) {{
if (s.src) return s.src;
}}
// 3. JSON-LD fallback
const scripts = document.querySelectorAll('script[type="application/ld+json"]');
for (const s of scripts) {{
try {{
const data = JSON.parse(s.innerText);
if (data.contentUrl) return data.contentUrl;
if (data.embedUrl) return data.embedUrl;
}} catch (e) {{}}
}}
return '';
}}""", video_selector)
return src.strip() if src else ""
finally:
page.remove_listener("response", handle_response)
async def close(self):
if self.context:
await self.context.close()
if self.browser:
await self.browser.close()
if self.playwright:
await self.playwright.stop()
def append_log(log_path: Path, line: str):
log_path.parent.mkdir(parents=True, exist_ok=True)
with open(log_path, "a", encoding="utf-8") as f:
f.write(line + "\n")
def is_already_downloaded(dest_path: Path) -> bool:
return dest_path.exists() and dest_path.stat().st_size > 0