Automated daily backup: 2026-10-10 03:08
This commit is contained in:
1 parent
005a717a5b
commit
87811e0a16
9 files changed
+433
-50
No files matched your search
+287
@@ -555,6 +555,12 @@ SITES = {
|
||||
"handler": "zootube1",
|
||||
"default_concurrency": 3,
|
||||
},
|
||||
"tickzoo.tv": {
|
||||
"folder": "tickzoo.tv",
|
||||
"domains": ["tickzoo.tv"],
|
||||
"handler": "tickzoo",
|
||||
"default_concurrency": 1,
|
||||
},
|
||||
"file.al": {
|
||||
"folder": "file.al",
|
||||
"domains": ["file.al"],
|
||||
@@ -1627,6 +1633,285 @@ async def zootubevip_process_urls(download_dir, urls, concurrency, skip_existing
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
# ----- tickzoo.tv ------------------------------------------------------------
|
||||
# Video pages have no <video> tag - they embed a 3rd-party player
|
||||
# (veev.to / firestream.to / hgcloud.to / rubyvidhub.com) via an "Embed Code"
|
||||
# <textarea id="iframe">, so each embed is resolved through headless Chrome.
|
||||
|
||||
_TICKZOO_AD_HOSTS = (
|
||||
"dtscout", "lijit", "doubleclick", "googlesyndication", "adservice",
|
||||
"pubmatic", "criteo", "taboola", "outbrain", "adsystem",
|
||||
)
|
||||
|
||||
|
||||
def _tickzoo_is_video_link(url):
|
||||
# tickzoo.tv video pages look like https://tickzoo.tv/<slug>/
|
||||
path = urlparse(url).path.strip("/")
|
||||
return bool(path) and not path.split("/")[-1].endswith(
|
||||
(".html", ".htm", ".php", ".txt", ".xml", ".json", ".css")
|
||||
)
|
||||
|
||||
|
||||
def _tickzoo_extract_embed(html: str) -> str:
|
||||
"""Pull the video iframe src out of the <textarea id="iframe"> embed box."""
|
||||
m = re.search(r"<textarea[^>]*id=[\"']iframe[\"'][^>]*>(.*?)</textarea>", html, re.S | re.I)
|
||||
if not m:
|
||||
return ""
|
||||
inner = m.group(1)
|
||||
m = re.search(r"<iframe\b[^>]*\bsrc\s*=\s*[\"']([^\"']+)[\"']", inner, re.I)
|
||||
if not m:
|
||||
m = re.search(r"<iframe\b[^>]*\bsrc\s*=\s*([^\"'>\s]+)", inner, re.I)
|
||||
if not m:
|
||||
return ""
|
||||
src = unescape(m.group(1).strip())
|
||||
if any(ad in src.lower() for ad in _TICKZOO_AD_HOSTS):
|
||||
return ""
|
||||
return src
|
||||
|
||||
|
||||
def _tickzoo_extract_title(html: str) -> str:
|
||||
m = re.search(r"<div[^>]*class=[\"']title[\"'][^>]*>(.*?)</div>", html, re.S)
|
||||
if m:
|
||||
txt = re.sub(r"<[^>]+>", "", m.group(1))
|
||||
return unescape(txt).strip()
|
||||
m = re.search(r"<meta[^>]*property=[\"']og:title[\"'][^>]*content=[\"']([^\"']*)", html, re.I)
|
||||
return unescape(m.group(1)).strip() if m else ""
|
||||
|
||||
|
||||
def _tickzoo_extract_uploader(html: str) -> str:
|
||||
"""Studio tag ('folder X') if the page carries one, else a stable default."""
|
||||
for m in re.finditer(r"folder\s*[:\-]?\s*([A-Za-z0-9][A-Za-z0-9_. \-\']{0,40})", html, re.I):
|
||||
name = re.sub(r"[<>\"]", "", m.group(1)).strip().strip(".")
|
||||
if name:
|
||||
return scraper_core.clean_filename(name)
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _tickzoo_score_media_url(url: str) -> int:
|
||||
"""Rank candidate media URLs; higher wins. Prefer HLS masters, then HLS, then mp4."""
|
||||
ul = url.lower()
|
||||
if "blank.mp4" in ul:
|
||||
return -1
|
||||
if "master.m3u8" in ul or "/master" in ul:
|
||||
return 100
|
||||
if ".m3u8" in ul or "manifest" in ul:
|
||||
return 90
|
||||
if "veevcdn" in ul or "veev.to" in ul:
|
||||
return 80
|
||||
if "/stream/" in ul:
|
||||
return 70
|
||||
if ".mp4" in ul or ".webm" in ul:
|
||||
return 50
|
||||
return 60
|
||||
|
||||
|
||||
async def _tickzoo_resolve_media(scraper, embed_url: str, referer_url: str) -> str:
|
||||
"""Resolve the direct media URL from a tickzoo embed host via headless Chrome.
|
||||
|
||||
Handles every embed host seen on tickzoo.tv:
|
||||
* firestream.to -> POST /api/videos/<id>/resolve returns JSON
|
||||
{signedVideoUrl: "<m3u8>", ...}
|
||||
* hgcloud.to -> HLS master manifest fires as a network response (jwplayer)
|
||||
* rubyvidhub.com -> HLS master manifest fires as a network response
|
||||
* veev.to -> the mp4 URL lives in <video>.currentSrc after muted
|
||||
play(); its CDN is TLS/network gated, so the download
|
||||
itself may fail from some networks.
|
||||
Returns the best scoring URL, or "" if nothing resolvable.
|
||||
"""
|
||||
page = await scraper.new_page()
|
||||
candidates = {}
|
||||
|
||||
def note(url, score):
|
||||
if not url or "blob:" in url or "data:" in url:
|
||||
return
|
||||
if score > candidates.get(url, -1):
|
||||
candidates[url] = score
|
||||
|
||||
async def on_response(resp):
|
||||
try:
|
||||
rurl = resp.url
|
||||
ct = resp.headers.get("content-type", "").lower()
|
||||
base = rurl.split("?")[0]
|
||||
if ("video/" in ct or "mpegurl" in ct or "octet-stream" in ct
|
||||
or base.endswith((".m3u8", ".mp4", ".webm"))):
|
||||
note(rurl, _tickzoo_score_media_url(rurl))
|
||||
if "/api/" in rurl.lower() or "resolve" in rurl.lower():
|
||||
try:
|
||||
body = await resp.text()
|
||||
except Exception:
|
||||
return
|
||||
try:
|
||||
payload = json.loads(body)
|
||||
except Exception:
|
||||
return
|
||||
for key in ("signedVideoUrl", "signedVideoSdUrl", "videoUrl", "url", "src"):
|
||||
val = payload.get(key)
|
||||
if isinstance(val, str) and val.startswith("http"):
|
||||
note(val, 90)
|
||||
break
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
page.on("response", lambda r: asyncio.create_task(on_response(r)))
|
||||
try:
|
||||
tqdm.write(f" resolving embed {embed_url}")
|
||||
await page.goto(embed_url, wait_until="domcontentloaded", timeout=45000, referer=referer_url)
|
||||
await page.wait_for_timeout(4000)
|
||||
|
||||
# Trigger playback so the player actually requests the media.
|
||||
for selector in ("button.vjs-big-play-button", ".play-button", "[class*='play']", "video"):
|
||||
try:
|
||||
el = await page.query_selector(selector)
|
||||
if el:
|
||||
await el.click(timeout=1200)
|
||||
await page.wait_for_timeout(1500)
|
||||
break
|
||||
except Exception:
|
||||
continue
|
||||
try:
|
||||
await page.evaluate(
|
||||
"""() => { const v = document.querySelector('video');
|
||||
if (v) { v.muted = true; v.play().catch(()=>{}); } }"""
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Some players (firestream) answer late / intermittently - poll a while.
|
||||
dom_js = """() => { const v = document.querySelector('video');
|
||||
if (!v) return '';
|
||||
return (v.currentSrc || v.src || ''); }"""
|
||||
for _ in range(8):
|
||||
await page.wait_for_timeout(4000)
|
||||
try:
|
||||
src = await page.evaluate(dom_js)
|
||||
if src:
|
||||
note(src, _tickzoo_score_media_url(src))
|
||||
except Exception:
|
||||
pass
|
||||
if candidates:
|
||||
break
|
||||
|
||||
good = {u: s for u, s in candidates.items() if s > 0}
|
||||
if not good:
|
||||
tqdm.write(f" x no media found for embed {embed_url}")
|
||||
return ""
|
||||
best = max(good.items(), key=lambda kv: (kv[1], kv[0]))[0]
|
||||
tqdm.write(f" -> {best}")
|
||||
return best
|
||||
finally:
|
||||
try:
|
||||
await page.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
async def _tickzoo_process_one(scraper, url, download_dir, skip_existing, overall_bar, bar_pool):
|
||||
"""Fetch the tickzoo page, resolve its embed, download the video."""
|
||||
if is_filtered(url):
|
||||
tqdm.write(f" x Filtered: {url}")
|
||||
overall_bar.update(1)
|
||||
return
|
||||
|
||||
html = await asyncio.to_thread(scraper_core.curl_fetch, url, referer="https://tickzoo.tv/")
|
||||
if not html:
|
||||
tqdm.write(f" x Failed to fetch page: {url}")
|
||||
overall_bar.update(1)
|
||||
return
|
||||
|
||||
embed_url = _tickzoo_extract_embed(html)
|
||||
if not embed_url:
|
||||
tqdm.write(f" x No embed iframe found on {url}")
|
||||
overall_bar.update(1)
|
||||
return
|
||||
|
||||
title = _tickzoo_extract_title(html) or url.rstrip("/").rsplit("/", 1)[-1]
|
||||
uploader = _tickzoo_extract_uploader(html)
|
||||
tqdm.write(f" [{title[:70]}]")
|
||||
tqdm.write(f" embed: {embed_url}")
|
||||
|
||||
media_url = await _tickzoo_resolve_media(scraper, embed_url, url)
|
||||
if not media_url:
|
||||
overall_bar.update(1)
|
||||
return
|
||||
|
||||
stem = scraper_core.clean_filename(url.rstrip("/").rsplit("/", 1)[-1])
|
||||
dest_path = download_dir / uploader / f"{stem}.mp4"
|
||||
is_hls = ".m3u8" in media_url.lower() or "manifest" in media_url.lower()
|
||||
|
||||
if skip_existing:
|
||||
if is_hls:
|
||||
if scraper_core.is_already_downloaded(dest_path):
|
||||
tqdm.write(f" [SKIP] '{uploader}/{stem}.mp4' already downloaded (HLS).")
|
||||
overall_bar.record_skip(dest_path.stat().st_size)
|
||||
return
|
||||
else:
|
||||
is_complete, existing_bytes = await check_existing_file(
|
||||
dest_path, media_url, {"Referer": embed_url}, None
|
||||
)
|
||||
if is_complete:
|
||||
overall_bar.record_skip(existing_bytes)
|
||||
return
|
||||
|
||||
pos = bar_pool.acquire()
|
||||
if pos is None:
|
||||
pos = 4
|
||||
headers = None if is_hls else {"Referer": embed_url}
|
||||
success = await asyncio.to_thread(
|
||||
scraper_core.download_file, media_url, dest_path, headers, pos, None, None
|
||||
)
|
||||
bar_pool.release(pos)
|
||||
|
||||
if success and dest_path.is_file() and dest_path.stat().st_size > 0:
|
||||
label = f"{dest_path.parent.name}/{dest_path.name}"
|
||||
overall_bar.record_download(dest_path.stat().st_size, name=label)
|
||||
scraper_core.append_log(download_dir / "urls.txt", url)
|
||||
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
|
||||
else:
|
||||
overall_bar.update(1)
|
||||
|
||||
|
||||
async def tickzoo_process_urls(download_dir, urls, concurrency, skip_existing):
|
||||
"""Download tickzoo.tv video pages.
|
||||
|
||||
Each page embeds a 3rd-party player (veev.to / firestream.to / hgcloud.to /
|
||||
rubyvidhub.com). Embeds are resolved through one shared headless Chrome
|
||||
session, then handed to the shared downloader (yt-dlp for HLS streams).
|
||||
Sequential on purpose: embed hosts are flaky and a browser session is reused.
|
||||
"""
|
||||
unique = [u for u in dedupe(urls) if _tickzoo_is_video_link(u)]
|
||||
if not unique:
|
||||
tqdm.write(" No tickzoo.tv video URLs provided.")
|
||||
return
|
||||
safe_mkdir(download_dir)
|
||||
|
||||
overall_bar = OverallProgressTracker(total=0, desc="tickzoo.tv")
|
||||
bar_pool = scraper_core.BarPositionPool(concurrency)
|
||||
bar_pool.available = [p + 3 for p in bar_pool.available]
|
||||
|
||||
for url in unique:
|
||||
tqdm.write(f"Queuing video URL: {url}")
|
||||
overall_bar.total += 1
|
||||
overall_bar.refresh()
|
||||
|
||||
scraper = scraper_core.PlaywrightScraper()
|
||||
await scraper.start()
|
||||
try:
|
||||
for url in unique:
|
||||
try:
|
||||
await _tickzoo_process_one(scraper, url, download_dir, skip_existing, overall_bar, bar_pool)
|
||||
except Exception as e:
|
||||
tqdm.write(f" Error processing {url}: {e}")
|
||||
overall_bar.update(1)
|
||||
finally:
|
||||
try:
|
||||
await scraper.close()
|
||||
except Exception:
|
||||
pass
|
||||
overall_bar.close()
|
||||
sys.stdout.write("\n" * (concurrency + 3))
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
# ----- Pornhub ---------------------------------------------------------------
|
||||
|
||||
def _pornhub_is_video_link(url):
|
||||
@@ -3052,6 +3337,8 @@ def main():
|
||||
asyncio.run(zootubevip_process_urls(download_dir, targets, concurrency, args.skip_existing))
|
||||
elif handler == "zootube1":
|
||||
asyncio.run(zootube1_process_urls(download_dir, targets, concurrency, args.skip_existing))
|
||||
elif handler == "tickzoo":
|
||||
asyncio.run(tickzoo_process_urls(download_dir, targets, concurrency, args.skip_existing))
|
||||
elif handler == "redgifs":
|
||||
output_dir = Path(args.output).resolve()
|
||||
creator_names = [t for t in targets if not looks_like_url(t)]
|
||||
|
||||
Reference in new issue
Block a user