Automated daily backup: 2026-10-10 03:08

This commit is contained in:
gooner committed 2026-10-10 03:09:46 -04:00
1 parent 005a717a5b
commit 87811e0a16
9 files changed
+433 -50

No files matched your search

+287
View File
@@ -555,6 +555,12 @@ SITES = {
"handler": "zootube1",
"default_concurrency": 3,
},
"tickzoo.tv": {
"folder": "tickzoo.tv",
"domains": ["tickzoo.tv"],
"handler": "tickzoo",
"default_concurrency": 1,
},
"file.al": {
"folder": "file.al",
"domains": ["file.al"],
@@ -1627,6 +1633,285 @@ async def zootubevip_process_urls(download_dir, urls, concurrency, skip_existing
sys.stdout.flush()
# ----- tickzoo.tv ------------------------------------------------------------
# Video pages have no <video> tag - they embed a 3rd-party player
# (veev.to / firestream.to / hgcloud.to / rubyvidhub.com) via an "Embed Code"
# <textarea id="iframe">, so each embed is resolved through headless Chrome.
_TICKZOO_AD_HOSTS = (
"dtscout", "lijit", "doubleclick", "googlesyndication", "adservice",
"pubmatic", "criteo", "taboola", "outbrain", "adsystem",
)
def _tickzoo_is_video_link(url):
# tickzoo.tv video pages look like https://tickzoo.tv/<slug>/
path = urlparse(url).path.strip("/")
return bool(path) and not path.split("/")[-1].endswith(
(".html", ".htm", ".php", ".txt", ".xml", ".json", ".css")
)
def _tickzoo_extract_embed(html: str) -> str:
"""Pull the video iframe src out of the <textarea id="iframe"> embed box."""
m = re.search(r"<textarea[^>]*id=[\"']iframe[\"'][^>]*>(.*?)</textarea>", html, re.S | re.I)
if not m:
return ""
inner = m.group(1)
m = re.search(r"<iframe\b[^>]*\bsrc\s*=\s*[\"']([^\"']+)[\"']", inner, re.I)
if not m:
m = re.search(r"<iframe\b[^>]*\bsrc\s*=\s*([^\"'>\s]+)", inner, re.I)
if not m:
return ""
src = unescape(m.group(1).strip())
if any(ad in src.lower() for ad in _TICKZOO_AD_HOSTS):
return ""
return src
def _tickzoo_extract_title(html: str) -> str:
m = re.search(r"<div[^>]*class=[\"']title[\"'][^>]*>(.*?)</div>", html, re.S)
if m:
txt = re.sub(r"<[^>]+>", "", m.group(1))
return unescape(txt).strip()
m = re.search(r"<meta[^>]*property=[\"']og:title[\"'][^>]*content=[\"']([^\"']*)", html, re.I)
return unescape(m.group(1)).strip() if m else ""
def _tickzoo_extract_uploader(html: str) -> str:
"""Studio tag ('folder X') if the page carries one, else a stable default."""
for m in re.finditer(r"folder\s*[:\-]?\s*([A-Za-z0-9][A-Za-z0-9_. \-\']{0,40})", html, re.I):
name = re.sub(r"[<>\"]", "", m.group(1)).strip().strip(".")
if name:
return scraper_core.clean_filename(name)
return "unknown"
def _tickzoo_score_media_url(url: str) -> int:
"""Rank candidate media URLs; higher wins. Prefer HLS masters, then HLS, then mp4."""
ul = url.lower()
if "blank.mp4" in ul:
return -1
if "master.m3u8" in ul or "/master" in ul:
return 100
if ".m3u8" in ul or "manifest" in ul:
return 90
if "veevcdn" in ul or "veev.to" in ul:
return 80
if "/stream/" in ul:
return 70
if ".mp4" in ul or ".webm" in ul:
return 50
return 60
async def _tickzoo_resolve_media(scraper, embed_url: str, referer_url: str) -> str:
"""Resolve the direct media URL from a tickzoo embed host via headless Chrome.
Handles every embed host seen on tickzoo.tv:
* firestream.to -> POST /api/videos/<id>/resolve returns JSON
{signedVideoUrl: "<m3u8>", ...}
* hgcloud.to -> HLS master manifest fires as a network response (jwplayer)
* rubyvidhub.com -> HLS master manifest fires as a network response
* veev.to -> the mp4 URL lives in <video>.currentSrc after muted
play(); its CDN is TLS/network gated, so the download
itself may fail from some networks.
Returns the best scoring URL, or "" if nothing resolvable.
"""
page = await scraper.new_page()
candidates = {}
def note(url, score):
if not url or "blob:" in url or "data:" in url:
return
if score > candidates.get(url, -1):
candidates[url] = score
async def on_response(resp):
try:
rurl = resp.url
ct = resp.headers.get("content-type", "").lower()
base = rurl.split("?")[0]
if ("video/" in ct or "mpegurl" in ct or "octet-stream" in ct
or base.endswith((".m3u8", ".mp4", ".webm"))):
note(rurl, _tickzoo_score_media_url(rurl))
if "/api/" in rurl.lower() or "resolve" in rurl.lower():
try:
body = await resp.text()
except Exception:
return
try:
payload = json.loads(body)
except Exception:
return
for key in ("signedVideoUrl", "signedVideoSdUrl", "videoUrl", "url", "src"):
val = payload.get(key)
if isinstance(val, str) and val.startswith("http"):
note(val, 90)
break
except Exception:
pass
page.on("response", lambda r: asyncio.create_task(on_response(r)))
try:
tqdm.write(f" resolving embed {embed_url}")
await page.goto(embed_url, wait_until="domcontentloaded", timeout=45000, referer=referer_url)
await page.wait_for_timeout(4000)
# Trigger playback so the player actually requests the media.
for selector in ("button.vjs-big-play-button", ".play-button", "[class*='play']", "video"):
try:
el = await page.query_selector(selector)
if el:
await el.click(timeout=1200)
await page.wait_for_timeout(1500)
break
except Exception:
continue
try:
await page.evaluate(
"""() => { const v = document.querySelector('video');
if (v) { v.muted = true; v.play().catch(()=>{}); } }"""
)
except Exception:
pass
# Some players (firestream) answer late / intermittently - poll a while.
dom_js = """() => { const v = document.querySelector('video');
if (!v) return '';
return (v.currentSrc || v.src || ''); }"""
for _ in range(8):
await page.wait_for_timeout(4000)
try:
src = await page.evaluate(dom_js)
if src:
note(src, _tickzoo_score_media_url(src))
except Exception:
pass
if candidates:
break
good = {u: s for u, s in candidates.items() if s > 0}
if not good:
tqdm.write(f" x no media found for embed {embed_url}")
return ""
best = max(good.items(), key=lambda kv: (kv[1], kv[0]))[0]
tqdm.write(f" -> {best}")
return best
finally:
try:
await page.close()
except Exception:
pass
async def _tickzoo_process_one(scraper, url, download_dir, skip_existing, overall_bar, bar_pool):
"""Fetch the tickzoo page, resolve its embed, download the video."""
if is_filtered(url):
tqdm.write(f" x Filtered: {url}")
overall_bar.update(1)
return
html = await asyncio.to_thread(scraper_core.curl_fetch, url, referer="https://tickzoo.tv/")
if not html:
tqdm.write(f" x Failed to fetch page: {url}")
overall_bar.update(1)
return
embed_url = _tickzoo_extract_embed(html)
if not embed_url:
tqdm.write(f" x No embed iframe found on {url}")
overall_bar.update(1)
return
title = _tickzoo_extract_title(html) or url.rstrip("/").rsplit("/", 1)[-1]
uploader = _tickzoo_extract_uploader(html)
tqdm.write(f" [{title[:70]}]")
tqdm.write(f" embed: {embed_url}")
media_url = await _tickzoo_resolve_media(scraper, embed_url, url)
if not media_url:
overall_bar.update(1)
return
stem = scraper_core.clean_filename(url.rstrip("/").rsplit("/", 1)[-1])
dest_path = download_dir / uploader / f"{stem}.mp4"
is_hls = ".m3u8" in media_url.lower() or "manifest" in media_url.lower()
if skip_existing:
if is_hls:
if scraper_core.is_already_downloaded(dest_path):
tqdm.write(f" [SKIP] '{uploader}/{stem}.mp4' already downloaded (HLS).")
overall_bar.record_skip(dest_path.stat().st_size)
return
else:
is_complete, existing_bytes = await check_existing_file(
dest_path, media_url, {"Referer": embed_url}, None
)
if is_complete:
overall_bar.record_skip(existing_bytes)
return
pos = bar_pool.acquire()
if pos is None:
pos = 4
headers = None if is_hls else {"Referer": embed_url}
success = await asyncio.to_thread(
scraper_core.download_file, media_url, dest_path, headers, pos, None, None
)
bar_pool.release(pos)
if success and dest_path.is_file() and dest_path.stat().st_size > 0:
label = f"{dest_path.parent.name}/{dest_path.name}"
overall_bar.record_download(dest_path.stat().st_size, name=label)
scraper_core.append_log(download_dir / "urls.txt", url)
scraper_core.append_log(download_dir / uploader / "urls.txt", url)
else:
overall_bar.update(1)
async def tickzoo_process_urls(download_dir, urls, concurrency, skip_existing):
"""Download tickzoo.tv video pages.
Each page embeds a 3rd-party player (veev.to / firestream.to / hgcloud.to /
rubyvidhub.com). Embeds are resolved through one shared headless Chrome
session, then handed to the shared downloader (yt-dlp for HLS streams).
Sequential on purpose: embed hosts are flaky and a browser session is reused.
"""
unique = [u for u in dedupe(urls) if _tickzoo_is_video_link(u)]
if not unique:
tqdm.write(" No tickzoo.tv video URLs provided.")
return
safe_mkdir(download_dir)
overall_bar = OverallProgressTracker(total=0, desc="tickzoo.tv")
bar_pool = scraper_core.BarPositionPool(concurrency)
bar_pool.available = [p + 3 for p in bar_pool.available]
for url in unique:
tqdm.write(f"Queuing video URL: {url}")
overall_bar.total += 1
overall_bar.refresh()
scraper = scraper_core.PlaywrightScraper()
await scraper.start()
try:
for url in unique:
try:
await _tickzoo_process_one(scraper, url, download_dir, skip_existing, overall_bar, bar_pool)
except Exception as e:
tqdm.write(f" Error processing {url}: {e}")
overall_bar.update(1)
finally:
try:
await scraper.close()
except Exception:
pass
overall_bar.close()
sys.stdout.write("\n" * (concurrency + 3))
sys.stdout.flush()
# ----- Pornhub ---------------------------------------------------------------
def _pornhub_is_video_link(url):
@@ -3052,6 +3337,8 @@ def main():
asyncio.run(zootubevip_process_urls(download_dir, targets, concurrency, args.skip_existing))
elif handler == "zootube1":
asyncio.run(zootube1_process_urls(download_dir, targets, concurrency, args.skip_existing))
elif handler == "tickzoo":
asyncio.run(tickzoo_process_urls(download_dir, targets, concurrency, args.skip_existing))
elif handler == "redgifs":
output_dir = Path(args.output).resolve()
creator_names = [t for t in targets if not looks_like_url(t)]