Automated daily backup: 2026-10-05 03:21

This commit is contained in:
gooner committed 2026-10-05 03:24:08 -04:00
1 parent 896be9ea75
commit a68bf86cf9
794 files changed
+1543740 -1975

No files matched your search

+192 -1
View File
@@ -36,7 +36,7 @@ import argparse
import random
import time
from pathlib import Path
from urllib.parse import urlparse
from urllib.parse import urlparse, unquote
from html import unescape
from concurrent.futures import ThreadPoolExecutor, as_completed
import threading
@@ -550,6 +550,12 @@ SITES = {
"handler": "zootube1",
"default_concurrency": 3,
},
"file.al": {
"folder": "file.al",
"domains": ["file.al"],
"handler": "fileal",
"default_concurrency": 1,
},
}
# ---------------------------------------------------------------------------
@@ -2735,6 +2741,181 @@ def redgifs_download_gif_urls(gif_urls, output_dir, concurrency, skip_existing,
finally:
api.stop()
# ----- File.al (XFS hoster) --------------------------------------------------
def _fileal_load_cookies(cookie_path: str) -> dict:
"""Load file.al cookies from a Netscape cookies.txt or EditThisCookie JSON export.
Login on file.al is reCAPTCHA-gated, so cookies must be exported from a
browser session. Netscape lines are tab-separated:
domain, includeSubdomains, path, secure, expiry, name, value.
"""
path = Path(cookie_path)
if not path.is_file():
tqdm.write(f' [!] Cookie file not found: {cookie_path}')
return {}
try:
text = path.read_text(encoding="utf-8", errors="replace")
except Exception as e:
tqdm.write(f' [!] Cannot read cookie file: {e}')
return {}
cookies = {}
if text.lstrip().startswith("["):
try:
for c in json.loads(text):
domain = c.get("domain") or ""
if c.get("name") and c.get("value") and "file.al" in domain:
cookies[c["name"]] = c["value"]
except Exception as e:
tqdm.write(f' [!] Failed to parse JSON cookies: {e}')
cookies = {}
else:
for line in text.splitlines():
if not line or line.startswith("#"):
continue
parts = line.split("\t")
if len(parts) >= 7 and "file.al" in parts[0]:
cookies[parts[5]] = parts[6]
return cookies
def _fileal_extract_form_payload(html):
"""Return the hidden-field payload of the download1/download2 form, or None.
XFS offers a free flow (submit method_free) and a subscribed flow (already
includes method_premium=1, which must be posted verbatim).
"""
for m in re.finditer(r"<[Ff]orm[^>]*>.*?</[Ff]orm>", html, re.S):
fields = dict(re.findall(r'<input type="hidden" name="([^"]+)" value="([^"]*)">', m.group(0)))
if fields.get("op") in ("download1", "download2"):
if "method_premium" not in fields and re.search(r'name="method_free"', m.group(0)):
fields["method_free"] = "Free Download"
return fields
return None
def _fileal_extract_direct_link(html):
"""Find the generated direct file URL on the download page.
Prefers the anchor labelled "Click here to download"; falls back to the
first off-site link that ends in a media/archive extension (the XFS CDN
pattern, excluding ad links).
"""
m = re.search(r'<a[^>]*href="([^"]+)"[^>]*>\s*Click here to download', html, re.S)
if m:
return m.group(1)
for m in re.finditer(r'<a[^>]*href="(https?://[^"]+)"', html):
host = urlparse(m.group(1)).netloc.lower()
path = urlparse(m.group(1)).path.lower()
if host != "file.al" and not host.endswith(".file.al") and path.endswith(
(".avi", ".mp4", ".mkv", ".webm", ".mov", ".flv", ".wmv", ".mpg", ".mpeg", ".zip", ".rar", ".7z")
):
return m.group(1)
return None
def _fileal_extract_filename(html, direct_url):
m = re.search(r"Filename:\s*([^<]+)", html)
if m:
name = m.group(1).strip()
if name:
return name
seg = urlparse(direct_url).path.rstrip("/").split("/")[-1]
return unquote(seg) if seg else None
def _fileal_download_one(url, output_dir, skip_existing, cookies, overall_bar, log_path):
from curl_cffi import requests as curl_req
session = curl_req.Session(impersonate="chrome")
if cookies:
session.cookies.update(cookies)
resp = session.get(url, timeout=60, allow_redirects=True)
resp.raise_for_status()
if "Premium Users only" in resp.text:
tqdm.write(f' [!] Paywalled (uploader subscription required): {url}')
overall_bar.update(1)
return
fallback_name = Path(urlparse(url).path).name
if fallback_name.lower().endswith(".html"):
fallback_name = fallback_name[:-5]
payload = _fileal_extract_form_payload(resp.text)
if not payload:
tqdm.write(f' [!] No download form on page: {url}')
overall_bar.update(1)
return
current = session.post("https://file.al/", data=payload, timeout=120, allow_redirects=True)
direct_url = filename = None
for _ in range(4):
if "Premium Users only" in current.text:
tqdm.write(f' [!] Paywalled (uploader subscription required): {url}')
overall_bar.update(1)
return
if not (current.headers.get("content-type") or "").startswith("text/html"):
direct_url = current.url
break
direct_url = _fileal_extract_direct_link(current.text)
if direct_url:
filename = _fileal_extract_filename(current.text, direct_url)
break
payload = _fileal_extract_form_payload(current.text)
if payload:
current = session.post("https://file.al/", data=payload, timeout=120, allow_redirects=True)
continue
break
if not direct_url:
tqdm.write(f' [!] No direct link resolved for: {url}')
overall_bar.update(1)
return
fname = scraper_core.clean_filename(filename or fallback_name or "file")
dest_path = output_dir / fname
dl_cookies = cookies or None
if skip_existing:
is_complete, existing_bytes = check_existing_file_sync(dest_path, direct_url, cookies=dl_cookies)
if is_complete:
overall_bar.record_skip(existing_bytes)
return
safe_append_text(log_path, direct_url + "\n")
success = scraper_core.download_file(direct_url, dest_path, cookies=dl_cookies, referer="https://file.al/")
if success:
size = dest_path.stat().st_size if dest_path.is_file() else 0
overall_bar.record_download(size, name=f"{dest_path.parent.name}/{dest_path.name}")
else:
overall_bar.update(1)
def fileal_download_urls(urls, output_dir, concurrency, skip_existing, cookies=None):
"""Download file.al links into output_dir/.
Sequential on purpose: XFS hosters are brittle under parallel requests, so
the concurrency argument is accepted only for CLI symmetry.
"""
unique = dedupe(urls)
if not unique:
tqdm.write(" No file.al URLs provided.")
return
safe_mkdir(output_dir)
overall_bar = OverallProgressTracker(total=len(unique), desc="file.al")
log_path = output_dir / "all_links.txt"
try:
for url in unique:
try:
_fileal_download_one(url, output_dir, skip_existing, cookies, overall_bar, log_path)
except Exception as e:
tqdm.write(f" [ERROR] {url}\n {e}")
overall_bar.update(1)
finally:
overall_bar.close()
sys.stdout.write("\n" * 2)
sys.stdout.flush()
# ---------------------------------------------------------------------------
# CLI
# ---------------------------------------------------------------------------
@@ -2804,6 +2985,11 @@ def main():
"--list-sites", action="store_true",
help="List supported websites and exit.",
)
parser.add_argument(
"--cookies", default=None,
help="Path to a browser cookie export (Netscape cookies.txt or EditThisCookie "
"JSON) for sites that need an authenticated session, e.g. file.al.",
)
args = parser.parse_args()
if args.list_sites:
@@ -2869,6 +3055,11 @@ def main():
redgifs_download_creators(creator_names, output_dir, concurrency, args.skip_existing, skip_cooldown=args.skip_redgifs_cooldown)
if gif_urls:
redgifs_download_gif_urls(gif_urls, output_dir, concurrency, args.skip_existing, skip_cooldown=args.skip_redgifs_cooldown)
elif handler == "fileal":
file_cookies = _fileal_load_cookies(args.cookies) if args.cookies else {}
if args.cookies and not file_cookies:
print(f" [!] No file.al cookies loaded from '{args.cookies}' - uploader paywalls will be skipped.", file=sys.stderr)
fileal_download_urls(targets, download_dir, concurrency, args.skip_existing, file_cookies)
else:
print(f" [!] Unknown handler '{handler}' for site '{site_key}' - skipped.", file=sys.stderr)