Automated daily backup: 2026-10-05 03:21
This commit is contained in:
1 parent
896be9ea75
commit
a68bf86cf9
794 files changed
+1543740
-1975
No files matched your search
+192
-1
@@ -36,7 +36,7 @@ import argparse
|
||||
import random
|
||||
import time
|
||||
from pathlib import Path
|
||||
from urllib.parse import urlparse
|
||||
from urllib.parse import urlparse, unquote
|
||||
from html import unescape
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
import threading
|
||||
@@ -550,6 +550,12 @@ SITES = {
|
||||
"handler": "zootube1",
|
||||
"default_concurrency": 3,
|
||||
},
|
||||
"file.al": {
|
||||
"folder": "file.al",
|
||||
"domains": ["file.al"],
|
||||
"handler": "fileal",
|
||||
"default_concurrency": 1,
|
||||
},
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -2735,6 +2741,181 @@ def redgifs_download_gif_urls(gif_urls, output_dir, concurrency, skip_existing,
|
||||
finally:
|
||||
api.stop()
|
||||
|
||||
# ----- File.al (XFS hoster) --------------------------------------------------
|
||||
|
||||
def _fileal_load_cookies(cookie_path: str) -> dict:
|
||||
"""Load file.al cookies from a Netscape cookies.txt or EditThisCookie JSON export.
|
||||
|
||||
Login on file.al is reCAPTCHA-gated, so cookies must be exported from a
|
||||
browser session. Netscape lines are tab-separated:
|
||||
domain, includeSubdomains, path, secure, expiry, name, value.
|
||||
"""
|
||||
path = Path(cookie_path)
|
||||
if not path.is_file():
|
||||
tqdm.write(f' [!] Cookie file not found: {cookie_path}')
|
||||
return {}
|
||||
try:
|
||||
text = path.read_text(encoding="utf-8", errors="replace")
|
||||
except Exception as e:
|
||||
tqdm.write(f' [!] Cannot read cookie file: {e}')
|
||||
return {}
|
||||
cookies = {}
|
||||
if text.lstrip().startswith("["):
|
||||
try:
|
||||
for c in json.loads(text):
|
||||
domain = c.get("domain") or ""
|
||||
if c.get("name") and c.get("value") and "file.al" in domain:
|
||||
cookies[c["name"]] = c["value"]
|
||||
except Exception as e:
|
||||
tqdm.write(f' [!] Failed to parse JSON cookies: {e}')
|
||||
cookies = {}
|
||||
else:
|
||||
for line in text.splitlines():
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
if len(parts) >= 7 and "file.al" in parts[0]:
|
||||
cookies[parts[5]] = parts[6]
|
||||
return cookies
|
||||
|
||||
|
||||
def _fileal_extract_form_payload(html):
|
||||
"""Return the hidden-field payload of the download1/download2 form, or None.
|
||||
|
||||
XFS offers a free flow (submit method_free) and a subscribed flow (already
|
||||
includes method_premium=1, which must be posted verbatim).
|
||||
"""
|
||||
for m in re.finditer(r"<[Ff]orm[^>]*>.*?</[Ff]orm>", html, re.S):
|
||||
fields = dict(re.findall(r'<input type="hidden" name="([^"]+)" value="([^"]*)">', m.group(0)))
|
||||
if fields.get("op") in ("download1", "download2"):
|
||||
if "method_premium" not in fields and re.search(r'name="method_free"', m.group(0)):
|
||||
fields["method_free"] = "Free Download"
|
||||
return fields
|
||||
return None
|
||||
|
||||
|
||||
def _fileal_extract_direct_link(html):
|
||||
"""Find the generated direct file URL on the download page.
|
||||
|
||||
Prefers the anchor labelled "Click here to download"; falls back to the
|
||||
first off-site link that ends in a media/archive extension (the XFS CDN
|
||||
pattern, excluding ad links).
|
||||
"""
|
||||
m = re.search(r'<a[^>]*href="([^"]+)"[^>]*>\s*Click here to download', html, re.S)
|
||||
if m:
|
||||
return m.group(1)
|
||||
for m in re.finditer(r'<a[^>]*href="(https?://[^"]+)"', html):
|
||||
host = urlparse(m.group(1)).netloc.lower()
|
||||
path = urlparse(m.group(1)).path.lower()
|
||||
if host != "file.al" and not host.endswith(".file.al") and path.endswith(
|
||||
(".avi", ".mp4", ".mkv", ".webm", ".mov", ".flv", ".wmv", ".mpg", ".mpeg", ".zip", ".rar", ".7z")
|
||||
):
|
||||
return m.group(1)
|
||||
return None
|
||||
|
||||
|
||||
def _fileal_extract_filename(html, direct_url):
|
||||
m = re.search(r"Filename:\s*([^<]+)", html)
|
||||
if m:
|
||||
name = m.group(1).strip()
|
||||
if name:
|
||||
return name
|
||||
seg = urlparse(direct_url).path.rstrip("/").split("/")[-1]
|
||||
return unquote(seg) if seg else None
|
||||
|
||||
|
||||
def _fileal_download_one(url, output_dir, skip_existing, cookies, overall_bar, log_path):
|
||||
from curl_cffi import requests as curl_req
|
||||
session = curl_req.Session(impersonate="chrome")
|
||||
if cookies:
|
||||
session.cookies.update(cookies)
|
||||
|
||||
resp = session.get(url, timeout=60, allow_redirects=True)
|
||||
resp.raise_for_status()
|
||||
if "Premium Users only" in resp.text:
|
||||
tqdm.write(f' [!] Paywalled (uploader subscription required): {url}')
|
||||
overall_bar.update(1)
|
||||
return
|
||||
|
||||
fallback_name = Path(urlparse(url).path).name
|
||||
if fallback_name.lower().endswith(".html"):
|
||||
fallback_name = fallback_name[:-5]
|
||||
|
||||
payload = _fileal_extract_form_payload(resp.text)
|
||||
if not payload:
|
||||
tqdm.write(f' [!] No download form on page: {url}')
|
||||
overall_bar.update(1)
|
||||
return
|
||||
|
||||
current = session.post("https://file.al/", data=payload, timeout=120, allow_redirects=True)
|
||||
direct_url = filename = None
|
||||
for _ in range(4):
|
||||
if "Premium Users only" in current.text:
|
||||
tqdm.write(f' [!] Paywalled (uploader subscription required): {url}')
|
||||
overall_bar.update(1)
|
||||
return
|
||||
if not (current.headers.get("content-type") or "").startswith("text/html"):
|
||||
direct_url = current.url
|
||||
break
|
||||
direct_url = _fileal_extract_direct_link(current.text)
|
||||
if direct_url:
|
||||
filename = _fileal_extract_filename(current.text, direct_url)
|
||||
break
|
||||
payload = _fileal_extract_form_payload(current.text)
|
||||
if payload:
|
||||
current = session.post("https://file.al/", data=payload, timeout=120, allow_redirects=True)
|
||||
continue
|
||||
break
|
||||
|
||||
if not direct_url:
|
||||
tqdm.write(f' [!] No direct link resolved for: {url}')
|
||||
overall_bar.update(1)
|
||||
return
|
||||
|
||||
fname = scraper_core.clean_filename(filename or fallback_name or "file")
|
||||
dest_path = output_dir / fname
|
||||
|
||||
dl_cookies = cookies or None
|
||||
if skip_existing:
|
||||
is_complete, existing_bytes = check_existing_file_sync(dest_path, direct_url, cookies=dl_cookies)
|
||||
if is_complete:
|
||||
overall_bar.record_skip(existing_bytes)
|
||||
return
|
||||
|
||||
safe_append_text(log_path, direct_url + "\n")
|
||||
success = scraper_core.download_file(direct_url, dest_path, cookies=dl_cookies, referer="https://file.al/")
|
||||
if success:
|
||||
size = dest_path.stat().st_size if dest_path.is_file() else 0
|
||||
overall_bar.record_download(size, name=f"{dest_path.parent.name}/{dest_path.name}")
|
||||
else:
|
||||
overall_bar.update(1)
|
||||
|
||||
|
||||
def fileal_download_urls(urls, output_dir, concurrency, skip_existing, cookies=None):
|
||||
"""Download file.al links into output_dir/.
|
||||
|
||||
Sequential on purpose: XFS hosters are brittle under parallel requests, so
|
||||
the concurrency argument is accepted only for CLI symmetry.
|
||||
"""
|
||||
unique = dedupe(urls)
|
||||
if not unique:
|
||||
tqdm.write(" No file.al URLs provided.")
|
||||
return
|
||||
safe_mkdir(output_dir)
|
||||
overall_bar = OverallProgressTracker(total=len(unique), desc="file.al")
|
||||
log_path = output_dir / "all_links.txt"
|
||||
try:
|
||||
for url in unique:
|
||||
try:
|
||||
_fileal_download_one(url, output_dir, skip_existing, cookies, overall_bar, log_path)
|
||||
except Exception as e:
|
||||
tqdm.write(f" [ERROR] {url}\n {e}")
|
||||
overall_bar.update(1)
|
||||
finally:
|
||||
overall_bar.close()
|
||||
sys.stdout.write("\n" * 2)
|
||||
sys.stdout.flush()
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# CLI
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -2804,6 +2985,11 @@ def main():
|
||||
"--list-sites", action="store_true",
|
||||
help="List supported websites and exit.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--cookies", default=None,
|
||||
help="Path to a browser cookie export (Netscape cookies.txt or EditThisCookie "
|
||||
"JSON) for sites that need an authenticated session, e.g. file.al.",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.list_sites:
|
||||
@@ -2869,6 +3055,11 @@ def main():
|
||||
redgifs_download_creators(creator_names, output_dir, concurrency, args.skip_existing, skip_cooldown=args.skip_redgifs_cooldown)
|
||||
if gif_urls:
|
||||
redgifs_download_gif_urls(gif_urls, output_dir, concurrency, args.skip_existing, skip_cooldown=args.skip_redgifs_cooldown)
|
||||
elif handler == "fileal":
|
||||
file_cookies = _fileal_load_cookies(args.cookies) if args.cookies else {}
|
||||
if args.cookies and not file_cookies:
|
||||
print(f" [!] No file.al cookies loaded from '{args.cookies}' - uploader paywalls will be skipped.", file=sys.stderr)
|
||||
fileal_download_urls(targets, download_dir, concurrency, args.skip_existing, file_cookies)
|
||||
else:
|
||||
print(f" [!] Unknown handler '{handler}' for site '{site_key}' - skipped.", file=sys.stderr)
|
||||
|
||||
|
||||
Reference in new issue
Block a user