Initial backup: folder structure manifest, tree summary, scraper scripts, and text metadata
This commit is contained in:
commit
f164832dcc
10504 files changed
+2423594
No files matched your search
@@ -0,0 +1,309 @@
|
||||
import os
|
||||
import sys
|
||||
import re
|
||||
import asyncio
|
||||
import argparse
|
||||
import random
|
||||
from pathlib import Path
|
||||
from tqdm import tqdm
|
||||
|
||||
sys.path.append(str(Path(__file__).resolve().parents[1]))
|
||||
import scraper_core
|
||||
|
||||
BASE_DIR = Path(__file__).resolve().parent
|
||||
DOWNLOAD_DIR = BASE_DIR / "videos"
|
||||
|
||||
FILTERED_WORDS = {"shit", "scat", "shitty", "poop", "horseshit", "bullshit", "cowshit", "shitting", "crap", "feces", "dung", "pungpile", "pungheap", "pooping", "crapping", "manure"}
|
||||
|
||||
def is_filtered(url: str) -> bool:
|
||||
lowered = url.lower()
|
||||
return any(w in lowered for w in FILTERED_WORDS)
|
||||
|
||||
|
||||
async def get_cloudflare_cookies(scraper, url):
|
||||
"""Use Playwright to solve one Cloudflare challenge and return browser cookies."""
|
||||
page = await scraper.new_page()
|
||||
try:
|
||||
await scraper.resolve_page(page, url)
|
||||
raw_cookies = await scraper.context.cookies()
|
||||
return {c["name"]: c["value"] for c in raw_cookies}
|
||||
finally:
|
||||
await page.close()
|
||||
|
||||
|
||||
async def scrape_playlist(playlist_url: str, cookies: dict, queue, counter, overall_bar):
|
||||
"""Scrape all video URLs from a playlist using curl_cffi with browser cookies."""
|
||||
base_check = playlist_url.lower()
|
||||
page_num = 1
|
||||
|
||||
while True:
|
||||
if page_num > 1:
|
||||
base_no_slash = playlist_url.rstrip("/")
|
||||
if "?" in playlist_url:
|
||||
url = f"{base_no_slash}&page={page_num}"
|
||||
elif "uploads-by-user" in base_check or "playlist-by-user" in base_check or "videos-commented-by-user" in base_check or "/channels/" in base_check:
|
||||
url = f"{base_no_slash}/page{page_num}.html"
|
||||
else:
|
||||
url = f"{base_no_slash}?page={page_num}"
|
||||
else:
|
||||
url = playlist_url
|
||||
|
||||
tqdm.write(f" Scraping playlist page {page_num}: {url}")
|
||||
|
||||
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.luxuretv.com/")
|
||||
if not html:
|
||||
tqdm.write(f" Failed to fetch page {page_num}")
|
||||
break
|
||||
|
||||
video_urls = scraper_core.extract_video_links_from_html(html)
|
||||
|
||||
if not video_urls:
|
||||
tqdm.write(f" No video links found on page.")
|
||||
break
|
||||
|
||||
tqdm.write(f" Found {len(video_urls)} unique video links")
|
||||
for video_url in video_urls:
|
||||
if is_filtered(video_url):
|
||||
tqdm.write(f" x Filtered: {video_url}")
|
||||
continue
|
||||
await queue.put(video_url)
|
||||
counter["total"] += 1
|
||||
overall_bar.total = counter["total"]
|
||||
overall_bar.refresh()
|
||||
|
||||
if not scraper_core.has_next_page(html):
|
||||
tqdm.write(f" No 'Next' button found, done.")
|
||||
break
|
||||
|
||||
page_num += 1
|
||||
await asyncio.sleep(random.uniform(1.0, 2.5))
|
||||
|
||||
|
||||
def get_remote_file_size(url: str, headers: dict = None, cookies: dict = None) -> int:
|
||||
"""Fetch the content-length of the remote video file."""
|
||||
req_headers = {
|
||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
|
||||
"Accept": "*/*",
|
||||
"Connection": "keep-alive",
|
||||
}
|
||||
if headers:
|
||||
req_headers.update(headers)
|
||||
|
||||
if cookies:
|
||||
try:
|
||||
from curl_cffi import requests as curl_req
|
||||
with curl_req.get(
|
||||
url, impersonate="chrome", headers=req_headers,
|
||||
cookies=cookies, stream=True, timeout=15,
|
||||
) as resp:
|
||||
if resp.status_code == 200:
|
||||
return int(resp.headers.get("content-length", 0))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
try:
|
||||
import requests
|
||||
with requests.get(url, headers=req_headers, stream=True, timeout=15, cookies=cookies) as resp:
|
||||
if resp.status_code == 200:
|
||||
return int(resp.headers.get("content-length", 0))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
async def worker(queue, cookies, skip_existing, bar_pool, overall_bar):
|
||||
"""Worker: fetch video page HTML via curl_cffi, extract source, download."""
|
||||
while True:
|
||||
url = await queue.get()
|
||||
if url is None:
|
||||
queue.task_done()
|
||||
break
|
||||
|
||||
try:
|
||||
if is_filtered(url):
|
||||
tqdm.write(f" x Filtered: {url}")
|
||||
continue
|
||||
|
||||
# Add a short polite delay to avoid rate limits during bulk scraping
|
||||
await asyncio.sleep(random.uniform(0.3, 0.8))
|
||||
|
||||
html = await asyncio.to_thread(scraper_core.curl_fetch, url, cookies=cookies, referer="https://en.luxuretv.com/")
|
||||
|
||||
if not html:
|
||||
tqdm.write(f" x Failed to fetch {url}")
|
||||
continue
|
||||
|
||||
src, uploader = scraper_core.extract_video_info_from_html(html)
|
||||
|
||||
if not src:
|
||||
tqdm.write(f" x Skipping {url} - no video source found")
|
||||
continue
|
||||
|
||||
stem = url.rstrip(".html").rstrip("/").rsplit("/", 1)[-1]
|
||||
filename = scraper_core.clean_filename(stem) + ".mp4"
|
||||
dest_path = DOWNLOAD_DIR / uploader / filename
|
||||
|
||||
file_exists = dest_path.exists()
|
||||
is_incomplete = False
|
||||
|
||||
if file_exists:
|
||||
local_size = dest_path.stat().st_size
|
||||
if local_size == 0:
|
||||
is_incomplete = True
|
||||
else:
|
||||
remote_size = await asyncio.to_thread(
|
||||
get_remote_file_size, src, {"Referer": url}, cookies
|
||||
)
|
||||
if remote_size > 0 and local_size < remote_size:
|
||||
is_incomplete = True
|
||||
tqdm.write(f" ! Incomplete download detected for {dest_path.parent.name}/{dest_path.name} (local: {local_size} B, remote: {remote_size} B). Redownloading...")
|
||||
|
||||
if skip_existing and file_exists and not is_incomplete:
|
||||
tqdm.write(f" > Already downloaded: {dest_path.parent.name}/{dest_path.name}")
|
||||
continue
|
||||
|
||||
pos = bar_pool.acquire()
|
||||
if pos is None:
|
||||
pos = 1
|
||||
|
||||
success = await asyncio.to_thread(
|
||||
scraper_core.download_file,
|
||||
src, dest_path, {"Referer": url}, pos, cookies=cookies
|
||||
)
|
||||
|
||||
bar_pool.release(pos)
|
||||
|
||||
if success:
|
||||
scraper_core.append_log(DOWNLOAD_DIR / "urls.txt", url)
|
||||
scraper_core.append_log(DOWNLOAD_DIR / uploader / "urls.txt", url)
|
||||
|
||||
except Exception as e:
|
||||
tqdm.write(f" Error processing {url}: {e}")
|
||||
finally:
|
||||
overall_bar.update(1)
|
||||
queue.task_done()
|
||||
|
||||
|
||||
async def producer(urls, cookies, queue, counter, overall_bar):
|
||||
"""Scrape playlists and feed discovered video URLs into the download queue immediately."""
|
||||
for url in urls:
|
||||
if "/videos/" in url and ".html" in url:
|
||||
if is_filtered(url):
|
||||
tqdm.write(f" x Filtered (not queued): {url}")
|
||||
continue
|
||||
tqdm.write(f"Queuing direct video URL: {url}")
|
||||
await queue.put(url)
|
||||
counter["total"] += 1
|
||||
overall_bar.total = counter["total"]
|
||||
overall_bar.refresh()
|
||||
else:
|
||||
tqdm.write(f"Scraping playlist: {url}")
|
||||
try:
|
||||
await scrape_playlist(url, cookies, queue, counter, overall_bar)
|
||||
except Exception as e:
|
||||
tqdm.write(f" Error scraping {url}: {e}")
|
||||
|
||||
await asyncio.sleep(random.uniform(1.0, 2.0))
|
||||
|
||||
|
||||
async def process_urls(urls, concurrency, skip_existing):
|
||||
scraper = scraper_core.PlaywrightScraper()
|
||||
await scraper.start()
|
||||
|
||||
tqdm.write("Solving initial Cloudflare challenge for cookies...")
|
||||
first_url = urls[0]
|
||||
cookies = await get_cloudflare_cookies(scraper, first_url)
|
||||
tqdm.write(f"Got {len(cookies)} cookies: {list(cookies.keys())}")
|
||||
await scraper.close()
|
||||
|
||||
queue = asyncio.Queue()
|
||||
counter = {"total": 0}
|
||||
overall_bar = tqdm(
|
||||
total=0,
|
||||
desc="Overall Progress",
|
||||
position=0,
|
||||
leave=True,
|
||||
ncols=80,
|
||||
bar_format="{desc}: {n_fmt}/{total_fmt} |{bar}| {percentage:.0f}%",
|
||||
)
|
||||
bar_pool = scraper_core.BarPositionPool(concurrency)
|
||||
|
||||
workers = [
|
||||
asyncio.create_task(worker(queue, cookies, skip_existing, bar_pool, overall_bar))
|
||||
for _ in range(concurrency)
|
||||
]
|
||||
|
||||
await producer(urls, cookies, queue, counter, overall_bar)
|
||||
|
||||
for _ in range(concurrency):
|
||||
await queue.put(None)
|
||||
|
||||
await asyncio.gather(*workers)
|
||||
overall_bar.close()
|
||||
|
||||
sys.stdout.write("\n" * (concurrency + 1))
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
def resolve_urls(args_or_files):
|
||||
"""Expand file arguments into URL lists."""
|
||||
urls = []
|
||||
for arg in args_or_files:
|
||||
if os.path.isfile(arg):
|
||||
with open(arg, "r") as f:
|
||||
for line in f:
|
||||
for token in line.strip().split():
|
||||
if token and not token.startswith("#"):
|
||||
urls.append(token)
|
||||
else:
|
||||
urls.append(arg)
|
||||
return urls
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Download videos from LuxureTV (headless, concurrent, multithreaded)."
|
||||
)
|
||||
parser.add_argument(
|
||||
"urls",
|
||||
nargs="*",
|
||||
help="Playlist URLs, direct video URLs, or .txt files containing them. Default: user playlist.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--concurrency",
|
||||
type=int,
|
||||
default=3,
|
||||
help="Number of concurrent downloads (default: 3).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--skip-existing",
|
||||
action="store_true",
|
||||
default=True,
|
||||
help="Skip already-downloaded files (default: true).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--no-skip-existing",
|
||||
action="store_false",
|
||||
dest="skip_existing",
|
||||
help="Re-download existing files.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--user-playlist",
|
||||
default="https://en.luxuretv.com/playlist-by-user/267285/",
|
||||
help="Default user playlist URL (used when no URLs or files given).",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
raw = args.urls if args.urls else [args.user_playlist]
|
||||
targets = resolve_urls(raw)
|
||||
|
||||
if not targets:
|
||||
sys.exit("No URLs found in arguments or file.")
|
||||
|
||||
asyncio.run(process_urls(targets, args.concurrency, args.skip_existing))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
Reference in new issue
Block a user