import os import requests import time import sys from concurrent.futures import ThreadPoolExecutor import threading # Root directory where folders with links.txt are located root_dir = r'V:\smalldata\niggers\redgifs\videos' # Configuration MAX_WORKERS = 10 # Number of simultaneous downloads HEADERS = { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36', 'Accept': 'video/webm,video/ogg,video/*;q=0.9,application/ogg;q=0.7,audio/*;q=0.6,*/*;q=0.5', 'Referer': 'https://www.redgifs.com/', 'Connection': 'keep-alive', } # Thread-safe progress reporting progress_lock = threading.Lock() completed_count = 0 def progress_bar(current, total, bar_length=40): fraction = current / total if total > 0 else 1 arrow = int(fraction * bar_length - 1) * '=' + '>' padding = int(bar_length - len(arrow)) * ' ' ending = '\n' if current == total else '\r' print(f'Progress: [{arrow}{padding}] {current}/{total} ({fraction*100:.2f}%)', end=ending) def download_url(url, download_dir, folder_failed_log, total): global completed_count filename = os.path.join(download_dir, os.path.basename(url)) # Skip if already downloaded if os.path.exists(filename) and os.path.getsize(filename) > 0: with progress_lock: completed_count += 1 progress_bar(completed_count, total) return retries = 3 success = False while retries > 0 and not success: try: response = requests.get(url, headers=HEADERS, stream=True, timeout=30) if response.status_code == 200: with open(filename, 'wb') as f: for chunk in response.iter_content(chunk_size=8192): f.write(chunk) success = True time.sleep(0.1) # Minimal sleep for parallel threads elif response.status_code == 429: # If rate limited, we wait longer but it affects all threads time.sleep(10) retries -= 1 else: retries = 0 except Exception: retries -= 1 time.sleep(2) if not success: with progress_lock: with open(folder_failed_log, 'a') as fail_log: fail_log.write(f"{url}\n") with progress_lock: completed_count += 1 progress_bar(completed_count, total) def download_from_file(file_path, download_dir): global completed_count if not os.path.exists(file_path): return folder_failed_log = os.path.join(download_dir, 'failed_downloads.txt') with open(file_path, 'r', encoding='utf-8') as f: urls = [line.strip() for line in f if line.strip() and not line.strip().startswith('#')] total = len(urls) folder_name = os.path.basename(download_dir) print(f"\nProcessing folder: {folder_name} ({total} files)") completed_count = 0 with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor: for url in urls: executor.submit(download_url, url, download_dir, folder_failed_log, total) # Main execution if not os.path.exists(root_dir): print(f"Root directory not found: {root_dir}") sys.exit(1) subfolders = [f.path for f in os.scandir(root_dir) if f.is_dir()] print(f"Found {len(subfolders)} folders. Starting parallel batch download...") for folder in subfolders: links_file = os.path.join(folder, 'links.txt') if os.path.exists(links_file): download_from_file(links_file, folder) print("\nBatch download process complete.")