Files
niggers/redgifs/redgifs_downloader.py
T

106 lines
3.6 KiB
Python

import os
import requests
import time
import sys
from concurrent.futures import ThreadPoolExecutor
import threading
# Root directory where folders with links.txt are located
root_dir = r'V:\smalldata\niggers\redgifs\videos'
# Configuration
MAX_WORKERS = 10 # Number of simultaneous downloads
HEADERS = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
'Accept': 'video/webm,video/ogg,video/*;q=0.9,application/ogg;q=0.7,audio/*;q=0.6,*/*;q=0.5',
'Referer': 'https://www.redgifs.com/',
'Connection': 'keep-alive',
}
# Thread-safe progress reporting
progress_lock = threading.Lock()
completed_count = 0
def progress_bar(current, total, bar_length=40):
fraction = current / total if total > 0 else 1
arrow = int(fraction * bar_length - 1) * '=' + '>'
padding = int(bar_length - len(arrow)) * ' '
ending = '\n' if current == total else '\r'
print(f'Progress: [{arrow}{padding}] {current}/{total} ({fraction*100:.2f}%)', end=ending)
def download_url(url, download_dir, folder_failed_log, total):
global completed_count
filename = os.path.join(download_dir, os.path.basename(url))
# Skip if already downloaded
if os.path.exists(filename) and os.path.getsize(filename) > 0:
with progress_lock:
completed_count += 1
progress_bar(completed_count, total)
return
retries = 3
success = False
while retries > 0 and not success:
try:
response = requests.get(url, headers=HEADERS, stream=True, timeout=30)
if response.status_code == 200:
with open(filename, 'wb') as f:
for chunk in response.iter_content(chunk_size=8192):
f.write(chunk)
success = True
time.sleep(0.1) # Minimal sleep for parallel threads
elif response.status_code == 429:
# If rate limited, we wait longer but it affects all threads
time.sleep(10)
retries -= 1
else:
retries = 0
except Exception:
retries -= 1
time.sleep(2)
if not success:
with progress_lock:
with open(folder_failed_log, 'a') as fail_log:
fail_log.write(f"{url}\n")
with progress_lock:
completed_count += 1
progress_bar(completed_count, total)
def download_from_file(file_path, download_dir):
global completed_count
if not os.path.exists(file_path):
return
folder_failed_log = os.path.join(download_dir, 'failed_downloads.txt')
with open(file_path, 'r', encoding='utf-8') as f:
urls = [line.strip() for line in f if line.strip() and not line.strip().startswith('#')]
total = len(urls)
folder_name = os.path.basename(download_dir)
print(f"\nProcessing folder: {folder_name} ({total} files)")
completed_count = 0
with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
for url in urls:
executor.submit(download_url, url, download_dir, folder_failed_log, total)
# Main execution
if not os.path.exists(root_dir):
print(f"Root directory not found: {root_dir}")
sys.exit(1)
subfolders = [f.path for f in os.scandir(root_dir) if f.is_dir()]
print(f"Found {len(subfolders)} folders. Starting parallel batch download...")
for folder in subfolders:
links_file = os.path.join(folder, 'links.txt')
if os.path.exists(links_file):
download_from_file(links_file, folder)
print("\nBatch download process complete.")