#!/usr/bin/env python3 import os import sys import json import re import argparse import time import requests from pathlib import Path from urllib.parse import urlparse from concurrent.futures import ThreadPoolExecutor, as_completed from tqdm import tqdm # Add parent workspace directory to path to import scraper_core sys.path.append(str(Path(__file__).resolve().parents[1])) import scraper_core HEADERS = { 'User-Agent': scraper_core.USER_AGENT, 'Accept': 'application/json, text/plain, */*', 'Accept-Language': 'en-US,en;q=0.9', 'Referer': 'https://www.redgifs.com/', 'Origin': 'https://www.redgifs.com', 'DNT': '1', 'Connection': 'keep-alive', } class RedGIFsAPI: """Headless browser wrapper for the RedGIFs API (token-bound to browser).""" def __init__(self): self._pw_cm = None self._browser = None self._page = None self._token = None def start(self): tqdm.write(' Obtaining RedGIFs API token ...') from playwright.sync_api import sync_playwright self._pw_cm = sync_playwright() pw = self._pw_cm.__enter__() self._browser = pw.chromium.launch(headless=True) context = self._browser.new_context( user_agent=HEADERS['User-Agent'], viewport={'width': 1920, 'height': 1080}, ) self._page = context.new_page() self._page.goto( 'https://www.redgifs.com/', wait_until='domcontentloaded', timeout=30000, ) self.obtain_token() def obtain_token(self, retries=3): """Fetches or refreshes the temporary token from the RedGIFs API.""" last_err = None for attempt in range(retries): try: # Ensure we're on a valid redgifs page (not a Cloudflare challenge redirect) cur = self._page.url if 'redgifs' not in cur or 'challenge' in cur: self._page.goto('https://www.redgifs.com/', wait_until='networkidle', timeout=30000) elif cur != 'https://www.redgifs.com/': self._page.goto('https://www.redgifs.com/', wait_until='domcontentloaded', timeout=30000) result = self._page.evaluate('''async () => { try { const r = await fetch('https://api.redgifs.com/v2/auth/temporary'); if (!r.ok) { return {error: "HTTP " + r.status + ": " + await r.text()}; } const d = await r.json(); return {token: d.token}; } catch(e) { return {error: e.message || String(e)}; } }''') if isinstance(result, dict) and 'error' in result: raise RuntimeError(result['error']) self._token = result if isinstance(result, str) else result.get('token') if isinstance(result, dict) else result return except Exception as e: last_err = e if attempt < retries - 1: wait = 5 * (attempt + 1) tqdm.write(f' Token fetch failed (attempt {attempt+1}/{retries}): {e}. Retrying in {wait}s...') time.sleep(wait) try: self._page.goto('https://www.redgifs.com/', wait_until='domcontentloaded', timeout=30000) except Exception: pass raise RuntimeError(f'Failed to obtain token after {retries} attempts: {last_err}') def stop(self): if self._browser: try: self._browser.close() except Exception: pass if self._pw_cm: try: self._pw_cm.__exit__(None, None, None) except Exception: pass def get_all_creator_videos(self, username, fresh_token=True): """Fetch all videos for a creator. Automatically refreshes tokens if expired.""" if fresh_token or not self._token: self.obtain_token() raw = self._page.evaluate('''async ({username, token}) => { let allGifs = []; let page = 1; let pages = 1; let total = 0; let currentToken = token; let errorMsg = ''; const sleep = ms => new Promise(r => setTimeout(r, ms)); try { while (page <= pages) { let r; while (true) { try { r = await fetch( 'https://api.redgifs.com/v2/users/' + encodeURIComponent(username) + '/search?page=' + page + '&count=80&order=latest&type=g', {headers: {'Authorization': 'Bearer ' + currentToken}} ); } catch (networkErr) { errorMsg = 'Network error: ' + (networkErr.message || String(networkErr)); r = null; break; } if (r.status === 401 || r.status === 403) { try { const tokRes = await fetch('https://api.redgifs.com/v2/auth/temporary'); if (tokRes.status === 200) { const tokData = await tokRes.json(); currentToken = tokData.token; continue; } } catch (e) {} } if (r.status === 429) { let rawBody = ''; try { rawBody = await r.text(); } catch (e) {} errorMsg = 'API status 429: ' + rawBody; break; } break; } if (errorMsg || !r || r.status !== 200) { if (!errorMsg && r) { errorMsg = 'API status ' + r.status + ': ' + await r.text(); } break; } const d = await r.json(); if (page === 1) { pages = Math.min(d.pages || 1, 100); total = d.total || 0; } for (const g of (d.gifs || [])) { allGifs.push({urls: g.urls, id: g.id}); } page++; await sleep(1500); // 1.5s delay between pages } } catch (outerErr) { errorMsg = 'Unexpected error: ' + (outerErr.message || String(outerErr)); } return JSON.stringify({gifs: allGifs, total: total, error: errorMsg}); }''', {'username': username, 'token': self._token}) data = json.loads(raw) if data.get('error'): raise RuntimeError(data['error']) return data.get('gifs', []), int(data.get('total', 0)) def extract_filename(url): return urlparse(url).path.split('/')[-1] def download_creator_videos(username, output_dir, gifs, total, concurrency, skip_existing): creator_dir = output_dir / username creator_dir.mkdir(parents=True, exist_ok=True) links_file = creator_dir / 'links.txt' master_links_file = output_dir / 'all_links.txt' download_urls = [] link_lines = [] for gif in gifs: urls = gif['urls'] dl_url = urls.get('hd') or urls.get('sd') or next(iter(urls.values())) filename = extract_filename(dl_url) filepath = creator_dir / filename download_urls.append((dl_url, filepath)) link_lines.append(f'{dl_url}\n') # Save url lists with open(links_file, 'w') as f: f.writelines(link_lines) with open(master_links_file, 'a') as f: f.writelines(link_lines) if len(gifs) < total: tqdm.write(f"Downloading {len(gifs)} of {total} video(s) (capped by API page limit) for {username} with concurrency {concurrency} ...") else: tqdm.write(f"Downloading {total} video(s) for {username} with concurrency {concurrency} ...") bar_pool = scraper_core.BarPositionPool(concurrency) overall_bar = tqdm( total=len(gifs), desc=f"Creator: {username}", position=0, leave=True, ncols=80, ) def download_one(dl_url, filepath): if skip_existing and filepath.exists() and filepath.stat().st_size > 0: try: head = requests.head(dl_url, headers=HEADERS, timeout=10, allow_redirects=True) expected = int(head.headers.get('Content-Length', 0)) if expected > 0: current = filepath.stat().st_size if current == expected: overall_bar.update(1) return if current < expected: filepath.unlink() except Exception: pass pos = bar_pool.acquire() or 1 success = scraper_core.download_file( dl_url, filepath, HEADERS, pos ) bar_pool.release(pos) overall_bar.update(1) with ThreadPoolExecutor(max_workers=concurrency) as executor: futures = [executor.submit(download_one, url, path) for url, path in download_urls] for f in as_completed(futures): f.result() overall_bar.close() # Clean up console lines sys.stdout.write("\n" * (concurrency + 1)) sys.stdout.flush() def main(): parser = argparse.ArgumentParser( description='Download all videos from RedGIFs creator(s)', ) parser.add_argument( 'creators', nargs='*', help='Creator username(s) - space-separated', ) parser.add_argument( '-o', '--output', default=str(Path(__file__).resolve().parent / 'videos'), help='Output directory (default: videos subfolder)', ) parser.add_argument( '--concurrency', type=int, default=5, help='Number of concurrent downloads (default: 5)', ) parser.add_argument( '--skip-existing', action='store_true', default=True, help='Skip already-downloaded files (default: true)', ) parser.add_argument( '--no-skip-existing', action='store_false', dest='skip_existing', help='Re-download existing files', ) parser.add_argument( '--update-all-creators', action='store_true', help='Gets the name of each creator folder, adds them to creators.txt, removes duplicate entries, and then uses creators.txt as the argument', ) args = parser.parse_args() creators_file = Path(__file__).resolve().parent / 'creators.txt' if args.update_all_creators: output_dir = Path(args.output).resolve() found_folders = [] if output_dir.exists(): for entry in output_dir.iterdir(): if entry.is_dir(): found_folders.append(entry.name) existing_creators = [] if creators_file.is_file(): try: with open(creators_file, 'r', encoding='utf-8') as f: for line in f: stripped = line.strip() if not stripped or stripped.startswith('#'): continue for part in re.split(r'[,\s]+', stripped): if part: existing_creators.append(part) except Exception as e: tqdm.write(f"Error reading creators.txt: {e}") all_creators = sorted(list(set(existing_creators + found_folders))) try: with open(creators_file, 'w', encoding='utf-8') as f: f.write(' '.join(all_creators)) tqdm.write(f"Updated creators.txt with {len(all_creators)} unique creators.") except Exception as e: tqdm.write(f"Error writing to creators.txt: {e}") args.creators = [str(creators_file)] creators = args.creators if not creators: raw = input('RedGIFs creator username(s) (comma/space separated): ').strip() creators = [c.strip() for c in re.split(r'[,\s]+', raw) if c.strip()] else: resolved = [] for c in creators: p = Path(c) if p.is_file(): try: with open(p, 'r', encoding='utf-8') as f: for line in f: stripped = line.strip() if not stripped or stripped.startswith('#'): continue for part in re.split(r'[,\s]+', stripped): if part: resolved.append(part) except Exception as e: tqdm.write(f"Error reading creators from file {c}: {e}") else: resolved.append(c) creators = resolved seen = set() creators = [c for c in creators if not (c in seen or seen.add(c))] if not creators: sys.exit('No creators provided.') output_dir = Path(args.output).resolve() output_dir.mkdir(parents=True, exist_ok=True) api = RedGIFsAPI() try: api.start() except Exception as e: sys.exit(f'Failed to initialise Playwright browser: {e}') try: total_creators = len(creators) for idx, username in enumerate(creators, 1): sep = '-' * 3 tqdm.write(f'\n{sep} [{idx}/{total_creators}] {username} {sep}') tqdm.write(f' Fetching video URLs for {username} ...') gifs, total = [], 0 success = False # Retry loop with browser reset on failure for attempt in range(1, 4): try: gifs, total = api.get_all_creator_videos(username, fresh_token=(attempt == 1)) success = True break except Exception as e: tqdm.write(f' [ERROR] Attempt {attempt}/3 failed for {username}: {e}') if attempt < 3: err_str = str(e) if '429' in err_str: m = re.search(r'"delay"\s*:\s*(\d+)', err_str) wait_seconds = int(m.group(1)) + 1 if m else 30 tqdm.write(f' Rate limited. Waiting {wait_seconds:.1f}s...') for _ in tqdm(range(int(wait_seconds)), desc=' Waiting', unit='s', ncols=80, leave=False): time.sleep(1) fractional = wait_seconds - int(wait_seconds) if fractional > 0: time.sleep(fractional) else: tqdm.write(' Resetting browser context and waiting 30s before retry...') try: api.stop() except Exception: pass time.sleep(30) try: api.start() except Exception as restart_err: tqdm.write(f' Failed to restart browser: {restart_err}') if success: tqdm.write(f' Found {total} video(s)') if total > 0: download_creator_videos( username, output_dir, gifs, total, args.concurrency, args.skip_existing ) else: tqdm.write(f' [ERROR] Skipping {username} after 3 failed attempts.') # Introduce a small pacing delay to avoid aggressive rate limits time.sleep(2) finally: api.stop() if __name__ == '__main__': main()