Files
niggers/redgifs/redgifs_creator_downloader.py

432 lines
17 KiB
Python

#!/usr/bin/env python3
import os
import sys
import json
import re
import argparse
import time
import requests
from pathlib import Path
from urllib.parse import urlparse
from concurrent.futures import ThreadPoolExecutor, as_completed
from tqdm import tqdm
# Add parent workspace directory to path to import scraper_core
sys.path.append(str(Path(__file__).resolve().parents[1]))
import scraper_core
HEADERS = {
'User-Agent': scraper_core.USER_AGENT,
'Accept': 'application/json, text/plain, */*',
'Accept-Language': 'en-US,en;q=0.9',
'Referer': 'https://www.redgifs.com/',
'Origin': 'https://www.redgifs.com',
'DNT': '1',
'Connection': 'keep-alive',
}
class RedGIFsAPI:
"""Headless browser wrapper for the RedGIFs API (token-bound to browser)."""
def __init__(self):
self._pw_cm = None
self._browser = None
self._page = None
self._token = None
def start(self):
tqdm.write(' Obtaining RedGIFs API token ...')
from playwright.sync_api import sync_playwright
self._pw_cm = sync_playwright()
pw = self._pw_cm.__enter__()
self._browser = pw.chromium.launch(headless=True)
context = self._browser.new_context(
user_agent=HEADERS['User-Agent'],
viewport={'width': 1920, 'height': 1080},
)
self._page = context.new_page()
self._page.goto(
'https://www.redgifs.com/',
wait_until='domcontentloaded',
timeout=30000,
)
self.obtain_token()
def obtain_token(self, retries=3):
"""Fetches or refreshes the temporary token from the RedGIFs API."""
last_err = None
for attempt in range(retries):
try:
# Ensure we're on a valid redgifs page (not a Cloudflare challenge redirect)
cur = self._page.url
if 'redgifs' not in cur or 'challenge' in cur:
self._page.goto('https://www.redgifs.com/', wait_until='networkidle', timeout=30000)
elif cur != 'https://www.redgifs.com/':
self._page.goto('https://www.redgifs.com/', wait_until='domcontentloaded', timeout=30000)
result = self._page.evaluate('''async () => {
try {
const r = await fetch('https://api.redgifs.com/v2/auth/temporary');
if (!r.ok) {
return {error: "HTTP " + r.status + ": " + await r.text()};
}
const d = await r.json();
return {token: d.token};
} catch(e) {
return {error: e.message || String(e)};
}
}''')
if isinstance(result, dict) and 'error' in result:
raise RuntimeError(result['error'])
self._token = result if isinstance(result, str) else result.get('token') if isinstance(result, dict) else result
return
except Exception as e:
last_err = e
if attempt < retries - 1:
wait = 5 * (attempt + 1)
tqdm.write(f' Token fetch failed (attempt {attempt+1}/{retries}): {e}. Retrying in {wait}s...')
time.sleep(wait)
try:
self._page.goto('https://www.redgifs.com/', wait_until='domcontentloaded', timeout=30000)
except Exception:
pass
raise RuntimeError(f'Failed to obtain token after {retries} attempts: {last_err}')
def stop(self):
if self._browser:
try:
self._browser.close()
except Exception:
pass
if self._pw_cm:
try:
self._pw_cm.__exit__(None, None, None)
except Exception:
pass
def get_all_creator_videos(self, username, fresh_token=True):
"""Fetch all videos for a creator. Automatically refreshes tokens if expired."""
if fresh_token or not self._token:
self.obtain_token()
raw = self._page.evaluate('''async ({username, token}) => {
let allGifs = [];
let page = 1;
let pages = 1;
let total = 0;
let currentToken = token;
let errorMsg = '';
const sleep = ms => new Promise(r => setTimeout(r, ms));
try {
while (page <= pages) {
let r;
while (true) {
try {
r = await fetch(
'https://api.redgifs.com/v2/users/'
+ encodeURIComponent(username)
+ '/search?page=' + page + '&count=80&order=latest&type=g',
{headers: {'Authorization': 'Bearer ' + currentToken}}
);
} catch (networkErr) {
errorMsg = 'Network error: ' + (networkErr.message || String(networkErr));
r = null;
break;
}
if (r.status === 401 || r.status === 403) {
try {
const tokRes = await fetch('https://api.redgifs.com/v2/auth/temporary');
if (tokRes.status === 200) {
const tokData = await tokRes.json();
currentToken = tokData.token;
continue;
}
} catch (e) {}
}
if (r.status === 429) {
let rawBody = '';
try { rawBody = await r.text(); } catch (e) {}
errorMsg = 'API status 429: ' + rawBody;
break;
}
break;
}
if (errorMsg || !r || r.status !== 200) {
if (!errorMsg && r) {
errorMsg = 'API status ' + r.status + ': ' + await r.text();
}
break;
}
const d = await r.json();
if (page === 1) {
pages = Math.min(d.pages || 1, 100);
total = d.total || 0;
}
for (const g of (d.gifs || [])) {
allGifs.push({urls: g.urls, id: g.id});
}
page++;
await sleep(1500); // 1.5s delay between pages
}
} catch (outerErr) {
errorMsg = 'Unexpected error: ' + (outerErr.message || String(outerErr));
}
return JSON.stringify({gifs: allGifs, total: total, error: errorMsg});
}''', {'username': username, 'token': self._token})
data = json.loads(raw)
if data.get('error'):
raise RuntimeError(data['error'])
return data.get('gifs', []), int(data.get('total', 0))
def extract_filename(url):
return urlparse(url).path.split('/')[-1]
def download_creator_videos(username, output_dir, gifs, total, concurrency, skip_existing):
creator_dir = output_dir / username
creator_dir.mkdir(parents=True, exist_ok=True)
links_file = creator_dir / 'links.txt'
master_links_file = output_dir / 'all_links.txt'
download_urls = []
link_lines = []
for gif in gifs:
urls = gif['urls']
dl_url = urls.get('hd') or urls.get('sd') or next(iter(urls.values()))
filename = extract_filename(dl_url)
filepath = creator_dir / filename
download_urls.append((dl_url, filepath))
link_lines.append(f'{dl_url}\n')
# Save url lists
with open(links_file, 'w') as f:
f.writelines(link_lines)
with open(master_links_file, 'a') as f:
f.writelines(link_lines)
if len(gifs) < total:
tqdm.write(f"Downloading {len(gifs)} of {total} video(s) (capped by API page limit) for {username} with concurrency {concurrency} ...")
else:
tqdm.write(f"Downloading {total} video(s) for {username} with concurrency {concurrency} ...")
bar_pool = scraper_core.BarPositionPool(concurrency)
overall_bar = tqdm(
total=len(gifs),
desc=f"Creator: {username}",
position=0,
leave=True,
ncols=80,
)
def download_one(dl_url, filepath):
if skip_existing and filepath.exists() and filepath.stat().st_size > 0:
try:
head = requests.head(dl_url, headers=HEADERS, timeout=10, allow_redirects=True)
expected = int(head.headers.get('Content-Length', 0))
if expected > 0:
current = filepath.stat().st_size
if current == expected:
overall_bar.update(1)
return
if current < expected:
filepath.unlink()
except Exception:
pass
pos = bar_pool.acquire() or 1
success = scraper_core.download_file(
dl_url, filepath, HEADERS, pos
)
bar_pool.release(pos)
overall_bar.update(1)
with ThreadPoolExecutor(max_workers=concurrency) as executor:
futures = [executor.submit(download_one, url, path) for url, path in download_urls]
for f in as_completed(futures):
f.result()
overall_bar.close()
# Clean up console lines
sys.stdout.write("\n" * (concurrency + 1))
sys.stdout.flush()
def main():
parser = argparse.ArgumentParser(
description='Download all videos from RedGIFs creator(s)',
)
parser.add_argument(
'creators', nargs='*',
help='Creator username(s) - space-separated',
)
parser.add_argument(
'-o', '--output', default=str(Path(__file__).resolve().parent / 'videos'),
help='Output directory (default: videos subfolder)',
)
parser.add_argument(
'--concurrency', type=int, default=5,
help='Number of concurrent downloads (default: 5)',
)
parser.add_argument(
'--skip-existing', action='store_true', default=True,
help='Skip already-downloaded files (default: true)',
)
parser.add_argument(
'--no-skip-existing', action='store_false', dest='skip_existing',
help='Re-download existing files',
)
parser.add_argument(
'--update-all-creators', action='store_true',
help='Gets the name of each creator folder, adds them to creators.txt, removes duplicate entries, and then uses creators.txt as the argument',
)
args = parser.parse_args()
creators_file = Path(__file__).resolve().parent / 'creators.txt'
if args.update_all_creators:
output_dir = Path(args.output).resolve()
found_folders = []
if output_dir.exists():
for entry in output_dir.iterdir():
if entry.is_dir():
found_folders.append(entry.name)
existing_creators = []
if creators_file.is_file():
try:
with open(creators_file, 'r', encoding='utf-8') as f:
for line in f:
stripped = line.strip()
if not stripped or stripped.startswith('#'):
continue
for part in re.split(r'[,\s]+', stripped):
if part:
existing_creators.append(part)
except Exception as e:
tqdm.write(f"Error reading creators.txt: {e}")
all_creators = sorted(list(set(existing_creators + found_folders)))
try:
with open(creators_file, 'w', encoding='utf-8') as f:
f.write(' '.join(all_creators))
tqdm.write(f"Updated creators.txt with {len(all_creators)} unique creators.")
except Exception as e:
tqdm.write(f"Error writing to creators.txt: {e}")
args.creators = [str(creators_file)]
creators = args.creators
if not creators:
raw = input('RedGIFs creator username(s) (comma/space separated): ').strip()
creators = [c.strip() for c in re.split(r'[,\s]+', raw) if c.strip()]
else:
resolved = []
for c in creators:
p = Path(c)
if p.is_file():
try:
with open(p, 'r', encoding='utf-8') as f:
for line in f:
stripped = line.strip()
if not stripped or stripped.startswith('#'):
continue
for part in re.split(r'[,\s]+', stripped):
if part:
resolved.append(part)
except Exception as e:
tqdm.write(f"Error reading creators from file {c}: {e}")
else:
resolved.append(c)
creators = resolved
seen = set()
creators = [c for c in creators if not (c in seen or seen.add(c))]
if not creators:
sys.exit('No creators provided.')
output_dir = Path(args.output).resolve()
output_dir.mkdir(parents=True, exist_ok=True)
api = RedGIFsAPI()
try:
api.start()
except Exception as e:
sys.exit(f'Failed to initialise Playwright browser: {e}')
try:
total_creators = len(creators)
for idx, username in enumerate(creators, 1):
sep = '-' * 3
tqdm.write(f'\n{sep} [{idx}/{total_creators}] {username} {sep}')
tqdm.write(f' Fetching video URLs for {username} ...')
gifs, total = [], 0
success = False
# Retry loop with browser reset on failure
for attempt in range(1, 4):
try:
gifs, total = api.get_all_creator_videos(username, fresh_token=(attempt == 1))
success = True
break
except Exception as e:
tqdm.write(f' [ERROR] Attempt {attempt}/3 failed for {username}: {e}')
if attempt < 3:
err_str = str(e)
if '429' in err_str:
m = re.search(r'"delay"\s*:\s*(\d+)', err_str)
wait_seconds = int(m.group(1)) + 1 if m else 30
tqdm.write(f' Rate limited. Waiting {wait_seconds:.1f}s...')
for _ in tqdm(range(int(wait_seconds)), desc=' Waiting', unit='s', ncols=80, leave=False):
time.sleep(1)
fractional = wait_seconds - int(wait_seconds)
if fractional > 0:
time.sleep(fractional)
else:
tqdm.write(' Resetting browser context and waiting 30s before retry...')
try:
api.stop()
except Exception:
pass
time.sleep(30)
try:
api.start()
except Exception as restart_err:
tqdm.write(f' Failed to restart browser: {restart_err}')
if success:
tqdm.write(f' Found {total} video(s)')
if total > 0:
download_creator_videos(
username, output_dir, gifs, total,
args.concurrency, args.skip_existing
)
else:
tqdm.write(f' [ERROR] Skipping {username} after 3 failed attempts.')
# Introduce a small pacing delay to avoid aggressive rate limits
time.sleep(2)
finally:
api.stop()
if __name__ == '__main__':
main()