432 lines
17 KiB
Python
432 lines
17 KiB
Python
#!/usr/bin/env python3
|
|
import os
|
|
import sys
|
|
import json
|
|
import re
|
|
import argparse
|
|
import time
|
|
import requests
|
|
from pathlib import Path
|
|
from urllib.parse import urlparse
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from tqdm import tqdm
|
|
|
|
# Add parent workspace directory to path to import scraper_core
|
|
sys.path.append(str(Path(__file__).resolve().parents[1]))
|
|
import scraper_core
|
|
|
|
HEADERS = {
|
|
'User-Agent': scraper_core.USER_AGENT,
|
|
'Accept': 'application/json, text/plain, */*',
|
|
'Accept-Language': 'en-US,en;q=0.9',
|
|
'Referer': 'https://www.redgifs.com/',
|
|
'Origin': 'https://www.redgifs.com',
|
|
'DNT': '1',
|
|
'Connection': 'keep-alive',
|
|
}
|
|
|
|
class RedGIFsAPI:
|
|
"""Headless browser wrapper for the RedGIFs API (token-bound to browser)."""
|
|
def __init__(self):
|
|
self._pw_cm = None
|
|
self._browser = None
|
|
self._page = None
|
|
self._token = None
|
|
|
|
def start(self):
|
|
tqdm.write(' Obtaining RedGIFs API token ...')
|
|
from playwright.sync_api import sync_playwright
|
|
self._pw_cm = sync_playwright()
|
|
pw = self._pw_cm.__enter__()
|
|
self._browser = pw.chromium.launch(headless=True)
|
|
context = self._browser.new_context(
|
|
user_agent=HEADERS['User-Agent'],
|
|
viewport={'width': 1920, 'height': 1080},
|
|
)
|
|
self._page = context.new_page()
|
|
self._page.goto(
|
|
'https://www.redgifs.com/',
|
|
wait_until='domcontentloaded',
|
|
timeout=30000,
|
|
)
|
|
self.obtain_token()
|
|
|
|
def obtain_token(self, retries=3):
|
|
"""Fetches or refreshes the temporary token from the RedGIFs API."""
|
|
last_err = None
|
|
for attempt in range(retries):
|
|
try:
|
|
# Ensure we're on a valid redgifs page (not a Cloudflare challenge redirect)
|
|
cur = self._page.url
|
|
if 'redgifs' not in cur or 'challenge' in cur:
|
|
self._page.goto('https://www.redgifs.com/', wait_until='networkidle', timeout=30000)
|
|
elif cur != 'https://www.redgifs.com/':
|
|
self._page.goto('https://www.redgifs.com/', wait_until='domcontentloaded', timeout=30000)
|
|
|
|
result = self._page.evaluate('''async () => {
|
|
try {
|
|
const r = await fetch('https://api.redgifs.com/v2/auth/temporary');
|
|
if (!r.ok) {
|
|
return {error: "HTTP " + r.status + ": " + await r.text()};
|
|
}
|
|
const d = await r.json();
|
|
return {token: d.token};
|
|
} catch(e) {
|
|
return {error: e.message || String(e)};
|
|
}
|
|
}''')
|
|
|
|
if isinstance(result, dict) and 'error' in result:
|
|
raise RuntimeError(result['error'])
|
|
self._token = result if isinstance(result, str) else result.get('token') if isinstance(result, dict) else result
|
|
return
|
|
except Exception as e:
|
|
last_err = e
|
|
if attempt < retries - 1:
|
|
wait = 5 * (attempt + 1)
|
|
tqdm.write(f' Token fetch failed (attempt {attempt+1}/{retries}): {e}. Retrying in {wait}s...')
|
|
time.sleep(wait)
|
|
try:
|
|
self._page.goto('https://www.redgifs.com/', wait_until='domcontentloaded', timeout=30000)
|
|
except Exception:
|
|
pass
|
|
raise RuntimeError(f'Failed to obtain token after {retries} attempts: {last_err}')
|
|
|
|
def stop(self):
|
|
if self._browser:
|
|
try:
|
|
self._browser.close()
|
|
except Exception:
|
|
pass
|
|
if self._pw_cm:
|
|
try:
|
|
self._pw_cm.__exit__(None, None, None)
|
|
except Exception:
|
|
pass
|
|
|
|
def get_all_creator_videos(self, username, fresh_token=True):
|
|
"""Fetch all videos for a creator. Automatically refreshes tokens if expired."""
|
|
if fresh_token or not self._token:
|
|
self.obtain_token()
|
|
|
|
raw = self._page.evaluate('''async ({username, token}) => {
|
|
let allGifs = [];
|
|
let page = 1;
|
|
let pages = 1;
|
|
let total = 0;
|
|
let currentToken = token;
|
|
let errorMsg = '';
|
|
|
|
const sleep = ms => new Promise(r => setTimeout(r, ms));
|
|
|
|
try {
|
|
while (page <= pages) {
|
|
let r;
|
|
|
|
while (true) {
|
|
try {
|
|
r = await fetch(
|
|
'https://api.redgifs.com/v2/users/'
|
|
+ encodeURIComponent(username)
|
|
+ '/search?page=' + page + '&count=80&order=latest&type=g',
|
|
{headers: {'Authorization': 'Bearer ' + currentToken}}
|
|
);
|
|
} catch (networkErr) {
|
|
errorMsg = 'Network error: ' + (networkErr.message || String(networkErr));
|
|
r = null;
|
|
break;
|
|
}
|
|
|
|
if (r.status === 401 || r.status === 403) {
|
|
try {
|
|
const tokRes = await fetch('https://api.redgifs.com/v2/auth/temporary');
|
|
if (tokRes.status === 200) {
|
|
const tokData = await tokRes.json();
|
|
currentToken = tokData.token;
|
|
continue;
|
|
}
|
|
} catch (e) {}
|
|
}
|
|
|
|
if (r.status === 429) {
|
|
let rawBody = '';
|
|
try { rawBody = await r.text(); } catch (e) {}
|
|
errorMsg = 'API status 429: ' + rawBody;
|
|
break;
|
|
}
|
|
|
|
break;
|
|
}
|
|
|
|
if (errorMsg || !r || r.status !== 200) {
|
|
if (!errorMsg && r) {
|
|
errorMsg = 'API status ' + r.status + ': ' + await r.text();
|
|
}
|
|
break;
|
|
}
|
|
|
|
const d = await r.json();
|
|
if (page === 1) {
|
|
pages = Math.min(d.pages || 1, 100);
|
|
total = d.total || 0;
|
|
}
|
|
for (const g of (d.gifs || [])) {
|
|
allGifs.push({urls: g.urls, id: g.id});
|
|
}
|
|
page++;
|
|
await sleep(1500); // 1.5s delay between pages
|
|
}
|
|
} catch (outerErr) {
|
|
errorMsg = 'Unexpected error: ' + (outerErr.message || String(outerErr));
|
|
}
|
|
|
|
return JSON.stringify({gifs: allGifs, total: total, error: errorMsg});
|
|
}''', {'username': username, 'token': self._token})
|
|
|
|
data = json.loads(raw)
|
|
if data.get('error'):
|
|
raise RuntimeError(data['error'])
|
|
|
|
return data.get('gifs', []), int(data.get('total', 0))
|
|
|
|
|
|
def extract_filename(url):
|
|
return urlparse(url).path.split('/')[-1]
|
|
|
|
|
|
def download_creator_videos(username, output_dir, gifs, total, concurrency, skip_existing):
|
|
creator_dir = output_dir / username
|
|
creator_dir.mkdir(parents=True, exist_ok=True)
|
|
links_file = creator_dir / 'links.txt'
|
|
master_links_file = output_dir / 'all_links.txt'
|
|
|
|
download_urls = []
|
|
link_lines = []
|
|
for gif in gifs:
|
|
urls = gif['urls']
|
|
dl_url = urls.get('hd') or urls.get('sd') or next(iter(urls.values()))
|
|
filename = extract_filename(dl_url)
|
|
filepath = creator_dir / filename
|
|
download_urls.append((dl_url, filepath))
|
|
link_lines.append(f'{dl_url}\n')
|
|
|
|
# Save url lists
|
|
with open(links_file, 'w') as f:
|
|
f.writelines(link_lines)
|
|
with open(master_links_file, 'a') as f:
|
|
f.writelines(link_lines)
|
|
|
|
if len(gifs) < total:
|
|
tqdm.write(f"Downloading {len(gifs)} of {total} video(s) (capped by API page limit) for {username} with concurrency {concurrency} ...")
|
|
else:
|
|
tqdm.write(f"Downloading {total} video(s) for {username} with concurrency {concurrency} ...")
|
|
|
|
bar_pool = scraper_core.BarPositionPool(concurrency)
|
|
overall_bar = tqdm(
|
|
total=len(gifs),
|
|
desc=f"Creator: {username}",
|
|
position=0,
|
|
leave=True,
|
|
ncols=80,
|
|
)
|
|
|
|
def download_one(dl_url, filepath):
|
|
if skip_existing and filepath.exists() and filepath.stat().st_size > 0:
|
|
try:
|
|
head = requests.head(dl_url, headers=HEADERS, timeout=10, allow_redirects=True)
|
|
expected = int(head.headers.get('Content-Length', 0))
|
|
if expected > 0:
|
|
current = filepath.stat().st_size
|
|
if current == expected:
|
|
overall_bar.update(1)
|
|
return
|
|
if current < expected:
|
|
filepath.unlink()
|
|
except Exception:
|
|
pass
|
|
|
|
pos = bar_pool.acquire() or 1
|
|
|
|
success = scraper_core.download_file(
|
|
dl_url, filepath, HEADERS, pos
|
|
)
|
|
|
|
bar_pool.release(pos)
|
|
overall_bar.update(1)
|
|
|
|
with ThreadPoolExecutor(max_workers=concurrency) as executor:
|
|
futures = [executor.submit(download_one, url, path) for url, path in download_urls]
|
|
for f in as_completed(futures):
|
|
f.result()
|
|
|
|
overall_bar.close()
|
|
|
|
# Clean up console lines
|
|
sys.stdout.write("\n" * (concurrency + 1))
|
|
sys.stdout.flush()
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(
|
|
description='Download all videos from RedGIFs creator(s)',
|
|
)
|
|
parser.add_argument(
|
|
'creators', nargs='*',
|
|
help='Creator username(s) - space-separated',
|
|
)
|
|
parser.add_argument(
|
|
'-o', '--output', default=str(Path(__file__).resolve().parent / 'videos'),
|
|
help='Output directory (default: videos subfolder)',
|
|
)
|
|
parser.add_argument(
|
|
'--concurrency', type=int, default=5,
|
|
help='Number of concurrent downloads (default: 5)',
|
|
)
|
|
parser.add_argument(
|
|
'--skip-existing', action='store_true', default=True,
|
|
help='Skip already-downloaded files (default: true)',
|
|
)
|
|
parser.add_argument(
|
|
'--no-skip-existing', action='store_false', dest='skip_existing',
|
|
help='Re-download existing files',
|
|
)
|
|
parser.add_argument(
|
|
'--update-all-creators', action='store_true',
|
|
help='Gets the name of each creator folder, adds them to creators.txt, removes duplicate entries, and then uses creators.txt as the argument',
|
|
)
|
|
args = parser.parse_args()
|
|
|
|
creators_file = Path(__file__).resolve().parent / 'creators.txt'
|
|
|
|
if args.update_all_creators:
|
|
output_dir = Path(args.output).resolve()
|
|
found_folders = []
|
|
if output_dir.exists():
|
|
for entry in output_dir.iterdir():
|
|
if entry.is_dir():
|
|
found_folders.append(entry.name)
|
|
|
|
existing_creators = []
|
|
if creators_file.is_file():
|
|
try:
|
|
with open(creators_file, 'r', encoding='utf-8') as f:
|
|
for line in f:
|
|
stripped = line.strip()
|
|
if not stripped or stripped.startswith('#'):
|
|
continue
|
|
for part in re.split(r'[,\s]+', stripped):
|
|
if part:
|
|
existing_creators.append(part)
|
|
except Exception as e:
|
|
tqdm.write(f"Error reading creators.txt: {e}")
|
|
|
|
all_creators = sorted(list(set(existing_creators + found_folders)))
|
|
try:
|
|
with open(creators_file, 'w', encoding='utf-8') as f:
|
|
f.write(' '.join(all_creators))
|
|
tqdm.write(f"Updated creators.txt with {len(all_creators)} unique creators.")
|
|
except Exception as e:
|
|
tqdm.write(f"Error writing to creators.txt: {e}")
|
|
|
|
args.creators = [str(creators_file)]
|
|
|
|
creators = args.creators
|
|
if not creators:
|
|
raw = input('RedGIFs creator username(s) (comma/space separated): ').strip()
|
|
creators = [c.strip() for c in re.split(r'[,\s]+', raw) if c.strip()]
|
|
else:
|
|
resolved = []
|
|
for c in creators:
|
|
p = Path(c)
|
|
if p.is_file():
|
|
try:
|
|
with open(p, 'r', encoding='utf-8') as f:
|
|
for line in f:
|
|
stripped = line.strip()
|
|
if not stripped or stripped.startswith('#'):
|
|
continue
|
|
for part in re.split(r'[,\s]+', stripped):
|
|
if part:
|
|
resolved.append(part)
|
|
except Exception as e:
|
|
tqdm.write(f"Error reading creators from file {c}: {e}")
|
|
else:
|
|
resolved.append(c)
|
|
creators = resolved
|
|
|
|
seen = set()
|
|
creators = [c for c in creators if not (c in seen or seen.add(c))]
|
|
|
|
if not creators:
|
|
sys.exit('No creators provided.')
|
|
|
|
output_dir = Path(args.output).resolve()
|
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
api = RedGIFsAPI()
|
|
try:
|
|
api.start()
|
|
except Exception as e:
|
|
sys.exit(f'Failed to initialise Playwright browser: {e}')
|
|
|
|
try:
|
|
total_creators = len(creators)
|
|
for idx, username in enumerate(creators, 1):
|
|
sep = '-' * 3
|
|
tqdm.write(f'\n{sep} [{idx}/{total_creators}] {username} {sep}')
|
|
tqdm.write(f' Fetching video URLs for {username} ...')
|
|
|
|
gifs, total = [], 0
|
|
success = False
|
|
|
|
# Retry loop with browser reset on failure
|
|
for attempt in range(1, 4):
|
|
try:
|
|
gifs, total = api.get_all_creator_videos(username, fresh_token=(attempt == 1))
|
|
success = True
|
|
break
|
|
except Exception as e:
|
|
tqdm.write(f' [ERROR] Attempt {attempt}/3 failed for {username}: {e}')
|
|
if attempt < 3:
|
|
err_str = str(e)
|
|
if '429' in err_str:
|
|
m = re.search(r'"delay"\s*:\s*(\d+)', err_str)
|
|
wait_seconds = int(m.group(1)) + 1 if m else 30
|
|
tqdm.write(f' Rate limited. Waiting {wait_seconds:.1f}s...')
|
|
for _ in tqdm(range(int(wait_seconds)), desc=' Waiting', unit='s', ncols=80, leave=False):
|
|
time.sleep(1)
|
|
fractional = wait_seconds - int(wait_seconds)
|
|
if fractional > 0:
|
|
time.sleep(fractional)
|
|
else:
|
|
tqdm.write(' Resetting browser context and waiting 30s before retry...')
|
|
try:
|
|
api.stop()
|
|
except Exception:
|
|
pass
|
|
time.sleep(30)
|
|
try:
|
|
api.start()
|
|
except Exception as restart_err:
|
|
tqdm.write(f' Failed to restart browser: {restart_err}')
|
|
|
|
if success:
|
|
tqdm.write(f' Found {total} video(s)')
|
|
if total > 0:
|
|
download_creator_videos(
|
|
username, output_dir, gifs, total,
|
|
args.concurrency, args.skip_existing
|
|
)
|
|
else:
|
|
tqdm.write(f' [ERROR] Skipping {username} after 3 failed attempts.')
|
|
|
|
# Introduce a small pacing delay to avoid aggressive rate limits
|
|
time.sleep(2)
|
|
|
|
finally:
|
|
api.stop()
|
|
|
|
|
|
if __name__ == '__main__':
|
|
main()
|