Initial backup: folder structure manifest, tree summary, scraper scripts, and text metadata
This commit is contained in:
commit
f164832dcc
10504 files changed
+2423594
No files matched your search
@@ -0,0 +1,431 @@
|
||||
#!/usr/bin/env python3
|
||||
import os
|
||||
import sys
|
||||
import json
|
||||
import re
|
||||
import argparse
|
||||
import time
|
||||
import requests
|
||||
from pathlib import Path
|
||||
from urllib.parse import urlparse
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from tqdm import tqdm
|
||||
|
||||
# Add parent workspace directory to path to import scraper_core
|
||||
sys.path.append(str(Path(__file__).resolve().parents[1]))
|
||||
import scraper_core
|
||||
|
||||
HEADERS = {
|
||||
'User-Agent': scraper_core.USER_AGENT,
|
||||
'Accept': 'application/json, text/plain, */*',
|
||||
'Accept-Language': 'en-US,en;q=0.9',
|
||||
'Referer': 'https://www.redgifs.com/',
|
||||
'Origin': 'https://www.redgifs.com',
|
||||
'DNT': '1',
|
||||
'Connection': 'keep-alive',
|
||||
}
|
||||
|
||||
class RedGIFsAPI:
|
||||
"""Headless browser wrapper for the RedGIFs API (token-bound to browser)."""
|
||||
def __init__(self):
|
||||
self._pw_cm = None
|
||||
self._browser = None
|
||||
self._page = None
|
||||
self._token = None
|
||||
|
||||
def start(self):
|
||||
tqdm.write(' Obtaining RedGIFs API token ...')
|
||||
from playwright.sync_api import sync_playwright
|
||||
self._pw_cm = sync_playwright()
|
||||
pw = self._pw_cm.__enter__()
|
||||
self._browser = pw.chromium.launch(headless=True)
|
||||
context = self._browser.new_context(
|
||||
user_agent=HEADERS['User-Agent'],
|
||||
viewport={'width': 1920, 'height': 1080},
|
||||
)
|
||||
self._page = context.new_page()
|
||||
self._page.goto(
|
||||
'https://www.redgifs.com/',
|
||||
wait_until='domcontentloaded',
|
||||
timeout=30000,
|
||||
)
|
||||
self.obtain_token()
|
||||
|
||||
def obtain_token(self, retries=3):
|
||||
"""Fetches or refreshes the temporary token from the RedGIFs API."""
|
||||
last_err = None
|
||||
for attempt in range(retries):
|
||||
try:
|
||||
# Ensure we're on a valid redgifs page (not a Cloudflare challenge redirect)
|
||||
cur = self._page.url
|
||||
if 'redgifs' not in cur or 'challenge' in cur:
|
||||
self._page.goto('https://www.redgifs.com/', wait_until='networkidle', timeout=30000)
|
||||
elif cur != 'https://www.redgifs.com/':
|
||||
self._page.goto('https://www.redgifs.com/', wait_until='domcontentloaded', timeout=30000)
|
||||
|
||||
result = self._page.evaluate('''async () => {
|
||||
try {
|
||||
const r = await fetch('https://api.redgifs.com/v2/auth/temporary');
|
||||
if (!r.ok) {
|
||||
return {error: "HTTP " + r.status + ": " + await r.text()};
|
||||
}
|
||||
const d = await r.json();
|
||||
return {token: d.token};
|
||||
} catch(e) {
|
||||
return {error: e.message || String(e)};
|
||||
}
|
||||
}''')
|
||||
|
||||
if isinstance(result, dict) and 'error' in result:
|
||||
raise RuntimeError(result['error'])
|
||||
self._token = result if isinstance(result, str) else result.get('token') if isinstance(result, dict) else result
|
||||
return
|
||||
except Exception as e:
|
||||
last_err = e
|
||||
if attempt < retries - 1:
|
||||
wait = 5 * (attempt + 1)
|
||||
tqdm.write(f' Token fetch failed (attempt {attempt+1}/{retries}): {e}. Retrying in {wait}s...')
|
||||
time.sleep(wait)
|
||||
try:
|
||||
self._page.goto('https://www.redgifs.com/', wait_until='domcontentloaded', timeout=30000)
|
||||
except Exception:
|
||||
pass
|
||||
raise RuntimeError(f'Failed to obtain token after {retries} attempts: {last_err}')
|
||||
|
||||
def stop(self):
|
||||
if self._browser:
|
||||
try:
|
||||
self._browser.close()
|
||||
except Exception:
|
||||
pass
|
||||
if self._pw_cm:
|
||||
try:
|
||||
self._pw_cm.__exit__(None, None, None)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def get_all_creator_videos(self, username, fresh_token=True):
|
||||
"""Fetch all videos for a creator. Automatically refreshes tokens if expired."""
|
||||
if fresh_token or not self._token:
|
||||
self.obtain_token()
|
||||
|
||||
raw = self._page.evaluate('''async ({username, token}) => {
|
||||
let allGifs = [];
|
||||
let page = 1;
|
||||
let pages = 1;
|
||||
let total = 0;
|
||||
let currentToken = token;
|
||||
let errorMsg = '';
|
||||
|
||||
const sleep = ms => new Promise(r => setTimeout(r, ms));
|
||||
|
||||
try {
|
||||
while (page <= pages) {
|
||||
let r;
|
||||
|
||||
while (true) {
|
||||
try {
|
||||
r = await fetch(
|
||||
'https://api.redgifs.com/v2/users/'
|
||||
+ encodeURIComponent(username)
|
||||
+ '/search?page=' + page + '&count=80&order=latest&type=g',
|
||||
{headers: {'Authorization': 'Bearer ' + currentToken}}
|
||||
);
|
||||
} catch (networkErr) {
|
||||
errorMsg = 'Network error: ' + (networkErr.message || String(networkErr));
|
||||
r = null;
|
||||
break;
|
||||
}
|
||||
|
||||
if (r.status === 401 || r.status === 403) {
|
||||
try {
|
||||
const tokRes = await fetch('https://api.redgifs.com/v2/auth/temporary');
|
||||
if (tokRes.status === 200) {
|
||||
const tokData = await tokRes.json();
|
||||
currentToken = tokData.token;
|
||||
continue;
|
||||
}
|
||||
} catch (e) {}
|
||||
}
|
||||
|
||||
if (r.status === 429) {
|
||||
let rawBody = '';
|
||||
try { rawBody = await r.text(); } catch (e) {}
|
||||
errorMsg = 'API status 429: ' + rawBody;
|
||||
break;
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
if (errorMsg || !r || r.status !== 200) {
|
||||
if (!errorMsg && r) {
|
||||
errorMsg = 'API status ' + r.status + ': ' + await r.text();
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
const d = await r.json();
|
||||
if (page === 1) {
|
||||
pages = Math.min(d.pages || 1, 100);
|
||||
total = d.total || 0;
|
||||
}
|
||||
for (const g of (d.gifs || [])) {
|
||||
allGifs.push({urls: g.urls, id: g.id});
|
||||
}
|
||||
page++;
|
||||
await sleep(1500); // 1.5s delay between pages
|
||||
}
|
||||
} catch (outerErr) {
|
||||
errorMsg = 'Unexpected error: ' + (outerErr.message || String(outerErr));
|
||||
}
|
||||
|
||||
return JSON.stringify({gifs: allGifs, total: total, error: errorMsg});
|
||||
}''', {'username': username, 'token': self._token})
|
||||
|
||||
data = json.loads(raw)
|
||||
if data.get('error'):
|
||||
raise RuntimeError(data['error'])
|
||||
|
||||
return data.get('gifs', []), int(data.get('total', 0))
|
||||
|
||||
|
||||
def extract_filename(url):
|
||||
return urlparse(url).path.split('/')[-1]
|
||||
|
||||
|
||||
def download_creator_videos(username, output_dir, gifs, total, concurrency, skip_existing):
|
||||
creator_dir = output_dir / username
|
||||
creator_dir.mkdir(parents=True, exist_ok=True)
|
||||
links_file = creator_dir / 'links.txt'
|
||||
master_links_file = output_dir / 'all_links.txt'
|
||||
|
||||
download_urls = []
|
||||
link_lines = []
|
||||
for gif in gifs:
|
||||
urls = gif['urls']
|
||||
dl_url = urls.get('hd') or urls.get('sd') or next(iter(urls.values()))
|
||||
filename = extract_filename(dl_url)
|
||||
filepath = creator_dir / filename
|
||||
download_urls.append((dl_url, filepath))
|
||||
link_lines.append(f'{dl_url}\n')
|
||||
|
||||
# Save url lists
|
||||
with open(links_file, 'w') as f:
|
||||
f.writelines(link_lines)
|
||||
with open(master_links_file, 'a') as f:
|
||||
f.writelines(link_lines)
|
||||
|
||||
if len(gifs) < total:
|
||||
tqdm.write(f"Downloading {len(gifs)} of {total} video(s) (capped by API page limit) for {username} with concurrency {concurrency} ...")
|
||||
else:
|
||||
tqdm.write(f"Downloading {total} video(s) for {username} with concurrency {concurrency} ...")
|
||||
|
||||
bar_pool = scraper_core.BarPositionPool(concurrency)
|
||||
overall_bar = tqdm(
|
||||
total=len(gifs),
|
||||
desc=f"Creator: {username}",
|
||||
position=0,
|
||||
leave=True,
|
||||
ncols=80,
|
||||
)
|
||||
|
||||
def download_one(dl_url, filepath):
|
||||
if skip_existing and filepath.exists() and filepath.stat().st_size > 0:
|
||||
try:
|
||||
head = requests.head(dl_url, headers=HEADERS, timeout=10, allow_redirects=True)
|
||||
expected = int(head.headers.get('Content-Length', 0))
|
||||
if expected > 0:
|
||||
current = filepath.stat().st_size
|
||||
if current == expected:
|
||||
overall_bar.update(1)
|
||||
return
|
||||
if current < expected:
|
||||
filepath.unlink()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
pos = bar_pool.acquire() or 1
|
||||
|
||||
success = scraper_core.download_file(
|
||||
dl_url, filepath, HEADERS, pos
|
||||
)
|
||||
|
||||
bar_pool.release(pos)
|
||||
overall_bar.update(1)
|
||||
|
||||
with ThreadPoolExecutor(max_workers=concurrency) as executor:
|
||||
futures = [executor.submit(download_one, url, path) for url, path in download_urls]
|
||||
for f in as_completed(futures):
|
||||
f.result()
|
||||
|
||||
overall_bar.close()
|
||||
|
||||
# Clean up console lines
|
||||
sys.stdout.write("\n" * (concurrency + 1))
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description='Download all videos from RedGIFs creator(s)',
|
||||
)
|
||||
parser.add_argument(
|
||||
'creators', nargs='*',
|
||||
help='Creator username(s) - space-separated',
|
||||
)
|
||||
parser.add_argument(
|
||||
'-o', '--output', default=str(Path(__file__).resolve().parent / 'videos'),
|
||||
help='Output directory (default: videos subfolder)',
|
||||
)
|
||||
parser.add_argument(
|
||||
'--concurrency', type=int, default=5,
|
||||
help='Number of concurrent downloads (default: 5)',
|
||||
)
|
||||
parser.add_argument(
|
||||
'--skip-existing', action='store_true', default=True,
|
||||
help='Skip already-downloaded files (default: true)',
|
||||
)
|
||||
parser.add_argument(
|
||||
'--no-skip-existing', action='store_false', dest='skip_existing',
|
||||
help='Re-download existing files',
|
||||
)
|
||||
parser.add_argument(
|
||||
'--update-all-creators', action='store_true',
|
||||
help='Gets the name of each creator folder, adds them to creators.txt, removes duplicate entries, and then uses creators.txt as the argument',
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
creators_file = Path(__file__).resolve().parent / 'creators.txt'
|
||||
|
||||
if args.update_all_creators:
|
||||
output_dir = Path(args.output).resolve()
|
||||
found_folders = []
|
||||
if output_dir.exists():
|
||||
for entry in output_dir.iterdir():
|
||||
if entry.is_dir():
|
||||
found_folders.append(entry.name)
|
||||
|
||||
existing_creators = []
|
||||
if creators_file.is_file():
|
||||
try:
|
||||
with open(creators_file, 'r', encoding='utf-8') as f:
|
||||
for line in f:
|
||||
stripped = line.strip()
|
||||
if not stripped or stripped.startswith('#'):
|
||||
continue
|
||||
for part in re.split(r'[,\s]+', stripped):
|
||||
if part:
|
||||
existing_creators.append(part)
|
||||
except Exception as e:
|
||||
tqdm.write(f"Error reading creators.txt: {e}")
|
||||
|
||||
all_creators = sorted(list(set(existing_creators + found_folders)))
|
||||
try:
|
||||
with open(creators_file, 'w', encoding='utf-8') as f:
|
||||
f.write(' '.join(all_creators))
|
||||
tqdm.write(f"Updated creators.txt with {len(all_creators)} unique creators.")
|
||||
except Exception as e:
|
||||
tqdm.write(f"Error writing to creators.txt: {e}")
|
||||
|
||||
args.creators = [str(creators_file)]
|
||||
|
||||
creators = args.creators
|
||||
if not creators:
|
||||
raw = input('RedGIFs creator username(s) (comma/space separated): ').strip()
|
||||
creators = [c.strip() for c in re.split(r'[,\s]+', raw) if c.strip()]
|
||||
else:
|
||||
resolved = []
|
||||
for c in creators:
|
||||
p = Path(c)
|
||||
if p.is_file():
|
||||
try:
|
||||
with open(p, 'r', encoding='utf-8') as f:
|
||||
for line in f:
|
||||
stripped = line.strip()
|
||||
if not stripped or stripped.startswith('#'):
|
||||
continue
|
||||
for part in re.split(r'[,\s]+', stripped):
|
||||
if part:
|
||||
resolved.append(part)
|
||||
except Exception as e:
|
||||
tqdm.write(f"Error reading creators from file {c}: {e}")
|
||||
else:
|
||||
resolved.append(c)
|
||||
creators = resolved
|
||||
|
||||
seen = set()
|
||||
creators = [c for c in creators if not (c in seen or seen.add(c))]
|
||||
|
||||
if not creators:
|
||||
sys.exit('No creators provided.')
|
||||
|
||||
output_dir = Path(args.output).resolve()
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
api = RedGIFsAPI()
|
||||
try:
|
||||
api.start()
|
||||
except Exception as e:
|
||||
sys.exit(f'Failed to initialise Playwright browser: {e}')
|
||||
|
||||
try:
|
||||
total_creators = len(creators)
|
||||
for idx, username in enumerate(creators, 1):
|
||||
sep = '-' * 3
|
||||
tqdm.write(f'\n{sep} [{idx}/{total_creators}] {username} {sep}')
|
||||
tqdm.write(f' Fetching video URLs for {username} ...')
|
||||
|
||||
gifs, total = [], 0
|
||||
success = False
|
||||
|
||||
# Retry loop with browser reset on failure
|
||||
for attempt in range(1, 4):
|
||||
try:
|
||||
gifs, total = api.get_all_creator_videos(username, fresh_token=(attempt == 1))
|
||||
success = True
|
||||
break
|
||||
except Exception as e:
|
||||
tqdm.write(f' [ERROR] Attempt {attempt}/3 failed for {username}: {e}')
|
||||
if attempt < 3:
|
||||
err_str = str(e)
|
||||
if '429' in err_str:
|
||||
m = re.search(r'"delay"\s*:\s*(\d+)', err_str)
|
||||
wait_seconds = int(m.group(1)) + 1 if m else 30
|
||||
tqdm.write(f' Rate limited. Waiting {wait_seconds:.1f}s...')
|
||||
for _ in tqdm(range(int(wait_seconds)), desc=' Waiting', unit='s', ncols=80, leave=False):
|
||||
time.sleep(1)
|
||||
fractional = wait_seconds - int(wait_seconds)
|
||||
if fractional > 0:
|
||||
time.sleep(fractional)
|
||||
else:
|
||||
tqdm.write(' Resetting browser context and waiting 30s before retry...')
|
||||
try:
|
||||
api.stop()
|
||||
except Exception:
|
||||
pass
|
||||
time.sleep(30)
|
||||
try:
|
||||
api.start()
|
||||
except Exception as restart_err:
|
||||
tqdm.write(f' Failed to restart browser: {restart_err}')
|
||||
|
||||
if success:
|
||||
tqdm.write(f' Found {total} video(s)')
|
||||
if total > 0:
|
||||
download_creator_videos(
|
||||
username, output_dir, gifs, total,
|
||||
args.concurrency, args.skip_existing
|
||||
)
|
||||
else:
|
||||
tqdm.write(f' [ERROR] Skipping {username} after 3 failed attempts.')
|
||||
|
||||
# Introduce a small pacing delay to avoid aggressive rate limits
|
||||
time.sleep(2)
|
||||
|
||||
finally:
|
||||
api.stop()
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Reference in new issue
Block a user