Initial backup: folder structure manifest, tree summary, scraper scripts, and text metadata
This commit is contained in:
commit
f164832dcc
10504 files changed
+2423594
No files matched your search
@@ -0,0 +1,156 @@
|
||||
import os
|
||||
import re
|
||||
import requests
|
||||
import sys
|
||||
import time
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
# --- Configuration ---
|
||||
ROOT_DIR = r'W:\smalldata\niggers\redgifs\videos' # Adjust this to your main directory containing creator folders
|
||||
LINKS_FILENAME = 'links.txt'
|
||||
HEADERS = {
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
|
||||
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7',
|
||||
'Referer': 'https://www.redgifs.com/',
|
||||
'Connection': 'keep-alive',
|
||||
}
|
||||
|
||||
# Regex to match redgifs media URLs that are potentially all lowercase
|
||||
LOWERCASE_MEDIA_URL_PATTERN = re.compile(r"https://media\.redgifs\.com/([a-z0-9]+)\.mp4")
|
||||
|
||||
def is_all_lowercase_media_url(url):
|
||||
"""Checks if a URL is a Redgifs media URL and if its identifier part is all lowercase."""
|
||||
match = LOWERCASE_MEDIA_URL_PATTERN.match(url)
|
||||
if match:
|
||||
identifier = match.group(1)
|
||||
return identifier.islower()
|
||||
return False
|
||||
|
||||
def get_watch_url(media_url):
|
||||
"""Constructs the redgifs.com/watch/ URL from a media.redgifs.com URL."""
|
||||
match = LOWERCASE_MEDIA_URL_PATTERN.match(media_url)
|
||||
if match:
|
||||
identifier = match.group(1)
|
||||
return f"https://www.redgifs.com/watch/{identifier}"
|
||||
return None
|
||||
|
||||
def fetch_and_extract_capitalized_url(watch_url):
|
||||
"""
|
||||
Fetches the watch page, extracts the capitalized media URL ending in -mobile.jpg,
|
||||
and converts it to an .mp4 URL.
|
||||
"""
|
||||
try:
|
||||
print(f" Fetching watch page: {watch_url}")
|
||||
response = requests.get(watch_url, headers=HEADERS, timeout=15)
|
||||
response.raise_for_status() # Raise an exception for HTTP errors (4xx or 5xx)
|
||||
|
||||
soup = BeautifulSoup(response.text, 'html.parser')
|
||||
|
||||
capitalized_jpg_url = None
|
||||
|
||||
# Strategy 1: Look for meta tags (often used for social media previews)
|
||||
meta_og_image = soup.find('meta', property='og:image', content=re.compile(r".*-mobile\.jpg$"))
|
||||
if meta_og_image:
|
||||
capitalized_jpg_url = meta_og_image.get('content')
|
||||
|
||||
if not capitalized_jpg_url:
|
||||
# Strategy 2: Look for video source tags, img, or video tags with src or data-src
|
||||
media_element = soup.find(['source', 'img', 'video'], src=re.compile(r".*-mobile\.jpg$"))
|
||||
if not media_element:
|
||||
media_element = soup.find(['source', 'img', 'video'], attrs={'data-src': re.compile(r".*-mobile\.jpg$")})
|
||||
|
||||
if media_element:
|
||||
capitalized_jpg_url = media_element.get('src') or media_element.get('data-src')
|
||||
|
||||
if capitalized_jpg_url:
|
||||
# Transform the .jpg URL to .mp4
|
||||
capitalized_mp4_url = capitalized_jpg_url.replace('-mobile.jpg', '.mp4')
|
||||
print(f" Found and transformed URL: {capitalized_mp4_url}")
|
||||
return capitalized_mp4_url
|
||||
|
||||
print(f" Could not find capitalized media URL on page: {watch_url}")
|
||||
return None
|
||||
|
||||
except requests.exceptions.RequestException as e:
|
||||
print(f" Error fetching {watch_url}: {e}")
|
||||
return None
|
||||
except Exception as e:
|
||||
print(f" An unexpected error occurred while parsing {watch_url}: {e}")
|
||||
return None
|
||||
|
||||
def process_links_file(links_file_path):
|
||||
"""
|
||||
Processes a single links.txt file, corrects capitalization for URLs,
|
||||
and updates the file in place.
|
||||
"""
|
||||
print(f"Processing links file: {links_file_path}")
|
||||
updated_lines = []
|
||||
changes_made = False
|
||||
|
||||
if not os.path.exists(links_file_path):
|
||||
print(f" '{LINKS_FILENAME}' not found. Skipping.")
|
||||
return False
|
||||
|
||||
with open(links_file_path, 'r', encoding='utf-8') as f:
|
||||
lines = f.readlines()
|
||||
|
||||
for line in lines:
|
||||
original_url = line.strip()
|
||||
if original_url and is_all_lowercase_media_url(original_url):
|
||||
print(f" Found lowercase URL: {original_url}")
|
||||
watch_url = get_watch_url(original_url)
|
||||
if watch_url:
|
||||
capitalized_mp4_url = fetch_and_extract_capitalized_url(watch_url)
|
||||
if capitalized_mp4_url:
|
||||
if capitalized_mp4_url != original_url: # Only update if different
|
||||
updated_lines.append(capitalized_mp4_url + "\n")
|
||||
print(f" Replaced with: {capitalized_mp4_url}")
|
||||
changes_made = True
|
||||
else:
|
||||
updated_lines.append(line) # No change needed
|
||||
print(" Capitalized URL is identical to original, no update.")
|
||||
else:
|
||||
updated_lines.append(line) # Keep original if new URL not found
|
||||
else:
|
||||
updated_lines.append(line) # Keep original if watch URL couldn't be constructed
|
||||
else:
|
||||
updated_lines.append(line) # Keep original if no change needed
|
||||
|
||||
if changes_made:
|
||||
print(f" Changes made to {links_file_path}. Writing updated content.")
|
||||
with open(links_file_path, 'w', encoding='utf-8') as f:
|
||||
f.writelines(updated_lines)
|
||||
return True
|
||||
else:
|
||||
print(f" No changes needed for {links_file_path}.")
|
||||
return False
|
||||
|
||||
def main():
|
||||
if not os.path.exists(ROOT_DIR):
|
||||
print(f"Error: Root directory not found at '{ROOT_DIR}'. Please ensure it exists and is correctly configured.")
|
||||
sys.exit(1)
|
||||
|
||||
creator_folders = [f.path for f in os.scandir(ROOT_DIR) if f.is_dir() and not f.name.startswith('.')]
|
||||
|
||||
if not creator_folders:
|
||||
print(f"No creator folders found in '{ROOT_DIR}'. Exiting.")
|
||||
return
|
||||
|
||||
print(f"Starting capitalization correction for {len(creator_folders)} creator folders...")
|
||||
|
||||
total_files_changed = 0
|
||||
total_creators = len(creator_folders)
|
||||
for i, folder_path in enumerate(creator_folders):
|
||||
creator_name = os.scandir(folder_path)
|
||||
creator_name = os.path.basename(folder_path)
|
||||
print(f"\n--- Processing creator folder [{i+1}/{total_creators}]: {creator_name} ---")
|
||||
|
||||
links_file = os.path.join(folder_path, LINKS_FILENAME)
|
||||
if process_links_file(links_file):
|
||||
total_files_changed += 1
|
||||
time.sleep(1) # Small delay to avoid hammering websites
|
||||
|
||||
print(f"\nCapitalization correction complete. Total links.txt files modified: {total_files_changed}.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in new issue
Block a user