Files
niggers/redgifs/correct_capitalization.py
T

157 lines
6.3 KiB
Python

import os
import re
import requests
import sys
import time
from bs4 import BeautifulSoup
# --- Configuration ---
ROOT_DIR = r'W:\smalldata\niggers\redgifs\videos' # Adjust this to your main directory containing creator folders
LINKS_FILENAME = 'links.txt'
HEADERS = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7',
'Referer': 'https://www.redgifs.com/',
'Connection': 'keep-alive',
}
# Regex to match redgifs media URLs that are potentially all lowercase
LOWERCASE_MEDIA_URL_PATTERN = re.compile(r"https://media\.redgifs\.com/([a-z0-9]+)\.mp4")
def is_all_lowercase_media_url(url):
"""Checks if a URL is a Redgifs media URL and if its identifier part is all lowercase."""
match = LOWERCASE_MEDIA_URL_PATTERN.match(url)
if match:
identifier = match.group(1)
return identifier.islower()
return False
def get_watch_url(media_url):
"""Constructs the redgifs.com/watch/ URL from a media.redgifs.com URL."""
match = LOWERCASE_MEDIA_URL_PATTERN.match(media_url)
if match:
identifier = match.group(1)
return f"https://www.redgifs.com/watch/{identifier}"
return None
def fetch_and_extract_capitalized_url(watch_url):
"""
Fetches the watch page, extracts the capitalized media URL ending in -mobile.jpg,
and converts it to an .mp4 URL.
"""
try:
print(f" Fetching watch page: {watch_url}")
response = requests.get(watch_url, headers=HEADERS, timeout=15)
response.raise_for_status() # Raise an exception for HTTP errors (4xx or 5xx)
soup = BeautifulSoup(response.text, 'html.parser')
capitalized_jpg_url = None
# Strategy 1: Look for meta tags (often used for social media previews)
meta_og_image = soup.find('meta', property='og:image', content=re.compile(r".*-mobile\.jpg$"))
if meta_og_image:
capitalized_jpg_url = meta_og_image.get('content')
if not capitalized_jpg_url:
# Strategy 2: Look for video source tags, img, or video tags with src or data-src
media_element = soup.find(['source', 'img', 'video'], src=re.compile(r".*-mobile\.jpg$"))
if not media_element:
media_element = soup.find(['source', 'img', 'video'], attrs={'data-src': re.compile(r".*-mobile\.jpg$")})
if media_element:
capitalized_jpg_url = media_element.get('src') or media_element.get('data-src')
if capitalized_jpg_url:
# Transform the .jpg URL to .mp4
capitalized_mp4_url = capitalized_jpg_url.replace('-mobile.jpg', '.mp4')
print(f" Found and transformed URL: {capitalized_mp4_url}")
return capitalized_mp4_url
print(f" Could not find capitalized media URL on page: {watch_url}")
return None
except requests.exceptions.RequestException as e:
print(f" Error fetching {watch_url}: {e}")
return None
except Exception as e:
print(f" An unexpected error occurred while parsing {watch_url}: {e}")
return None
def process_links_file(links_file_path):
"""
Processes a single links.txt file, corrects capitalization for URLs,
and updates the file in place.
"""
print(f"Processing links file: {links_file_path}")
updated_lines = []
changes_made = False
if not os.path.exists(links_file_path):
print(f" '{LINKS_FILENAME}' not found. Skipping.")
return False
with open(links_file_path, 'r', encoding='utf-8') as f:
lines = f.readlines()
for line in lines:
original_url = line.strip()
if original_url and is_all_lowercase_media_url(original_url):
print(f" Found lowercase URL: {original_url}")
watch_url = get_watch_url(original_url)
if watch_url:
capitalized_mp4_url = fetch_and_extract_capitalized_url(watch_url)
if capitalized_mp4_url:
if capitalized_mp4_url != original_url: # Only update if different
updated_lines.append(capitalized_mp4_url + "\n")
print(f" Replaced with: {capitalized_mp4_url}")
changes_made = True
else:
updated_lines.append(line) # No change needed
print(" Capitalized URL is identical to original, no update.")
else:
updated_lines.append(line) # Keep original if new URL not found
else:
updated_lines.append(line) # Keep original if watch URL couldn't be constructed
else:
updated_lines.append(line) # Keep original if no change needed
if changes_made:
print(f" Changes made to {links_file_path}. Writing updated content.")
with open(links_file_path, 'w', encoding='utf-8') as f:
f.writelines(updated_lines)
return True
else:
print(f" No changes needed for {links_file_path}.")
return False
def main():
if not os.path.exists(ROOT_DIR):
print(f"Error: Root directory not found at '{ROOT_DIR}'. Please ensure it exists and is correctly configured.")
sys.exit(1)
creator_folders = [f.path for f in os.scandir(ROOT_DIR) if f.is_dir() and not f.name.startswith('.')]
if not creator_folders:
print(f"No creator folders found in '{ROOT_DIR}'. Exiting.")
return
print(f"Starting capitalization correction for {len(creator_folders)} creator folders...")
total_files_changed = 0
total_creators = len(creator_folders)
for i, folder_path in enumerate(creator_folders):
creator_name = os.scandir(folder_path)
creator_name = os.path.basename(folder_path)
print(f"\n--- Processing creator folder [{i+1}/{total_creators}]: {creator_name} ---")
links_file = os.path.join(folder_path, LINKS_FILENAME)
if process_links_file(links_file):
total_files_changed += 1
time.sleep(1) # Small delay to avoid hammering websites
print(f"\nCapitalization correction complete. Total links.txt files modified: {total_files_changed}.")
if __name__ == "__main__":
main()