import os import re import requests import sys import time from bs4 import BeautifulSoup # --- Configuration --- ROOT_DIR = r'W:\smalldata\niggers\redgifs\videos' # Adjust this to your main directory containing creator folders LINKS_FILENAME = 'links.txt' HEADERS = { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36', 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7', 'Referer': 'https://www.redgifs.com/', 'Connection': 'keep-alive', } # Regex to match redgifs media URLs that are potentially all lowercase LOWERCASE_MEDIA_URL_PATTERN = re.compile(r"https://media\.redgifs\.com/([a-z0-9]+)\.mp4") def is_all_lowercase_media_url(url): """Checks if a URL is a Redgifs media URL and if its identifier part is all lowercase.""" match = LOWERCASE_MEDIA_URL_PATTERN.match(url) if match: identifier = match.group(1) return identifier.islower() return False def get_watch_url(media_url): """Constructs the redgifs.com/watch/ URL from a media.redgifs.com URL.""" match = LOWERCASE_MEDIA_URL_PATTERN.match(media_url) if match: identifier = match.group(1) return f"https://www.redgifs.com/watch/{identifier}" return None def fetch_and_extract_capitalized_url(watch_url): """ Fetches the watch page, extracts the capitalized media URL ending in -mobile.jpg, and converts it to an .mp4 URL. """ try: print(f" Fetching watch page: {watch_url}") response = requests.get(watch_url, headers=HEADERS, timeout=15) response.raise_for_status() # Raise an exception for HTTP errors (4xx or 5xx) soup = BeautifulSoup(response.text, 'html.parser') capitalized_jpg_url = None # Strategy 1: Look for meta tags (often used for social media previews) meta_og_image = soup.find('meta', property='og:image', content=re.compile(r".*-mobile\.jpg$")) if meta_og_image: capitalized_jpg_url = meta_og_image.get('content') if not capitalized_jpg_url: # Strategy 2: Look for video source tags, img, or video tags with src or data-src media_element = soup.find(['source', 'img', 'video'], src=re.compile(r".*-mobile\.jpg$")) if not media_element: media_element = soup.find(['source', 'img', 'video'], attrs={'data-src': re.compile(r".*-mobile\.jpg$")}) if media_element: capitalized_jpg_url = media_element.get('src') or media_element.get('data-src') if capitalized_jpg_url: # Transform the .jpg URL to .mp4 capitalized_mp4_url = capitalized_jpg_url.replace('-mobile.jpg', '.mp4') print(f" Found and transformed URL: {capitalized_mp4_url}") return capitalized_mp4_url print(f" Could not find capitalized media URL on page: {watch_url}") return None except requests.exceptions.RequestException as e: print(f" Error fetching {watch_url}: {e}") return None except Exception as e: print(f" An unexpected error occurred while parsing {watch_url}: {e}") return None def process_links_file(links_file_path): """ Processes a single links.txt file, corrects capitalization for URLs, and updates the file in place. """ print(f"Processing links file: {links_file_path}") updated_lines = [] changes_made = False if not os.path.exists(links_file_path): print(f" '{LINKS_FILENAME}' not found. Skipping.") return False with open(links_file_path, 'r', encoding='utf-8') as f: lines = f.readlines() for line in lines: original_url = line.strip() if original_url and is_all_lowercase_media_url(original_url): print(f" Found lowercase URL: {original_url}") watch_url = get_watch_url(original_url) if watch_url: capitalized_mp4_url = fetch_and_extract_capitalized_url(watch_url) if capitalized_mp4_url: if capitalized_mp4_url != original_url: # Only update if different updated_lines.append(capitalized_mp4_url + "\n") print(f" Replaced with: {capitalized_mp4_url}") changes_made = True else: updated_lines.append(line) # No change needed print(" Capitalized URL is identical to original, no update.") else: updated_lines.append(line) # Keep original if new URL not found else: updated_lines.append(line) # Keep original if watch URL couldn't be constructed else: updated_lines.append(line) # Keep original if no change needed if changes_made: print(f" Changes made to {links_file_path}. Writing updated content.") with open(links_file_path, 'w', encoding='utf-8') as f: f.writelines(updated_lines) return True else: print(f" No changes needed for {links_file_path}.") return False def main(): if not os.path.exists(ROOT_DIR): print(f"Error: Root directory not found at '{ROOT_DIR}'. Please ensure it exists and is correctly configured.") sys.exit(1) creator_folders = [f.path for f in os.scandir(ROOT_DIR) if f.is_dir() and not f.name.startswith('.')] if not creator_folders: print(f"No creator folders found in '{ROOT_DIR}'. Exiting.") return print(f"Starting capitalization correction for {len(creator_folders)} creator folders...") total_files_changed = 0 total_creators = len(creator_folders) for i, folder_path in enumerate(creator_folders): creator_name = os.scandir(folder_path) creator_name = os.path.basename(folder_path) print(f"\n--- Processing creator folder [{i+1}/{total_creators}]: {creator_name} ---") links_file = os.path.join(folder_path, LINKS_FILENAME) if process_links_file(links_file): total_files_changed += 1 time.sleep(1) # Small delay to avoid hammering websites print(f"\nCapitalization correction complete. Total links.txt files modified: {total_files_changed}.") if __name__ == "__main__": main()