157 lines
6.3 KiB
Python
157 lines
6.3 KiB
Python
import os
|
|
import re
|
|
import requests
|
|
import sys
|
|
import time
|
|
from bs4 import BeautifulSoup
|
|
|
|
# --- Configuration ---
|
|
ROOT_DIR = r'W:\smalldata\niggers\redgifs\videos' # Adjust this to your main directory containing creator folders
|
|
LINKS_FILENAME = 'links.txt'
|
|
HEADERS = {
|
|
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
|
|
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7',
|
|
'Referer': 'https://www.redgifs.com/',
|
|
'Connection': 'keep-alive',
|
|
}
|
|
|
|
# Regex to match redgifs media URLs that are potentially all lowercase
|
|
LOWERCASE_MEDIA_URL_PATTERN = re.compile(r"https://media\.redgifs\.com/([a-z0-9]+)\.mp4")
|
|
|
|
def is_all_lowercase_media_url(url):
|
|
"""Checks if a URL is a Redgifs media URL and if its identifier part is all lowercase."""
|
|
match = LOWERCASE_MEDIA_URL_PATTERN.match(url)
|
|
if match:
|
|
identifier = match.group(1)
|
|
return identifier.islower()
|
|
return False
|
|
|
|
def get_watch_url(media_url):
|
|
"""Constructs the redgifs.com/watch/ URL from a media.redgifs.com URL."""
|
|
match = LOWERCASE_MEDIA_URL_PATTERN.match(media_url)
|
|
if match:
|
|
identifier = match.group(1)
|
|
return f"https://www.redgifs.com/watch/{identifier}"
|
|
return None
|
|
|
|
def fetch_and_extract_capitalized_url(watch_url):
|
|
"""
|
|
Fetches the watch page, extracts the capitalized media URL ending in -mobile.jpg,
|
|
and converts it to an .mp4 URL.
|
|
"""
|
|
try:
|
|
print(f" Fetching watch page: {watch_url}")
|
|
response = requests.get(watch_url, headers=HEADERS, timeout=15)
|
|
response.raise_for_status() # Raise an exception for HTTP errors (4xx or 5xx)
|
|
|
|
soup = BeautifulSoup(response.text, 'html.parser')
|
|
|
|
capitalized_jpg_url = None
|
|
|
|
# Strategy 1: Look for meta tags (often used for social media previews)
|
|
meta_og_image = soup.find('meta', property='og:image', content=re.compile(r".*-mobile\.jpg$"))
|
|
if meta_og_image:
|
|
capitalized_jpg_url = meta_og_image.get('content')
|
|
|
|
if not capitalized_jpg_url:
|
|
# Strategy 2: Look for video source tags, img, or video tags with src or data-src
|
|
media_element = soup.find(['source', 'img', 'video'], src=re.compile(r".*-mobile\.jpg$"))
|
|
if not media_element:
|
|
media_element = soup.find(['source', 'img', 'video'], attrs={'data-src': re.compile(r".*-mobile\.jpg$")})
|
|
|
|
if media_element:
|
|
capitalized_jpg_url = media_element.get('src') or media_element.get('data-src')
|
|
|
|
if capitalized_jpg_url:
|
|
# Transform the .jpg URL to .mp4
|
|
capitalized_mp4_url = capitalized_jpg_url.replace('-mobile.jpg', '.mp4')
|
|
print(f" Found and transformed URL: {capitalized_mp4_url}")
|
|
return capitalized_mp4_url
|
|
|
|
print(f" Could not find capitalized media URL on page: {watch_url}")
|
|
return None
|
|
|
|
except requests.exceptions.RequestException as e:
|
|
print(f" Error fetching {watch_url}: {e}")
|
|
return None
|
|
except Exception as e:
|
|
print(f" An unexpected error occurred while parsing {watch_url}: {e}")
|
|
return None
|
|
|
|
def process_links_file(links_file_path):
|
|
"""
|
|
Processes a single links.txt file, corrects capitalization for URLs,
|
|
and updates the file in place.
|
|
"""
|
|
print(f"Processing links file: {links_file_path}")
|
|
updated_lines = []
|
|
changes_made = False
|
|
|
|
if not os.path.exists(links_file_path):
|
|
print(f" '{LINKS_FILENAME}' not found. Skipping.")
|
|
return False
|
|
|
|
with open(links_file_path, 'r', encoding='utf-8') as f:
|
|
lines = f.readlines()
|
|
|
|
for line in lines:
|
|
original_url = line.strip()
|
|
if original_url and is_all_lowercase_media_url(original_url):
|
|
print(f" Found lowercase URL: {original_url}")
|
|
watch_url = get_watch_url(original_url)
|
|
if watch_url:
|
|
capitalized_mp4_url = fetch_and_extract_capitalized_url(watch_url)
|
|
if capitalized_mp4_url:
|
|
if capitalized_mp4_url != original_url: # Only update if different
|
|
updated_lines.append(capitalized_mp4_url + "\n")
|
|
print(f" Replaced with: {capitalized_mp4_url}")
|
|
changes_made = True
|
|
else:
|
|
updated_lines.append(line) # No change needed
|
|
print(" Capitalized URL is identical to original, no update.")
|
|
else:
|
|
updated_lines.append(line) # Keep original if new URL not found
|
|
else:
|
|
updated_lines.append(line) # Keep original if watch URL couldn't be constructed
|
|
else:
|
|
updated_lines.append(line) # Keep original if no change needed
|
|
|
|
if changes_made:
|
|
print(f" Changes made to {links_file_path}. Writing updated content.")
|
|
with open(links_file_path, 'w', encoding='utf-8') as f:
|
|
f.writelines(updated_lines)
|
|
return True
|
|
else:
|
|
print(f" No changes needed for {links_file_path}.")
|
|
return False
|
|
|
|
def main():
|
|
if not os.path.exists(ROOT_DIR):
|
|
print(f"Error: Root directory not found at '{ROOT_DIR}'. Please ensure it exists and is correctly configured.")
|
|
sys.exit(1)
|
|
|
|
creator_folders = [f.path for f in os.scandir(ROOT_DIR) if f.is_dir() and not f.name.startswith('.')]
|
|
|
|
if not creator_folders:
|
|
print(f"No creator folders found in '{ROOT_DIR}'. Exiting.")
|
|
return
|
|
|
|
print(f"Starting capitalization correction for {len(creator_folders)} creator folders...")
|
|
|
|
total_files_changed = 0
|
|
total_creators = len(creator_folders)
|
|
for i, folder_path in enumerate(creator_folders):
|
|
creator_name = os.scandir(folder_path)
|
|
creator_name = os.path.basename(folder_path)
|
|
print(f"\n--- Processing creator folder [{i+1}/{total_creators}]: {creator_name} ---")
|
|
|
|
links_file = os.path.join(folder_path, LINKS_FILENAME)
|
|
if process_links_file(links_file):
|
|
total_files_changed += 1
|
|
time.sleep(1) # Small delay to avoid hammering websites
|
|
|
|
print(f"\nCapitalization correction complete. Total links.txt files modified: {total_files_changed}.")
|
|
|
|
if __name__ == "__main__":
|
|
main()
|