Initial backup: folder structure manifest, tree summary, scraper scripts, and text metadata
This commit is contained in:
commit
f164832dcc
10504 files changed
+2423594
No files matched your search
@@ -0,0 +1,40 @@
|
||||
import sys
|
||||
import re
|
||||
from pathlib import Path
|
||||
from urllib.parse import urlparse
|
||||
|
||||
# Add parent workspace directory to path to import generic_downloader
|
||||
sys.path.append(str(Path(__file__).resolve().parents[1]))
|
||||
import generic_downloader
|
||||
|
||||
def is_video_link(url):
|
||||
# Match Motherless video URL patterns like:
|
||||
# https://motherless.com/AC493CD
|
||||
# https://motherless.com/g/group_name/AC493CD
|
||||
path = urlparse(url).path.rstrip("/")
|
||||
if not path:
|
||||
return False
|
||||
parts = path.split("/")
|
||||
last_part = parts[-1]
|
||||
# Motherless IDs are typically 7 or 8 characters of uppercase letters and numbers
|
||||
return bool(re.match(r'^[A-Z0-9]{6,9}$', last_part))
|
||||
|
||||
if __name__ == "__main__":
|
||||
uploader_eval = """() => {
|
||||
const memberLink = document.querySelector('a[href*="/members/"]');
|
||||
if (memberLink) return memberLink.innerText.trim();
|
||||
const uploadText = document.querySelector('.upload-info');
|
||||
if (uploadText) {
|
||||
const m = uploadText.innerText.match(/by\\s+(\\S+)/i);
|
||||
if (m) return m[1];
|
||||
}
|
||||
return 'unknown';
|
||||
}"""
|
||||
|
||||
generic_downloader.run(
|
||||
site_name="Motherless",
|
||||
is_video_link_fn=is_video_link,
|
||||
next_page_selector="a:has-text('Next'), a.next, a.pagination-next",
|
||||
video_selector="video",
|
||||
uploader_eval_js=uploader_eval
|
||||
)
|
||||
Reference in new issue
Block a user