Initial backup: folder structure manifest, tree summary, scraper scripts, and text metadata

This commit is contained in:
gooner committed 2026-10-03 16:08:16 -04:00
commit f164832dcc
10504 files changed
+2423594

No files matched your search

File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,87 @@
{
"11392": {
"filename": "Piss marking curtains in public store display.mp4",
"size": 8917618,
"timestamp": "2026-09-22 03:32:20"
},
"11393": {
"filename": "Piss marking in public store isle.mp4",
"size": 9657311,
"timestamp": "2026-09-22 03:32:21"
},
"11394": {
"filename": "Piss marking public shop products.mp4",
"size": 14868518,
"timestamp": "2026-09-22 03:32:24"
},
"11395": {
"filename": "Piss marking public store plastic plants.mp4",
"size": 9743461,
"timestamp": "2026-09-22 03:32:26"
},
"11396": {
"filename": "Piss marking store tool shelf.mp4",
"size": 5746973,
"timestamp": "2026-09-22 03:32:27"
},
"11397": {
"filename": "Piss on store cushion shelf.mp4",
"size": 11506623,
"timestamp": "2026-09-22 03:32:29"
},
"11398": {
"filename": "Pissmarking pisstrashing pisstagging public store shelf.mp4",
"size": 5242276,
"timestamp": "2026-09-22 03:32:30"
},
"11399": {
"filename": "Poked by the family dog.mp4",
"size": 42708485,
"timestamp": "2026-09-22 03:32:37"
},
"11400": {
"filename": "Slut Fisted.mp4",
"size": 55603897,
"timestamp": "2026-09-22 03:32:45"
},
"11401": {
"filename": "SLUTS Babecock Dogbabecock.mp4",
"size": 27741185,
"timestamp": "2026-09-29 15:43:57"
},
"11402": {
"filename": "Teen boy gets fucked by dog.mp4",
"size": 19474961,
"timestamp": "2026-09-29 15:43:59"
},
"11403": {
"filename": "Tight assed twink gets knotted.mp4",
"size": 37933931,
"timestamp": "2026-09-22 03:33:01"
},
"11404": {
"filename": "toilet licker.mp4",
"size": 13355478,
"timestamp": "2026-09-22 03:33:03"
},
"11405": {
"filename": "Used asshole fucks trailer hitch on back of pickup.mp4",
"size": 21990986,
"timestamp": "2026-09-22 03:33:07"
},
"11406": {
"filename": "Woman pissing on shirt in a store.mp4",
"size": 14836034,
"timestamp": "2026-09-22 03:33:09"
},
"11407": {
"filename": "Young pussy home with the family dog.mp4",
"size": 8537003,
"timestamp": "2026-09-29 15:44:00"
},
"11408": {
"filename": "Yummy fuck with the dog with pullout.mp4",
"size": 4902392,
"timestamp": "2026-09-29 15:44:01"
}
}
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
@@ -0,0 +1,122 @@
{
"4": {
"filename": "1.mp4",
"size": 3985475674,
"timestamp": "2026-09-22 01:41:59"
},
"5": {
"filename": "Luna Knots Public_1_prob4.mp4",
"size": 2143382637,
"timestamp": "2026-09-22 01:44:01"
},
"6": {
"filename": "Beastluna - Exotic whore.mp4",
"size": 184781451,
"timestamp": "2026-09-22 01:44:34"
},
"7": {
"filename": "Isabella Productions - Pole Dance.mp4",
"size": 1388511206,
"timestamp": "2026-09-22 01:51:18"
},
"8": {
"filename": "Coral - Pain & Pleasure (Unmasked).mp4",
"size": 338594451,
"timestamp": "2026-09-22 01:52:17"
},
"9": {
"filename": "ida-dog-01-1024-kbps_processed.mp4",
"size": 336775091,
"timestamp": "2026-09-22 01:53:19"
},
"10": {
"filename": "ida-dog-06-1024-kbps_processed.mp4",
"size": 369164316,
"timestamp": "2026-09-22 01:54:23"
},
"11": {
"filename": "Isabella - Enter The Dog.mp4",
"size": 454529982,
"timestamp": "2026-09-22 01:55:47"
},
"12": {
"filename": "1_4929445926826673411.mp4",
"size": 518245769,
"timestamp": "2026-09-22 01:56:41"
},
"13": {
"filename": "1_4929445926826673412.mp4",
"size": 1345147005,
"timestamp": "2026-09-22 01:58:04"
},
"14": {
"filename": "1_5019591947430397220.MOV",
"size": 148046883,
"timestamp": "2026-09-22 01:58:20"
},
"15": {
"filename": "1_5019591947430397268.MOV",
"size": 186380058,
"timestamp": "2026-09-22 01:58:38"
},
"16": {
"filename": "1_16.mp4",
"size": 4003447897,
"timestamp": "2026-09-22 02:11:55"
},
"17": {
"filename": "2.mp4",
"size": 1224567277,
"timestamp": "2026-09-22 02:15:57"
},
"18": {
"filename": "Wanwan - Chinese Burn Artofzoo 1080p.mp4",
"size": 1560341210,
"timestamp": "2026-09-22 02:23:21"
},
"19": {
"filename": "ivana-project-grindhound-chapter-three-two-dogs-exclusive.mp4",
"size": 2289901816,
"timestamp": "2026-09-22 02:25:46"
},
"20": {
"filename": "Boar Corps 4.mp4",
"size": 1547122707,
"timestamp": "2026-09-22 02:27:23"
},
"21": {
"filename": "Ivana in Project GrindHound - Chapter Four.mp4",
"size": 1536386785,
"timestamp": "2026-09-22 02:28:57"
},
"22": {
"filename": "aoz.hot.rocks.mp4",
"size": 976756414,
"timestamp": "2026-09-22 02:31:52"
},
"23": {
"filename": "photo_23.jpg",
"size": 350017,
"timestamp": "2026-09-29 15:43:51"
},
"35": {
"filename": "Coral & Mia - Horse Sensation.mp4",
"size": 1174824933,
"timestamp": "2026-09-22 02:35:33"
},
"36": {
"filename": "SquirtSquid - Rider Nun.mp4",
"size": 1088016143,
"timestamp": "2026-09-22 02:38:56"
},
"58": {
"filename": "MakingcummyDoberman Kittykn9ne - Custom - EXCLUSIVE.mp4",
"size": 3195353718,
"timestamp": "2026-09-22 02:48:40"
},
"59": {
"filename": "Magnum_Equus.mp4",
"size": 3608631632,
"timestamp": "2026-09-22 03:08:49"
}
}
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
+9
View File
@@ -0,0 +1,9 @@
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-3742774514" "V:\smalldata\niggers\Telegram Desktop\KPS Premium" 1 8
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "7522570451" "V:\smalldata\niggers\Telegram Desktop\Deleted Account" 1 8
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "8025093129" "V:\smalldata\niggers\Telegram Desktop\Garcia Sarah" 1 8
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-4418305717" "V:\smalldata\niggers\Telegram Desktop\Shared Spaces" 1 8
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "@Knotslut303" "V:\smalldata\niggers\Telegram Desktop\Knottyknot💦💦❣️" 1 8
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-2880246536" "V:\smalldata\niggers\Telegram Desktop\LW Zoo" 1 8
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-3738137252" "V:\smalldata\niggers\Telegram Desktop\Stock Shares" 1 8
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-4240729023" "V:\smalldata\niggers\Telegram Desktop\Super Cool People Club" 1 8
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "@Dumdumisdumb" "V:\smalldata\niggers\Telegram Desktop\Dumdumisdumb" 1 8
+1
View File
@@ -0,0 +1 @@
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-3738137252" "V:\smalldata\niggers\Telegram Desktop\Stock Shares" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-4240729023" "V:\smalldata\niggers\Telegram Desktop\Super Cool People Club" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-4418305717" "V:\smalldata\niggers\Telegram Desktop\Shared Spaces" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "8025093129" "V:\smalldata\niggers\Telegram Desktop\Garcia Sarah" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-2880246536" "V:\smalldata\niggers\Telegram Desktop\LW Zoo" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "@Knotslut303" "V:\smalldata\niggers\Telegram Desktop\Knottyknot💦💦❣️" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "7522570451" "V:\smalldata\niggers\Telegram Desktop\Deleted Account" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-3742774514" "V:\smalldata\niggers\Telegram Desktop\KPS Premium" 1 8
+513
View File
@@ -0,0 +1,513 @@
import os
import sys
import json
import time
import math
import asyncio
import inspect
import logging
from telethon import TelegramClient, errors, utils
from telethon.tl.functions.auth import ExportAuthorizationRequest, ImportAuthorizationRequest
from telethon.tl.functions import InvokeWithLayerRequest
from telethon.tl.types import MessageMediaPhoto, MessageMediaDocument
from telethon.network import MTProtoSender
from telethon.tl.alltlobjects import LAYER
from FastTelethonhelper.FastTelethon import DownloadSender
# Suppress Telethon's internal network warnings (e.g., transient server-closed socket resets that auto-reconnect)
logging.basicConfig(level=logging.ERROR)
logging.getLogger('telethon').setLevel(logging.ERROR)
logging.getLogger('asyncio').setLevel(logging.ERROR)
# Global cache for persistent downloaders per DC ID to prevent connection churn
downloaders = {}
downloader_lock = asyncio.Lock()
# Helper function to format sizes
def format_size(bytes_count):
if bytes_count < 1024:
return f"{bytes_count} B"
elif bytes_count < 1024 * 1024:
return f"{bytes_count / 1024:.2f} KB"
elif bytes_count < 1024 * 1024 * 1024:
return f"{bytes_count / (1024 * 1024):.2f} MB"
else:
return f"{bytes_count / (1024 * 1024 * 1024):.2f} GB"
# Shared statistics and UI rendering
class ConcurrentProgressRenderer:
def __init__(self, total_files, concurrency):
self.total_files = total_files
self.concurrency = concurrency
self.completed_files = 0
self.downloaded_files = 0
self.skipped_files = 0
self.skipped_bytes = 0
self.total_downloaded_bytes = 0
self.start_time = time.time()
self.slots = [""] * concurrency
self.lock = asyncio.Lock()
self.initialized = False
async def get_free_slot(self):
async with self.lock:
for i in range(self.concurrency):
if self.slots[i] == "":
self.slots[i] = "Initializing..."
return i
return -1
async def update_slot(self, slot_idx, filename, received, total, start_time):
if not total:
total = 1
percent = (received / total) * 100
bar_len = 15
filled_len = int(bar_len * received // total)
bar = '█' * filled_len + '░' * (bar_len - filled_len)
now = time.time()
elapsed = now - start_time
speed = received / elapsed if elapsed > 0 else 0
speed_str = f"{format_size(speed)}/s"
size_str = f"{format_size(received)}/{format_size(total)}"
display_name = filename
if len(display_name) > 20:
display_name = display_name[:9] + "..." + display_name[-8:]
async with self.lock:
self.slots[slot_idx] = f"Slot {slot_idx+1}: {display_name} [{bar}] {percent:.1f}% ({size_str}) @ {speed_str}"
async def release_slot(self, slot_idx):
async with self.lock:
self.slots[slot_idx] = ""
async def add_bytes(self, size):
async with self.lock:
self.total_downloaded_bytes += size
self.downloaded_files += 1
self.completed_files += 1
async def increment_skipped(self, size=0):
async with self.lock:
self.skipped_files += 1
self.skipped_bytes += size
self.completed_files += 1
async def log(self, message):
async with self.lock:
if self.initialized:
num_lines = 2 + len(self.slots)
sys.stdout.write(f"\033[{num_lines}A")
sys.stdout.write("\033[J")
sys.stdout.write(message + "\n")
sys.stdout.flush()
self.initialized = False
async def render(self):
async with self.lock:
elapsed = time.time() - self.start_time
overall_speed = self.total_downloaded_bytes / elapsed if elapsed > 0 else 0
percent = (self.completed_files / self.total_files * 100) if self.total_files > 0 else 0.0
bar_len = 50
filled_len = int(bar_len * self.completed_files // self.total_files) if self.total_files > 0 else 0
bar = '█' * filled_len + ' ' * (bar_len - filled_len)
lines = []
lines.append(
f"Overall Progress: {self.completed_files}/{self.total_files} |{bar}| {percent:.2f}%"
)
lines.append(
f"Skipped: {self.skipped_files} files equaling {format_size(self.skipped_bytes)} | "
f"Downloaded: {self.downloaded_files} files equaling {format_size(self.total_downloaded_bytes)} (Avg: {format_size(overall_speed)}/s)"
)
for slot_str in self.slots:
lines.append(slot_str.ljust(100))
if self.initialized:
sys.stdout.write(f"\033[{len(lines)}A")
else:
self.initialized = True
sys.stdout.write("\n".join(lines) + "\n")
sys.stdout.flush()
async def ui_loop(renderer):
if sys.platform == 'win32':
import ctypes
kernel32 = ctypes.windll.kernel32
kernel32.SetConsoleMode(kernel32.GetStdHandle(-11), 7)
while True:
await renderer.render()
await asyncio.sleep(0.2)
# Persistent parallel connection downloader to avoid connection churn
class PersistentParallelDownloader:
def __init__(self, client, dc_id, connection_count):
self.client = client
self.dc_id = dc_id
self.connection_count = connection_count
self.senders = []
self.auth_key = None
async def initialize(self):
self.auth_key = (
None
if self.dc_id and self.client.session.dc_id != self.dc_id
else self.client.session.auth_key
)
# Connect persistent MTProtoSenders sequentially to prevent session ID conflicts
for i in range(self.connection_count):
dc = await self.client._get_dc(self.dc_id)
sender = MTProtoSender(self.auth_key, loggers=self.client._log, retries=10, delay=1, auto_reconnect=True)
await sender.connect(
self.client._connection(
dc.ip_address,
dc.port,
dc.id,
loggers=self.client._log,
proxy=self.client._proxy,
)
)
if not self.auth_key:
auth = await self.client(ExportAuthorizationRequest(self.dc_id))
self.client._init_request.query = ImportAuthorizationRequest(
id=auth.id, bytes=auth.bytes
)
req = InvokeWithLayerRequest(LAYER, self.client._init_request)
await sender.send(req)
self.auth_key = sender.auth_key
elif i > 0 and not sender.auth_key:
sender.auth_key = self.auth_key
self.senders.append(sender)
async def download_file(self, input_file_location, size, out, progress_callback=None):
part_size_kb = utils.get_appropriated_part_size(size)
part_size = part_size_kb * 1024
part_count = math.ceil(size / part_size)
connections = self.connection_count
minimum, remainder = divmod(part_count, connections)
def get_part_count():
nonlocal remainder
if remainder > 0:
remainder -= 1
return minimum + 1
return minimum
download_senders = []
for i in range(connections):
ds = DownloadSender(
self.client,
self.senders[i],
input_file_location, # Pass the correct InputFileLocation subclass, not the Document TLObject
offset=i * part_size,
limit=part_size,
stride=connections * part_size,
count=get_part_count()
)
download_senders.append(ds)
part = 0
while part < part_count:
tasks = []
for ds in download_senders:
tasks.append(self.client.loop.create_task(ds.next()))
try:
for task in tasks:
data = await task
if not data:
break
out.write(data)
part += 1
if progress_callback:
r = progress_callback(out.tell(), size)
if inspect.isawaitable(r):
await r
except Exception:
for task in tasks:
if not task.done():
task.cancel()
await asyncio.gather(*tasks, return_exceptions=True)
raise
async def disconnect_all(self):
await asyncio.gather(*[sender.disconnect() for sender in self.senders if sender], return_exceptions=True)
self.senders = []
async def get_downloader(client, dc_id, connection_count):
async with downloader_lock:
if dc_id in downloaders:
dl = downloaders[dc_id]
# Ensure existing downloaders have all active senders matching connection count
all_alive = len(dl.senders) == connection_count and all(s.is_connected() for s in dl.senders)
if not all_alive:
await dl.disconnect_all()
del downloaders[dc_id]
if dc_id not in downloaders:
dl = PersistentParallelDownloader(client, dc_id, connection_count)
await dl.initialize()
downloaders[dc_id] = dl
return downloaders[dc_id]
async def download_media_fast(client, chat, message, filepath, callback, connection_count=8):
if message.document:
dc_id, input_file_location = utils.get_input_location(message.document)
dl = await get_downloader(client, dc_id, connection_count)
with open(filepath, "wb") as f:
await dl.download_file(input_file_location, message.document.size, f, progress_callback=callback)
else:
# Photos are small enough that native download works perfectly
await client.download_media(message, file=filepath, progress_callback=callback)
async def download_file_task(client, chat, message, filepath, filename, file_size, index,
renderer, semaphore, history_file, downloaded_history, history_lock, connection_count=8):
msg_id_str = str(message.id)
slot_idx = await renderer.get_free_slot()
async with semaphore:
start_time = time.time()
def callback(received, total):
asyncio.create_task(renderer.update_slot(slot_idx, filename, received, total or file_size, start_time))
success = False
retries = 0
max_retries = 5
while retries < max_retries:
try:
await download_media_fast(client, chat, message, filepath, callback, connection_count=connection_count)
success = True
break
except errors.FileReferenceExpiredError:
retries += 1
try:
refreshed = await client.get_messages(chat, ids=message.id)
if refreshed and refreshed.media:
message = refreshed
except Exception:
pass
if retries >= max_retries:
break
try:
if os.path.exists(filepath):
os.truncate(filepath, 0)
except Exception:
pass
await asyncio.sleep(1)
except errors.FloodWaitError as e:
await asyncio.sleep(e.seconds)
except (errors.RPCError, asyncio.TimeoutError, ConnectionError, Exception):
retries += 1
try:
if message.document:
dc_id, _ = utils.get_input_location(message.document)
async with downloader_lock:
dl = downloaders.pop(dc_id, None)
if dl:
await dl.disconnect_all()
except Exception:
pass
if retries >= max_retries:
break
try:
if os.path.exists(filepath):
os.truncate(filepath, 0)
except Exception:
pass
await asyncio.sleep(retries * 3)
actual_size = os.path.getsize(filepath) if success and os.path.exists(filepath) else 0
await renderer.release_slot(slot_idx)
if success:
await renderer.add_bytes(actual_size)
await renderer.log(f"[FINISHED] '{filename}' downloaded successfully ({format_size(actual_size)}).")
async with history_lock:
downloaded_history[msg_id_str] = {
"filename": filename,
"size": actual_size,
"timestamp": time.strftime("%Y-%m-%d %H:%M:%S")
}
try:
with open(history_file, 'w', encoding='utf-8') as f:
json.dump(downloaded_history, f, indent=4, ensure_ascii=False)
except Exception:
pass
else:
await renderer.add_bytes(0)
async def main():
if len(sys.argv) < 5:
print("Usage: python download_telegram.py <api_id> <api_hash> <phone> <chat_username> [output_dir] [concurrency] [connections_per_file]")
sys.exit(1)
api_id = int(sys.argv[1])
api_hash = sys.argv[2]
phone = sys.argv[3]
try:
chat_username = int(sys.argv[4])
except ValueError:
chat_username = sys.argv[4]
output_dir = sys.argv[5] if len(sys.argv) > 5 else "telegram_downloads"
concurrency = int(sys.argv[6]) if len(sys.argv) > 6 else 1
connections_per_file = int(sys.argv[7]) if len(sys.argv) > 7 else 8
os.makedirs(output_dir, exist_ok=True)
history_file = os.path.join(output_dir, "download_history.json")
downloaded_history = {}
if os.path.exists(history_file):
try:
with open(history_file, 'r', encoding='utf-8') as f:
downloaded_history = json.load(f)
except Exception as e:
print(f"Warning: Failed to load download history: {e}")
client = TelegramClient('session_dumdum', api_id, api_hash)
await client.start(phone=phone)
print("LOGGED_IN")
chat = None
try:
chat = await client.get_entity(chat_username)
except Exception as e:
if isinstance(chat_username, int) and chat_username < 0 and not str(chat_username).startswith("-100"):
try:
alternative_id = int(f"-100{abs(chat_username)}")
print(f"Failed to resolve {chat_username}. Retrying with channel ID format {alternative_id}...")
chat = await client.get_entity(alternative_id)
except Exception:
pass
# If entity not found in session cache (common for deleted accounts or raw user IDs), iterate dialogs to find entity and access hash
if not chat:
print(f"Direct get_entity failed ({e}). Searching dialogs for ID {chat_username}...")
target_id = chat_username if isinstance(chat_username, int) else None
if target_id is None:
try:
target_id = int(str(chat_username).strip())
except ValueError:
target_id = None
async for dialog in client.iter_dialogs():
if target_id is not None and dialog.id == target_id:
chat = dialog.input_entity
break
elif getattr(dialog.entity, 'username', None) and dialog.entity.username.lower() == str(chat_username).lower().lstrip('@'):
chat = dialog.input_entity
break
elif target_id is not None and getattr(dialog.entity, 'id', None) == target_id:
chat = dialog.input_entity
break
if not chat:
print(f"Error getting chat: {e}")
await client.disconnect()
sys.exit(1)
print(f"Connected to chat: {chat_username}")
print("Scanning chat history to count files. Please wait...")
media_messages = []
async for message in client.iter_messages(chat):
if message.media:
media_messages.append(message)
media_messages.reverse()
total_files = len(media_messages)
print(f"Found {total_files} media files in total.")
renderer = ConcurrentProgressRenderer(total_files, concurrency)
semaphore = asyncio.Semaphore(concurrency)
history_lock = asyncio.Lock()
tasks_to_run = []
assigned_filenames = {}
for message in media_messages:
msg_id_str = str(message.id)
filename = None
file_size = 0
if isinstance(message.media, MessageMediaPhoto):
filename = f"photo_{message.id}.jpg"
if hasattr(message.media, 'photo') and message.media.photo:
file_size = getattr(message.media.photo, 'sizes', [None])[-1]
file_size = getattr(file_size, 'size', 0) if file_size else 0
else:
if message.file:
filename = message.file.name
file_size = message.file.size
if not filename:
ext = message.file.ext if message.file and message.file.ext else '.bin'
filename = f"file_{message.id}{ext}"
if filename in assigned_filenames and assigned_filenames[filename] != msg_id_str:
base, extension = os.path.splitext(filename)
filename = f"{base}_{message.id}{extension}"
assigned_filenames[filename] = msg_id_str
filepath = os.path.join(output_dir, filename)
if os.path.exists(filepath):
local_size = os.path.getsize(filepath)
if file_size > 0 and local_size == file_size:
print(f"[SKIP] '{filename}' already exists and is complete ({format_size(local_size)}).")
if msg_id_str not in downloaded_history:
downloaded_history[msg_id_str] = {
"filename": filename,
"size": local_size,
"timestamp": time.strftime("%Y-%m-%d %H:%M:%S")
}
await renderer.increment_skipped(local_size)
continue
elif file_size > 0 and local_size != file_size:
print(f"[OVERWRITE] '{filename}' is incomplete (local: {format_size(local_size)}, expected: {format_size(file_size)}). Overwriting...")
else:
print(f"[OVERWRITE] '{filename}' size unknown or conflict. Overwriting...")
task = download_file_task(
client, chat, message, filepath, filename, file_size, len(tasks_to_run) + 1,
renderer, semaphore, history_file, downloaded_history, history_lock,
connection_count=connections_per_file
)
tasks_to_run.append(task)
print("\nStarting downloads...")
ui_task = asyncio.create_task(ui_loop(renderer))
if tasks_to_run:
await asyncio.gather(*tasks_to_run)
ui_task.cancel()
await renderer.render()
# Close persistent downloaders
for dl in downloaders.values():
await dl.disconnect_all()
try:
with open(history_file, 'w', encoding='utf-8') as f:
json.dump(downloaded_history, f, indent=4, ensure_ascii=False)
except Exception:
pass
print(f"\n\nFINISHED.")
print(f"Total media files: {total_files}")
print(f"Newly downloaded/retried: {len(tasks_to_run)}")
print(f"Skipped: {renderer.skipped_files}")
print(f"Output folder: {os.path.abspath(output_dir)}")
await client.disconnect()
if __name__ == '__main__':
asyncio.run(main())
+224
View File
@@ -0,0 +1,224 @@
import os
import sys
import re
import asyncio
import argparse
from pathlib import Path
from urllib.parse import urlparse
from tqdm import tqdm
# Add parent workspace directory to path to import scraper_core
sys.path.append(str(Path(__file__).resolve().parents[1]))
import scraper_core
BASE_DIR = Path(__file__).resolve().parent
DOWNLOAD_DIR = BASE_DIR / "videos"
async def extract_telegram_media(page):
"""Scrape the current page for video and image links, returning (media_list, min_msg_id)."""
messages = await page.query_selector_all(".tgme_widget_message")
media_list = []
msg_ids = []
for msg in messages:
# Extract message link to get the ID for pagination
link_el = await msg.query_selector("a.tgme_widget_message_date")
if not link_el:
continue
href = await link_el.get_attribute("href")
if not href:
continue
parsed_path = urlparse(href).path.rstrip("/").split("/")
if not parsed_path or not parsed_path[-1].isdigit():
continue
msg_id = int(parsed_path[-1])
msg_ids.append(msg_id)
# 1. Look for Video
video_el = await msg.query_selector(".tgme_widget_message_video_player video")
if video_el:
video_src = await video_el.get_attribute("src")
if video_src:
media_list.append({
"id": msg_id,
"url": video_src,
"type": "video",
"ext": ".mp4"
})
continue
# 2. Look for Image (Photo)
photo_el = await msg.query_selector(".tgme_widget_message_photo_wrap")
if photo_el:
style = await photo_el.get_attribute("style")
if style:
# Extract URL from background-image: url('...')
m = re.search(r"background-image:\s*url\(['\"]?(https://[^'\"]+)['\"]?\)", style)
if m:
media_list.append({
"id": msg_id,
"url": m.group(1),
"type": "photo",
"ext": ".jpg"
})
continue
min_id = min(msg_ids) if msg_ids else None
return media_list, min_id
async def scrape_channel(page, channel_name: str, limit: int):
"""Crawl a public Telegram channel backwards in time to gather media URLs."""
tqdm.write(f"Scraping channel '{channel_name}' ...")
base_url = f"https://t.me/s/{channel_name}"
all_media = []
seen_ids = set()
current_url = base_url
while len(all_media) < limit:
tqdm.write(f" Fetching page: {current_url}")
try:
await page.goto(current_url, wait_until="domcontentloaded", timeout=30000)
await page.wait_for_timeout(3000)
except Exception as e:
tqdm.write(f" Error loading Telegram web page: {e}")
break
page_media, min_id = await extract_telegram_media(page)
# Filter new media
new_items = []
for item in page_media:
if item["id"] not in seen_ids:
seen_ids.add(item["id"])
new_items.append(item)
if not new_items:
tqdm.write(" No new media found on this page.")
break
all_media.extend(new_items)
tqdm.write(f" Found {len(new_items)} new media items (Total collected: {len(all_media)})")
if not min_id:
break
# Paginate to messages before the minimum ID we've seen
current_url = f"{base_url}?before={min_id}"
await page.wait_for_timeout(1000)
return all_media[:limit]
async def worker(queue, scraper, skip_existing, bar_pool, overall_bar, channel_name):
"""Worker task that downloads media concurrently."""
while True:
item = await queue.get()
if item is None:
queue.task_done()
break
msg_id, url, mtype, ext = item
filename = f"msg_{msg_id}{ext}"
dest_path = DOWNLOAD_DIR / channel_name / filename
if skip_existing and scraper_core.is_already_downloaded(dest_path):
overall_bar.update(1)
queue.task_done()
continue
pos = bar_pool.acquire() or 1
success = await asyncio.to_thread(
scraper_core.download_file,
url, dest_path, None, pos, f"https://t.me/s/{channel_name}"
)
bar_pool.release(pos)
overall_bar.update(1)
queue.task_done()
async def process_channel(channel_name, limit, concurrency, skip_existing):
scraper = scraper_core.PlaywrightScraper()
await scraper.start()
page = await scraper.new_page()
media_items = await scrape_channel(page, channel_name, limit)
await page.close()
if not media_items:
tqdm.write("No media files found to download.")
await scraper.close()
return
tqdm.write(f"Downloading {len(media_items)} files with concurrency {concurrency} ...")
queue = asyncio.Queue()
for item in media_items:
await queue.put((item["id"], item["url"], item["type"], item["ext"]))
for _ in range(concurrency):
await queue.put(None)
bar_pool = scraper_core.BarPositionPool(concurrency)
overall_bar = tqdm(
total=len(media_items),
desc=f"Channel: {channel_name}",
position=0,
leave=True,
ncols=80,
)
workers = [
asyncio.create_task(worker(queue, scraper, skip_existing, bar_pool, overall_bar, channel_name))
for _ in range(concurrency)
]
await asyncio.gather(*workers)
overall_bar.close()
sys.stdout.write("\n" * (concurrency + 1))
sys.stdout.flush()
await scraper.close()
def main():
parser = argparse.ArgumentParser(
description="Download media from public Telegram channels (headless, concurrent, multithreaded)."
)
parser.add_argument(
"channel",
help="Telegram channel username (e.g. 'durov')",
)
parser.add_argument(
"--limit",
type=int,
default=50,
help="Maximum number of media items to download (default: 50).",
)
parser.add_argument(
"--concurrency",
type=int,
default=3,
help="Number of concurrent downloads (default: 3).",
)
parser.add_argument(
"--skip-existing",
action="store_true",
default=True,
help="Skip already-downloaded files (default: true).",
)
parser.add_argument(
"--no-skip-existing",
action="store_false",
dest="skip_existing",
help="Re-download existing files.",
)
args = parser.parse_args()
asyncio.run(process_channel(args.channel, args.limit, args.concurrency, args.skip_existing))
if __name__ == "__main__":
main()
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff