Initial backup: folder structure manifest, tree summary, scraper scripts, and text metadata
This commit is contained in:
commit
f164832dcc
10504 files changed
+2423594
No files matched your search
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,87 @@
|
||||
{
|
||||
"11392": {
|
||||
"filename": "Piss marking curtains in public store display.mp4",
|
||||
"size": 8917618,
|
||||
"timestamp": "2026-09-22 03:32:20"
|
||||
},
|
||||
"11393": {
|
||||
"filename": "Piss marking in public store isle.mp4",
|
||||
"size": 9657311,
|
||||
"timestamp": "2026-09-22 03:32:21"
|
||||
},
|
||||
"11394": {
|
||||
"filename": "Piss marking public shop products.mp4",
|
||||
"size": 14868518,
|
||||
"timestamp": "2026-09-22 03:32:24"
|
||||
},
|
||||
"11395": {
|
||||
"filename": "Piss marking public store plastic plants.mp4",
|
||||
"size": 9743461,
|
||||
"timestamp": "2026-09-22 03:32:26"
|
||||
},
|
||||
"11396": {
|
||||
"filename": "Piss marking store tool shelf.mp4",
|
||||
"size": 5746973,
|
||||
"timestamp": "2026-09-22 03:32:27"
|
||||
},
|
||||
"11397": {
|
||||
"filename": "Piss on store cushion shelf.mp4",
|
||||
"size": 11506623,
|
||||
"timestamp": "2026-09-22 03:32:29"
|
||||
},
|
||||
"11398": {
|
||||
"filename": "Pissmarking pisstrashing pisstagging public store shelf.mp4",
|
||||
"size": 5242276,
|
||||
"timestamp": "2026-09-22 03:32:30"
|
||||
},
|
||||
"11399": {
|
||||
"filename": "Poked by the family dog.mp4",
|
||||
"size": 42708485,
|
||||
"timestamp": "2026-09-22 03:32:37"
|
||||
},
|
||||
"11400": {
|
||||
"filename": "Slut Fisted.mp4",
|
||||
"size": 55603897,
|
||||
"timestamp": "2026-09-22 03:32:45"
|
||||
},
|
||||
"11401": {
|
||||
"filename": "SLUTS Babecock Dogbabecock.mp4",
|
||||
"size": 27741185,
|
||||
"timestamp": "2026-09-29 15:43:57"
|
||||
},
|
||||
"11402": {
|
||||
"filename": "Teen boy gets fucked by dog.mp4",
|
||||
"size": 19474961,
|
||||
"timestamp": "2026-09-29 15:43:59"
|
||||
},
|
||||
"11403": {
|
||||
"filename": "Tight assed twink gets knotted.mp4",
|
||||
"size": 37933931,
|
||||
"timestamp": "2026-09-22 03:33:01"
|
||||
},
|
||||
"11404": {
|
||||
"filename": "toilet licker.mp4",
|
||||
"size": 13355478,
|
||||
"timestamp": "2026-09-22 03:33:03"
|
||||
},
|
||||
"11405": {
|
||||
"filename": "Used asshole fucks trailer hitch on back of pickup.mp4",
|
||||
"size": 21990986,
|
||||
"timestamp": "2026-09-22 03:33:07"
|
||||
},
|
||||
"11406": {
|
||||
"filename": "Woman pissing on shirt in a store.mp4",
|
||||
"size": 14836034,
|
||||
"timestamp": "2026-09-22 03:33:09"
|
||||
},
|
||||
"11407": {
|
||||
"filename": "Young pussy home with the family dog.mp4",
|
||||
"size": 8537003,
|
||||
"timestamp": "2026-09-29 15:44:00"
|
||||
},
|
||||
"11408": {
|
||||
"filename": "Yummy fuck with the dog with pullout.mp4",
|
||||
"size": 4902392,
|
||||
"timestamp": "2026-09-29 15:44:01"
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,122 @@
|
||||
{
|
||||
"4": {
|
||||
"filename": "1.mp4",
|
||||
"size": 3985475674,
|
||||
"timestamp": "2026-09-22 01:41:59"
|
||||
},
|
||||
"5": {
|
||||
"filename": "Luna Knots Public_1_prob4.mp4",
|
||||
"size": 2143382637,
|
||||
"timestamp": "2026-09-22 01:44:01"
|
||||
},
|
||||
"6": {
|
||||
"filename": "Beastluna - Exotic whore.mp4",
|
||||
"size": 184781451,
|
||||
"timestamp": "2026-09-22 01:44:34"
|
||||
},
|
||||
"7": {
|
||||
"filename": "Isabella Productions - Pole Dance.mp4",
|
||||
"size": 1388511206,
|
||||
"timestamp": "2026-09-22 01:51:18"
|
||||
},
|
||||
"8": {
|
||||
"filename": "Coral - Pain & Pleasure (Unmasked).mp4",
|
||||
"size": 338594451,
|
||||
"timestamp": "2026-09-22 01:52:17"
|
||||
},
|
||||
"9": {
|
||||
"filename": "ida-dog-01-1024-kbps_processed.mp4",
|
||||
"size": 336775091,
|
||||
"timestamp": "2026-09-22 01:53:19"
|
||||
},
|
||||
"10": {
|
||||
"filename": "ida-dog-06-1024-kbps_processed.mp4",
|
||||
"size": 369164316,
|
||||
"timestamp": "2026-09-22 01:54:23"
|
||||
},
|
||||
"11": {
|
||||
"filename": "Isabella - Enter The Dog.mp4",
|
||||
"size": 454529982,
|
||||
"timestamp": "2026-09-22 01:55:47"
|
||||
},
|
||||
"12": {
|
||||
"filename": "1_4929445926826673411.mp4",
|
||||
"size": 518245769,
|
||||
"timestamp": "2026-09-22 01:56:41"
|
||||
},
|
||||
"13": {
|
||||
"filename": "1_4929445926826673412.mp4",
|
||||
"size": 1345147005,
|
||||
"timestamp": "2026-09-22 01:58:04"
|
||||
},
|
||||
"14": {
|
||||
"filename": "1_5019591947430397220.MOV",
|
||||
"size": 148046883,
|
||||
"timestamp": "2026-09-22 01:58:20"
|
||||
},
|
||||
"15": {
|
||||
"filename": "1_5019591947430397268.MOV",
|
||||
"size": 186380058,
|
||||
"timestamp": "2026-09-22 01:58:38"
|
||||
},
|
||||
"16": {
|
||||
"filename": "1_16.mp4",
|
||||
"size": 4003447897,
|
||||
"timestamp": "2026-09-22 02:11:55"
|
||||
},
|
||||
"17": {
|
||||
"filename": "2.mp4",
|
||||
"size": 1224567277,
|
||||
"timestamp": "2026-09-22 02:15:57"
|
||||
},
|
||||
"18": {
|
||||
"filename": "Wanwan - Chinese Burn Artofzoo 1080p.mp4",
|
||||
"size": 1560341210,
|
||||
"timestamp": "2026-09-22 02:23:21"
|
||||
},
|
||||
"19": {
|
||||
"filename": "ivana-project-grindhound-chapter-three-two-dogs-exclusive.mp4",
|
||||
"size": 2289901816,
|
||||
"timestamp": "2026-09-22 02:25:46"
|
||||
},
|
||||
"20": {
|
||||
"filename": "Boar Corps 4.mp4",
|
||||
"size": 1547122707,
|
||||
"timestamp": "2026-09-22 02:27:23"
|
||||
},
|
||||
"21": {
|
||||
"filename": "Ivana in Project GrindHound - Chapter Four.mp4",
|
||||
"size": 1536386785,
|
||||
"timestamp": "2026-09-22 02:28:57"
|
||||
},
|
||||
"22": {
|
||||
"filename": "aoz.hot.rocks.mp4",
|
||||
"size": 976756414,
|
||||
"timestamp": "2026-09-22 02:31:52"
|
||||
},
|
||||
"23": {
|
||||
"filename": "photo_23.jpg",
|
||||
"size": 350017,
|
||||
"timestamp": "2026-09-29 15:43:51"
|
||||
},
|
||||
"35": {
|
||||
"filename": "Coral & Mia - Horse Sensation.mp4",
|
||||
"size": 1174824933,
|
||||
"timestamp": "2026-09-22 02:35:33"
|
||||
},
|
||||
"36": {
|
||||
"filename": "SquirtSquid - Rider Nun.mp4",
|
||||
"size": 1088016143,
|
||||
"timestamp": "2026-09-22 02:38:56"
|
||||
},
|
||||
"58": {
|
||||
"filename": "MakingcummyDoberman Kittykn9ne - Custom - EXCLUSIVE.mp4",
|
||||
"size": 3195353718,
|
||||
"timestamp": "2026-09-22 02:48:40"
|
||||
},
|
||||
"59": {
|
||||
"filename": "Magnum_Equus.mp4",
|
||||
"size": 3608631632,
|
||||
"timestamp": "2026-09-22 03:08:49"
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,9 @@
|
||||
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-3742774514" "V:\smalldata\niggers\Telegram Desktop\KPS Premium" 1 8
|
||||
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "7522570451" "V:\smalldata\niggers\Telegram Desktop\Deleted Account" 1 8
|
||||
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "8025093129" "V:\smalldata\niggers\Telegram Desktop\Garcia Sarah" 1 8
|
||||
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-4418305717" "V:\smalldata\niggers\Telegram Desktop\Shared Spaces" 1 8
|
||||
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "@Knotslut303" "V:\smalldata\niggers\Telegram Desktop\Knottyknot💦💦❣️" 1 8
|
||||
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-2880246536" "V:\smalldata\niggers\Telegram Desktop\LW Zoo" 1 8
|
||||
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-3738137252" "V:\smalldata\niggers\Telegram Desktop\Stock Shares" 1 8
|
||||
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-4240729023" "V:\smalldata\niggers\Telegram Desktop\Super Cool People Club" 1 8
|
||||
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "@Dumdumisdumb" "V:\smalldata\niggers\Telegram Desktop\Dumdumisdumb" 1 8
|
||||
@@ -0,0 +1 @@
|
||||
python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-3738137252" "V:\smalldata\niggers\Telegram Desktop\Stock Shares" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-4240729023" "V:\smalldata\niggers\Telegram Desktop\Super Cool People Club" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-4418305717" "V:\smalldata\niggers\Telegram Desktop\Shared Spaces" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "8025093129" "V:\smalldata\niggers\Telegram Desktop\Garcia Sarah" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-2880246536" "V:\smalldata\niggers\Telegram Desktop\LW Zoo" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "@Knotslut303" "V:\smalldata\niggers\Telegram Desktop\Knottyknot💦💦❣️" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "7522570451" "V:\smalldata\niggers\Telegram Desktop\Deleted Account" 1 8 && python3 download_telegram.py 24463886 1e5a30a6edcc5ab25608f83d0d290788 "+19788060912" "-3742774514" "V:\smalldata\niggers\Telegram Desktop\KPS Premium" 1 8
|
||||
@@ -0,0 +1,513 @@
|
||||
import os
|
||||
import sys
|
||||
import json
|
||||
import time
|
||||
import math
|
||||
import asyncio
|
||||
import inspect
|
||||
import logging
|
||||
from telethon import TelegramClient, errors, utils
|
||||
from telethon.tl.functions.auth import ExportAuthorizationRequest, ImportAuthorizationRequest
|
||||
from telethon.tl.functions import InvokeWithLayerRequest
|
||||
from telethon.tl.types import MessageMediaPhoto, MessageMediaDocument
|
||||
from telethon.network import MTProtoSender
|
||||
from telethon.tl.alltlobjects import LAYER
|
||||
from FastTelethonhelper.FastTelethon import DownloadSender
|
||||
|
||||
# Suppress Telethon's internal network warnings (e.g., transient server-closed socket resets that auto-reconnect)
|
||||
logging.basicConfig(level=logging.ERROR)
|
||||
logging.getLogger('telethon').setLevel(logging.ERROR)
|
||||
logging.getLogger('asyncio').setLevel(logging.ERROR)
|
||||
|
||||
# Global cache for persistent downloaders per DC ID to prevent connection churn
|
||||
downloaders = {}
|
||||
downloader_lock = asyncio.Lock()
|
||||
|
||||
# Helper function to format sizes
|
||||
def format_size(bytes_count):
|
||||
if bytes_count < 1024:
|
||||
return f"{bytes_count} B"
|
||||
elif bytes_count < 1024 * 1024:
|
||||
return f"{bytes_count / 1024:.2f} KB"
|
||||
elif bytes_count < 1024 * 1024 * 1024:
|
||||
return f"{bytes_count / (1024 * 1024):.2f} MB"
|
||||
else:
|
||||
return f"{bytes_count / (1024 * 1024 * 1024):.2f} GB"
|
||||
|
||||
# Shared statistics and UI rendering
|
||||
class ConcurrentProgressRenderer:
|
||||
def __init__(self, total_files, concurrency):
|
||||
self.total_files = total_files
|
||||
self.concurrency = concurrency
|
||||
self.completed_files = 0
|
||||
self.downloaded_files = 0
|
||||
self.skipped_files = 0
|
||||
self.skipped_bytes = 0
|
||||
self.total_downloaded_bytes = 0
|
||||
self.start_time = time.time()
|
||||
self.slots = [""] * concurrency
|
||||
self.lock = asyncio.Lock()
|
||||
self.initialized = False
|
||||
|
||||
async def get_free_slot(self):
|
||||
async with self.lock:
|
||||
for i in range(self.concurrency):
|
||||
if self.slots[i] == "":
|
||||
self.slots[i] = "Initializing..."
|
||||
return i
|
||||
return -1
|
||||
|
||||
async def update_slot(self, slot_idx, filename, received, total, start_time):
|
||||
if not total:
|
||||
total = 1
|
||||
percent = (received / total) * 100
|
||||
bar_len = 15
|
||||
filled_len = int(bar_len * received // total)
|
||||
bar = '█' * filled_len + '░' * (bar_len - filled_len)
|
||||
|
||||
now = time.time()
|
||||
elapsed = now - start_time
|
||||
speed = received / elapsed if elapsed > 0 else 0
|
||||
speed_str = f"{format_size(speed)}/s"
|
||||
size_str = f"{format_size(received)}/{format_size(total)}"
|
||||
|
||||
display_name = filename
|
||||
if len(display_name) > 20:
|
||||
display_name = display_name[:9] + "..." + display_name[-8:]
|
||||
|
||||
async with self.lock:
|
||||
self.slots[slot_idx] = f"Slot {slot_idx+1}: {display_name} [{bar}] {percent:.1f}% ({size_str}) @ {speed_str}"
|
||||
|
||||
async def release_slot(self, slot_idx):
|
||||
async with self.lock:
|
||||
self.slots[slot_idx] = ""
|
||||
|
||||
async def add_bytes(self, size):
|
||||
async with self.lock:
|
||||
self.total_downloaded_bytes += size
|
||||
self.downloaded_files += 1
|
||||
self.completed_files += 1
|
||||
|
||||
async def increment_skipped(self, size=0):
|
||||
async with self.lock:
|
||||
self.skipped_files += 1
|
||||
self.skipped_bytes += size
|
||||
self.completed_files += 1
|
||||
|
||||
async def log(self, message):
|
||||
async with self.lock:
|
||||
if self.initialized:
|
||||
num_lines = 2 + len(self.slots)
|
||||
sys.stdout.write(f"\033[{num_lines}A")
|
||||
sys.stdout.write("\033[J")
|
||||
sys.stdout.write(message + "\n")
|
||||
sys.stdout.flush()
|
||||
self.initialized = False
|
||||
|
||||
async def render(self):
|
||||
async with self.lock:
|
||||
elapsed = time.time() - self.start_time
|
||||
overall_speed = self.total_downloaded_bytes / elapsed if elapsed > 0 else 0
|
||||
|
||||
percent = (self.completed_files / self.total_files * 100) if self.total_files > 0 else 0.0
|
||||
bar_len = 50
|
||||
filled_len = int(bar_len * self.completed_files // self.total_files) if self.total_files > 0 else 0
|
||||
bar = '█' * filled_len + ' ' * (bar_len - filled_len)
|
||||
|
||||
lines = []
|
||||
lines.append(
|
||||
f"Overall Progress: {self.completed_files}/{self.total_files} |{bar}| {percent:.2f}%"
|
||||
)
|
||||
lines.append(
|
||||
f"Skipped: {self.skipped_files} files equaling {format_size(self.skipped_bytes)} | "
|
||||
f"Downloaded: {self.downloaded_files} files equaling {format_size(self.total_downloaded_bytes)} (Avg: {format_size(overall_speed)}/s)"
|
||||
)
|
||||
|
||||
for slot_str in self.slots:
|
||||
lines.append(slot_str.ljust(100))
|
||||
|
||||
if self.initialized:
|
||||
sys.stdout.write(f"\033[{len(lines)}A")
|
||||
else:
|
||||
self.initialized = True
|
||||
|
||||
sys.stdout.write("\n".join(lines) + "\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
async def ui_loop(renderer):
|
||||
if sys.platform == 'win32':
|
||||
import ctypes
|
||||
kernel32 = ctypes.windll.kernel32
|
||||
kernel32.SetConsoleMode(kernel32.GetStdHandle(-11), 7)
|
||||
|
||||
while True:
|
||||
await renderer.render()
|
||||
await asyncio.sleep(0.2)
|
||||
|
||||
# Persistent parallel connection downloader to avoid connection churn
|
||||
class PersistentParallelDownloader:
|
||||
def __init__(self, client, dc_id, connection_count):
|
||||
self.client = client
|
||||
self.dc_id = dc_id
|
||||
self.connection_count = connection_count
|
||||
self.senders = []
|
||||
self.auth_key = None
|
||||
|
||||
async def initialize(self):
|
||||
self.auth_key = (
|
||||
None
|
||||
if self.dc_id and self.client.session.dc_id != self.dc_id
|
||||
else self.client.session.auth_key
|
||||
)
|
||||
|
||||
# Connect persistent MTProtoSenders sequentially to prevent session ID conflicts
|
||||
for i in range(self.connection_count):
|
||||
dc = await self.client._get_dc(self.dc_id)
|
||||
sender = MTProtoSender(self.auth_key, loggers=self.client._log, retries=10, delay=1, auto_reconnect=True)
|
||||
await sender.connect(
|
||||
self.client._connection(
|
||||
dc.ip_address,
|
||||
dc.port,
|
||||
dc.id,
|
||||
loggers=self.client._log,
|
||||
proxy=self.client._proxy,
|
||||
)
|
||||
)
|
||||
if not self.auth_key:
|
||||
auth = await self.client(ExportAuthorizationRequest(self.dc_id))
|
||||
self.client._init_request.query = ImportAuthorizationRequest(
|
||||
id=auth.id, bytes=auth.bytes
|
||||
)
|
||||
req = InvokeWithLayerRequest(LAYER, self.client._init_request)
|
||||
await sender.send(req)
|
||||
self.auth_key = sender.auth_key
|
||||
elif i > 0 and not sender.auth_key:
|
||||
sender.auth_key = self.auth_key
|
||||
|
||||
self.senders.append(sender)
|
||||
|
||||
async def download_file(self, input_file_location, size, out, progress_callback=None):
|
||||
part_size_kb = utils.get_appropriated_part_size(size)
|
||||
part_size = part_size_kb * 1024
|
||||
part_count = math.ceil(size / part_size)
|
||||
|
||||
connections = self.connection_count
|
||||
minimum, remainder = divmod(part_count, connections)
|
||||
|
||||
def get_part_count():
|
||||
nonlocal remainder
|
||||
if remainder > 0:
|
||||
remainder -= 1
|
||||
return minimum + 1
|
||||
return minimum
|
||||
|
||||
download_senders = []
|
||||
for i in range(connections):
|
||||
ds = DownloadSender(
|
||||
self.client,
|
||||
self.senders[i],
|
||||
input_file_location, # Pass the correct InputFileLocation subclass, not the Document TLObject
|
||||
offset=i * part_size,
|
||||
limit=part_size,
|
||||
stride=connections * part_size,
|
||||
count=get_part_count()
|
||||
)
|
||||
download_senders.append(ds)
|
||||
|
||||
part = 0
|
||||
while part < part_count:
|
||||
tasks = []
|
||||
for ds in download_senders:
|
||||
tasks.append(self.client.loop.create_task(ds.next()))
|
||||
try:
|
||||
for task in tasks:
|
||||
data = await task
|
||||
if not data:
|
||||
break
|
||||
out.write(data)
|
||||
part += 1
|
||||
if progress_callback:
|
||||
r = progress_callback(out.tell(), size)
|
||||
if inspect.isawaitable(r):
|
||||
await r
|
||||
except Exception:
|
||||
for task in tasks:
|
||||
if not task.done():
|
||||
task.cancel()
|
||||
await asyncio.gather(*tasks, return_exceptions=True)
|
||||
raise
|
||||
|
||||
async def disconnect_all(self):
|
||||
await asyncio.gather(*[sender.disconnect() for sender in self.senders if sender], return_exceptions=True)
|
||||
self.senders = []
|
||||
|
||||
async def get_downloader(client, dc_id, connection_count):
|
||||
async with downloader_lock:
|
||||
if dc_id in downloaders:
|
||||
dl = downloaders[dc_id]
|
||||
# Ensure existing downloaders have all active senders matching connection count
|
||||
all_alive = len(dl.senders) == connection_count and all(s.is_connected() for s in dl.senders)
|
||||
if not all_alive:
|
||||
await dl.disconnect_all()
|
||||
del downloaders[dc_id]
|
||||
if dc_id not in downloaders:
|
||||
dl = PersistentParallelDownloader(client, dc_id, connection_count)
|
||||
await dl.initialize()
|
||||
downloaders[dc_id] = dl
|
||||
return downloaders[dc_id]
|
||||
|
||||
async def download_media_fast(client, chat, message, filepath, callback, connection_count=8):
|
||||
if message.document:
|
||||
dc_id, input_file_location = utils.get_input_location(message.document)
|
||||
dl = await get_downloader(client, dc_id, connection_count)
|
||||
with open(filepath, "wb") as f:
|
||||
await dl.download_file(input_file_location, message.document.size, f, progress_callback=callback)
|
||||
else:
|
||||
# Photos are small enough that native download works perfectly
|
||||
await client.download_media(message, file=filepath, progress_callback=callback)
|
||||
|
||||
async def download_file_task(client, chat, message, filepath, filename, file_size, index,
|
||||
renderer, semaphore, history_file, downloaded_history, history_lock, connection_count=8):
|
||||
msg_id_str = str(message.id)
|
||||
slot_idx = await renderer.get_free_slot()
|
||||
|
||||
async with semaphore:
|
||||
start_time = time.time()
|
||||
|
||||
def callback(received, total):
|
||||
asyncio.create_task(renderer.update_slot(slot_idx, filename, received, total or file_size, start_time))
|
||||
|
||||
success = False
|
||||
retries = 0
|
||||
max_retries = 5
|
||||
|
||||
while retries < max_retries:
|
||||
try:
|
||||
await download_media_fast(client, chat, message, filepath, callback, connection_count=connection_count)
|
||||
success = True
|
||||
break
|
||||
except errors.FileReferenceExpiredError:
|
||||
retries += 1
|
||||
try:
|
||||
refreshed = await client.get_messages(chat, ids=message.id)
|
||||
if refreshed and refreshed.media:
|
||||
message = refreshed
|
||||
except Exception:
|
||||
pass
|
||||
if retries >= max_retries:
|
||||
break
|
||||
try:
|
||||
if os.path.exists(filepath):
|
||||
os.truncate(filepath, 0)
|
||||
except Exception:
|
||||
pass
|
||||
await asyncio.sleep(1)
|
||||
except errors.FloodWaitError as e:
|
||||
await asyncio.sleep(e.seconds)
|
||||
except (errors.RPCError, asyncio.TimeoutError, ConnectionError, Exception):
|
||||
retries += 1
|
||||
try:
|
||||
if message.document:
|
||||
dc_id, _ = utils.get_input_location(message.document)
|
||||
async with downloader_lock:
|
||||
dl = downloaders.pop(dc_id, None)
|
||||
if dl:
|
||||
await dl.disconnect_all()
|
||||
except Exception:
|
||||
pass
|
||||
if retries >= max_retries:
|
||||
break
|
||||
try:
|
||||
if os.path.exists(filepath):
|
||||
os.truncate(filepath, 0)
|
||||
except Exception:
|
||||
pass
|
||||
await asyncio.sleep(retries * 3)
|
||||
|
||||
actual_size = os.path.getsize(filepath) if success and os.path.exists(filepath) else 0
|
||||
await renderer.release_slot(slot_idx)
|
||||
|
||||
if success:
|
||||
await renderer.add_bytes(actual_size)
|
||||
await renderer.log(f"[FINISHED] '{filename}' downloaded successfully ({format_size(actual_size)}).")
|
||||
async with history_lock:
|
||||
downloaded_history[msg_id_str] = {
|
||||
"filename": filename,
|
||||
"size": actual_size,
|
||||
"timestamp": time.strftime("%Y-%m-%d %H:%M:%S")
|
||||
}
|
||||
try:
|
||||
with open(history_file, 'w', encoding='utf-8') as f:
|
||||
json.dump(downloaded_history, f, indent=4, ensure_ascii=False)
|
||||
except Exception:
|
||||
pass
|
||||
else:
|
||||
await renderer.add_bytes(0)
|
||||
|
||||
async def main():
|
||||
if len(sys.argv) < 5:
|
||||
print("Usage: python download_telegram.py <api_id> <api_hash> <phone> <chat_username> [output_dir] [concurrency] [connections_per_file]")
|
||||
sys.exit(1)
|
||||
|
||||
api_id = int(sys.argv[1])
|
||||
api_hash = sys.argv[2]
|
||||
phone = sys.argv[3]
|
||||
try:
|
||||
chat_username = int(sys.argv[4])
|
||||
except ValueError:
|
||||
chat_username = sys.argv[4]
|
||||
output_dir = sys.argv[5] if len(sys.argv) > 5 else "telegram_downloads"
|
||||
concurrency = int(sys.argv[6]) if len(sys.argv) > 6 else 1
|
||||
connections_per_file = int(sys.argv[7]) if len(sys.argv) > 7 else 8
|
||||
|
||||
os.makedirs(output_dir, exist_ok=True)
|
||||
history_file = os.path.join(output_dir, "download_history.json")
|
||||
|
||||
downloaded_history = {}
|
||||
if os.path.exists(history_file):
|
||||
try:
|
||||
with open(history_file, 'r', encoding='utf-8') as f:
|
||||
downloaded_history = json.load(f)
|
||||
except Exception as e:
|
||||
print(f"Warning: Failed to load download history: {e}")
|
||||
|
||||
client = TelegramClient('session_dumdum', api_id, api_hash)
|
||||
|
||||
await client.start(phone=phone)
|
||||
print("LOGGED_IN")
|
||||
|
||||
chat = None
|
||||
try:
|
||||
chat = await client.get_entity(chat_username)
|
||||
except Exception as e:
|
||||
if isinstance(chat_username, int) and chat_username < 0 and not str(chat_username).startswith("-100"):
|
||||
try:
|
||||
alternative_id = int(f"-100{abs(chat_username)}")
|
||||
print(f"Failed to resolve {chat_username}. Retrying with channel ID format {alternative_id}...")
|
||||
chat = await client.get_entity(alternative_id)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# If entity not found in session cache (common for deleted accounts or raw user IDs), iterate dialogs to find entity and access hash
|
||||
if not chat:
|
||||
print(f"Direct get_entity failed ({e}). Searching dialogs for ID {chat_username}...")
|
||||
target_id = chat_username if isinstance(chat_username, int) else None
|
||||
if target_id is None:
|
||||
try:
|
||||
target_id = int(str(chat_username).strip())
|
||||
except ValueError:
|
||||
target_id = None
|
||||
|
||||
async for dialog in client.iter_dialogs():
|
||||
if target_id is not None and dialog.id == target_id:
|
||||
chat = dialog.input_entity
|
||||
break
|
||||
elif getattr(dialog.entity, 'username', None) and dialog.entity.username.lower() == str(chat_username).lower().lstrip('@'):
|
||||
chat = dialog.input_entity
|
||||
break
|
||||
elif target_id is not None and getattr(dialog.entity, 'id', None) == target_id:
|
||||
chat = dialog.input_entity
|
||||
break
|
||||
|
||||
if not chat:
|
||||
print(f"Error getting chat: {e}")
|
||||
await client.disconnect()
|
||||
sys.exit(1)
|
||||
|
||||
print(f"Connected to chat: {chat_username}")
|
||||
print("Scanning chat history to count files. Please wait...")
|
||||
|
||||
media_messages = []
|
||||
async for message in client.iter_messages(chat):
|
||||
if message.media:
|
||||
media_messages.append(message)
|
||||
|
||||
media_messages.reverse()
|
||||
total_files = len(media_messages)
|
||||
print(f"Found {total_files} media files in total.")
|
||||
|
||||
renderer = ConcurrentProgressRenderer(total_files, concurrency)
|
||||
semaphore = asyncio.Semaphore(concurrency)
|
||||
history_lock = asyncio.Lock()
|
||||
|
||||
tasks_to_run = []
|
||||
assigned_filenames = {}
|
||||
|
||||
for message in media_messages:
|
||||
msg_id_str = str(message.id)
|
||||
filename = None
|
||||
file_size = 0
|
||||
|
||||
if isinstance(message.media, MessageMediaPhoto):
|
||||
filename = f"photo_{message.id}.jpg"
|
||||
if hasattr(message.media, 'photo') and message.media.photo:
|
||||
file_size = getattr(message.media.photo, 'sizes', [None])[-1]
|
||||
file_size = getattr(file_size, 'size', 0) if file_size else 0
|
||||
else:
|
||||
if message.file:
|
||||
filename = message.file.name
|
||||
file_size = message.file.size
|
||||
if not filename:
|
||||
ext = message.file.ext if message.file and message.file.ext else '.bin'
|
||||
filename = f"file_{message.id}{ext}"
|
||||
|
||||
if filename in assigned_filenames and assigned_filenames[filename] != msg_id_str:
|
||||
base, extension = os.path.splitext(filename)
|
||||
filename = f"{base}_{message.id}{extension}"
|
||||
|
||||
assigned_filenames[filename] = msg_id_str
|
||||
filepath = os.path.join(output_dir, filename)
|
||||
|
||||
if os.path.exists(filepath):
|
||||
local_size = os.path.getsize(filepath)
|
||||
|
||||
if file_size > 0 and local_size == file_size:
|
||||
print(f"[SKIP] '{filename}' already exists and is complete ({format_size(local_size)}).")
|
||||
if msg_id_str not in downloaded_history:
|
||||
downloaded_history[msg_id_str] = {
|
||||
"filename": filename,
|
||||
"size": local_size,
|
||||
"timestamp": time.strftime("%Y-%m-%d %H:%M:%S")
|
||||
}
|
||||
await renderer.increment_skipped(local_size)
|
||||
continue
|
||||
elif file_size > 0 and local_size != file_size:
|
||||
print(f"[OVERWRITE] '{filename}' is incomplete (local: {format_size(local_size)}, expected: {format_size(file_size)}). Overwriting...")
|
||||
else:
|
||||
print(f"[OVERWRITE] '{filename}' size unknown or conflict. Overwriting...")
|
||||
|
||||
task = download_file_task(
|
||||
client, chat, message, filepath, filename, file_size, len(tasks_to_run) + 1,
|
||||
renderer, semaphore, history_file, downloaded_history, history_lock,
|
||||
connection_count=connections_per_file
|
||||
)
|
||||
tasks_to_run.append(task)
|
||||
|
||||
print("\nStarting downloads...")
|
||||
ui_task = asyncio.create_task(ui_loop(renderer))
|
||||
|
||||
if tasks_to_run:
|
||||
await asyncio.gather(*tasks_to_run)
|
||||
|
||||
ui_task.cancel()
|
||||
await renderer.render()
|
||||
|
||||
# Close persistent downloaders
|
||||
for dl in downloaders.values():
|
||||
await dl.disconnect_all()
|
||||
|
||||
try:
|
||||
with open(history_file, 'w', encoding='utf-8') as f:
|
||||
json.dump(downloaded_history, f, indent=4, ensure_ascii=False)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
print(f"\n\nFINISHED.")
|
||||
print(f"Total media files: {total_files}")
|
||||
print(f"Newly downloaded/retried: {len(tasks_to_run)}")
|
||||
print(f"Skipped: {renderer.skipped_files}")
|
||||
print(f"Output folder: {os.path.abspath(output_dir)}")
|
||||
await client.disconnect()
|
||||
|
||||
if __name__ == '__main__':
|
||||
asyncio.run(main())
|
||||
@@ -0,0 +1,224 @@
|
||||
import os
|
||||
import sys
|
||||
import re
|
||||
import asyncio
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
from urllib.parse import urlparse
|
||||
from tqdm import tqdm
|
||||
|
||||
# Add parent workspace directory to path to import scraper_core
|
||||
sys.path.append(str(Path(__file__).resolve().parents[1]))
|
||||
import scraper_core
|
||||
|
||||
BASE_DIR = Path(__file__).resolve().parent
|
||||
DOWNLOAD_DIR = BASE_DIR / "videos"
|
||||
|
||||
async def extract_telegram_media(page):
|
||||
"""Scrape the current page for video and image links, returning (media_list, min_msg_id)."""
|
||||
messages = await page.query_selector_all(".tgme_widget_message")
|
||||
media_list = []
|
||||
msg_ids = []
|
||||
|
||||
for msg in messages:
|
||||
# Extract message link to get the ID for pagination
|
||||
link_el = await msg.query_selector("a.tgme_widget_message_date")
|
||||
if not link_el:
|
||||
continue
|
||||
href = await link_el.get_attribute("href")
|
||||
if not href:
|
||||
continue
|
||||
|
||||
parsed_path = urlparse(href).path.rstrip("/").split("/")
|
||||
if not parsed_path or not parsed_path[-1].isdigit():
|
||||
continue
|
||||
msg_id = int(parsed_path[-1])
|
||||
msg_ids.append(msg_id)
|
||||
|
||||
# 1. Look for Video
|
||||
video_el = await msg.query_selector(".tgme_widget_message_video_player video")
|
||||
if video_el:
|
||||
video_src = await video_el.get_attribute("src")
|
||||
if video_src:
|
||||
media_list.append({
|
||||
"id": msg_id,
|
||||
"url": video_src,
|
||||
"type": "video",
|
||||
"ext": ".mp4"
|
||||
})
|
||||
continue
|
||||
|
||||
# 2. Look for Image (Photo)
|
||||
photo_el = await msg.query_selector(".tgme_widget_message_photo_wrap")
|
||||
if photo_el:
|
||||
style = await photo_el.get_attribute("style")
|
||||
if style:
|
||||
# Extract URL from background-image: url('...')
|
||||
m = re.search(r"background-image:\s*url\(['\"]?(https://[^'\"]+)['\"]?\)", style)
|
||||
if m:
|
||||
media_list.append({
|
||||
"id": msg_id,
|
||||
"url": m.group(1),
|
||||
"type": "photo",
|
||||
"ext": ".jpg"
|
||||
})
|
||||
continue
|
||||
|
||||
min_id = min(msg_ids) if msg_ids else None
|
||||
return media_list, min_id
|
||||
|
||||
|
||||
async def scrape_channel(page, channel_name: str, limit: int):
|
||||
"""Crawl a public Telegram channel backwards in time to gather media URLs."""
|
||||
tqdm.write(f"Scraping channel '{channel_name}' ...")
|
||||
base_url = f"https://t.me/s/{channel_name}"
|
||||
all_media = []
|
||||
seen_ids = set()
|
||||
current_url = base_url
|
||||
|
||||
while len(all_media) < limit:
|
||||
tqdm.write(f" Fetching page: {current_url}")
|
||||
try:
|
||||
await page.goto(current_url, wait_until="domcontentloaded", timeout=30000)
|
||||
await page.wait_for_timeout(3000)
|
||||
except Exception as e:
|
||||
tqdm.write(f" Error loading Telegram web page: {e}")
|
||||
break
|
||||
|
||||
page_media, min_id = await extract_telegram_media(page)
|
||||
|
||||
# Filter new media
|
||||
new_items = []
|
||||
for item in page_media:
|
||||
if item["id"] not in seen_ids:
|
||||
seen_ids.add(item["id"])
|
||||
new_items.append(item)
|
||||
|
||||
if not new_items:
|
||||
tqdm.write(" No new media found on this page.")
|
||||
break
|
||||
|
||||
all_media.extend(new_items)
|
||||
tqdm.write(f" Found {len(new_items)} new media items (Total collected: {len(all_media)})")
|
||||
|
||||
if not min_id:
|
||||
break
|
||||
|
||||
# Paginate to messages before the minimum ID we've seen
|
||||
current_url = f"{base_url}?before={min_id}"
|
||||
await page.wait_for_timeout(1000)
|
||||
|
||||
return all_media[:limit]
|
||||
|
||||
|
||||
async def worker(queue, scraper, skip_existing, bar_pool, overall_bar, channel_name):
|
||||
"""Worker task that downloads media concurrently."""
|
||||
while True:
|
||||
item = await queue.get()
|
||||
if item is None:
|
||||
queue.task_done()
|
||||
break
|
||||
|
||||
msg_id, url, mtype, ext = item
|
||||
filename = f"msg_{msg_id}{ext}"
|
||||
dest_path = DOWNLOAD_DIR / channel_name / filename
|
||||
|
||||
if skip_existing and scraper_core.is_already_downloaded(dest_path):
|
||||
overall_bar.update(1)
|
||||
queue.task_done()
|
||||
continue
|
||||
|
||||
pos = bar_pool.acquire() or 1
|
||||
|
||||
success = await asyncio.to_thread(
|
||||
scraper_core.download_file,
|
||||
url, dest_path, None, pos, f"https://t.me/s/{channel_name}"
|
||||
)
|
||||
|
||||
bar_pool.release(pos)
|
||||
overall_bar.update(1)
|
||||
queue.task_done()
|
||||
|
||||
|
||||
async def process_channel(channel_name, limit, concurrency, skip_existing):
|
||||
scraper = scraper_core.PlaywrightScraper()
|
||||
await scraper.start()
|
||||
|
||||
page = await scraper.new_page()
|
||||
media_items = await scrape_channel(page, channel_name, limit)
|
||||
await page.close()
|
||||
|
||||
if not media_items:
|
||||
tqdm.write("No media files found to download.")
|
||||
await scraper.close()
|
||||
return
|
||||
|
||||
tqdm.write(f"Downloading {len(media_items)} files with concurrency {concurrency} ...")
|
||||
|
||||
queue = asyncio.Queue()
|
||||
for item in media_items:
|
||||
await queue.put((item["id"], item["url"], item["type"], item["ext"]))
|
||||
for _ in range(concurrency):
|
||||
await queue.put(None)
|
||||
|
||||
bar_pool = scraper_core.BarPositionPool(concurrency)
|
||||
overall_bar = tqdm(
|
||||
total=len(media_items),
|
||||
desc=f"Channel: {channel_name}",
|
||||
position=0,
|
||||
leave=True,
|
||||
ncols=80,
|
||||
)
|
||||
|
||||
workers = [
|
||||
asyncio.create_task(worker(queue, scraper, skip_existing, bar_pool, overall_bar, channel_name))
|
||||
for _ in range(concurrency)
|
||||
]
|
||||
|
||||
await asyncio.gather(*workers)
|
||||
overall_bar.close()
|
||||
|
||||
sys.stdout.write("\n" * (concurrency + 1))
|
||||
sys.stdout.flush()
|
||||
await scraper.close()
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Download media from public Telegram channels (headless, concurrent, multithreaded)."
|
||||
)
|
||||
parser.add_argument(
|
||||
"channel",
|
||||
help="Telegram channel username (e.g. 'durov')",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--limit",
|
||||
type=int,
|
||||
default=50,
|
||||
help="Maximum number of media items to download (default: 50).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--concurrency",
|
||||
type=int,
|
||||
default=3,
|
||||
help="Number of concurrent downloads (default: 3).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--skip-existing",
|
||||
action="store_true",
|
||||
default=True,
|
||||
help="Skip already-downloaded files (default: true).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--no-skip-existing",
|
||||
action="store_false",
|
||||
dest="skip_existing",
|
||||
help="Re-download existing files.",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
asyncio.run(process_channel(args.channel, args.limit, args.concurrency, args.skip_existing))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
Reference in new issue
Block a user