Major overhaul to the theme and now functioning.
This commit is contained in:
@@ -0,0 +1,359 @@
|
||||
import hashlib
|
||||
import os
|
||||
import requests
|
||||
import time
|
||||
from bs4 import BeautifulSoup
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from urllib.parse import urlparse, urljoin
|
||||
import uuid
|
||||
import re
|
||||
import threading
|
||||
|
||||
class BunkrDownloader:
|
||||
def __init__(self, download_folder, log_callback=None, enable_widgets_callback=None, update_progress_callback=None, update_global_progress_callback=None, headers=None, max_workers=5, translations=None):
|
||||
self.download_folder = download_folder
|
||||
self.log_callback = log_callback
|
||||
self.enable_widgets_callback = enable_widgets_callback
|
||||
self.update_progress_callback = update_progress_callback
|
||||
self.update_global_progress_callback = update_global_progress_callback
|
||||
self.session = requests.Session()
|
||||
self.headers = headers or {
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/127.0.0.0 Safari/537.36',
|
||||
'Referer': 'https://bunkr.site/',
|
||||
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8,application/signed-exchange;v=b3;q=0.9',
|
||||
'Accept-Language': 'en-US,en;q=0.9',
|
||||
}
|
||||
self.cancel_requested = False # Flag to indicate if a cancellation request has been made
|
||||
self.executor = ThreadPoolExecutor(max_workers=max_workers) # Thread pool executor for concurrent downloads
|
||||
self.total_files = 0
|
||||
self.completed_files = 0
|
||||
self.max_downloads = 5 # Valor por defecto
|
||||
self.log_messages = [] # Cola para almacenar mensajes de log
|
||||
self.notification_interval = 10 # Intervalo de notificación en segundos
|
||||
self.start_notification_thread()
|
||||
self.translations = translations or {}
|
||||
|
||||
def start_notification_thread(self):
|
||||
def notify_user():
|
||||
while not self.cancel_requested:
|
||||
if self.log_messages:
|
||||
# Enviar todos los mensajes acumulados
|
||||
if self.log_callback:
|
||||
self.log_callback("\n".join(self.log_messages))
|
||||
self.log_messages.clear()
|
||||
time.sleep(self.notification_interval)
|
||||
|
||||
# Iniciar un hilo para notificaciones periódicas
|
||||
notification_thread = threading.Thread(target=notify_user, daemon=True)
|
||||
notification_thread.start()
|
||||
|
||||
def tr(self, key):
|
||||
# Obtener la traducción para la clave dada
|
||||
return self.translations.get(key, key)
|
||||
|
||||
def log(self, message_key, url=None):
|
||||
message = self.tr(message_key)
|
||||
domain = urlparse(url).netloc if url else "General"
|
||||
full_message = f"{domain}: {message}"
|
||||
self.log_messages.append(full_message) # Agregar mensaje a la cola
|
||||
|
||||
def request_cancel(self):
|
||||
self.cancel_requested = True
|
||||
self.log("Download has been cancelled.")
|
||||
self.shutdown_executor()
|
||||
|
||||
def shutdown_executor(self):
|
||||
self.executor.shutdown(wait=False)
|
||||
self.log("Executor shut down.")
|
||||
|
||||
def clean_filename(self, filename):
|
||||
return re.sub(r'[<>:"/\\|?*\u200b]', '_', filename)
|
||||
|
||||
def get_consistent_folder_name(self, url, default_name):
|
||||
# Genera un hash de la URL para crear un nombre único y consistente
|
||||
url_hash = hashlib.md5(url.encode()).hexdigest()[:8]
|
||||
folder_name = f"{default_name}_{url_hash}"
|
||||
return self.clean_filename(folder_name)
|
||||
|
||||
def download_file(self, url_media, ruta_carpeta, file_id):
|
||||
if self.cancel_requested:
|
||||
self.log("Descarga cancelada", url=url_media)
|
||||
return
|
||||
|
||||
file_name = os.path.basename(urlparse(url_media).path)
|
||||
file_path = os.path.join(ruta_carpeta, file_name)
|
||||
|
||||
if os.path.exists(file_path):
|
||||
self.log(f"El archivo ya existe, omitiendo: {file_path}")
|
||||
self.completed_files += 1
|
||||
if self.update_global_progress_callback:
|
||||
self.update_global_progress_callback(self.completed_files, self.total_files)
|
||||
return
|
||||
|
||||
max_attempts = 3
|
||||
delay = 1
|
||||
for attempt in range(max_attempts):
|
||||
try:
|
||||
self.log(f"Intentando descargar {url_media} (Intento {attempt + 1}/{max_attempts})")
|
||||
response = self.session.get(url_media, headers=self.headers, stream=True)
|
||||
response.raise_for_status()
|
||||
|
||||
total_size = int(response.headers.get('content-length', 0))
|
||||
downloaded_size = 0
|
||||
|
||||
# Descargar el archivo en fragmentos
|
||||
with open(file_path, 'wb') as file:
|
||||
for chunk in response.iter_content(chunk_size=65536): # Fragmentos de 64KB
|
||||
if self.cancel_requested:
|
||||
self.log("Descarga cancelada durante la descarga del archivo.", url=url_media)
|
||||
file.close()
|
||||
os.remove(file_path)
|
||||
return
|
||||
file.write(chunk)
|
||||
downloaded_size += len(chunk)
|
||||
if self.update_progress_callback:
|
||||
self.update_progress_callback(downloaded_size, total_size, file_id=file_id, file_path=file_path)
|
||||
|
||||
self.log("Archivo descargado", url=url_media)
|
||||
# Notificar al usuario al completar la descarga
|
||||
if self.log_callback:
|
||||
self.log_callback(f"Descarga completada: {file_name}")
|
||||
self.completed_files += 1
|
||||
if self.update_global_progress_callback:
|
||||
self.update_global_progress_callback(self.completed_files, self.total_files)
|
||||
break
|
||||
except requests.RequestException as e:
|
||||
if response.status_code == 429:
|
||||
self.log(f"Límite de tasa excedido. Reintentando después de {delay} segundos.")
|
||||
time.sleep(delay)
|
||||
delay *= 2 # Retroceso exponencial para limitación de tasa
|
||||
else:
|
||||
self.log(f"Error al descargar de {url_media}: {e}. Intento {attempt + 1} de {max_attempts}", url=url_media)
|
||||
if attempt < max_attempts - 1:
|
||||
time.sleep(3)
|
||||
|
||||
def descargar_post_bunkr(self, url_post):
|
||||
try:
|
||||
self.log(f"Iniciando descarga para el post: {url_post}")
|
||||
|
||||
# Si se trata de una URL tipo '/f/', seguimos el flujo en dos pasos:
|
||||
if '/f/' in url_post:
|
||||
self.log("Detectado URL tipo '/f/'. Procediendo a extraer el enlace intermedio.")
|
||||
# Paso 1: Accedemos a la URL original para obtener el primer enlace (intermedio)
|
||||
response = self.session.get(url_post, headers=self.headers)
|
||||
if response.status_code != 200:
|
||||
self.log(f"Error al acceder al post {url_post}: Estado {response.status_code}")
|
||||
return
|
||||
|
||||
soup = BeautifulSoup(response.text, 'html.parser')
|
||||
first_anchor = soup.find('a', {
|
||||
'class': 'btn btn-main btn-lg rounded-full px-6 font-semibold flex-1 ic-download-01 ic-before before:text-lg'
|
||||
})
|
||||
|
||||
if not first_anchor or 'href' not in first_anchor.attrs:
|
||||
self.log("No se encontró el primer enlace de descarga en la página original.")
|
||||
return
|
||||
|
||||
intermediate_url = first_anchor['href']
|
||||
self.log(f"Enlace intermedio encontrado: {intermediate_url}")
|
||||
|
||||
# Paso 2: Accedemos a la URL intermedia para extraer el enlace final de descarga
|
||||
intermediate_response = self.session.get(intermediate_url, headers=self.headers)
|
||||
if intermediate_response.status_code != 200:
|
||||
self.log(f"Error al acceder a la URL intermedia: {intermediate_url} (Estado {intermediate_response.status_code})")
|
||||
return
|
||||
|
||||
soup2 = BeautifulSoup(intermediate_response.text, 'html.parser')
|
||||
# Buscamos la etiqueta <p> con la clase esperada y, dentro, el <a> con el enlace final
|
||||
p_tag = soup2.find('p', class_="mt-3 text-center")
|
||||
if not p_tag:
|
||||
self.log("No se encontró la etiqueta <p> con clase 'mt-3 text-center' en la página intermedia.")
|
||||
return
|
||||
|
||||
download_anchor = p_tag.find('a', {
|
||||
'class': 'btn btn-main btn-lg rounded-full px-6 font-semibold ic-download-01 ic-before before:text-lg'
|
||||
})
|
||||
if not download_anchor or 'href' not in download_anchor.attrs:
|
||||
self.log("No se encontró el enlace de descarga final en la página intermedia.")
|
||||
return
|
||||
|
||||
final_download_url = download_anchor['href']
|
||||
self.log(f"Enlace de descarga final encontrado: {final_download_url}")
|
||||
|
||||
# Creamos la carpeta de destino para este post
|
||||
file_name = "bunkr_post"
|
||||
folder_name = self.get_consistent_folder_name(url_post, file_name)
|
||||
ruta_carpeta = os.path.join(self.download_folder, folder_name)
|
||||
os.makedirs(ruta_carpeta, exist_ok=True)
|
||||
|
||||
# Preparamos la lista de medios con el enlace final
|
||||
media_urls = [(final_download_url, ruta_carpeta)]
|
||||
|
||||
else:
|
||||
# Lógica original para posts que contienen imágenes y videos
|
||||
|
||||
response = self.session.get(url_post, headers=self.headers)
|
||||
if response.status_code != 200:
|
||||
self.log(f"Error al acceder al post {url_post}: Estado {response.status_code}")
|
||||
return
|
||||
|
||||
soup = BeautifulSoup(response.text, 'html.parser')
|
||||
|
||||
# Extraer y sanitizar el nombre de la carpeta para el post
|
||||
file_name_tag = soup.find('h1', {'class': 'truncate'})
|
||||
if file_name_tag:
|
||||
file_name = file_name_tag.text.strip()
|
||||
file_name = self.clean_filename(file_name)[:50] # Limitar a 50 caracteres
|
||||
else:
|
||||
file_name = "bunkr_post"
|
||||
|
||||
folder_name = self.get_consistent_folder_name(url_post, file_name)
|
||||
ruta_carpeta = os.path.join(self.download_folder, folder_name)
|
||||
os.makedirs(ruta_carpeta, exist_ok=True)
|
||||
|
||||
media_urls = []
|
||||
|
||||
# Buscar imágenes en etiquetas <figure>
|
||||
media_divs = soup.find_all('figure', {'class': 'relative rounded-lg overflow-hidden flex justify-center items-center aspect-video bg-soft'})
|
||||
for div in media_divs:
|
||||
img_tags = div.find_all('img')
|
||||
for img_tag in img_tags:
|
||||
if 'src' in img_tag.attrs:
|
||||
img_url = img_tag['src']
|
||||
self.log(f"URL de imagen encontrada: {img_url}")
|
||||
media_urls.append((img_url, ruta_carpeta))
|
||||
|
||||
# Buscar videos: se recorre cada div que pueda contener el enlace intermedio de descarga
|
||||
video_divs = soup.find_all('div', {'class': 'flex w-full md:w-auto gap-4'})
|
||||
self.log(f"Se encontraron {len(video_divs)} divs de video.")
|
||||
for video_div in video_divs:
|
||||
self.log("Buscando enlace de página de descarga en el div de video.")
|
||||
download_page_link = video_div.find('a', {
|
||||
'class': 'btn btn-main btn-lg rounded-full px-6 font-semibold flex-1 ic-download-01 ic-before before:text-lg'
|
||||
})
|
||||
if download_page_link and 'href' in download_page_link.attrs:
|
||||
video_page_url = download_page_link['href']
|
||||
self.log(f"URL de la página de descarga encontrada: {video_page_url}. Accediendo ahora.")
|
||||
video_page_response = self.session.get(video_page_url, headers=self.headers)
|
||||
self.log(f"Estado de la respuesta de la página de video: {video_page_response.status_code} para {video_page_url}")
|
||||
|
||||
if video_page_response.status_code == 200:
|
||||
video_page_soup = BeautifulSoup(video_page_response.text, 'html.parser')
|
||||
self.log("Buscando enlace de descarga real en la página de video.")
|
||||
download_link = video_page_soup.find('a', {
|
||||
'class': 'btn btn-main btn-lg rounded-full px-6 font-semibold ic-download-01 ic-before before:text-lg'
|
||||
})
|
||||
if download_link and 'href' in download_link.attrs:
|
||||
video_url = download_link['href']
|
||||
self.log(f"URL de descarga de video encontrada: {video_url}")
|
||||
media_urls.append((video_url, ruta_carpeta))
|
||||
else:
|
||||
self.log(f"No se encontró enlace de descarga en la página de video: {video_page_url}")
|
||||
else:
|
||||
self.log(f"Error al acceder a la página de video: {video_page_url} con estado {video_page_response.status_code}")
|
||||
|
||||
# Proceder a la descarga de todos los medios encontrados
|
||||
self.total_files = len(media_urls)
|
||||
if media_urls: # Solo proceder si hay URLs para descargar
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
with ThreadPoolExecutor(max_workers=self.max_downloads) as executor:
|
||||
futures = [executor.submit(self.download_file, url, folder, str(uuid.uuid4())) for url, folder in media_urls]
|
||||
for future in as_completed(futures):
|
||||
if self.cancel_requested:
|
||||
self.log("Cancelando descargas restantes.")
|
||||
break
|
||||
future.result()
|
||||
|
||||
self.log("Descarga iniciada para todos los medios.")
|
||||
if self.enable_widgets_callback:
|
||||
self.enable_widgets_callback()
|
||||
|
||||
except Exception as e:
|
||||
self.log(f"Error al procesar el post {url_post}: {e}")
|
||||
if self.enable_widgets_callback:
|
||||
self.enable_widgets_callback()
|
||||
|
||||
|
||||
def descargar_perfil_bunkr(self, url_perfil):
|
||||
try:
|
||||
self.log(f"Iniciando descarga para el perfil: {url_perfil}")
|
||||
response = self.session.get(url_perfil, headers=self.headers)
|
||||
self.log(f"Código de estado de la respuesta: {response.status_code} para {url_perfil}")
|
||||
if response.status_code == 200:
|
||||
soup = BeautifulSoup(response.text, 'html.parser')
|
||||
|
||||
# Extraer y sanitizar el nombre de la carpeta para el perfil
|
||||
file_name_tag = soup.find('h1', {'class': 'truncate'})
|
||||
if file_name_tag:
|
||||
folder_name = file_name_tag.text.strip()
|
||||
else:
|
||||
folder_name = "bunkr_profile"
|
||||
|
||||
# Usar el nuevo método para obtener un nombre de carpeta consistente
|
||||
folder_name = self.get_consistent_folder_name(url_perfil, folder_name)
|
||||
ruta_carpeta = os.path.join(self.download_folder, folder_name)
|
||||
os.makedirs(ruta_carpeta, exist_ok=True)
|
||||
|
||||
# Find media URLs in the profile
|
||||
media_urls = []
|
||||
grid_div = soup.find('div', {'class': 'grid gap-4 grid-cols-repeat [--size:11rem] lg:[--size:14rem] grid-images'})
|
||||
if grid_div:
|
||||
links = grid_div.find_all('a', {'class': 'after:absolute after:z-10 after:inset-0'})
|
||||
total_links = len(links)
|
||||
for idx, link in enumerate(links):
|
||||
if self.cancel_requested:
|
||||
self.log("Cancelling remaining downloads.")
|
||||
break
|
||||
|
||||
# Resolve relative URLs to full URLs
|
||||
image_page_url = urljoin(url_perfil, link['href'])
|
||||
self.log(f"Processing media page URL: {image_page_url}")
|
||||
|
||||
# Visit the page to get the media URL
|
||||
image_response = self.session.get(image_page_url, headers=self.headers)
|
||||
if image_response.status_code == 200:
|
||||
image_soup = BeautifulSoup(image_response.text, 'html.parser')
|
||||
|
||||
# Search for image URL
|
||||
media_tag = image_soup.select_one("figure.relative img[class='w-full h-full absolute opacity-20 object-cover blur-sm z-10']")
|
||||
if media_tag and 'src' in media_tag.attrs:
|
||||
media_url = urljoin(image_page_url, media_tag['src']) # Resolve media URL
|
||||
self.log(f"Found image URL: {media_url}")
|
||||
media_urls.append((media_url, ruta_carpeta))
|
||||
|
||||
# Search for video URL
|
||||
video_tag = image_soup.select_one("video#player")
|
||||
if video_tag and 'src' in video_tag.attrs:
|
||||
video_url = urljoin(image_page_url, video_tag['src']) # Resolve video URL
|
||||
self.log(f"Found video URL: {video_url}")
|
||||
media_urls.append((video_url, ruta_carpeta))
|
||||
else:
|
||||
source_tag = video_tag.find('source') if video_tag else None
|
||||
if source_tag and 'src' in source_tag.attrs:
|
||||
video_url = urljoin(image_page_url, source_tag['src']) # Resolve video URL from source
|
||||
self.log(f"Found video URL from source: {video_url}")
|
||||
media_urls.append((video_url, ruta_carpeta))
|
||||
|
||||
self.total_files = len(media_urls)
|
||||
futures = [self.executor.submit(self.download_file, url, folder, str(uuid.uuid4())) for url, folder in media_urls]
|
||||
|
||||
# Only after all futures are done, enable widgets again
|
||||
for future in as_completed(futures):
|
||||
if self.cancel_requested:
|
||||
self.log("Cancelling remaining downloads.")
|
||||
break
|
||||
future.result()
|
||||
|
||||
self.log("Download completed for all media.")
|
||||
if self.enable_widgets_callback:
|
||||
self.enable_widgets_callback() # Only enable after all downloads are done
|
||||
else:
|
||||
self.log(f"Failed to access the profile {url_perfil}: Status {response.status_code}")
|
||||
except Exception as e:
|
||||
self.log(f"Failed to access the profile {url_perfil}: {e}")
|
||||
if self.enable_widgets_callback:
|
||||
self.enable_widgets_callback()
|
||||
|
||||
def set_max_downloads(self, max_downloads):
|
||||
self.max_downloads = max_downloads
|
||||
|
||||
@@ -0,0 +1,724 @@
|
||||
from collections import defaultdict
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from threading import Semaphore
|
||||
from urllib.parse import quote_plus, urlencode, urljoin, urlparse
|
||||
import os
|
||||
import re
|
||||
import requests
|
||||
import threading
|
||||
import time
|
||||
import sqlite3
|
||||
|
||||
class Downloader:
|
||||
def __init__(self, download_folder, max_workers=5, log_callback=None,
|
||||
enable_widgets_callback=None, update_progress_callback=None,
|
||||
update_global_progress_callback=None, headers=None,
|
||||
max_retries=999999, retry_interval=1.0, stream_read_timeout=10,
|
||||
download_images=True, download_videos=True, download_compressed=True,
|
||||
tr=None, folder_structure='default', rate_limit_interval=1.0):
|
||||
|
||||
self.download_folder = download_folder
|
||||
self.log_callback = log_callback
|
||||
self.enable_widgets_callback = enable_widgets_callback
|
||||
self.update_progress_callback = update_progress_callback
|
||||
self.update_global_progress_callback = update_global_progress_callback
|
||||
self.cancel_requested = threading.Event()
|
||||
self.headers = headers or {
|
||||
'User-Agent': 'Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)',
|
||||
'Referer': 'https://coomer.st/',
|
||||
"Accept": "text/css"
|
||||
}
|
||||
self.media_counter = 0
|
||||
self.session = requests.Session()
|
||||
self.max_workers = max_workers
|
||||
self.executor = ThreadPoolExecutor(max_workers=self.max_workers)
|
||||
self.rate_limit = Semaphore(self.max_workers)
|
||||
self.domain_locks = defaultdict(lambda: Semaphore(self.max_workers))
|
||||
self.domain_last_request = defaultdict(float)
|
||||
self.rate_limit_interval = rate_limit_interval
|
||||
self.download_mode = "multi"
|
||||
self.video_extensions = ('.mp4', '.mkv', '.webm', '.mov', '.avi', '.flv', '.wmv', '.m4v')
|
||||
self.image_extensions = ('.jpg', '.jpeg', '.png', '.gif', '.bmp', '.tiff')
|
||||
self.document_extensions = ('.pdf', '.doc', '.docx', '.xls', '.xlsx', '.ppt', '.pptx')
|
||||
self.compressed_extensions = ('.zip', '.rar', '.7z', '.tar', '.gz')
|
||||
self.download_images = download_images
|
||||
self.download_videos = download_videos
|
||||
self.download_compressed = download_compressed
|
||||
self.futures = []
|
||||
self.total_files = 0
|
||||
self.completed_files = 0
|
||||
self.skipped_files = []
|
||||
self.failed_files = []
|
||||
self.start_time = None
|
||||
self.tr = tr
|
||||
self.shutdown_called = False
|
||||
self.folder_structure = folder_structure
|
||||
self.failed_retry_count = {}
|
||||
self.max_retries = max_retries
|
||||
self.retry_interval = retry_interval
|
||||
self.file_lock = threading.Lock()
|
||||
self.post_attachment_counter = defaultdict(int)
|
||||
self.subdomain_cache = {}
|
||||
self.subdomain_locks = defaultdict(threading.Lock)
|
||||
self.stream_read_timeout = stream_read_timeout
|
||||
|
||||
|
||||
db_folder = os.path.join("resources", "config")
|
||||
os.makedirs(db_folder, exist_ok=True)
|
||||
self.db_path = os.path.join(db_folder, "downloads.db")
|
||||
self.db_lock = threading.Lock()
|
||||
self.init_db()
|
||||
self.load_download_cache()
|
||||
|
||||
|
||||
def init_db(self):
|
||||
self.db_connection = sqlite3.connect(self.db_path, check_same_thread=False)
|
||||
self.db_cursor = self.db_connection.cursor()
|
||||
self.db_cursor.execute("""
|
||||
CREATE TABLE IF NOT EXISTS downloads (
|
||||
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
media_url TEXT UNIQUE,
|
||||
file_path TEXT,
|
||||
file_size INTEGER,
|
||||
user_id TEXT,
|
||||
post_id TEXT,
|
||||
downloaded_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
|
||||
)
|
||||
""")
|
||||
self.db_connection.commit()
|
||||
|
||||
def load_download_cache(self):
|
||||
with self.db_lock:
|
||||
self.db_cursor.execute("SELECT media_url, file_path, file_size FROM downloads")
|
||||
rows = self.db_cursor.fetchall()
|
||||
self.download_cache = {row[0]: (row[1], row[2]) for row in rows}
|
||||
|
||||
|
||||
def log(self, message):
|
||||
if self.log_callback:
|
||||
self.log_callback(self.tr(message) if self.tr else message)
|
||||
|
||||
def set_download_mode(self, mode, max_workers):
|
||||
|
||||
if mode == 'queue':
|
||||
max_workers = 1
|
||||
|
||||
self.download_mode = mode
|
||||
self.max_workers = max_workers
|
||||
|
||||
|
||||
if self.executor:
|
||||
self.executor.shutdown(wait=True)
|
||||
|
||||
|
||||
self.executor = ThreadPoolExecutor(max_workers=max_workers)
|
||||
self.rate_limit = Semaphore(max_workers)
|
||||
|
||||
self.log(f"Updated download mode to {mode} with max_workers = {max_workers}")
|
||||
|
||||
def set_retry_settings(self, max_retries, retry_interval):
|
||||
self.max_retries = max_retries
|
||||
self.rate_limit_interval = retry_interval
|
||||
|
||||
def request_cancel(self):
|
||||
self.cancel_requested.set()
|
||||
self.log(self.tr("Download cancellation requested."))
|
||||
for future in self.futures:
|
||||
future.cancel()
|
||||
|
||||
def shutdown_executor(self):
|
||||
if not self.shutdown_called:
|
||||
self.shutdown_called = True
|
||||
if self.executor:
|
||||
self.executor.shutdown(wait=True)
|
||||
if self.enable_widgets_callback:
|
||||
self.enable_widgets_callback()
|
||||
self.log(self.tr("All downloads completed or cancelled."))
|
||||
|
||||
def safe_request(self, url, max_retries=None, headers=None):
|
||||
if max_retries is None:
|
||||
max_retries = self.max_retries
|
||||
if headers is None:
|
||||
headers = self.headers
|
||||
|
||||
parsed = urlparse(url)
|
||||
domain = parsed.netloc
|
||||
path = parsed.path
|
||||
|
||||
for attempt in range(max_retries + 1):
|
||||
if self.cancel_requested.is_set():
|
||||
return None
|
||||
|
||||
with self.domain_locks[domain]:
|
||||
elapsed_time = time.time() - self.domain_last_request[domain]
|
||||
if elapsed_time < self.rate_limit_interval:
|
||||
time.sleep(self.rate_limit_interval - elapsed_time)
|
||||
|
||||
try:
|
||||
self.domain_last_request[domain] = time.time()
|
||||
|
||||
response = self.session.get(url, stream=True, headers=headers, timeout=self.stream_read_timeout)
|
||||
sc = response.status_code
|
||||
|
||||
if sc in (403, 404) and ("coomer" in domain or "kemono" in domain):
|
||||
if self.update_progress_callback:
|
||||
self.update_progress_callback(0, 0, status=f"{sc} - probing subdomains")
|
||||
|
||||
with self.subdomain_locks[path]:
|
||||
if path in self.subdomain_cache:
|
||||
alt_url = self.subdomain_cache[path]
|
||||
else:
|
||||
alt_url = self._find_valid_subdomain(url)
|
||||
self.subdomain_cache[path] = alt_url
|
||||
|
||||
if alt_url != url:
|
||||
found = urlparse(alt_url).netloc
|
||||
if self.update_progress_callback:
|
||||
self.update_progress_callback(0, 0, status=f"Subdomain found: {found}")
|
||||
|
||||
|
||||
response = self.session.get(alt_url, stream=True, headers=headers)
|
||||
response.raise_for_status()
|
||||
return response
|
||||
else:
|
||||
if self.update_progress_callback:
|
||||
self.update_progress_callback(0, 0, status="Exhausted subdomains")
|
||||
return None
|
||||
|
||||
response.raise_for_status()
|
||||
return response
|
||||
|
||||
except requests.exceptions.RequestException as e:
|
||||
status_code = getattr(e.response, 'status_code', None)
|
||||
if status_code in (429, 500, 502, 503, 504):
|
||||
self.log(log_message)
|
||||
time.sleep(self.retry_interval)
|
||||
elif status_code not in (403, 404):
|
||||
url_display = getattr(e.request, 'url', url)
|
||||
if len(url_display) > 60:
|
||||
url_display = url_display[:60] + "..."
|
||||
self.log(self.tr("Intento {attempt}/{max_retries_val}: Error al acceder a {url} - {error}").format(
|
||||
attempt=attempt + 1, max_retries_val=max_retries + 1, url=url_display, error=e))
|
||||
if attempt < max_retries:
|
||||
time.sleep(self.retry_interval)
|
||||
else:
|
||||
if isinstance(e, requests.exceptions.ReadTimeout):
|
||||
self.log(self.tr("Intento {attempt}/{max_retries_val}: Read timeout ({stream_timeout}s) - Reintentando...").format(
|
||||
attempt=attempt + 1,
|
||||
max_retries_val=max_retries + 1,
|
||||
stream_timeout=self.stream_read_timeout
|
||||
))
|
||||
time.sleep(self.retry_interval)
|
||||
else:
|
||||
log_message = self.tr("Intento {attempt}/{max_retries_val}: Error {status_code} - Reintentando...").format(
|
||||
attempt=attempt + 1, max_retries_val=max_retries + 1, status_code=status_code)
|
||||
|
||||
if status_code in (403, 404) and ("coomer" in domain or "kemono" in domain) and attempt == max_retries:
|
||||
self.log(self.tr("Fallo final al acceder a {url} con error {status_code}").format(url=url, status_code=status_code))
|
||||
|
||||
|
||||
return None
|
||||
|
||||
def _find_valid_subdomain(self, url, max_subdomains=10):
|
||||
parsed = urlparse(url)
|
||||
original_path = parsed.path
|
||||
|
||||
path = original_path
|
||||
if not original_path.startswith("/data/"):
|
||||
path = ("/data" + original_path) if not original_path.startswith("/data") else original_path
|
||||
|
||||
host = parsed.netloc
|
||||
|
||||
if "coomer" in host:
|
||||
base_domains = ["coomer.st"]
|
||||
elif "kemono" in host:
|
||||
|
||||
base_domains = ["kemono.cr", "kemono.su"]
|
||||
else:
|
||||
base_domains = [host]
|
||||
|
||||
for base in base_domains:
|
||||
for i in range(1, max_subdomains + 1):
|
||||
domain = f"n{i}.{base}"
|
||||
test_url = parsed._replace(netloc=domain, path=path).geturl()
|
||||
|
||||
if self.update_progress_callback:
|
||||
self.update_progress_callback(0, 0, status=f"Testing subdomain: {domain}")
|
||||
|
||||
try:
|
||||
resp = self.session.get(test_url, headers=self.headers,
|
||||
timeout=self.stream_read_timeout, stream=True)
|
||||
if resp.status_code == 200:
|
||||
return test_url
|
||||
else:
|
||||
if self.update_progress_callback:
|
||||
self.update_progress_callback(0, 0, status=f"Invalid subdomain: {domain}")
|
||||
except requests.exceptions.ReadTimeout:
|
||||
if self.update_progress_callback:
|
||||
self.update_progress_callback(0, 0, status=f"Timeout in: {domain}")
|
||||
except Exception:
|
||||
if self.update_progress_callback:
|
||||
self.update_progress_callback(0, 0, status=f"Invalid subdomain: {domain}")
|
||||
|
||||
return url
|
||||
|
||||
def fetch_user_posts(self, site, user_id, service, query=None, specific_post_id=None, initial_offset=0, log_fetching=True):
|
||||
all_posts = []
|
||||
offset = initial_offset
|
||||
user_id_encoded = quote_plus(user_id)
|
||||
while True:
|
||||
if self.cancel_requested.is_set():
|
||||
return all_posts
|
||||
|
||||
api_url = f"https://{site}/api/v1/{service}/user/{user_id_encoded}/posts"
|
||||
url_query = {"o": offset}
|
||||
if query not in (None, "", 0, "0"):
|
||||
url_query["q"] = query
|
||||
api_url += "?" + urlencode(url_query)
|
||||
if log_fetching:
|
||||
self.log(self.tr("Fetching user posts from {api_url}", api_url=api_url))
|
||||
try:
|
||||
|
||||
response = self.session.get(api_url, headers=self.headers)
|
||||
|
||||
if response.status_code == 400:
|
||||
self.log(self.tr("End of posts at offset {offset}.", offset=offset))
|
||||
break
|
||||
|
||||
response.raise_for_status()
|
||||
try:
|
||||
posts_data = response.json()
|
||||
except ValueError as e:
|
||||
self.log(self.tr("Error al parsear JSON: {e}", e=e))
|
||||
break
|
||||
|
||||
if isinstance(posts_data, dict) and 'data' in posts_data:
|
||||
posts = posts_data['data']
|
||||
else:
|
||||
posts = posts_data
|
||||
if not posts:
|
||||
break
|
||||
if specific_post_id:
|
||||
post = next((p for p in posts if p['id'] == specific_post_id), None)
|
||||
if post:
|
||||
return [post]
|
||||
all_posts.extend(posts)
|
||||
offset += 50
|
||||
except Exception as e:
|
||||
self.log(self.tr("Error fetching user posts: {e}", e=e))
|
||||
break
|
||||
if specific_post_id:
|
||||
return [post for post in all_posts if post['id'] == specific_post_id]
|
||||
return all_posts
|
||||
|
||||
|
||||
def get_filename(self, media_url, post_id=None, post_name=None, attachment_index=1, post_time=None):
|
||||
base_name = os.path.basename(media_url).split('?')[0]
|
||||
name_no_ext, extension = os.path.splitext(base_name)
|
||||
if not hasattr(self, 'file_naming_mode'):
|
||||
self.file_naming_mode = 0
|
||||
mode = self.file_naming_mode
|
||||
|
||||
def sanitize(name):
|
||||
|
||||
sanitized = self.sanitize_filename(name)
|
||||
return sanitized.strip()
|
||||
|
||||
if mode == 0:
|
||||
|
||||
sanitized = sanitize(name_no_ext)
|
||||
if not sanitized:
|
||||
sanitized = "file"
|
||||
final_name = f"{sanitized}_{attachment_index}{extension}"
|
||||
elif mode == 1:
|
||||
|
||||
sanitized_post = sanitize(post_name or "")
|
||||
if not sanitized_post:
|
||||
sanitized_post = f"post_{post_id}" if post_id else "post"
|
||||
short_hash = f"{hash(media_url) & 0xFFFF:04x}"
|
||||
final_name = f"{sanitized_post}_{attachment_index}_{short_hash}{extension}"
|
||||
elif mode == 2:
|
||||
|
||||
sanitized_post = sanitize(post_name or "")
|
||||
if not sanitized_post:
|
||||
sanitized_post = f"post_{post_id}" if post_id else "post"
|
||||
if post_id:
|
||||
final_name = f"{sanitized_post} - {post_id}_{attachment_index}{extension}"
|
||||
else:
|
||||
final_name = f"{sanitized_post}_{attachment_index}{extension}"
|
||||
elif mode == 3:
|
||||
|
||||
sanitized_post = sanitize(post_name or "")
|
||||
if not sanitized_post:
|
||||
sanitized_post = f"post_{post_id}" if post_id else "post"
|
||||
sanitized_time = sanitize(post_time or "")
|
||||
short_hash = f"{hash(media_url) & 0xFFFF:04x}"
|
||||
final_name = f"{sanitized_time} - {sanitized_post}_{attachment_index}_{short_hash}{extension}"
|
||||
else:
|
||||
final_name = sanitize(name_no_ext) + extension
|
||||
|
||||
return final_name
|
||||
|
||||
|
||||
|
||||
def process_post(self, post, site):
|
||||
base = f"https://{site}/"
|
||||
|
||||
def _full(path):
|
||||
if not path:
|
||||
return None
|
||||
p = path if str(path).startswith('/') else f'/{path}'
|
||||
return urljoin(base, p)
|
||||
|
||||
media_urls = []
|
||||
|
||||
f = post.get('file') or {}
|
||||
u = _full(f.get('path') or f.get('url') or f.get('name'))
|
||||
if u:
|
||||
media_urls.append(u)
|
||||
|
||||
for att in (post.get('attachments') or []):
|
||||
u = _full(att.get('path') or att.get('url') or att.get('name'))
|
||||
if u:
|
||||
media_urls.append(u)
|
||||
|
||||
return media_urls
|
||||
|
||||
def sanitize_filename(self, filename):
|
||||
return re.sub(r'[<>:"/\\|?*]', '_', filename)
|
||||
|
||||
def get_media_folder(self, extension, user_id, post_id=None):
|
||||
if extension in self.video_extensions:
|
||||
folder_name = "videos"
|
||||
elif extension in self.image_extensions:
|
||||
folder_name = "images"
|
||||
elif extension in self.document_extensions:
|
||||
folder_name = "documents"
|
||||
elif extension in self.compressed_extensions:
|
||||
folder_name = "compressed"
|
||||
else:
|
||||
folder_name = "other"
|
||||
if self.folder_structure == 'post_number' and post_id:
|
||||
media_folder = os.path.join(self.download_folder, user_id, f'post_{post_id}', folder_name)
|
||||
else:
|
||||
media_folder = os.path.join(self.download_folder, user_id, folder_name)
|
||||
return media_folder
|
||||
|
||||
def process_media_element(self, media_url, user_id, post_id=None,
|
||||
post_name=None, post_time=None, download_id=None):
|
||||
|
||||
|
||||
if self.cancel_requested.is_set():
|
||||
return
|
||||
|
||||
extension = os.path.splitext(media_url)[1].lower()
|
||||
if (extension in self.image_extensions and not self.download_images) or \
|
||||
(extension in self.video_extensions and not self.download_videos) or \
|
||||
(extension in self.compressed_extensions and not self.download_compressed):
|
||||
self.log(f"Skipping {media_url} due to settings.")
|
||||
return
|
||||
|
||||
|
||||
if post_id:
|
||||
self.post_attachment_counter[post_id] += 1
|
||||
attachment_index = self.post_attachment_counter[post_id]
|
||||
else:
|
||||
attachment_index = 1
|
||||
|
||||
filename = self.get_filename(media_url, post_id=post_id, post_name=post_name, post_time=post_time,
|
||||
attachment_index=attachment_index)
|
||||
media_folder = self.get_media_folder(extension, user_id, post_id)
|
||||
os.makedirs(media_folder, exist_ok=True)
|
||||
|
||||
final_path = os.path.normpath(os.path.join(media_folder, filename))
|
||||
tmp_path = final_path + ".tmp"
|
||||
|
||||
|
||||
if media_url in self.download_cache:
|
||||
self.log(f"File from {media_url} is in DB, skipping.")
|
||||
with self.file_lock:
|
||||
self.skipped_files.append(final_path)
|
||||
return
|
||||
|
||||
self.log(f"Starting download from {media_url}")
|
||||
|
||||
for attempt in range(self.max_retries + 1):
|
||||
if self.cancel_requested.is_set():
|
||||
if os.path.exists(tmp_path):
|
||||
os.remove(tmp_path)
|
||||
self.log(f"Download cancelled from {media_url}")
|
||||
return
|
||||
|
||||
response = self.safe_request(media_url, max_retries=self.max_retries)
|
||||
|
||||
if response is None:
|
||||
if attempt < self.max_retries:
|
||||
self.log(f"Initial request failed for {media_url}. Resuming download in {self.retry_interval}s. (Attempt {attempt+1}/{self.max_retries + 1})")
|
||||
time.sleep(self.retry_interval)
|
||||
continue
|
||||
else:
|
||||
break
|
||||
|
||||
try:
|
||||
try:
|
||||
total_size = int(response.headers.get('content-length', 0))
|
||||
except Exception as e:
|
||||
self.log(f"Error getting total size: {e}")
|
||||
total_size = 0
|
||||
|
||||
downloaded_size = 0
|
||||
self.start_time = time.time()
|
||||
|
||||
|
||||
with open(tmp_path, 'wb') as f:
|
||||
for chunk in response.iter_content(chunk_size=1048576):
|
||||
if self.cancel_requested.is_set():
|
||||
raise Exception("Cancellation Requested")
|
||||
if chunk:
|
||||
f.write(chunk)
|
||||
downloaded_size += len(chunk)
|
||||
if self.update_progress_callback:
|
||||
elapsed_time = time.time() - self.start_time
|
||||
speed = downloaded_size / elapsed_time if elapsed_time > 0 else 0
|
||||
remaining_time = (total_size - downloaded_size) / speed if speed > 0 else 0
|
||||
self.update_progress_callback(downloaded_size, total_size,
|
||||
file_id=download_id,
|
||||
file_path=tmp_path,
|
||||
speed=speed,
|
||||
eta=remaining_time)
|
||||
|
||||
|
||||
while total_size and downloaded_size < total_size:
|
||||
|
||||
resume_headers = self.headers.copy()
|
||||
resume_headers['Range'] = f'bytes={downloaded_size}-'
|
||||
self.log(f"Resuming download at byte {downloaded_size} for {media_url}")
|
||||
|
||||
part_response = self.safe_request(media_url, max_retries=self.max_retries, headers=resume_headers)
|
||||
|
||||
if part_response is None:
|
||||
raise Exception("Resumption Failed after retries")
|
||||
|
||||
with open(tmp_path, 'ab') as f:
|
||||
for chunk in part_response.iter_content(chunk_size=1048576):
|
||||
if self.cancel_requested.is_set():
|
||||
raise Exception("Cancellation Requested")
|
||||
if chunk:
|
||||
f.write(chunk)
|
||||
downloaded_size += len(chunk)
|
||||
if self.update_progress_callback:
|
||||
elapsed_time = time.time() - self.start_time
|
||||
speed = downloaded_size / elapsed_time if elapsed_time > 0 else 0
|
||||
remaining_time = (total_size - downloaded_size) / speed if speed > 0 else 0
|
||||
self.update_progress_callback(downloaded_size, total_size,
|
||||
file_id=download_id,
|
||||
file_path=tmp_path,
|
||||
speed=speed,
|
||||
eta=remaining_time)
|
||||
|
||||
|
||||
if total_size > 0 and downloaded_size != total_size:
|
||||
raise Exception(f"Final size mismatch: expected {total_size}, got {downloaded_size}")
|
||||
|
||||
|
||||
with self.file_lock:
|
||||
if os.path.exists(final_path):
|
||||
os.remove(final_path)
|
||||
os.rename(tmp_path, final_path)
|
||||
|
||||
|
||||
with self.file_lock:
|
||||
self.completed_files += 1
|
||||
|
||||
self.log(f"Download success from {media_url}")
|
||||
if self.update_global_progress_callback:
|
||||
self.update_global_progress_callback(self.completed_files, self.total_files)
|
||||
|
||||
|
||||
with self.db_lock:
|
||||
self.db_cursor.execute(
|
||||
"""INSERT OR REPLACE INTO downloads (media_url, file_path, file_size, user_id, post_id)
|
||||
VALUES (?, ?, ?, ?, ?)""",
|
||||
(media_url, final_path, total_size, user_id, post_id)
|
||||
)
|
||||
self.db_connection.commit()
|
||||
|
||||
self.download_cache[media_url] = (final_path, total_size)
|
||||
return
|
||||
except Exception as e:
|
||||
|
||||
if str(e) == "Cancellation Requested":
|
||||
if os.path.exists(tmp_path):
|
||||
os.remove(tmp_path)
|
||||
self.log(f"Download cancelled from {media_url}")
|
||||
return
|
||||
|
||||
if attempt < self.max_retries:
|
||||
time.sleep(self.retry_interval)
|
||||
continue
|
||||
|
||||
self.log(f"Failed to download {media_url} after {self.max_retries + 1} total download attempts.")
|
||||
with self.file_lock:
|
||||
self.failed_files.append(media_url)
|
||||
|
||||
|
||||
def get_remote_file_size(self, media_url, filename):
|
||||
try:
|
||||
response = requests.head(media_url, allow_redirects=True)
|
||||
if response.status_code == 200:
|
||||
size = int(response.headers.get('Content-Length', 0))
|
||||
return media_url, filename, size
|
||||
else:
|
||||
self.log(self.tr(f"Failed to get size for {filename}: HTTP {response.status_code}"))
|
||||
return media_url, filename, None
|
||||
except Exception as e:
|
||||
self.log(self.tr(f"Error getting size for {filename}: {e}"))
|
||||
return media_url, filename, None
|
||||
|
||||
def download_media(self, site, user_id, service, query=None, download_all=False, initial_offset=0):
|
||||
try:
|
||||
self.log(self.tr("Starting download process..."))
|
||||
|
||||
posts = self.fetch_user_posts(
|
||||
site, user_id, service,
|
||||
query=query,
|
||||
initial_offset=initial_offset,
|
||||
log_fetching=download_all
|
||||
)
|
||||
if not posts:
|
||||
self.log(self.tr("No posts found for this user."))
|
||||
return
|
||||
|
||||
if not download_all:
|
||||
|
||||
posts = posts[:50]
|
||||
|
||||
self.total_files = 0
|
||||
for post in posts:
|
||||
|
||||
current_post_id = post.get('id') or "unknown_id"
|
||||
|
||||
title = post.get('title') or ""
|
||||
|
||||
|
||||
media_urls = self.process_post(post, site)
|
||||
|
||||
|
||||
for media_url in media_urls:
|
||||
ext = os.path.splitext(media_url)[1].lower()
|
||||
if (ext in self.image_extensions and not self.download_images) or \
|
||||
(ext in self.video_extensions and not self.download_videos) or \
|
||||
(ext in self.compressed_extensions and not self.download_compressed):
|
||||
continue
|
||||
|
||||
|
||||
self.total_files += 1
|
||||
|
||||
|
||||
futures = []
|
||||
for post in posts:
|
||||
current_post_id = post.get('id') or "unknown_id"
|
||||
title = post.get('title') or ""
|
||||
time = post.get('published') or ""
|
||||
|
||||
media_urls = self.process_post(post, site)
|
||||
for media_url in media_urls:
|
||||
ext = os.path.splitext(media_url)[1].lower()
|
||||
if (ext in self.image_extensions and not self.download_images) or \
|
||||
(ext in self.video_extensions and not self.download_videos) or \
|
||||
(ext in self.compressed_extensions and not self.download_compressed):
|
||||
continue
|
||||
|
||||
|
||||
if self.download_mode == 'queue':
|
||||
|
||||
self.process_media_element(
|
||||
media_url,
|
||||
user_id,
|
||||
post_id=current_post_id,
|
||||
post_name=title,
|
||||
post_time=time
|
||||
)
|
||||
else:
|
||||
|
||||
future = self.executor.submit(
|
||||
self.process_media_element,
|
||||
media_url,
|
||||
user_id,
|
||||
current_post_id,
|
||||
title,
|
||||
time,
|
||||
media_url
|
||||
)
|
||||
futures.append(future)
|
||||
|
||||
|
||||
if self.download_mode == 'multi':
|
||||
for future in as_completed(futures):
|
||||
if self.cancel_requested.is_set():
|
||||
break
|
||||
|
||||
except Exception as e:
|
||||
self.log(self.tr(f"Error during download: {e}"))
|
||||
finally:
|
||||
self.shutdown_executor()
|
||||
|
||||
def download_single_post(self, site, post_id, service, user_id):
|
||||
try:
|
||||
post = self.fetch_user_posts(site, user_id, service, specific_post_id=post_id)
|
||||
if not post:
|
||||
self.log(self.tr("No post found for this ID."))
|
||||
return
|
||||
media_urls = self.process_post(post[0], site)
|
||||
futures = []
|
||||
grouped_media_urls = defaultdict(list)
|
||||
for media_url in media_urls:
|
||||
grouped_media_urls[post[0]['id']].append(media_url)
|
||||
self.total_files = len(media_urls)
|
||||
self.completed_files = 0
|
||||
for post_id, media_urls in grouped_media_urls.items():
|
||||
for media_url in media_urls:
|
||||
if self.download_mode == 'queue':
|
||||
self.process_media_element(media_url, user_id, post_id)
|
||||
else:
|
||||
future = self.executor.submit(self.process_media_element, media_url, user_id, post_id)
|
||||
futures.append(future)
|
||||
if self.download_mode == 'multi':
|
||||
for future in as_completed(futures):
|
||||
if self.cancel_requested.is_set():
|
||||
break
|
||||
except Exception as e:
|
||||
self.log(self.tr(f"Error during download: {e}"))
|
||||
finally:
|
||||
self.shutdown_executor()
|
||||
|
||||
def fetch_single_post(self, site, post_id, service):
|
||||
api_url = f"https://{site}/api/v1/{service}/post/{post_id}"
|
||||
self.log(self.tr(f"Fetching post from {api_url}"))
|
||||
try:
|
||||
with self.rate_limit:
|
||||
response = self.session.get(api_url, headers=self.headers)
|
||||
response.raise_for_status()
|
||||
return response.json()
|
||||
except Exception as e:
|
||||
self.log(self.tr(f"Error fetching post: {e}"))
|
||||
return None
|
||||
|
||||
def clear_database(self):
|
||||
|
||||
with self.db_lock:
|
||||
self.db_cursor.execute("DELETE FROM downloads")
|
||||
self.db_connection.commit()
|
||||
self.log(self.tr("Database cleared."))
|
||||
|
||||
def update_max_downloads(self, new_max):
|
||||
|
||||
|
||||
if self.executor:
|
||||
self.executor.shutdown(wait=True)
|
||||
|
||||
self.max_workers = new_max
|
||||
self.executor = ThreadPoolExecutor(max_workers=new_max)
|
||||
self.rate_limit = Semaphore(new_max)
|
||||
|
||||
self.log(f"Updated max_workers to {new_max}")
|
||||
@@ -0,0 +1,287 @@
|
||||
import re
|
||||
import uuid
|
||||
import requests
|
||||
import os
|
||||
import time
|
||||
import datetime
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
from urllib.parse import urljoin, quote
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from pathlib import Path
|
||||
from requests.exceptions import ChunkedEncodingError
|
||||
from tkinter import messagebox, simpledialog
|
||||
|
||||
class EromeDownloader:
|
||||
def __init__(self, root, log_callback=None, enable_widgets_callback=None, update_progress_callback=None, update_global_progress_callback=None, download_images=True, download_videos=True, headers=None, language="en", is_profile_download=False, direct_download=False, tr=None, max_workers=5):
|
||||
self.root = root
|
||||
self.session = requests.Session()
|
||||
self.headers = {k: str(v).encode('ascii', 'ignore').decode('ascii') for k, v in (headers or {
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/127.0.0.0 Safari/537.36',
|
||||
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8'
|
||||
}).items()}
|
||||
self.log_messages = [] # Store log messages
|
||||
self.log_callback = log_callback
|
||||
self.enable_widgets_callback = enable_widgets_callback
|
||||
self.update_progress_callback = update_progress_callback
|
||||
self.update_global_progress_callback = update_global_progress_callback
|
||||
self.download_images = download_images
|
||||
self.download_videos = download_videos
|
||||
self.cancel_requested = False
|
||||
self.language = language
|
||||
self.executor = ThreadPoolExecutor(max_workers=max_workers) # Thread pool for concurrent downloads
|
||||
self.total_files = 0
|
||||
self.completed_files = 0
|
||||
self.is_profile_download = is_profile_download
|
||||
self.direct_download = direct_download # Option for direct downloads without folder creation
|
||||
self.tr = tr if tr else lambda x, **kwargs: x.format(**kwargs) # Translation function
|
||||
|
||||
def request_cancel(self):
|
||||
self.cancel_requested = True
|
||||
self.log(self.tr("Download cancelled"))
|
||||
if self.is_profile_download:
|
||||
self.enable_widgets_callback()
|
||||
|
||||
def log(self, message):
|
||||
if self.log_callback is not None:
|
||||
self.log_callback(message)
|
||||
self.log_messages.append(message)
|
||||
|
||||
def shutdown_executor(self):
|
||||
self.executor.shutdown(wait=False)
|
||||
self.log(self.tr("Executor shut down."))
|
||||
if self.is_profile_download:
|
||||
self.enable_widgets_callback()
|
||||
|
||||
@staticmethod
|
||||
def clean_filename(filename):
|
||||
return re.sub(r'[<>:"/\\|?*]', '_', filename.split('?')[0])
|
||||
|
||||
def create_folder(self, folder_name):
|
||||
try:
|
||||
os.makedirs(folder_name, exist_ok=True)
|
||||
except OSError as e:
|
||||
self.log(self.tr("Error creating folder: {error}", error=e))
|
||||
response = messagebox.askyesno(self.tr("Error"), self.tr("Couldn't create folder: {folder_name}\nWould you like to choose a new name?", folder_name=folder_name), parent=self.root)
|
||||
if response:
|
||||
new_folder_name = simpledialog.askstring(self.tr("New folder name"), self.tr("Enter new folder name:"), parent=self.root)
|
||||
if new_folder_name:
|
||||
folder_name = os.path.join(os.path.dirname(folder_name), self.clean_filename(new_folder_name))
|
||||
try:
|
||||
os.makedirs(folder_name, exist_ok=True)
|
||||
except OSError as e:
|
||||
messagebox.showerror(self.tr("Error"), self.tr("Could not create folder: {folder_name}\nError: {error}", folder_name=folder_name, error=e), parent=self.root)
|
||||
return folder_name
|
||||
|
||||
def download_file(self, url, file_path, resource_type, file_id=None, max_retries=999999):
|
||||
if self.cancel_requested:
|
||||
return
|
||||
|
||||
# Evita sobreescrituras y descargas duplicadas
|
||||
if os.path.exists(file_path):
|
||||
self.log(self.tr("File already exists, skipping: {file_path}",
|
||||
file_path=file_path))
|
||||
return
|
||||
|
||||
os.makedirs(os.path.dirname(file_path), exist_ok=True)
|
||||
self.log(self.tr("Start downloading {resource_type}: {file_path}",
|
||||
resource_type=resource_type, file_path=file_path))
|
||||
|
||||
retries = 0
|
||||
while retries <= max_retries:
|
||||
try:
|
||||
with requests.get(url, headers=self.headers,
|
||||
stream=True, timeout=15) as response:
|
||||
if response.status_code != 200:
|
||||
self.log(self.tr("Error downloading {resource_type}, "
|
||||
"status code: {status_code}",
|
||||
resource_type=resource_type,
|
||||
status_code=response.status_code))
|
||||
break
|
||||
|
||||
total_size = int(response.headers.get("content-length", 0))
|
||||
downloaded_size = 0
|
||||
start_time = last_update = time.time()
|
||||
|
||||
with open(file_path, "wb") as f:
|
||||
for chunk in response.iter_content(chunk_size=65536):
|
||||
if self.cancel_requested:
|
||||
return
|
||||
f.write(chunk)
|
||||
downloaded_size += len(chunk)
|
||||
|
||||
# Envía actualización cada 0.5 s máx.
|
||||
now = time.time()
|
||||
if now - last_update >= 0.5:
|
||||
elapsed = now - start_time
|
||||
speed = downloaded_size / elapsed if elapsed else 0
|
||||
eta = ((total_size - downloaded_size) / speed
|
||||
if speed else None)
|
||||
self.update_progress_callback(
|
||||
downloaded_size, total_size,
|
||||
file_id=file_id,
|
||||
file_path=file_path,
|
||||
speed=speed,
|
||||
eta=eta
|
||||
)
|
||||
last_update = now
|
||||
|
||||
# ─────── Fin de la descarga ───────
|
||||
elapsed = time.time() - start_time
|
||||
final_speed = total_size / elapsed if elapsed else 0
|
||||
|
||||
# 1· Fuerza la barra al 100 % (sin status)
|
||||
self.update_progress_callback(
|
||||
total_size, total_size,
|
||||
file_id=file_id,
|
||||
file_path=file_path,
|
||||
speed=final_speed,
|
||||
eta=0
|
||||
)
|
||||
|
||||
# 2· Notifica “Completed” (no altera la barra)
|
||||
self.update_progress_callback(
|
||||
total_size, total_size,
|
||||
file_id=file_id,
|
||||
file_path=file_path,
|
||||
status="Completed"
|
||||
)
|
||||
|
||||
# Contabiliza y avanza la barra global
|
||||
self.completed_files += 1
|
||||
if self.update_global_progress_callback:
|
||||
self.update_global_progress_callback(
|
||||
self.completed_files, self.total_files
|
||||
)
|
||||
|
||||
self.log(self.tr("Download successful: {resource_type}, "
|
||||
"{file_path}",
|
||||
resource_type=resource_type,
|
||||
file_path=file_path))
|
||||
break # Éxito ⇒ sal del bucle
|
||||
|
||||
except (requests.exceptions.ChunkedEncodingError,
|
||||
requests.exceptions.ConnectionError,
|
||||
requests.exceptions.Timeout) as e:
|
||||
retries += 1
|
||||
self.log(self.tr("Error downloading {resource_type}, "
|
||||
"attempt {retries}/{max_retries}: {error}",
|
||||
resource_type=resource_type,
|
||||
retries=retries,
|
||||
max_retries=max_retries,
|
||||
error=e))
|
||||
if retries == max_retries:
|
||||
self.log(self.tr("Max retries reached. Failed to "
|
||||
"download {resource_type}: {file_path}",
|
||||
resource_type=resource_type,
|
||||
file_path=file_path))
|
||||
|
||||
|
||||
def process_album_page(self, page_url, base_folder, download_images=True, download_videos=True):
|
||||
try:
|
||||
if self.cancel_requested:
|
||||
return
|
||||
self.log(self.tr("Processing album URL: {page_url}", page_url=page_url))
|
||||
response = requests.get(page_url, headers=self.headers)
|
||||
if response.status_code == 200:
|
||||
soup = BeautifulSoup(response.text, 'html.parser')
|
||||
if not self.direct_download:
|
||||
folder_name = self.clean_filename(soup.find('h1').text if soup.find('h1') else self.tr("Unknown Album"))
|
||||
folder_path = self.create_folder(os.path.join(base_folder, folder_name))
|
||||
else:
|
||||
folder_path = base_folder # Use the base folder directly
|
||||
|
||||
media_urls = []
|
||||
seen_urls = set()
|
||||
|
||||
# --- vídeos ---
|
||||
if download_videos:
|
||||
for video in soup.find_all('video'):
|
||||
source = video.find('source')
|
||||
if source:
|
||||
abs_video_src = urljoin(page_url, source['src'])
|
||||
if abs_video_src in seen_urls: # ya lo tenemos
|
||||
continue
|
||||
seen_urls.add(abs_video_src)
|
||||
video_name = os.path.join(
|
||||
folder_path,
|
||||
self.clean_filename(os.path.basename(abs_video_src))
|
||||
)
|
||||
media_urls.append(
|
||||
(abs_video_src, video_name, 'Video')
|
||||
)
|
||||
|
||||
# --- imágenes ---
|
||||
if download_images:
|
||||
for div in soup.select('div.img'):
|
||||
img = div.find('img', attrs={'data-src': True})
|
||||
if img:
|
||||
abs_img_src = urljoin(page_url, img['data-src'])
|
||||
if abs_img_src in seen_urls:
|
||||
continue
|
||||
seen_urls.add(abs_img_src)
|
||||
img_name = os.path.join(
|
||||
folder_path,
|
||||
self.clean_filename(os.path.basename(abs_img_src))
|
||||
)
|
||||
media_urls.append(
|
||||
(abs_img_src, img_name, 'Image')
|
||||
)
|
||||
|
||||
self.total_files += len(media_urls)
|
||||
futures = [self.executor.submit(self.download_file, url, file_path, resource_type, str(uuid.uuid4())) for url, file_path, resource_type in media_urls]
|
||||
for future in as_completed(futures):
|
||||
if self.cancel_requested:
|
||||
self.log(self.tr("Cancelling remaining downloads."))
|
||||
break
|
||||
future.result()
|
||||
|
||||
self.log(self.tr("Album download complete: {folder_name}", folder_name=folder_name) if not self.direct_download else self.tr("Album download complete"))
|
||||
if not self.is_profile_download:
|
||||
self.enable_widgets_callback()
|
||||
else:
|
||||
self.log(self.tr("Error accessing page: {page_url}, status code: {status_code}", page_url=page_url, status_code=response.status_code))
|
||||
if not self.is_profile_download:
|
||||
self.enable_widgets_callback()
|
||||
finally:
|
||||
if not self.is_profile_download:
|
||||
self.enable_widgets_callback()
|
||||
self.export_logs()
|
||||
|
||||
def process_profile_page(self, url, download_folder, download_images, download_videos):
|
||||
try:
|
||||
if self.cancel_requested:
|
||||
return
|
||||
self.log(self.tr("Processing profile URL: {url}", url=url))
|
||||
response = requests.get(url, headers=self.headers)
|
||||
if response.status_code == 200:
|
||||
soup = BeautifulSoup(response.text, 'html.parser')
|
||||
username = soup.find('h1', class_='username').text.strip() if soup.find('h1', class_='username') else self.tr("Unknown Profile")
|
||||
base_folder = self.create_folder(os.path.join(download_folder, self.clean_filename(username)))
|
||||
|
||||
album_links = soup.find_all('a', class_='album-link')
|
||||
for album_link in album_links:
|
||||
album_href = album_link.get('href')
|
||||
album_full_url = urljoin(url, album_href)
|
||||
self.process_album_page(album_full_url, base_folder, download_images, download_videos)
|
||||
|
||||
self.log(self.tr("Profile download complete: {username}", username=username))
|
||||
self.enable_widgets_callback()
|
||||
else:
|
||||
self.log(self.tr("Error accessing page: {url}, status code: {status_code}", url=url, status_code=response.status_code))
|
||||
self.enable_widgets_callback()
|
||||
finally:
|
||||
if not self.is_profile_download:
|
||||
self.enable_widgets_callback()
|
||||
self.export_logs()
|
||||
|
||||
def export_logs(self):
|
||||
log_folder = "resources/config/logs/"
|
||||
Path(log_folder).mkdir(parents=True, exist_ok=True)
|
||||
log_file_path = Path(log_folder) / f"log_{datetime.datetime.now().strftime('%Y%m%d_%H%M%S')}.txt"
|
||||
try:
|
||||
with open(log_file_path, 'w') as file:
|
||||
file.write("\n".join(self.log_messages))
|
||||
self.log(self.tr("Logs exported successfully to {path}", path=log_file_path))
|
||||
except Exception as e:
|
||||
self.log(self.tr(f"Failed to export logs: {e}"))
|
||||
@@ -0,0 +1,112 @@
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
import os
|
||||
import threading
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from app import progress_manager
|
||||
|
||||
class Jpg5Downloader:
|
||||
def __init__(self, url, carpeta_destino, progress_manager, log_callback=None, tr=None, update_progress_callback=None, update_global_progress_callback=None, max_workers=3):
|
||||
self.url = url
|
||||
self.carpeta_destino = carpeta_destino
|
||||
self.log_callback = log_callback
|
||||
self.tr = tr if tr else lambda x: x # Función de traducción por defecto
|
||||
self.cancel_requested = threading.Event() # Usar un evento para manejar la cancelación
|
||||
self.update_progress_callback = update_progress_callback
|
||||
self.update_global_progress_callback = update_global_progress_callback
|
||||
self.max_workers = max_workers
|
||||
self.progress_manager = progress_manager
|
||||
|
||||
def log(self, message):
|
||||
if self.log_callback:
|
||||
self.log_callback(message)
|
||||
else:
|
||||
print(message)
|
||||
|
||||
def request_cancel(self):
|
||||
self.cancel_requested.set() # Activar el evento de cancelación
|
||||
self.log(self.tr("Descarga cancelada por el usuario."))
|
||||
|
||||
def descargar_imagenes(self):
|
||||
if not os.path.exists(self.carpeta_destino):
|
||||
os.makedirs(self.carpeta_destino)
|
||||
|
||||
self.log(self.tr(f"Iniciando descarga desde: {self.url}"))
|
||||
respuesta = requests.get(self.url)
|
||||
if self.cancel_requested.is_set():
|
||||
self.log(self.tr("Descarga cancelada por el usuario."))
|
||||
return
|
||||
|
||||
soup = BeautifulSoup(respuesta.content, 'html.parser')
|
||||
|
||||
divs = soup.find_all('div', class_='list-item c8 gutter-margin-right-bottom')
|
||||
total_divs = len(divs)
|
||||
self.log(self.tr(f"Total de elementos a procesar: {total_divs}"))
|
||||
|
||||
with ThreadPoolExecutor(max_workers=self.max_workers) as executor:
|
||||
futures = []
|
||||
for i, div in enumerate(divs):
|
||||
if self.cancel_requested.is_set():
|
||||
self.log(self.tr("Descarga cancelada por el usuario."))
|
||||
return
|
||||
|
||||
enlaces = div.find_all('a', class_='image-container --media')
|
||||
for enlace in enlaces:
|
||||
if self.cancel_requested.is_set():
|
||||
self.log(self.tr("Descarga cancelada por el usuario."))
|
||||
return
|
||||
|
||||
futures.append(executor.submit(self.descargar_enlace, enlace, i, total_divs))
|
||||
|
||||
for future in futures:
|
||||
future.result() # Esperar a que todas las descargas terminen
|
||||
|
||||
def descargar_enlace(self, enlace, i, total_divs):
|
||||
try:
|
||||
media_url = enlace['href']
|
||||
self.log(self.tr(f"Procesando enlace: {media_url}"))
|
||||
|
||||
media_respuesta = requests.get(media_url)
|
||||
if self.cancel_requested.is_set():
|
||||
self.log(self.tr("Descarga cancelada por el usuario."))
|
||||
return
|
||||
|
||||
media_soup = BeautifulSoup(media_respuesta.content, 'html.parser')
|
||||
|
||||
header_content = media_soup.find('div', class_='header-content-right')
|
||||
if header_content:
|
||||
btn_descarga = header_content.find('a', class_='btn btn-download default')
|
||||
if btn_descarga and 'href' in btn_descarga.attrs:
|
||||
descarga_url = btn_descarga['href']
|
||||
self.log(self.tr(f"Descargando desde: {descarga_url}"))
|
||||
|
||||
img_respuesta = requests.get(descarga_url, stream=True)
|
||||
if self.cancel_requested.is_set():
|
||||
self.log(self.tr("Descarga cancelada por el usuario."))
|
||||
return
|
||||
|
||||
img_nombre = os.path.basename(descarga_url)
|
||||
img_path = os.path.join(self.carpeta_destino, img_nombre)
|
||||
total_size = int(img_respuesta.headers.get('content-length', 0))
|
||||
downloaded_size = 0
|
||||
|
||||
with open(img_path, 'wb') as f:
|
||||
for chunk in img_respuesta.iter_content(chunk_size=1024):
|
||||
if self.cancel_requested.is_set():
|
||||
self.log(self.tr("Descarga cancelada por el usuario."))
|
||||
return
|
||||
f.write(chunk)
|
||||
downloaded_size += len(chunk)
|
||||
if self.update_progress_callback:
|
||||
self.update_progress_callback(downloaded_size, total_size)
|
||||
|
||||
self.log(self.tr(f"Imagen descargada: {img_nombre}"))
|
||||
|
||||
if self.update_global_progress_callback:
|
||||
self.update_global_progress_callback(i + 1, total_divs)
|
||||
else:
|
||||
self.log(self.tr("No se encontró el enlace de descarga."))
|
||||
else:
|
||||
self.log(self.tr("No se encontró la clase 'header-content-right'."))
|
||||
except Exception as e:
|
||||
self.log(self.tr(f"Error al procesar el enlace: {e}"))
|
||||
@@ -0,0 +1,137 @@
|
||||
import os
|
||||
import json
|
||||
import re
|
||||
import queue
|
||||
from pathlib import Path
|
||||
from bs4 import BeautifulSoup
|
||||
from urllib.parse import urlparse
|
||||
import cloudscraper
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
|
||||
class SimpCity:
|
||||
def __init__(self, download_folder, max_workers=5, log_callback=None, enable_widgets_callback=None, update_progress_callback=None, update_global_progress_callback=None, tr=None):
|
||||
self.download_folder = download_folder
|
||||
self.max_workers = max_workers
|
||||
self.descargadas = set()
|
||||
self.log_callback = log_callback
|
||||
self.enable_widgets_callback = enable_widgets_callback
|
||||
self.update_progress_callback = update_progress_callback
|
||||
self.update_global_progress_callback = update_global_progress_callback
|
||||
self.cancel_requested = False
|
||||
self.total_files = 0
|
||||
self.completed_files = 0
|
||||
self.download_queue = queue.Queue()
|
||||
self.scraper = cloudscraper.create_scraper(browser={'browser': 'chrome', 'platform': 'windows', 'mobile': False})
|
||||
self.tr = tr
|
||||
|
||||
# Selectors from original crawler
|
||||
self.title_selector = "h1[class=p-title-value]"
|
||||
self.posts_selector = "div[class*=message-main]"
|
||||
self.post_content_selector = "div[class*=message-userContent]"
|
||||
self.images_selector = "img[class*=bbImage]"
|
||||
self.videos_selector = "video source"
|
||||
self.iframe_selector = "iframe[class=saint-iframe]"
|
||||
self.attachments_block_selector = "section[class=message-attachments]"
|
||||
self.attachments_selector = "a"
|
||||
self.next_page_selector = "a[class*=pageNav-jump--next]"
|
||||
self.cookies_path = "resources/config/cookies/simpcity.json"
|
||||
self.set_cookies()
|
||||
|
||||
def log(self, message):
|
||||
if self.log_callback:
|
||||
self.log_callback(message)
|
||||
|
||||
def sanitize_folder_name(self, name):
|
||||
return re.sub(r'[<>:"/\\|?*]', '_', name)
|
||||
|
||||
def set_cookies(self):
|
||||
if os.path.exists(self.cookies_path):
|
||||
with open(self.cookies_path, "r", encoding="utf-8") as f:
|
||||
cookies = json.load(f)
|
||||
|
||||
if isinstance(cookies, dict):
|
||||
cookies = [cookies]
|
||||
|
||||
for c in cookies:
|
||||
if isinstance(c, dict) and "name" in c and "value" in c:
|
||||
self.scraper.cookies.set(c["name"], c["value"])
|
||||
|
||||
def fetch_page(self, url):
|
||||
try:
|
||||
response = self.scraper.get(url)
|
||||
if response.status_code == 200:
|
||||
return BeautifulSoup(response.content, 'html.parser')
|
||||
else:
|
||||
self.log(self.tr(f"Error: {response.status_code} al acceder a {url}"))
|
||||
return None
|
||||
except Exception as e:
|
||||
self.log(self.tr(f"Error al acceder a {url}: {e}"))
|
||||
return None
|
||||
|
||||
def save_file(self, file_url, path):
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
response = self.scraper.get(file_url, stream=True)
|
||||
if response.status_code == 200:
|
||||
with open(path, 'wb') as file:
|
||||
for chunk in response.iter_content(1024):
|
||||
file.write(chunk)
|
||||
self.log(self.tr(f"Archivo descargado: {path}"))
|
||||
else:
|
||||
self.log(self.tr(f"Error al descargar {file_url}: {response.status_code}"))
|
||||
|
||||
def process_post(self, post_content, download_folder):
|
||||
# Procesar imágenes
|
||||
images = post_content.select(self.images_selector)
|
||||
for img in images:
|
||||
src = img.get('src')
|
||||
if src:
|
||||
file_name = os.path.basename(urlparse(src).path)
|
||||
file_path = os.path.join(download_folder, file_name)
|
||||
self.save_file(src, file_path)
|
||||
|
||||
# Procesar videos
|
||||
videos = post_content.select(self.videos_selector)
|
||||
for video in videos:
|
||||
src = video.get('src')
|
||||
if src:
|
||||
file_name = os.path.basename(urlparse(src).path)
|
||||
file_path = os.path.join(download_folder, file_name)
|
||||
self.save_file(src, file_path)
|
||||
|
||||
# Procesar archivos adjuntos
|
||||
attachments_block = post_content.select_one(self.attachments_block_selector)
|
||||
if attachments_block:
|
||||
attachments = attachments_block.select(self.attachments_selector)
|
||||
for attachment in attachments:
|
||||
href = attachment.get('href')
|
||||
if href:
|
||||
file_name = os.path.basename(urlparse(href).path)
|
||||
file_path = os.path.join(download_folder, file_name)
|
||||
self.save_file(href, file_path)
|
||||
|
||||
def process_page(self, url):
|
||||
soup = self.fetch_page(url)
|
||||
if not soup:
|
||||
return
|
||||
|
||||
title_element = soup.select_one(self.title_selector)
|
||||
folder_name = self.sanitize_folder_name(title_element.text.strip()) if title_element else 'SimpCity_Download'
|
||||
download_folder = os.path.join(self.download_folder, folder_name)
|
||||
os.makedirs(download_folder, exist_ok=True)
|
||||
|
||||
message_inners = soup.select(self.posts_selector)
|
||||
for post in message_inners:
|
||||
post_content = post.select_one(self.post_content_selector)
|
||||
if post_content:
|
||||
self.process_post(post_content, download_folder)
|
||||
|
||||
next_page = soup.select_one(self.next_page_selector)
|
||||
if next_page:
|
||||
next_page_url = next_page.get('href')
|
||||
if next_page_url:
|
||||
self.process_page(self.base_url + next_page_url)
|
||||
|
||||
def download_images_from_simpcity(self, url):
|
||||
self.log(self.tr(f"Procesando hilo: {url}"))
|
||||
self.process_page(url)
|
||||
self.log(self.tr("Descarga completada."))
|
||||
Reference in New Issue
Block a user