Files
2026-01-09 19:24:41 +00:00

697 lines
31 KiB
Python

import os
import re
import shutil
import zipfile
import tempfile
import logging
import cloudscraper
import difflib
from bs4 import BeautifulSoup
from collections import defaultdict
from .samba_manager import SambaManager
from .config import settings
from smb.smb_structs import OperationFailure
# --- Dedicated logger for ComicsManager ---
comics_logger = logging.getLogger('comics_manager')
comics_logger.setLevel(logging.INFO)
comics_logger.propagate = False
if not comics_logger.handlers:
os.makedirs("logs", exist_ok=True)
comics_log_handler = logging.FileHandler("logs/comics_organization.log", mode='a')
comics_log_handler.setFormatter(logging.Formatter('%(asctime)s - %(levelname)s - %(message)s'))
comics_logger.addHandler(comics_log_handler)
comics_logger.addHandler(logging.StreamHandler())
class ComicsManager:
def __init__(self):
# Staging settings (on 'isolation' share)
self.comics_root = "/comics/pendingorganization"
# Library defaults
self.library_share_default = "isolation"
self.library_path_default = "/comics/manga"
self.scraper = cloudscraper.create_scraper()
def _get_series_info(self, filename):
base_name, _ = os.path.splitext(filename)
artist = "Unknown"
doujin_id = None
id_match = re.search(r'[[\\](](\\d{5,7})[[\\])]', base_name)
if id_match:
doujin_id = id_match.group(1)
tags = re.findall(r'[[\\](](.*?)[[\\])]', base_name)
tags = [t for t in tags if t != doujin_id]
if tags:
artist = tags[0].strip()
name = re.sub(r'[[\\](].*?[[\\])]', '', base_name).strip()
pattern = re.compile(r'[-_\\s]*(v(ol)?\\. ?|c(h)?\\. ?|chapter|issue|ep(isode)?)\\s*\\d+.*$', re.IGNORECASE)
match = pattern.search(name)
if match:
name = name[:match.start()]
else:
name = re.sub(r'[-_\\s]+\\d+\\s*$', '', name)
name = name.strip(' -_')
if not name and tags:
name = tags[0]
elif not name:
name = base_name.strip()
name = name.replace(' ', '_')
artist = artist.replace(' ', '_')
return name, artist, doujin_id
def _fetch_metadata_from_url(self, target_url):
try:
comics_logger.info(f"Fetching metadata from: {target_url}")
response = self.scraper.get(target_url)
if response.status_code == 200:
soup = BeautifulSoup(response.content, 'html.parser')
data = {}
title_info = soup.select_one('#info')
if title_info:
data['title'] = title_info.select_one('h1.title').text.strip() if title_info.select_one('h1.title') else ""
data['original_title'] = title_info.select_one('h2.title').text.strip() if title_info.select_one('h2.title') else ""
tag_containers = soup.select('.tag-container')
for container in tag_containers:
label = container.text.split(':')[0].strip().lower() if ':' in container.text else ""
tags = [t.select_one('.name').text for t in container.select('.tag')]
if 'tags' in label: data['tags'] = tags
elif 'artists' in label: data['artist'] = tags
elif 'groups' in label: data['circle'] = tags
elif 'parodies' in label: data['parody'] = tags
time_tag = soup.select_one('#info time')
if time_tag and time_tag.get('datetime'):
year_match = re.search(r'(\\d{4})', time_tag['datetime'])
if year_match: data['year'] = year_match.group(1)
data['url'] = target_url
data['id'] = target_url.split('/g/')[1].strip('/')
return data
except Exception as e:
comics_logger.error(f"Error parsing metadata: {e}")
return None
def _lookup_doujin_metadata(self, title, artist="Unknown", doujin_id=None):
base_url = "https://nhentai.net"
if doujin_id:
return self._fetch_metadata_from_url(f"{base_url}/g/{doujin_id}/")
clean_title = title.replace('_', ' ').strip()
clean_artist = artist.replace('_', ' ').strip() if artist != "Unknown" else ""
queries = []
if clean_artist: queries.append(f"{clean_artist} {clean_title}")
queries.append(clean_title)
for query in queries:
try:
search_url = f"{base_url}/search/?q={query.replace(' ', '+')}"
comics_logger.info(f"Searching metadata for: {query}")
response = self.scraper.get(search_url)
if response.status_code == 200:
soup = BeautifulSoup(response.content, 'html.parser')
results = soup.select('.gallery a.cover')
if results:
target_url = f"{base_url}{results[0]['href']}"
data = self._fetch_metadata_from_url(target_url)
if data: return data
except Exception as e:
comics_logger.error(f"Search failed for query '{query}': {e}")
return None
def _download_directory(self, samba_manager, remote_path, local_path):
os.makedirs(local_path, exist_ok=True)
items = samba_manager.list_path(remote_path)
if isinstance(items, dict) and "error" in items: raise Exception(items["error"])
for item in items:
if item['name'] in ['.', '..']: continue
local_item_path = os.path.join(local_path, item['name'])
if item['is_directory']: self._download_directory(samba_manager, item['path'], local_item_path)
else:
with open(local_item_path, 'wb') as f:
samba_manager.download_file(item['path'], f)
def _extract_archive(self, archive_path, extract_path):
if zipfile.is_zipfile(archive_path):
with zipfile.ZipFile(archive_path, 'r') as zf:
zf.extractall(extract_path)
else: raise Exception("Unsupported archive format")
def _create_cbz(self, source_folder, output_path):
with zipfile.ZipFile(output_path, 'w', zipfile.ZIP_DEFLATED) as zf:
for root, dirs, files in os.walk(source_folder):
for file in files:
file_path = os.path.join(root, file)
arcname = os.path.relpath(file_path, source_folder)
zf.write(file_path, arcname)
def _create_metadata_file(self, series_name, author="Unknown", artist="Unknown", api_data=None):
content = f"Title: {series_name}\n"
if api_data:
if api_data.get('title'): content = f"Title: {api_data['title']}\n"
if api_data.get('original_title'): content += f"Original Title: {api_data['original_title']}\n"
artists = ", ".join(api_data.get('artist', [])) or artist
circles = ", ".join(api_data.get('circle', [])) or "Unknown"
parodies = ", ".join(api_data.get('parody', [])) or "Original"
tags = ", ".join(api_data.get('tags', []))
content += f"Artist: {artists}\nCircle: {circles}\nParody: {parodies}\nTags: {tags}\nURL: {api_data.get('url', 'N/A')}\n"
else:
content += f"Artist: {artist}\nAuthor: {author}\n"
content += f"Processed by: ServerManagerWebApp\n"
return content
def _lookup_mangadex_metadata(self, title):
"""
Queries Mangadex API.
"""
try:
url = "https://api.mangadex.org/manga"
clean_title = title.replace('_', ' ').strip()
comics_logger.info(f"Searching Mangadex for: {clean_title}")
# Mangadex requires separate calls for author/artist usually, but we'll start with basic info
params = {
"title": clean_title,
"limit": 1,
"includes[]": ["author", "artist", "cover_art"]
}
response = self.scraper.get(url, params=params, timeout=10)
if response.status_code == 200:
data = response.json()
if data.get('data'):
manga = data['data'][0]
attr = manga['attributes']
# Extract authors/artists
authors = []
artists = []
for rel in manga['relationships']:
if rel['type'] == 'author': authors.append(rel.get('attributes', {}).get('name'))
if rel['type'] == 'artist': artists.append(rel.get('attributes', {}).get('name'))
# Tags
tags = [t['attributes']['name']['en'] for t in attr.get('tags', [])]
return {
"title": attr['title'].get('en') or list(attr['title'].values())[0],
"original_title": attr.get('altTitles', [{}])[0].get('en', "") if attr.get('altTitles') else "",
"year": attr.get('year'),
"url": f"https://mangadex.org/title/{manga['id']}",
"id": manga['id'],
"tags": tags,
"author": authors,
"artist": artists,
"source": "Mangadex"
}
except Exception as e:
comics_logger.error(f"Mangadex lookup failed: {e}")
return None
def _lookup_hentai2read_metadata(self, title):
"""
Scrapes Hentai2Read.
"""
try:
base_url = "https://hentai2read.com"
clean_title = title.replace('_', '+').strip()
search_url = f"{base_url}/search/?cmd={clean_title}"
comics_logger.info(f"Searching Hentai2Read for: {clean_title}")
response = self.scraper.get(search_url)
if response.status_code == 200:
soup = BeautifulSoup(response.content, 'html.parser')
# Hentai2Read search results structure
result_link = soup.select_one('.book-grid-item a')
if result_link:
target_url = result_link['href']
comics_logger.info(f"Fetching Hentai2Read details: {target_url}")
resp = self.scraper.get(target_url)
if resp.status_code == 200:
soup = BeautifulSoup(resp.content, 'html.parser')
data = {"source": "Hentai2Read", "url": target_url}
# Title
title_tag = soup.select_one('h3.block-title a')
if title_tag: data['title'] = title_tag.text.strip()
# Info list
info_items = soup.select('ul.list-simple-mini li')
for item in info_items:
text = item.text.strip()
if "Author" in text:
data['author'] = [a.strip() for a in text.replace("Author", "").strip(" :" ).split(',')]
elif "Artist" in text:
data['artist'] = [a.strip() for a in text.replace("Artist", "").strip(" :" ).split(',')]
elif "Parody" in text:
data['parody'] = [a.strip() for a in text.replace("Parody", "").strip(" :" ).split(',')]
elif "Storyline" in text or "Content" in text: # Tags
data['tags'] = [a.text.strip() for a in item.select('a')]
elif "Release" in text:
year_match = re.search(r'\\d{4}', text)
if year_match: data['year'] = year_match.group(0)
return data
except Exception as e:
comics_logger.error(f"Hentai2Read lookup failed: {e}")
return None
def _lookup_mangaupdates_metadata(self, title):
"""
Queries MangaUpdates API for metadata (good for Manhwa/Webtoons).
"""
try:
url = "https://api.mangaupdates.com/v1/series/search"
# Replace underscores with spaces for better search
clean_title = title.replace('_', ' ').strip()
comics_logger.info(f"Searching MangaUpdates for: {clean_title}")
response = self.scraper.post(url, json={"search": clean_title}, timeout=10)
if response.status_code == 200:
data = response.json()
if data.get('results'):
series = data['results'][0]['record']
return {
"title": series.get('title'),
"year": series.get('year'),
"url": series.get('url'),
"id": series.get('series_id'),
"tags": [g.get('genre') for g in series.get('genres', [])] if series.get('genres') else [],
"source": "MangaUpdates"
}
except Exception as e:
comics_logger.error(f"MangaUpdates lookup failed: {e}")
return None
def _process_item(self, samba_manager, item, is_archive=False):
item_path = item['path']
item_name = item['name']
name_part, artist_part, doujin_id = self._get_series_info(item_name)
api_data = None
# 1. Try Mangadex (Standard/Manhwa)
if not api_data:
api_data = self._lookup_mangadex_metadata(name_part)
# 2. Try nhentai (Doujinshi)
if not api_data:
# Need to re-add "source": "nhentai" to the existing method logic or wrapper
data = self._lookup_doujin_metadata(name_part, artist_part, doujin_id)
if data:
data['source'] = "nhentai"
api_data = data
# 3. Try Hentai2Read (Fallback Doujin)
if not api_data:
api_data = self._lookup_hentai2read_metadata(name_part)
# 4. Try MangaUpdates (Final Fallback)
if not api_data:
api_data = self._lookup_mangaupdates_metadata(name_part)
final_series_name = name_part
if api_data:
if api_data.get('title'):
title_clean = re.sub(r'[<>:"/\\|?*]', '', api_data['title']).strip()
final_series_name = title_clean.replace(' ', '_')
if api_data.get('year'): final_series_name = f"{final_series_name}_({api_data['year']})"
elif api_data.get('id'): final_series_name = f"{final_series_name}_({api_data['id']})"
final_series_name = final_series_name.replace(' ', '_')
series_folder_path = f"{self.comics_root}/{final_series_name}"
try:
samba_manager.create_directory(series_folder_path)
except OperationFailure: pass
if item_name == final_series_name: return
comics_logger.info(f"Processing: {item_name} -> {final_series_name}")
with tempfile.TemporaryDirectory() as temp_dir:
extraction_path = os.path.join(temp_dir, "extracted")
os.makedirs(extraction_path, exist_ok=True)
if is_archive:
local_archive = os.path.join(temp_dir, item_name)
with open(local_archive, 'wb') as f: samba_manager.download_file(item_path, f)
try: self._extract_archive(local_archive, extraction_path)
except Exception as e:
comics_logger.error(f"Failed to extract {item_name}: {e}")
return
else: self._download_directory(samba_manager, item_path, extraction_path)
has_images = False
for root, _, files in os.walk(extraction_path):
if any(f.lower().endswith(('.jpg', '.jpeg', '.png', '.webp')) for f in files):
has_images = True
break
if has_images:
cbz_name = f"{final_series_name}.cbz"
temp_cbz = os.path.join(temp_dir, cbz_name)
self._create_cbz(extraction_path, temp_cbz)
dest_cbz_path = f"{series_folder_path}/{cbz_name}"
with open(temp_cbz, 'rb') as f:
res = samba_manager.upload_file(dest_cbz_path, f)
if isinstance(res, dict) and "error" in res: raise Exception(f"Upload failed: {res['error']}")
meta_content = self._create_metadata_file(final_series_name, artist=artist_part, api_data=api_data)
meta_path = f"{series_folder_path}/{final_series_name}_info.txt"
with tempfile.NamedTemporaryFile(mode='w+', delete=False) as tmp_meta:
tmp_meta.write(meta_content)
tmp_meta.flush()
tmp_meta.seek(0)
with open(tmp_meta.name, 'rb') as f: samba_manager.upload_file(meta_path, f)
os.unlink(tmp_meta.name)
if is_archive: samba_manager.delete_file(item_path)
else: samba_manager.delete_directory_recursive(item_path)
else: comics_logger.warning(f"No images found in {item_name}, skipping.")
def organize_comics(self, samba_manager: SambaManager):
comics_logger.info(f"Starting organization in {self.comics_root}...")
self._recursive_scan_and_process(samba_manager, self.comics_root)
def _recursive_scan_and_process(self, samba_manager, current_path):
try:
items = samba_manager.list_path(current_path)
if isinstance(items, dict) and "error" in items: return
for item in items:
if item['name'] in ['.', '..']: continue
if item['is_directory']:
sub_items = samba_manager.list_path(item['path'])
if any(sub['name'].lower().endswith(('.jpg', '.jpeg', '.png', '.webp')) for sub in sub_items):
self._process_item(samba_manager, item, is_archive=False)
else: self._recursive_scan_and_process(samba_manager, item['path'])
elif item['name'].lower().endswith(('.zip', '.cbz')):
self._process_item(samba_manager, item, is_archive=True)
except Exception as e: comics_logger.error(f"Error scanning {current_path}: {e}")
def get_pending_comics(self, samba_manager: SambaManager):
try:
items = samba_manager.list_path(self.comics_root)
if isinstance(items, dict) and "error" in items: return []
return [{"name": i['name'], "path": i['path']} for i in items if i['is_directory'] and i['name'] not in ['.', '..']]
except Exception: return []
def move_series(self, src_samba: SambaManager, series_names, dest_share=None, dest_path=None):
"""Moves folders from isolation/staging to library share."""
share = dest_share if dest_share else self.library_share_default
path = dest_path if dest_path else self.library_path_default
# Connect to destination share
dest_samba = SambaManager(src_samba.server_ip, share, src_samba.username, src_samba.password)
results = {"success": [], "failed": []}
for series in series_names:
src_folder = f"{self.comics_root}/{series}"
target_folder = f"{path}/{series}"
try:
# Ensure target directory exists
dest_samba.create_directory(target_folder)
# List files in source (on isolation share)
items = src_samba.list_path(src_folder)
for item in items:
if item['name'] in ['.', '..']: continue
# Cross-share move: Download -> Upload -> Delete
with tempfile.TemporaryDirectory() as temp_dir:
local_file = os.path.join(temp_dir, item['name'])
with open(local_file, 'wb') as f: src_samba.download_file(item['path'], f)
with open(local_file, 'rb') as f: dest_samba.upload_file(f"{target_folder}/{item['name']}", f)
src_samba.delete_file(item['path'])
src_samba.delete_directory(src_folder)
results["success"].append(series)
comics_logger.info(f"Moved {series} to {share}:{target_folder}")
except Exception as e:
results["failed"].append({"name": series, "error": str(e)})
comics_logger.error(f"Failed to move {series}: {e}")
dest_samba.close()
return results
def update_existing_metadata(self, samba_manager, target_path="/comics/manga", force=False):
"""
Scans an existing library directory for series folders and updates/creates metadata files.
"""
comics_logger.info(f"Starting metadata update in {target_path} (Force: {force})...")
try:
items = samba_manager.list_path(target_path)
if isinstance(items, dict) and "error" in items:
comics_logger.error(f"Error listing {target_path}: {items['error']}")
return
for item in items:
if item['name'] in ['.', '..']: continue
if not item['is_directory']: continue
series_name = item['name']
series_path = item['path']
# Check for existing metadata
meta_filename = f"{series_name}_info.txt"
meta_path = f"{series_path}/{meta_filename}"
# Check if meta exists
series_contents = samba_manager.list_path(series_path)
has_meta = False
if isinstance(series_contents, list):
for sub in series_contents:
if sub['name'] == meta_filename:
has_meta = True
break
if has_meta and not force:
continue
comics_logger.info(f"Updating metadata for: {series_name}")
# Parse series info from FOLDER NAME
clean_name = series_name.replace('_', ' ')
clean_name = re.sub(r'\s*\\(\\d+\\)$', '', clean_name).strip()
artist = "Unknown"
artist_match = re.match(r'^\\[(.*?)\\]', clean_name)
if artist_match:
artist = artist_match.group(1)
clean_name = clean_name[artist_match.end():].strip()
doujin_id = None
id_match = re.search(r'\\(\\d{5,7}\\)$', series_name)
if id_match:
doujin_id = id_match.group(1)
# 1. Try Mangadex
api_data = self._lookup_mangadex_metadata(clean_name)
# 2. Try nhentai
if not api_data:
data = self._lookup_doujin_metadata(clean_name, artist, doujin_id)
if data:
data['source'] = "nhentai"
api_data = data
# 3. Try Hentai2Read
if not api_data:
api_data = self._lookup_hentai2read_metadata(clean_name)
# 4. Try MangaUpdates
if not api_data:
api_data = self._lookup_mangaupdates_metadata(clean_name)
meta_content = self._create_metadata_file(clean_name, artist=artist, api_data=api_data)
with tempfile.NamedTemporaryFile(mode='w+', delete=False) as tmp_meta:
tmp_meta.write(meta_content)
tmp_meta.flush()
tmp_meta.seek(0)
with open(tmp_meta.name, 'rb') as f:
samba_manager.upload_file(meta_path, f)
os.unlink(tmp_meta.name)
except Exception as e:
comics_logger.error(f"Error updating metadata in {target_path}: {e}")
def _collect_all_folders(self, samba_manager, path):
folders = []
try:
items = samba_manager.list_path(path)
for item in items:
if item['name'] in ['.', '..']: continue
if item['is_directory']:
folders.append({'name': item['name'], 'path': item['path']})
# Recurse
folders.extend(self._collect_all_folders(samba_manager, item['path']))
except Exception as e:
comics_logger.error(f"Error listing path {path}: {e}")
return folders
def find_similar_folders(self, samba_manager: SambaManager, root_path="/comics/manga", threshold=0.9):
"""
Scans for folders with similar names.
"""
comics_logger.info(f"Scanning for duplicate folders in {root_path}...")
all_dirs = self._collect_all_folders(samba_manager, root_path)
comics_logger.info(f"Found {len(all_dirs)} directories. Comparing...")
groups = []
processed_indices = set()
for i in range(len(all_dirs)):
if i in processed_indices: continue
current_group = [all_dirs[i]]
for j in range(i + 1, len(all_dirs)):
if j in processed_indices: continue
name1 = all_dirs[i]['name'].lower().replace('_', ' ')
name2 = all_dirs[j]['name'].lower().replace('_', ' ')
ratio = difflib.SequenceMatcher(None, name1, name2).ratio()
if ratio >= threshold:
current_group.append(all_dirs[j])
processed_indices.add(j)
if len(current_group) > 1:
groups.append({
"name": all_dirs[i]['name'],
"folders": current_group
})
processed_indices.add(i)
return groups
def delete_folder(self, samba_manager: SambaManager, folder_path):
"""
Deletes a specific folder.
"""
try:
comics_logger.info(f"Deleting duplicate folder: {folder_path}")
samba_manager.delete_directory_recursive(folder_path)
return {"success": True}
except Exception as e:
comics_logger.error(f"Failed to delete folder {folder_path}: {e}")
return {"error": str(e)}
def _get_artist_from_info(self, samba_manager, series_path, series_name):
info_filename = f"{series_name}_info.txt"
info_path = f"{series_path}/{info_filename}"
try:
with tempfile.NamedTemporaryFile(mode='w+b', delete=False) as tmp:
samba_manager.download_file(info_path, tmp)
tmp.seek(0)
content = tmp.read().decode('utf-8', errors='ignore')
# Parse content
artist = "Unknown"
author = "Unknown"
for line in content.splitlines():
if line.startswith("Artist:"):
val = line.split(":", 1)[1].strip()
if val and val.lower() != "unknown":
artist = val.split(',')[0].strip() # Take first artist if multiple
elif line.startswith("Author:"):
val = line.split(":", 1)[1].strip()
if val and val.lower() != "unknown":
author = val.split(',')[0].strip()
if artist != "Unknown": return artist
if author != "Unknown": return author
except Exception as e:
# File might not exist or other error
pass
finally:
if 'tmp' in locals() and os.path.exists(tmp.name):
os.unlink(tmp.name)
return "_Unknown"
def sort_by_artist(self, samba_manager, root_path="/comics/manga"):
comics_logger.info(f"Sorting by Artist in {root_path}...")
try:
items = samba_manager.list_path(root_path)
if isinstance(items, dict) and "error" in items:
comics_logger.error(f"Error listing {root_path}: {items['error']}")
return
for item in items:
if item['name'] in ['.', '..', '_Unknown']: continue
if not item['is_directory']: continue
series_name = item['name']
series_path = item['path']
# Check if this is a Series Folder
# Criteria: Contains .cbz, .zip, or _info.txt
try:
sub_items = samba_manager.list_path(series_path)
if isinstance(sub_items, dict) and "error" in sub_items: continue
is_series = False
for sub in sub_items:
if sub['name'].lower().endswith(('.cbz', '.zip', '_info.txt')):
is_series = True
break
if not is_series:
comics_logger.info(f"Skipping potential Artist folder or empty folder: {series_name}")
continue
except Exception:
continue
# Attempt to get artist
artist = self._get_artist_from_info(samba_manager, series_path, series_name)
# Sanitize artist name for folder
clean_artist = re.sub(r'[<>:"/\\|?*]', '', artist).strip().replace(' ', '_')
if not clean_artist: clean_artist = "_Unknown"
# Target path: /comics/manga/Artist/Series
artist_folder = f"{root_path}/{clean_artist}"
target_path = f"{artist_folder}/{series_name}"
# Skip if already in place
if series_name == clean_artist:
continue
# Check if we are moving into itself (e.g. Root/Artist -> Root/Artist/Artist)
# This happens if 'SeriesName' == 'ArtistName' and it was already sorted?
# But we checked is_series. An Artist folder usually doesn't have cbz inside directly.
comics_logger.info(f"Moving '{series_name}' to Artist folder '{clean_artist}'")
try:
# Create Artist folder
try:
samba_manager.create_directory(artist_folder)
except OperationFailure: pass # Exists
# Move Series folder
samba_manager.rename_file(series_path, target_path)
except Exception as e:
comics_logger.error(f"Failed to move {series_name}: {e}")
except Exception as e:
comics_logger.error(f"Sort by artist failed: {e}")