major server manager overhaul
This commit is contained in:
@@ -1,9 +1,15 @@
|
||||
import os
|
||||
import re
|
||||
from collections import defaultdict
|
||||
import shutil
|
||||
import zipfile
|
||||
import tempfile
|
||||
import logging
|
||||
import cloudscraper
|
||||
import difflib
|
||||
from bs4 import BeautifulSoup
|
||||
from collections import defaultdict
|
||||
from .samba_manager import SambaManager
|
||||
from .config import Settings
|
||||
from .config import settings
|
||||
from smb.smb_structs import OperationFailure
|
||||
|
||||
# --- Dedicated logger for ComicsManager ---
|
||||
@@ -11,117 +17,681 @@ comics_logger = logging.getLogger('comics_manager')
|
||||
comics_logger.setLevel(logging.INFO)
|
||||
comics_logger.propagate = False
|
||||
if not comics_logger.handlers:
|
||||
os.makedirs("logs", exist_ok=True)
|
||||
comics_log_handler = logging.FileHandler("logs/comics_organization.log", mode='a')
|
||||
comics_log_handler.setFormatter(logging.Formatter('%(asctime)s - %(levelname)s - %(message)s'))
|
||||
comics_logger.addHandler(comics_log_handler)
|
||||
comics_logger.addHandler(logging.StreamHandler())
|
||||
|
||||
class ComicsManager:
|
||||
def __init__(self):
|
||||
# Staging settings (on 'isolation' share)
|
||||
self.comics_root = "/comics/pendingorganization"
|
||||
|
||||
# Library defaults
|
||||
self.library_share_default = "isolation"
|
||||
self.library_path_default = "/comics/manga"
|
||||
|
||||
self.scraper = cloudscraper.create_scraper()
|
||||
|
||||
@staticmethod
|
||||
def _get_series_name(filename):
|
||||
def _get_series_info(self, filename):
|
||||
base_name, _ = os.path.splitext(filename)
|
||||
pattern = re.compile(r'[-_\s]*(v(ol)?|c(h)?|chapter|issue|ep|episode)[-_\s]*\d+.*|[-_\s]+\d+$', re.IGNORECASE)
|
||||
match = pattern.search(base_name)
|
||||
artist = "Unknown"
|
||||
doujin_id = None
|
||||
|
||||
id_match = re.search(r'[[\\](](\\d{5,7})[[\\])]', base_name)
|
||||
if id_match:
|
||||
doujin_id = id_match.group(1)
|
||||
|
||||
tags = re.findall(r'[[\\](](.*?)[[\\])]', base_name)
|
||||
tags = [t for t in tags if t != doujin_id]
|
||||
if tags:
|
||||
artist = tags[0].strip()
|
||||
|
||||
name = re.sub(r'[[\\](].*?[[\\])]', '', base_name).strip()
|
||||
pattern = re.compile(r'[-_\\s]*(v(ol)?\\. ?|c(h)?\\. ?|chapter|issue|ep(isode)?)\\s*\\d+.*$', re.IGNORECASE)
|
||||
match = pattern.search(name)
|
||||
if match:
|
||||
return base_name[:match.start()].strip('-_ ')
|
||||
return base_name.strip('-_ ')
|
||||
name = name[:match.start()]
|
||||
else:
|
||||
name = re.sub(r'[-_\\s]+\\d+\\s*$', '', name)
|
||||
|
||||
def cleanup_toberead(self, samba_manager: SambaManager):
|
||||
comics_logger.info("Starting cleanup of 'toberead' directory...")
|
||||
comics_root = "/comics/toberead"
|
||||
try:
|
||||
items_in_toberead = samba_manager.list_path(comics_root)
|
||||
if "error" in items_in_toberead:
|
||||
raise Exception(f"Failed to list files in 'toberead': {items_in_toberead['error']}")
|
||||
|
||||
for item in items_in_toberead:
|
||||
item_path = item["path"]
|
||||
item_name = item["name"]
|
||||
if item["is_directory"]:
|
||||
comics_logger.info(f"Deleting directory '{item_name}' and all its contents.")
|
||||
try:
|
||||
samba_manager.delete_directory_recursive(item_path)
|
||||
comics_logger.info(f"Successfully deleted directory: {item_name}")
|
||||
except OperationFailure as e:
|
||||
comics_logger.error(f"Failed to delete directory {item_name}: {e}")
|
||||
else:
|
||||
comics_logger.info(f"Deleting file '{item_name}'.")
|
||||
try:
|
||||
samba_manager.delete_file(item_path)
|
||||
comics_logger.info(f"Successfully deleted file: {item_name}")
|
||||
except OperationFailure as e:
|
||||
comics_logger.error(f"Failed to delete file {item_name}: {e}")
|
||||
name = name.strip(' -_')
|
||||
if not name and tags:
|
||||
name = tags[0]
|
||||
elif not name:
|
||||
name = base_name.strip()
|
||||
|
||||
comics_logger.info("Cleanup of 'toberead' finished.")
|
||||
name = name.replace(' ', '_')
|
||||
artist = artist.replace(' ', '_')
|
||||
|
||||
return name, artist, doujin_id
|
||||
|
||||
def _fetch_metadata_from_url(self, target_url):
|
||||
try:
|
||||
comics_logger.info(f"Fetching metadata from: {target_url}")
|
||||
response = self.scraper.get(target_url)
|
||||
if response.status_code == 200:
|
||||
soup = BeautifulSoup(response.content, 'html.parser')
|
||||
data = {}
|
||||
title_info = soup.select_one('#info')
|
||||
if title_info:
|
||||
data['title'] = title_info.select_one('h1.title').text.strip() if title_info.select_one('h1.title') else ""
|
||||
data['original_title'] = title_info.select_one('h2.title').text.strip() if title_info.select_one('h2.title') else ""
|
||||
|
||||
tag_containers = soup.select('.tag-container')
|
||||
for container in tag_containers:
|
||||
label = container.text.split(':')[0].strip().lower() if ':' in container.text else ""
|
||||
tags = [t.select_one('.name').text for t in container.select('.tag')]
|
||||
if 'tags' in label: data['tags'] = tags
|
||||
elif 'artists' in label: data['artist'] = tags
|
||||
elif 'groups' in label: data['circle'] = tags
|
||||
elif 'parodies' in label: data['parody'] = tags
|
||||
|
||||
time_tag = soup.select_one('#info time')
|
||||
if time_tag and time_tag.get('datetime'):
|
||||
year_match = re.search(r'(\\d{4})', time_tag['datetime'])
|
||||
if year_match: data['year'] = year_match.group(1)
|
||||
|
||||
data['url'] = target_url
|
||||
data['id'] = target_url.split('/g/')[1].strip('/')
|
||||
return data
|
||||
except Exception as e:
|
||||
comics_logger.error(f"An error occurred during 'toberead' cleanup: {e}", exc_info=True)
|
||||
comics_logger.error(f"Error parsing metadata: {e}")
|
||||
return None
|
||||
|
||||
def _lookup_doujin_metadata(self, title, artist="Unknown", doujin_id=None):
|
||||
base_url = "https://nhentai.net"
|
||||
if doujin_id:
|
||||
return self._fetch_metadata_from_url(f"{base_url}/g/{doujin_id}/")
|
||||
|
||||
clean_title = title.replace('_', ' ').strip()
|
||||
clean_artist = artist.replace('_', ' ').strip() if artist != "Unknown" else ""
|
||||
queries = []
|
||||
if clean_artist: queries.append(f"{clean_artist} {clean_title}")
|
||||
queries.append(clean_title)
|
||||
|
||||
for query in queries:
|
||||
try:
|
||||
search_url = f"{base_url}/search/?q={query.replace(' ', '+')}"
|
||||
comics_logger.info(f"Searching metadata for: {query}")
|
||||
response = self.scraper.get(search_url)
|
||||
if response.status_code == 200:
|
||||
soup = BeautifulSoup(response.content, 'html.parser')
|
||||
results = soup.select('.gallery a.cover')
|
||||
if results:
|
||||
target_url = f"{base_url}{results[0]['href']}"
|
||||
data = self._fetch_metadata_from_url(target_url)
|
||||
if data: return data
|
||||
except Exception as e:
|
||||
comics_logger.error(f"Search failed for query '{query}': {e}")
|
||||
return None
|
||||
|
||||
def _download_directory(self, samba_manager, remote_path, local_path):
|
||||
os.makedirs(local_path, exist_ok=True)
|
||||
items = samba_manager.list_path(remote_path)
|
||||
if isinstance(items, dict) and "error" in items: raise Exception(items["error"])
|
||||
for item in items:
|
||||
if item['name'] in ['.', '..']: continue
|
||||
local_item_path = os.path.join(local_path, item['name'])
|
||||
if item['is_directory']: self._download_directory(samba_manager, item['path'], local_item_path)
|
||||
else:
|
||||
with open(local_item_path, 'wb') as f:
|
||||
samba_manager.download_file(item['path'], f)
|
||||
|
||||
def _extract_archive(self, archive_path, extract_path):
|
||||
if zipfile.is_zipfile(archive_path):
|
||||
with zipfile.ZipFile(archive_path, 'r') as zf:
|
||||
zf.extractall(extract_path)
|
||||
else: raise Exception("Unsupported archive format")
|
||||
|
||||
def _create_cbz(self, source_folder, output_path):
|
||||
with zipfile.ZipFile(output_path, 'w', zipfile.ZIP_DEFLATED) as zf:
|
||||
for root, dirs, files in os.walk(source_folder):
|
||||
for file in files:
|
||||
file_path = os.path.join(root, file)
|
||||
arcname = os.path.relpath(file_path, source_folder)
|
||||
zf.write(file_path, arcname)
|
||||
|
||||
def _create_metadata_file(self, series_name, author="Unknown", artist="Unknown", api_data=None):
|
||||
content = f"Title: {series_name}\n"
|
||||
if api_data:
|
||||
if api_data.get('title'): content = f"Title: {api_data['title']}\n"
|
||||
if api_data.get('original_title'): content += f"Original Title: {api_data['original_title']}\n"
|
||||
artists = ", ".join(api_data.get('artist', [])) or artist
|
||||
circles = ", ".join(api_data.get('circle', [])) or "Unknown"
|
||||
parodies = ", ".join(api_data.get('parody', [])) or "Original"
|
||||
tags = ", ".join(api_data.get('tags', []))
|
||||
content += f"Artist: {artists}\nCircle: {circles}\nParody: {parodies}\nTags: {tags}\nURL: {api_data.get('url', 'N/A')}\n"
|
||||
else:
|
||||
content += f"Artist: {artist}\nAuthor: {author}\n"
|
||||
content += f"Processed by: ServerManagerWebApp\n"
|
||||
return content
|
||||
|
||||
def _lookup_mangadex_metadata(self, title):
|
||||
"""
|
||||
Queries Mangadex API.
|
||||
"""
|
||||
try:
|
||||
url = "https://api.mangadex.org/manga"
|
||||
clean_title = title.replace('_', ' ').strip()
|
||||
comics_logger.info(f"Searching Mangadex for: {clean_title}")
|
||||
|
||||
# Mangadex requires separate calls for author/artist usually, but we'll start with basic info
|
||||
params = {
|
||||
"title": clean_title,
|
||||
"limit": 1,
|
||||
"includes[]": ["author", "artist", "cover_art"]
|
||||
}
|
||||
|
||||
response = self.scraper.get(url, params=params, timeout=10)
|
||||
if response.status_code == 200:
|
||||
data = response.json()
|
||||
if data.get('data'):
|
||||
manga = data['data'][0]
|
||||
attr = manga['attributes']
|
||||
|
||||
# Extract authors/artists
|
||||
authors = []
|
||||
artists = []
|
||||
for rel in manga['relationships']:
|
||||
if rel['type'] == 'author': authors.append(rel.get('attributes', {}).get('name'))
|
||||
if rel['type'] == 'artist': artists.append(rel.get('attributes', {}).get('name'))
|
||||
|
||||
# Tags
|
||||
tags = [t['attributes']['name']['en'] for t in attr.get('tags', [])]
|
||||
|
||||
return {
|
||||
"title": attr['title'].get('en') or list(attr['title'].values())[0],
|
||||
"original_title": attr.get('altTitles', [{}])[0].get('en', "") if attr.get('altTitles') else "",
|
||||
"year": attr.get('year'),
|
||||
"url": f"https://mangadex.org/title/{manga['id']}",
|
||||
"id": manga['id'],
|
||||
"tags": tags,
|
||||
"author": authors,
|
||||
"artist": artists,
|
||||
"source": "Mangadex"
|
||||
}
|
||||
except Exception as e:
|
||||
comics_logger.error(f"Mangadex lookup failed: {e}")
|
||||
return None
|
||||
|
||||
def _lookup_hentai2read_metadata(self, title):
|
||||
"""
|
||||
Scrapes Hentai2Read.
|
||||
"""
|
||||
try:
|
||||
base_url = "https://hentai2read.com"
|
||||
clean_title = title.replace('_', '+').strip()
|
||||
search_url = f"{base_url}/search/?cmd={clean_title}"
|
||||
comics_logger.info(f"Searching Hentai2Read for: {clean_title}")
|
||||
|
||||
response = self.scraper.get(search_url)
|
||||
if response.status_code == 200:
|
||||
soup = BeautifulSoup(response.content, 'html.parser')
|
||||
# Hentai2Read search results structure
|
||||
result_link = soup.select_one('.book-grid-item a')
|
||||
|
||||
if result_link:
|
||||
target_url = result_link['href']
|
||||
comics_logger.info(f"Fetching Hentai2Read details: {target_url}")
|
||||
|
||||
resp = self.scraper.get(target_url)
|
||||
if resp.status_code == 200:
|
||||
soup = BeautifulSoup(resp.content, 'html.parser')
|
||||
data = {"source": "Hentai2Read", "url": target_url}
|
||||
|
||||
# Title
|
||||
title_tag = soup.select_one('h3.block-title a')
|
||||
if title_tag: data['title'] = title_tag.text.strip()
|
||||
|
||||
# Info list
|
||||
info_items = soup.select('ul.list-simple-mini li')
|
||||
for item in info_items:
|
||||
text = item.text.strip()
|
||||
if "Author" in text:
|
||||
data['author'] = [a.strip() for a in text.replace("Author", "").strip(" :" ).split(',')]
|
||||
elif "Artist" in text:
|
||||
data['artist'] = [a.strip() for a in text.replace("Artist", "").strip(" :" ).split(',')]
|
||||
elif "Parody" in text:
|
||||
data['parody'] = [a.strip() for a in text.replace("Parody", "").strip(" :" ).split(',')]
|
||||
elif "Storyline" in text or "Content" in text: # Tags
|
||||
data['tags'] = [a.text.strip() for a in item.select('a')]
|
||||
elif "Release" in text:
|
||||
year_match = re.search(r'\\d{4}', text)
|
||||
if year_match: data['year'] = year_match.group(0)
|
||||
|
||||
return data
|
||||
except Exception as e:
|
||||
comics_logger.error(f"Hentai2Read lookup failed: {e}")
|
||||
return None
|
||||
|
||||
def _lookup_mangaupdates_metadata(self, title):
|
||||
"""
|
||||
Queries MangaUpdates API for metadata (good for Manhwa/Webtoons).
|
||||
"""
|
||||
try:
|
||||
url = "https://api.mangaupdates.com/v1/series/search"
|
||||
# Replace underscores with spaces for better search
|
||||
clean_title = title.replace('_', ' ').strip()
|
||||
comics_logger.info(f"Searching MangaUpdates for: {clean_title}")
|
||||
|
||||
response = self.scraper.post(url, json={"search": clean_title}, timeout=10)
|
||||
if response.status_code == 200:
|
||||
data = response.json()
|
||||
if data.get('results'):
|
||||
series = data['results'][0]['record']
|
||||
return {
|
||||
"title": series.get('title'),
|
||||
"year": series.get('year'),
|
||||
"url": series.get('url'),
|
||||
"id": series.get('series_id'),
|
||||
"tags": [g.get('genre') for g in series.get('genres', [])] if series.get('genres') else [],
|
||||
"source": "MangaUpdates"
|
||||
}
|
||||
except Exception as e:
|
||||
comics_logger.error(f"MangaUpdates lookup failed: {e}")
|
||||
return None
|
||||
|
||||
def _process_item(self, samba_manager, item, is_archive=False):
|
||||
item_path = item['path']
|
||||
item_name = item['name']
|
||||
name_part, artist_part, doujin_id = self._get_series_info(item_name)
|
||||
|
||||
api_data = None
|
||||
|
||||
# 1. Try Mangadex (Standard/Manhwa)
|
||||
if not api_data:
|
||||
api_data = self._lookup_mangadex_metadata(name_part)
|
||||
|
||||
# 2. Try nhentai (Doujinshi)
|
||||
if not api_data:
|
||||
# Need to re-add "source": "nhentai" to the existing method logic or wrapper
|
||||
data = self._lookup_doujin_metadata(name_part, artist_part, doujin_id)
|
||||
if data:
|
||||
data['source'] = "nhentai"
|
||||
api_data = data
|
||||
|
||||
# 3. Try Hentai2Read (Fallback Doujin)
|
||||
if not api_data:
|
||||
api_data = self._lookup_hentai2read_metadata(name_part)
|
||||
|
||||
# 4. Try MangaUpdates (Final Fallback)
|
||||
if not api_data:
|
||||
api_data = self._lookup_mangaupdates_metadata(name_part)
|
||||
|
||||
final_series_name = name_part
|
||||
if api_data:
|
||||
if api_data.get('title'):
|
||||
title_clean = re.sub(r'[<>:"/\\|?*]', '', api_data['title']).strip()
|
||||
final_series_name = title_clean.replace(' ', '_')
|
||||
if api_data.get('year'): final_series_name = f"{final_series_name}_({api_data['year']})"
|
||||
elif api_data.get('id'): final_series_name = f"{final_series_name}_({api_data['id']})"
|
||||
|
||||
final_series_name = final_series_name.replace(' ', '_')
|
||||
series_folder_path = f"{self.comics_root}/{final_series_name}"
|
||||
|
||||
try:
|
||||
samba_manager.create_directory(series_folder_path)
|
||||
except OperationFailure: pass
|
||||
|
||||
if item_name == final_series_name: return
|
||||
|
||||
comics_logger.info(f"Processing: {item_name} -> {final_series_name}")
|
||||
with tempfile.TemporaryDirectory() as temp_dir:
|
||||
extraction_path = os.path.join(temp_dir, "extracted")
|
||||
os.makedirs(extraction_path, exist_ok=True)
|
||||
if is_archive:
|
||||
local_archive = os.path.join(temp_dir, item_name)
|
||||
with open(local_archive, 'wb') as f: samba_manager.download_file(item_path, f)
|
||||
try: self._extract_archive(local_archive, extraction_path)
|
||||
except Exception as e:
|
||||
comics_logger.error(f"Failed to extract {item_name}: {e}")
|
||||
return
|
||||
else: self._download_directory(samba_manager, item_path, extraction_path)
|
||||
|
||||
has_images = False
|
||||
for root, _, files in os.walk(extraction_path):
|
||||
if any(f.lower().endswith(('.jpg', '.jpeg', '.png', '.webp')) for f in files):
|
||||
has_images = True
|
||||
break
|
||||
|
||||
if has_images:
|
||||
cbz_name = f"{final_series_name}.cbz"
|
||||
temp_cbz = os.path.join(temp_dir, cbz_name)
|
||||
self._create_cbz(extraction_path, temp_cbz)
|
||||
dest_cbz_path = f"{series_folder_path}/{cbz_name}"
|
||||
|
||||
with open(temp_cbz, 'rb') as f:
|
||||
res = samba_manager.upload_file(dest_cbz_path, f)
|
||||
if isinstance(res, dict) and "error" in res: raise Exception(f"Upload failed: {res['error']}")
|
||||
|
||||
meta_content = self._create_metadata_file(final_series_name, artist=artist_part, api_data=api_data)
|
||||
meta_path = f"{series_folder_path}/{final_series_name}_info.txt"
|
||||
with tempfile.NamedTemporaryFile(mode='w+', delete=False) as tmp_meta:
|
||||
tmp_meta.write(meta_content)
|
||||
tmp_meta.flush()
|
||||
tmp_meta.seek(0)
|
||||
with open(tmp_meta.name, 'rb') as f: samba_manager.upload_file(meta_path, f)
|
||||
os.unlink(tmp_meta.name)
|
||||
|
||||
if is_archive: samba_manager.delete_file(item_path)
|
||||
else: samba_manager.delete_directory_recursive(item_path)
|
||||
else: comics_logger.warning(f"No images found in {item_name}, skipping.")
|
||||
|
||||
def organize_comics(self, samba_manager: SambaManager):
|
||||
comics_logger.info("Starting comics organization process...")
|
||||
comics_root = "/comics/toberead"
|
||||
comics_dest_root = "/comics/manga"
|
||||
comics_logger.info(f"Starting organization in {self.comics_root}...")
|
||||
self._recursive_scan_and_process(samba_manager, self.comics_root)
|
||||
|
||||
def _recursive_scan_and_process(self, samba_manager, current_path):
|
||||
try:
|
||||
items_in_toberead = samba_manager.list_path(comics_root)
|
||||
if "error" in items_in_toberead:
|
||||
raise Exception(f"Failed to list files in 'toberead': {items_in_toberead['error']}")
|
||||
items = samba_manager.list_path(current_path)
|
||||
if isinstance(items, dict) and "error" in items: return
|
||||
for item in items:
|
||||
if item['name'] in ['.', '..']: continue
|
||||
if item['is_directory']:
|
||||
sub_items = samba_manager.list_path(item['path'])
|
||||
if any(sub['name'].lower().endswith(('.jpg', '.jpeg', '.png', '.webp')) for sub in sub_items):
|
||||
self._process_item(samba_manager, item, is_archive=False)
|
||||
else: self._recursive_scan_and_process(samba_manager, item['path'])
|
||||
elif item['name'].lower().endswith(('.zip', '.cbz')):
|
||||
self._process_item(samba_manager, item, is_archive=True)
|
||||
except Exception as e: comics_logger.error(f"Error scanning {current_path}: {e}")
|
||||
|
||||
series_chapters = defaultdict(list)
|
||||
for item in items_in_toberead:
|
||||
if not item["is_directory"]:
|
||||
series_name = self._get_series_name(item["name"])
|
||||
series_chapters[series_name].append(item)
|
||||
def get_pending_comics(self, samba_manager: SambaManager):
|
||||
try:
|
||||
items = samba_manager.list_path(self.comics_root)
|
||||
if isinstance(items, dict) and "error" in items: return []
|
||||
return [{"name": i['name'], "path": i['path']} for i in items if i['is_directory'] and i['name'] not in ['.', '..']]
|
||||
except Exception: return []
|
||||
|
||||
def move_series(self, src_samba: SambaManager, series_names, dest_share=None, dest_path=None):
|
||||
"""Moves folders from isolation/staging to library share."""
|
||||
share = dest_share if dest_share else self.library_share_default
|
||||
path = dest_path if dest_path else self.library_path_default
|
||||
|
||||
# Connect to destination share
|
||||
dest_samba = SambaManager(src_samba.server_ip, share, src_samba.username, src_samba.password)
|
||||
results = {"success": [], "failed": []}
|
||||
|
||||
for series in series_names:
|
||||
src_folder = f"{self.comics_root}/{series}"
|
||||
target_folder = f"{path}/{series}"
|
||||
|
||||
comics_logger.info(f"Found {len(series_chapters)} series to process.")
|
||||
try:
|
||||
# Ensure target directory exists
|
||||
dest_samba.create_directory(target_folder)
|
||||
|
||||
# List files in source (on isolation share)
|
||||
items = src_samba.list_path(src_folder)
|
||||
for item in items:
|
||||
if item['name'] in ['.', '..']: continue
|
||||
|
||||
# Cross-share move: Download -> Upload -> Delete
|
||||
with tempfile.TemporaryDirectory() as temp_dir:
|
||||
local_file = os.path.join(temp_dir, item['name'])
|
||||
with open(local_file, 'wb') as f: src_samba.download_file(item['path'], f)
|
||||
with open(local_file, 'rb') as f: dest_samba.upload_file(f"{target_folder}/{item['name']}", f)
|
||||
src_samba.delete_file(item['path'])
|
||||
|
||||
src_samba.delete_directory(src_folder)
|
||||
results["success"].append(series)
|
||||
comics_logger.info(f"Moved {series} to {share}:{target_folder}")
|
||||
except Exception as e:
|
||||
results["failed"].append({"name": series, "error": str(e)})
|
||||
comics_logger.error(f"Failed to move {series}: {e}")
|
||||
|
||||
dest_samba.close()
|
||||
return results
|
||||
|
||||
for series_name, chapters in series_chapters.items():
|
||||
comics_logger.info(f"Processing series: {series_name}")
|
||||
final_series_folder_path = f"{comics_dest_root}/{series_name}".replace("\\", "/")
|
||||
def update_existing_metadata(self, samba_manager, target_path="/comics/manga", force=False):
|
||||
"""
|
||||
Scans an existing library directory for series folders and updates/creates metadata files.
|
||||
"""
|
||||
comics_logger.info(f"Starting metadata update in {target_path} (Force: {force})...")
|
||||
|
||||
try:
|
||||
items = samba_manager.list_path(target_path)
|
||||
if isinstance(items, dict) and "error" in items:
|
||||
comics_logger.error(f"Error listing {target_path}: {items['error']}")
|
||||
return
|
||||
|
||||
for item in items:
|
||||
if item['name'] in ['.', '..']: continue
|
||||
if not item['is_directory']: continue
|
||||
|
||||
series_name = item['name']
|
||||
series_path = item['path']
|
||||
|
||||
# Check for existing metadata
|
||||
meta_filename = f"{series_name}_info.txt"
|
||||
meta_path = f"{series_path}/{meta_filename}"
|
||||
|
||||
# Check if meta exists
|
||||
series_contents = samba_manager.list_path(series_path)
|
||||
has_meta = False
|
||||
if isinstance(series_contents, list):
|
||||
for sub in series_contents:
|
||||
if sub['name'] == meta_filename:
|
||||
has_meta = True
|
||||
break
|
||||
|
||||
if has_meta and not force:
|
||||
continue
|
||||
|
||||
comics_logger.info(f"Updating metadata for: {series_name}")
|
||||
|
||||
# Parse series info from FOLDER NAME
|
||||
clean_name = series_name.replace('_', ' ')
|
||||
clean_name = re.sub(r'\s*\\(\\d+\\)$', '', clean_name).strip()
|
||||
|
||||
artist = "Unknown"
|
||||
artist_match = re.match(r'^\\[(.*?)\\]', clean_name)
|
||||
if artist_match:
|
||||
artist = artist_match.group(1)
|
||||
clean_name = clean_name[artist_match.end():].strip()
|
||||
|
||||
doujin_id = None
|
||||
id_match = re.search(r'\\(\\d{5,7}\\)$', series_name)
|
||||
if id_match:
|
||||
doujin_id = id_match.group(1)
|
||||
|
||||
# 1. Try Mangadex
|
||||
api_data = self._lookup_mangadex_metadata(clean_name)
|
||||
|
||||
# 2. Try nhentai
|
||||
if not api_data:
|
||||
data = self._lookup_doujin_metadata(clean_name, artist, doujin_id)
|
||||
if data:
|
||||
data['source'] = "nhentai"
|
||||
api_data = data
|
||||
|
||||
# 3. Try Hentai2Read
|
||||
if not api_data:
|
||||
api_data = self._lookup_hentai2read_metadata(clean_name)
|
||||
|
||||
# 4. Try MangaUpdates
|
||||
if not api_data:
|
||||
api_data = self._lookup_mangaupdates_metadata(clean_name)
|
||||
|
||||
meta_content = self._create_metadata_file(clean_name, artist=artist, api_data=api_data)
|
||||
|
||||
with tempfile.NamedTemporaryFile(mode='w+', delete=False) as tmp_meta:
|
||||
tmp_meta.write(meta_content)
|
||||
tmp_meta.flush()
|
||||
tmp_meta.seek(0)
|
||||
with open(tmp_meta.name, 'rb') as f:
|
||||
samba_manager.upload_file(meta_path, f)
|
||||
os.unlink(tmp_meta.name)
|
||||
|
||||
except Exception as e:
|
||||
comics_logger.error(f"Error updating metadata in {target_path}: {e}")
|
||||
|
||||
def _collect_all_folders(self, samba_manager, path):
|
||||
folders = []
|
||||
try:
|
||||
items = samba_manager.list_path(path)
|
||||
for item in items:
|
||||
if item['name'] in ['.', '..']: continue
|
||||
if item['is_directory']:
|
||||
folders.append({'name': item['name'], 'path': item['path']})
|
||||
# Recurse
|
||||
folders.extend(self._collect_all_folders(samba_manager, item['path']))
|
||||
except Exception as e:
|
||||
comics_logger.error(f"Error listing path {path}: {e}")
|
||||
return folders
|
||||
|
||||
def find_similar_folders(self, samba_manager: SambaManager, root_path="/comics/manga", threshold=0.9):
|
||||
"""
|
||||
Scans for folders with similar names.
|
||||
"""
|
||||
comics_logger.info(f"Scanning for duplicate folders in {root_path}...")
|
||||
all_dirs = self._collect_all_folders(samba_manager, root_path)
|
||||
comics_logger.info(f"Found {len(all_dirs)} directories. Comparing...")
|
||||
|
||||
groups = []
|
||||
processed_indices = set()
|
||||
|
||||
for i in range(len(all_dirs)):
|
||||
if i in processed_indices: continue
|
||||
|
||||
current_group = [all_dirs[i]]
|
||||
|
||||
for j in range(i + 1, len(all_dirs)):
|
||||
if j in processed_indices: continue
|
||||
|
||||
name1 = all_dirs[i]['name'].lower().replace('_', ' ')
|
||||
name2 = all_dirs[j]['name'].lower().replace('_', ' ')
|
||||
|
||||
ratio = difflib.SequenceMatcher(None, name1, name2).ratio()
|
||||
|
||||
if ratio >= threshold:
|
||||
current_group.append(all_dirs[j])
|
||||
processed_indices.add(j)
|
||||
|
||||
if len(current_group) > 1:
|
||||
groups.append({
|
||||
"name": all_dirs[i]['name'],
|
||||
"folders": current_group
|
||||
})
|
||||
processed_indices.add(i)
|
||||
|
||||
return groups
|
||||
|
||||
def delete_folder(self, samba_manager: SambaManager, folder_path):
|
||||
"""
|
||||
Deletes a specific folder.
|
||||
"""
|
||||
try:
|
||||
comics_logger.info(f"Deleting duplicate folder: {folder_path}")
|
||||
samba_manager.delete_directory_recursive(folder_path)
|
||||
return {"success": True}
|
||||
except Exception as e:
|
||||
comics_logger.error(f"Failed to delete folder {folder_path}: {e}")
|
||||
return {"error": str(e)}
|
||||
|
||||
def _get_artist_from_info(self, samba_manager, series_path, series_name):
|
||||
info_filename = f"{series_name}_info.txt"
|
||||
info_path = f"{series_path}/{info_filename}"
|
||||
|
||||
try:
|
||||
with tempfile.NamedTemporaryFile(mode='w+b', delete=False) as tmp:
|
||||
samba_manager.download_file(info_path, tmp)
|
||||
tmp.seek(0)
|
||||
content = tmp.read().decode('utf-8', errors='ignore')
|
||||
|
||||
# Parse content
|
||||
artist = "Unknown"
|
||||
author = "Unknown"
|
||||
|
||||
for line in content.splitlines():
|
||||
if line.startswith("Artist:"):
|
||||
val = line.split(":", 1)[1].strip()
|
||||
if val and val.lower() != "unknown":
|
||||
artist = val.split(',')[0].strip() # Take first artist if multiple
|
||||
elif line.startswith("Author:"):
|
||||
val = line.split(":", 1)[1].strip()
|
||||
if val and val.lower() != "unknown":
|
||||
author = val.split(',')[0].strip()
|
||||
|
||||
if artist != "Unknown": return artist
|
||||
if author != "Unknown": return author
|
||||
|
||||
except Exception as e:
|
||||
# File might not exist or other error
|
||||
pass
|
||||
finally:
|
||||
if 'tmp' in locals() and os.path.exists(tmp.name):
|
||||
os.unlink(tmp.name)
|
||||
|
||||
return "_Unknown"
|
||||
|
||||
def sort_by_artist(self, samba_manager, root_path="/comics/manga"):
|
||||
comics_logger.info(f"Sorting by Artist in {root_path}...")
|
||||
|
||||
try:
|
||||
items = samba_manager.list_path(root_path)
|
||||
if isinstance(items, dict) and "error" in items:
|
||||
comics_logger.error(f"Error listing {root_path}: {items['error']}")
|
||||
return
|
||||
|
||||
for item in items:
|
||||
if item['name'] in ['.', '..', '_Unknown']: continue
|
||||
if not item['is_directory']: continue
|
||||
|
||||
series_name = item['name']
|
||||
series_path = item['path']
|
||||
|
||||
# Check if this is a Series Folder
|
||||
# Criteria: Contains .cbz, .zip, or _info.txt
|
||||
try:
|
||||
sub_items = samba_manager.list_path(series_path)
|
||||
if isinstance(sub_items, dict) and "error" in sub_items: continue
|
||||
|
||||
is_series = False
|
||||
for sub in sub_items:
|
||||
if sub['name'].lower().endswith(('.cbz', '.zip', '_info.txt')):
|
||||
is_series = True
|
||||
break
|
||||
|
||||
if not is_series:
|
||||
comics_logger.info(f"Skipping potential Artist folder or empty folder: {series_name}")
|
||||
continue
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
# Attempt to get artist
|
||||
artist = self._get_artist_from_info(samba_manager, series_path, series_name)
|
||||
|
||||
# Sanitize artist name for folder
|
||||
clean_artist = re.sub(r'[<>:"/\\|?*]', '', artist).strip().replace(' ', '_')
|
||||
if not clean_artist: clean_artist = "_Unknown"
|
||||
|
||||
# Target path: /comics/manga/Artist/Series
|
||||
artist_folder = f"{root_path}/{clean_artist}"
|
||||
target_path = f"{artist_folder}/{series_name}"
|
||||
|
||||
# Skip if already in place
|
||||
if series_name == clean_artist:
|
||||
continue
|
||||
|
||||
# Check if we are moving into itself (e.g. Root/Artist -> Root/Artist/Artist)
|
||||
# This happens if 'SeriesName' == 'ArtistName' and it was already sorted?
|
||||
# But we checked is_series. An Artist folder usually doesn't have cbz inside directly.
|
||||
|
||||
comics_logger.info(f"Moving '{series_name}' to Artist folder '{clean_artist}'")
|
||||
|
||||
try:
|
||||
samba_manager.create_directory(final_series_folder_path)
|
||||
except OperationFailure as e:
|
||||
if "FILE_OBJECT_NAME_COLLISION" not in str(e):
|
||||
comics_logger.error(f"Could not create directory {final_series_folder_path}: {e}")
|
||||
|
||||
for chapter_item in chapters:
|
||||
original_path = chapter_item["path"].replace("\\", "/")
|
||||
original_filename = chapter_item["name"]
|
||||
|
||||
path_to_move = ""
|
||||
filename_to_move = ""
|
||||
|
||||
if original_filename.lower().endswith('.zip'):
|
||||
cbz_filename = os.path.splitext(original_filename)[0] + '.cbz'
|
||||
temp_cbz_path = f"{comics_root}/{cbz_filename}".replace("\\", "/")
|
||||
|
||||
try:
|
||||
samba_manager.rename_file(original_path, temp_cbz_path)
|
||||
path_to_move = temp_cbz_path
|
||||
filename_to_move = cbz_filename
|
||||
comics_logger.info(f"Renamed {original_filename} to {cbz_filename}")
|
||||
except OperationFailure as e:
|
||||
comics_logger.error(f"Failed to rename ZIP {original_filename}: {e}")
|
||||
continue
|
||||
|
||||
elif original_filename.lower().endswith('.cbz'):
|
||||
path_to_move = original_path
|
||||
filename_to_move = original_filename
|
||||
else:
|
||||
continue
|
||||
|
||||
final_cbz_path = f"{final_series_folder_path}/{filename_to_move}".replace("\\", "/")
|
||||
|
||||
# Create Artist folder
|
||||
try:
|
||||
samba_manager.rename_file(path_to_move, final_cbz_path)
|
||||
comics_logger.info(f"Moved {filename_to_move} to {final_series_folder_path}")
|
||||
except OperationFailure as e:
|
||||
comics_logger.error(f"Failed to move {filename_to_move} to {final_series_folder_path}: {e}")
|
||||
samba_manager.create_directory(artist_folder)
|
||||
except OperationFailure: pass # Exists
|
||||
|
||||
# Move Series folder
|
||||
samba_manager.rename_file(series_path, target_path)
|
||||
|
||||
except Exception as e:
|
||||
comics_logger.error(f"Failed to move {series_name}: {e}")
|
||||
|
||||
except Exception as e:
|
||||
comics_logger.error(f"An error occurred during comics organization: {e}", exc_info=True)
|
||||
|
||||
comics_logger.info("Comics organization process finished.")
|
||||
|
||||
return
|
||||
comics_logger.error(f"Sort by artist failed: {e}")
|
||||
Reference in New Issue
Block a user