import os import re import shutil import zipfile import tempfile import logging import cloudscraper import difflib from bs4 import BeautifulSoup from collections import defaultdict from .samba_manager import SambaManager from .config import settings from smb.smb_structs import OperationFailure # --- Dedicated logger for ComicsManager --- comics_logger = logging.getLogger('comics_manager') comics_logger.setLevel(logging.INFO) comics_logger.propagate = False if not comics_logger.handlers: os.makedirs("logs", exist_ok=True) comics_log_handler = logging.FileHandler("logs/comics_organization.log", mode='a') comics_log_handler.setFormatter(logging.Formatter('%(asctime)s - %(levelname)s - %(message)s')) comics_logger.addHandler(comics_log_handler) comics_logger.addHandler(logging.StreamHandler()) class ComicsManager: def __init__(self): # Staging settings (on 'isolation' share) self.comics_root = "/comics/pendingorganization" # Library defaults self.library_share_default = "isolation" self.library_path_default = "/comics/manga" self.scraper = cloudscraper.create_scraper() def _get_series_info(self, filename): base_name, _ = os.path.splitext(filename) artist = "Unknown" doujin_id = None id_match = re.search(r'[[\\](](\\d{5,7})[[\\])]', base_name) if id_match: doujin_id = id_match.group(1) tags = re.findall(r'[[\\](](.*?)[[\\])]', base_name) tags = [t for t in tags if t != doujin_id] if tags: artist = tags[0].strip() name = re.sub(r'[[\\](].*?[[\\])]', '', base_name).strip() pattern = re.compile(r'[-_\\s]*(v(ol)?\\. ?|c(h)?\\. ?|chapter|issue|ep(isode)?)\\s*\\d+.*$', re.IGNORECASE) match = pattern.search(name) if match: name = name[:match.start()] else: name = re.sub(r'[-_\\s]+\\d+\\s*$', '', name) name = name.strip(' -_') if not name and tags: name = tags[0] elif not name: name = base_name.strip() name = name.replace(' ', '_') artist = artist.replace(' ', '_') return name, artist, doujin_id def _fetch_metadata_from_url(self, target_url): try: comics_logger.info(f"Fetching metadata from: {target_url}") response = self.scraper.get(target_url) if response.status_code == 200: soup = BeautifulSoup(response.content, 'html.parser') data = {} title_info = soup.select_one('#info') if title_info: data['title'] = title_info.select_one('h1.title').text.strip() if title_info.select_one('h1.title') else "" data['original_title'] = title_info.select_one('h2.title').text.strip() if title_info.select_one('h2.title') else "" tag_containers = soup.select('.tag-container') for container in tag_containers: label = container.text.split(':')[0].strip().lower() if ':' in container.text else "" tags = [t.select_one('.name').text for t in container.select('.tag')] if 'tags' in label: data['tags'] = tags elif 'artists' in label: data['artist'] = tags elif 'groups' in label: data['circle'] = tags elif 'parodies' in label: data['parody'] = tags time_tag = soup.select_one('#info time') if time_tag and time_tag.get('datetime'): year_match = re.search(r'(\\d{4})', time_tag['datetime']) if year_match: data['year'] = year_match.group(1) data['url'] = target_url data['id'] = target_url.split('/g/')[1].strip('/') return data except Exception as e: comics_logger.error(f"Error parsing metadata: {e}") return None def _lookup_doujin_metadata(self, title, artist="Unknown", doujin_id=None): base_url = "https://nhentai.net" if doujin_id: return self._fetch_metadata_from_url(f"{base_url}/g/{doujin_id}/") clean_title = title.replace('_', ' ').strip() clean_artist = artist.replace('_', ' ').strip() if artist != "Unknown" else "" queries = [] if clean_artist: queries.append(f"{clean_artist} {clean_title}") queries.append(clean_title) for query in queries: try: search_url = f"{base_url}/search/?q={query.replace(' ', '+')}" comics_logger.info(f"Searching metadata for: {query}") response = self.scraper.get(search_url) if response.status_code == 200: soup = BeautifulSoup(response.content, 'html.parser') results = soup.select('.gallery a.cover') if results: target_url = f"{base_url}{results[0]['href']}" data = self._fetch_metadata_from_url(target_url) if data: return data except Exception as e: comics_logger.error(f"Search failed for query '{query}': {e}") return None def _download_directory(self, samba_manager, remote_path, local_path): os.makedirs(local_path, exist_ok=True) items = samba_manager.list_path(remote_path) if isinstance(items, dict) and "error" in items: raise Exception(items["error"]) for item in items: if item['name'] in ['.', '..']: continue local_item_path = os.path.join(local_path, item['name']) if item['is_directory']: self._download_directory(samba_manager, item['path'], local_item_path) else: with open(local_item_path, 'wb') as f: samba_manager.download_file(item['path'], f) def _extract_archive(self, archive_path, extract_path): if zipfile.is_zipfile(archive_path): with zipfile.ZipFile(archive_path, 'r') as zf: zf.extractall(extract_path) else: raise Exception("Unsupported archive format") def _create_cbz(self, source_folder, output_path): with zipfile.ZipFile(output_path, 'w', zipfile.ZIP_DEFLATED) as zf: for root, dirs, files in os.walk(source_folder): for file in files: file_path = os.path.join(root, file) arcname = os.path.relpath(file_path, source_folder) zf.write(file_path, arcname) def _create_metadata_file(self, series_name, author="Unknown", artist="Unknown", api_data=None): content = f"Title: {series_name}\n" if api_data: if api_data.get('title'): content = f"Title: {api_data['title']}\n" if api_data.get('original_title'): content += f"Original Title: {api_data['original_title']}\n" artists = ", ".join(api_data.get('artist', [])) or artist circles = ", ".join(api_data.get('circle', [])) or "Unknown" parodies = ", ".join(api_data.get('parody', [])) or "Original" tags = ", ".join(api_data.get('tags', [])) content += f"Artist: {artists}\nCircle: {circles}\nParody: {parodies}\nTags: {tags}\nURL: {api_data.get('url', 'N/A')}\n" else: content += f"Artist: {artist}\nAuthor: {author}\n" content += f"Processed by: ServerManagerWebApp\n" return content def _lookup_mangadex_metadata(self, title): """ Queries Mangadex API. """ try: url = "https://api.mangadex.org/manga" clean_title = title.replace('_', ' ').strip() comics_logger.info(f"Searching Mangadex for: {clean_title}") # Mangadex requires separate calls for author/artist usually, but we'll start with basic info params = { "title": clean_title, "limit": 1, "includes[]": ["author", "artist", "cover_art"] } response = self.scraper.get(url, params=params, timeout=10) if response.status_code == 200: data = response.json() if data.get('data'): manga = data['data'][0] attr = manga['attributes'] # Extract authors/artists authors = [] artists = [] for rel in manga['relationships']: if rel['type'] == 'author': authors.append(rel.get('attributes', {}).get('name')) if rel['type'] == 'artist': artists.append(rel.get('attributes', {}).get('name')) # Tags tags = [t['attributes']['name']['en'] for t in attr.get('tags', [])] return { "title": attr['title'].get('en') or list(attr['title'].values())[0], "original_title": attr.get('altTitles', [{}])[0].get('en', "") if attr.get('altTitles') else "", "year": attr.get('year'), "url": f"https://mangadex.org/title/{manga['id']}", "id": manga['id'], "tags": tags, "author": authors, "artist": artists, "source": "Mangadex" } except Exception as e: comics_logger.error(f"Mangadex lookup failed: {e}") return None def _lookup_hentai2read_metadata(self, title): """ Scrapes Hentai2Read. """ try: base_url = "https://hentai2read.com" clean_title = title.replace('_', '+').strip() search_url = f"{base_url}/search/?cmd={clean_title}" comics_logger.info(f"Searching Hentai2Read for: {clean_title}") response = self.scraper.get(search_url) if response.status_code == 200: soup = BeautifulSoup(response.content, 'html.parser') # Hentai2Read search results structure result_link = soup.select_one('.book-grid-item a') if result_link: target_url = result_link['href'] comics_logger.info(f"Fetching Hentai2Read details: {target_url}") resp = self.scraper.get(target_url) if resp.status_code == 200: soup = BeautifulSoup(resp.content, 'html.parser') data = {"source": "Hentai2Read", "url": target_url} # Title title_tag = soup.select_one('h3.block-title a') if title_tag: data['title'] = title_tag.text.strip() # Info list info_items = soup.select('ul.list-simple-mini li') for item in info_items: text = item.text.strip() if "Author" in text: data['author'] = [a.strip() for a in text.replace("Author", "").strip(" :" ).split(',')] elif "Artist" in text: data['artist'] = [a.strip() for a in text.replace("Artist", "").strip(" :" ).split(',')] elif "Parody" in text: data['parody'] = [a.strip() for a in text.replace("Parody", "").strip(" :" ).split(',')] elif "Storyline" in text or "Content" in text: # Tags data['tags'] = [a.text.strip() for a in item.select('a')] elif "Release" in text: year_match = re.search(r'\\d{4}', text) if year_match: data['year'] = year_match.group(0) return data except Exception as e: comics_logger.error(f"Hentai2Read lookup failed: {e}") return None def _lookup_mangaupdates_metadata(self, title): """ Queries MangaUpdates API for metadata (good for Manhwa/Webtoons). """ try: url = "https://api.mangaupdates.com/v1/series/search" # Replace underscores with spaces for better search clean_title = title.replace('_', ' ').strip() comics_logger.info(f"Searching MangaUpdates for: {clean_title}") response = self.scraper.post(url, json={"search": clean_title}, timeout=10) if response.status_code == 200: data = response.json() if data.get('results'): series = data['results'][0]['record'] return { "title": series.get('title'), "year": series.get('year'), "url": series.get('url'), "id": series.get('series_id'), "tags": [g.get('genre') for g in series.get('genres', [])] if series.get('genres') else [], "source": "MangaUpdates" } except Exception as e: comics_logger.error(f"MangaUpdates lookup failed: {e}") return None def _process_item(self, samba_manager, item, is_archive=False): item_path = item['path'] item_name = item['name'] name_part, artist_part, doujin_id = self._get_series_info(item_name) api_data = None # 1. Try Mangadex (Standard/Manhwa) if not api_data: api_data = self._lookup_mangadex_metadata(name_part) # 2. Try nhentai (Doujinshi) if not api_data: # Need to re-add "source": "nhentai" to the existing method logic or wrapper data = self._lookup_doujin_metadata(name_part, artist_part, doujin_id) if data: data['source'] = "nhentai" api_data = data # 3. Try Hentai2Read (Fallback Doujin) if not api_data: api_data = self._lookup_hentai2read_metadata(name_part) # 4. Try MangaUpdates (Final Fallback) if not api_data: api_data = self._lookup_mangaupdates_metadata(name_part) final_series_name = name_part if api_data: if api_data.get('title'): title_clean = re.sub(r'[<>:"/\\|?*]', '', api_data['title']).strip() final_series_name = title_clean.replace(' ', '_') if api_data.get('year'): final_series_name = f"{final_series_name}_({api_data['year']})" elif api_data.get('id'): final_series_name = f"{final_series_name}_({api_data['id']})" final_series_name = final_series_name.replace(' ', '_') series_folder_path = f"{self.comics_root}/{final_series_name}" try: samba_manager.create_directory(series_folder_path) except OperationFailure: pass if item_name == final_series_name: return comics_logger.info(f"Processing: {item_name} -> {final_series_name}") with tempfile.TemporaryDirectory() as temp_dir: extraction_path = os.path.join(temp_dir, "extracted") os.makedirs(extraction_path, exist_ok=True) if is_archive: local_archive = os.path.join(temp_dir, item_name) with open(local_archive, 'wb') as f: samba_manager.download_file(item_path, f) try: self._extract_archive(local_archive, extraction_path) except Exception as e: comics_logger.error(f"Failed to extract {item_name}: {e}") return else: self._download_directory(samba_manager, item_path, extraction_path) has_images = False for root, _, files in os.walk(extraction_path): if any(f.lower().endswith(('.jpg', '.jpeg', '.png', '.webp')) for f in files): has_images = True break if has_images: cbz_name = f"{final_series_name}.cbz" temp_cbz = os.path.join(temp_dir, cbz_name) self._create_cbz(extraction_path, temp_cbz) dest_cbz_path = f"{series_folder_path}/{cbz_name}" with open(temp_cbz, 'rb') as f: res = samba_manager.upload_file(dest_cbz_path, f) if isinstance(res, dict) and "error" in res: raise Exception(f"Upload failed: {res['error']}") meta_content = self._create_metadata_file(final_series_name, artist=artist_part, api_data=api_data) meta_path = f"{series_folder_path}/{final_series_name}_info.txt" with tempfile.NamedTemporaryFile(mode='w+', delete=False) as tmp_meta: tmp_meta.write(meta_content) tmp_meta.flush() tmp_meta.seek(0) with open(tmp_meta.name, 'rb') as f: samba_manager.upload_file(meta_path, f) os.unlink(tmp_meta.name) if is_archive: samba_manager.delete_file(item_path) else: samba_manager.delete_directory_recursive(item_path) else: comics_logger.warning(f"No images found in {item_name}, skipping.") def organize_comics(self, samba_manager: SambaManager): comics_logger.info(f"Starting organization in {self.comics_root}...") self._recursive_scan_and_process(samba_manager, self.comics_root) def _recursive_scan_and_process(self, samba_manager, current_path): try: items = samba_manager.list_path(current_path) if isinstance(items, dict) and "error" in items: return for item in items: if item['name'] in ['.', '..']: continue if item['is_directory']: sub_items = samba_manager.list_path(item['path']) if any(sub['name'].lower().endswith(('.jpg', '.jpeg', '.png', '.webp')) for sub in sub_items): self._process_item(samba_manager, item, is_archive=False) else: self._recursive_scan_and_process(samba_manager, item['path']) elif item['name'].lower().endswith(('.zip', '.cbz')): self._process_item(samba_manager, item, is_archive=True) except Exception as e: comics_logger.error(f"Error scanning {current_path}: {e}") def get_pending_comics(self, samba_manager: SambaManager): try: items = samba_manager.list_path(self.comics_root) if isinstance(items, dict) and "error" in items: return [] return [{"name": i['name'], "path": i['path']} for i in items if i['is_directory'] and i['name'] not in ['.', '..']] except Exception: return [] def move_series(self, src_samba: SambaManager, series_names, dest_share=None, dest_path=None): """Moves folders from isolation/staging to library share.""" share = dest_share if dest_share else self.library_share_default path = dest_path if dest_path else self.library_path_default # Connect to destination share dest_samba = SambaManager(src_samba.server_ip, share, src_samba.username, src_samba.password) results = {"success": [], "failed": []} for series in series_names: src_folder = f"{self.comics_root}/{series}" target_folder = f"{path}/{series}" try: # Ensure target directory exists dest_samba.create_directory(target_folder) # List files in source (on isolation share) items = src_samba.list_path(src_folder) for item in items: if item['name'] in ['.', '..']: continue # Cross-share move: Download -> Upload -> Delete with tempfile.TemporaryDirectory() as temp_dir: local_file = os.path.join(temp_dir, item['name']) with open(local_file, 'wb') as f: src_samba.download_file(item['path'], f) with open(local_file, 'rb') as f: dest_samba.upload_file(f"{target_folder}/{item['name']}", f) src_samba.delete_file(item['path']) src_samba.delete_directory(src_folder) results["success"].append(series) comics_logger.info(f"Moved {series} to {share}:{target_folder}") except Exception as e: results["failed"].append({"name": series, "error": str(e)}) comics_logger.error(f"Failed to move {series}: {e}") dest_samba.close() return results def update_existing_metadata(self, samba_manager, target_path="/comics/manga", force=False): """ Scans an existing library directory for series folders and updates/creates metadata files. """ comics_logger.info(f"Starting metadata update in {target_path} (Force: {force})...") try: items = samba_manager.list_path(target_path) if isinstance(items, dict) and "error" in items: comics_logger.error(f"Error listing {target_path}: {items['error']}") return for item in items: if item['name'] in ['.', '..']: continue if not item['is_directory']: continue series_name = item['name'] series_path = item['path'] # Check for existing metadata meta_filename = f"{series_name}_info.txt" meta_path = f"{series_path}/{meta_filename}" # Check if meta exists series_contents = samba_manager.list_path(series_path) has_meta = False if isinstance(series_contents, list): for sub in series_contents: if sub['name'] == meta_filename: has_meta = True break if has_meta and not force: continue comics_logger.info(f"Updating metadata for: {series_name}") # Parse series info from FOLDER NAME clean_name = series_name.replace('_', ' ') clean_name = re.sub(r'\s*\\(\\d+\\)$', '', clean_name).strip() artist = "Unknown" artist_match = re.match(r'^\\[(.*?)\\]', clean_name) if artist_match: artist = artist_match.group(1) clean_name = clean_name[artist_match.end():].strip() doujin_id = None id_match = re.search(r'\\(\\d{5,7}\\)$', series_name) if id_match: doujin_id = id_match.group(1) # 1. Try Mangadex api_data = self._lookup_mangadex_metadata(clean_name) # 2. Try nhentai if not api_data: data = self._lookup_doujin_metadata(clean_name, artist, doujin_id) if data: data['source'] = "nhentai" api_data = data # 3. Try Hentai2Read if not api_data: api_data = self._lookup_hentai2read_metadata(clean_name) # 4. Try MangaUpdates if not api_data: api_data = self._lookup_mangaupdates_metadata(clean_name) meta_content = self._create_metadata_file(clean_name, artist=artist, api_data=api_data) with tempfile.NamedTemporaryFile(mode='w+', delete=False) as tmp_meta: tmp_meta.write(meta_content) tmp_meta.flush() tmp_meta.seek(0) with open(tmp_meta.name, 'rb') as f: samba_manager.upload_file(meta_path, f) os.unlink(tmp_meta.name) except Exception as e: comics_logger.error(f"Error updating metadata in {target_path}: {e}") def _collect_all_folders(self, samba_manager, path): folders = [] try: items = samba_manager.list_path(path) for item in items: if item['name'] in ['.', '..']: continue if item['is_directory']: folders.append({'name': item['name'], 'path': item['path']}) # Recurse folders.extend(self._collect_all_folders(samba_manager, item['path'])) except Exception as e: comics_logger.error(f"Error listing path {path}: {e}") return folders def find_similar_folders(self, samba_manager: SambaManager, root_path="/comics/manga", threshold=0.9): """ Scans for folders with similar names. """ comics_logger.info(f"Scanning for duplicate folders in {root_path}...") all_dirs = self._collect_all_folders(samba_manager, root_path) comics_logger.info(f"Found {len(all_dirs)} directories. Comparing...") groups = [] processed_indices = set() for i in range(len(all_dirs)): if i in processed_indices: continue current_group = [all_dirs[i]] for j in range(i + 1, len(all_dirs)): if j in processed_indices: continue name1 = all_dirs[i]['name'].lower().replace('_', ' ') name2 = all_dirs[j]['name'].lower().replace('_', ' ') ratio = difflib.SequenceMatcher(None, name1, name2).ratio() if ratio >= threshold: current_group.append(all_dirs[j]) processed_indices.add(j) if len(current_group) > 1: groups.append({ "name": all_dirs[i]['name'], "folders": current_group }) processed_indices.add(i) return groups def delete_folder(self, samba_manager: SambaManager, folder_path): """ Deletes a specific folder. """ try: comics_logger.info(f"Deleting duplicate folder: {folder_path}") samba_manager.delete_directory_recursive(folder_path) return {"success": True} except Exception as e: comics_logger.error(f"Failed to delete folder {folder_path}: {e}") return {"error": str(e)} def _get_artist_from_info(self, samba_manager, series_path, series_name): info_filename = f"{series_name}_info.txt" info_path = f"{series_path}/{info_filename}" try: with tempfile.NamedTemporaryFile(mode='w+b', delete=False) as tmp: samba_manager.download_file(info_path, tmp) tmp.seek(0) content = tmp.read().decode('utf-8', errors='ignore') # Parse content artist = "Unknown" author = "Unknown" for line in content.splitlines(): if line.startswith("Artist:"): val = line.split(":", 1)[1].strip() if val and val.lower() != "unknown": artist = val.split(',')[0].strip() # Take first artist if multiple elif line.startswith("Author:"): val = line.split(":", 1)[1].strip() if val and val.lower() != "unknown": author = val.split(',')[0].strip() if artist != "Unknown": return artist if author != "Unknown": return author except Exception as e: # File might not exist or other error pass finally: if 'tmp' in locals() and os.path.exists(tmp.name): os.unlink(tmp.name) return "_Unknown" def sort_by_artist(self, samba_manager, root_path="/comics/manga"): comics_logger.info(f"Sorting by Artist in {root_path}...") try: items = samba_manager.list_path(root_path) if isinstance(items, dict) and "error" in items: comics_logger.error(f"Error listing {root_path}: {items['error']}") return for item in items: if item['name'] in ['.', '..', '_Unknown']: continue if not item['is_directory']: continue series_name = item['name'] series_path = item['path'] # Check if this is a Series Folder # Criteria: Contains .cbz, .zip, or _info.txt try: sub_items = samba_manager.list_path(series_path) if isinstance(sub_items, dict) and "error" in sub_items: continue is_series = False for sub in sub_items: if sub['name'].lower().endswith(('.cbz', '.zip', '_info.txt')): is_series = True break if not is_series: comics_logger.info(f"Skipping potential Artist folder or empty folder: {series_name}") continue except Exception: continue # Attempt to get artist artist = self._get_artist_from_info(samba_manager, series_path, series_name) # Sanitize artist name for folder clean_artist = re.sub(r'[<>:"/\\|?*]', '', artist).strip().replace(' ', '_') if not clean_artist: clean_artist = "_Unknown" # Target path: /comics/manga/Artist/Series artist_folder = f"{root_path}/{clean_artist}" target_path = f"{artist_folder}/{series_name}" # Skip if already in place if series_name == clean_artist: continue # Check if we are moving into itself (e.g. Root/Artist -> Root/Artist/Artist) # This happens if 'SeriesName' == 'ArtistName' and it was already sorted? # But we checked is_series. An Artist folder usually doesn't have cbz inside directly. comics_logger.info(f"Moving '{series_name}' to Artist folder '{clean_artist}'") try: # Create Artist folder try: samba_manager.create_directory(artist_folder) except OperationFailure: pass # Exists # Move Series folder samba_manager.rename_file(series_path, target_path) except Exception as e: comics_logger.error(f"Failed to move {series_name}: {e}") except Exception as e: comics_logger.error(f"Sort by artist failed: {e}")