major server manager overhaul

This commit is contained in:
2026-01-09 19:24:41 +00:00
parent 6a667d1019
commit c0ac650625
202 changed files with 325323 additions and 6929 deletions
+662 -92
View File
@@ -1,9 +1,15 @@
import os
import re
from collections import defaultdict
import shutil
import zipfile
import tempfile
import logging
import cloudscraper
import difflib
from bs4 import BeautifulSoup
from collections import defaultdict
from .samba_manager import SambaManager
from .config import Settings
from .config import settings
from smb.smb_structs import OperationFailure
# --- Dedicated logger for ComicsManager ---
@@ -11,117 +17,681 @@ comics_logger = logging.getLogger('comics_manager')
comics_logger.setLevel(logging.INFO)
comics_logger.propagate = False
if not comics_logger.handlers:
os.makedirs("logs", exist_ok=True)
comics_log_handler = logging.FileHandler("logs/comics_organization.log", mode='a')
comics_log_handler.setFormatter(logging.Formatter('%(asctime)s - %(levelname)s - %(message)s'))
comics_logger.addHandler(comics_log_handler)
comics_logger.addHandler(logging.StreamHandler())
class ComicsManager:
def __init__(self):
# Staging settings (on 'isolation' share)
self.comics_root = "/comics/pendingorganization"
# Library defaults
self.library_share_default = "isolation"
self.library_path_default = "/comics/manga"
self.scraper = cloudscraper.create_scraper()
@staticmethod
def _get_series_name(filename):
def _get_series_info(self, filename):
base_name, _ = os.path.splitext(filename)
pattern = re.compile(r'[-_\s]*(v(ol)?|c(h)?|chapter|issue|ep|episode)[-_\s]*\d+.*|[-_\s]+\d+$', re.IGNORECASE)
match = pattern.search(base_name)
artist = "Unknown"
doujin_id = None
id_match = re.search(r'[[\\](](\\d{5,7})[[\\])]', base_name)
if id_match:
doujin_id = id_match.group(1)
tags = re.findall(r'[[\\](](.*?)[[\\])]', base_name)
tags = [t for t in tags if t != doujin_id]
if tags:
artist = tags[0].strip()
name = re.sub(r'[[\\](].*?[[\\])]', '', base_name).strip()
pattern = re.compile(r'[-_\\s]*(v(ol)?\\. ?|c(h)?\\. ?|chapter|issue|ep(isode)?)\\s*\\d+.*$', re.IGNORECASE)
match = pattern.search(name)
if match:
return base_name[:match.start()].strip('-_ ')
return base_name.strip('-_ ')
name = name[:match.start()]
else:
name = re.sub(r'[-_\\s]+\\d+\\s*$', '', name)
def cleanup_toberead(self, samba_manager: SambaManager):
comics_logger.info("Starting cleanup of 'toberead' directory...")
comics_root = "/comics/toberead"
try:
items_in_toberead = samba_manager.list_path(comics_root)
if "error" in items_in_toberead:
raise Exception(f"Failed to list files in 'toberead': {items_in_toberead['error']}")
for item in items_in_toberead:
item_path = item["path"]
item_name = item["name"]
if item["is_directory"]:
comics_logger.info(f"Deleting directory '{item_name}' and all its contents.")
try:
samba_manager.delete_directory_recursive(item_path)
comics_logger.info(f"Successfully deleted directory: {item_name}")
except OperationFailure as e:
comics_logger.error(f"Failed to delete directory {item_name}: {e}")
else:
comics_logger.info(f"Deleting file '{item_name}'.")
try:
samba_manager.delete_file(item_path)
comics_logger.info(f"Successfully deleted file: {item_name}")
except OperationFailure as e:
comics_logger.error(f"Failed to delete file {item_name}: {e}")
name = name.strip(' -_')
if not name and tags:
name = tags[0]
elif not name:
name = base_name.strip()
comics_logger.info("Cleanup of 'toberead' finished.")
name = name.replace(' ', '_')
artist = artist.replace(' ', '_')
return name, artist, doujin_id
def _fetch_metadata_from_url(self, target_url):
try:
comics_logger.info(f"Fetching metadata from: {target_url}")
response = self.scraper.get(target_url)
if response.status_code == 200:
soup = BeautifulSoup(response.content, 'html.parser')
data = {}
title_info = soup.select_one('#info')
if title_info:
data['title'] = title_info.select_one('h1.title').text.strip() if title_info.select_one('h1.title') else ""
data['original_title'] = title_info.select_one('h2.title').text.strip() if title_info.select_one('h2.title') else ""
tag_containers = soup.select('.tag-container')
for container in tag_containers:
label = container.text.split(':')[0].strip().lower() if ':' in container.text else ""
tags = [t.select_one('.name').text for t in container.select('.tag')]
if 'tags' in label: data['tags'] = tags
elif 'artists' in label: data['artist'] = tags
elif 'groups' in label: data['circle'] = tags
elif 'parodies' in label: data['parody'] = tags
time_tag = soup.select_one('#info time')
if time_tag and time_tag.get('datetime'):
year_match = re.search(r'(\\d{4})', time_tag['datetime'])
if year_match: data['year'] = year_match.group(1)
data['url'] = target_url
data['id'] = target_url.split('/g/')[1].strip('/')
return data
except Exception as e:
comics_logger.error(f"An error occurred during 'toberead' cleanup: {e}", exc_info=True)
comics_logger.error(f"Error parsing metadata: {e}")
return None
def _lookup_doujin_metadata(self, title, artist="Unknown", doujin_id=None):
base_url = "https://nhentai.net"
if doujin_id:
return self._fetch_metadata_from_url(f"{base_url}/g/{doujin_id}/")
clean_title = title.replace('_', ' ').strip()
clean_artist = artist.replace('_', ' ').strip() if artist != "Unknown" else ""
queries = []
if clean_artist: queries.append(f"{clean_artist} {clean_title}")
queries.append(clean_title)
for query in queries:
try:
search_url = f"{base_url}/search/?q={query.replace(' ', '+')}"
comics_logger.info(f"Searching metadata for: {query}")
response = self.scraper.get(search_url)
if response.status_code == 200:
soup = BeautifulSoup(response.content, 'html.parser')
results = soup.select('.gallery a.cover')
if results:
target_url = f"{base_url}{results[0]['href']}"
data = self._fetch_metadata_from_url(target_url)
if data: return data
except Exception as e:
comics_logger.error(f"Search failed for query '{query}': {e}")
return None
def _download_directory(self, samba_manager, remote_path, local_path):
os.makedirs(local_path, exist_ok=True)
items = samba_manager.list_path(remote_path)
if isinstance(items, dict) and "error" in items: raise Exception(items["error"])
for item in items:
if item['name'] in ['.', '..']: continue
local_item_path = os.path.join(local_path, item['name'])
if item['is_directory']: self._download_directory(samba_manager, item['path'], local_item_path)
else:
with open(local_item_path, 'wb') as f:
samba_manager.download_file(item['path'], f)
def _extract_archive(self, archive_path, extract_path):
if zipfile.is_zipfile(archive_path):
with zipfile.ZipFile(archive_path, 'r') as zf:
zf.extractall(extract_path)
else: raise Exception("Unsupported archive format")
def _create_cbz(self, source_folder, output_path):
with zipfile.ZipFile(output_path, 'w', zipfile.ZIP_DEFLATED) as zf:
for root, dirs, files in os.walk(source_folder):
for file in files:
file_path = os.path.join(root, file)
arcname = os.path.relpath(file_path, source_folder)
zf.write(file_path, arcname)
def _create_metadata_file(self, series_name, author="Unknown", artist="Unknown", api_data=None):
content = f"Title: {series_name}\n"
if api_data:
if api_data.get('title'): content = f"Title: {api_data['title']}\n"
if api_data.get('original_title'): content += f"Original Title: {api_data['original_title']}\n"
artists = ", ".join(api_data.get('artist', [])) or artist
circles = ", ".join(api_data.get('circle', [])) or "Unknown"
parodies = ", ".join(api_data.get('parody', [])) or "Original"
tags = ", ".join(api_data.get('tags', []))
content += f"Artist: {artists}\nCircle: {circles}\nParody: {parodies}\nTags: {tags}\nURL: {api_data.get('url', 'N/A')}\n"
else:
content += f"Artist: {artist}\nAuthor: {author}\n"
content += f"Processed by: ServerManagerWebApp\n"
return content
def _lookup_mangadex_metadata(self, title):
"""
Queries Mangadex API.
"""
try:
url = "https://api.mangadex.org/manga"
clean_title = title.replace('_', ' ').strip()
comics_logger.info(f"Searching Mangadex for: {clean_title}")
# Mangadex requires separate calls for author/artist usually, but we'll start with basic info
params = {
"title": clean_title,
"limit": 1,
"includes[]": ["author", "artist", "cover_art"]
}
response = self.scraper.get(url, params=params, timeout=10)
if response.status_code == 200:
data = response.json()
if data.get('data'):
manga = data['data'][0]
attr = manga['attributes']
# Extract authors/artists
authors = []
artists = []
for rel in manga['relationships']:
if rel['type'] == 'author': authors.append(rel.get('attributes', {}).get('name'))
if rel['type'] == 'artist': artists.append(rel.get('attributes', {}).get('name'))
# Tags
tags = [t['attributes']['name']['en'] for t in attr.get('tags', [])]
return {
"title": attr['title'].get('en') or list(attr['title'].values())[0],
"original_title": attr.get('altTitles', [{}])[0].get('en', "") if attr.get('altTitles') else "",
"year": attr.get('year'),
"url": f"https://mangadex.org/title/{manga['id']}",
"id": manga['id'],
"tags": tags,
"author": authors,
"artist": artists,
"source": "Mangadex"
}
except Exception as e:
comics_logger.error(f"Mangadex lookup failed: {e}")
return None
def _lookup_hentai2read_metadata(self, title):
"""
Scrapes Hentai2Read.
"""
try:
base_url = "https://hentai2read.com"
clean_title = title.replace('_', '+').strip()
search_url = f"{base_url}/search/?cmd={clean_title}"
comics_logger.info(f"Searching Hentai2Read for: {clean_title}")
response = self.scraper.get(search_url)
if response.status_code == 200:
soup = BeautifulSoup(response.content, 'html.parser')
# Hentai2Read search results structure
result_link = soup.select_one('.book-grid-item a')
if result_link:
target_url = result_link['href']
comics_logger.info(f"Fetching Hentai2Read details: {target_url}")
resp = self.scraper.get(target_url)
if resp.status_code == 200:
soup = BeautifulSoup(resp.content, 'html.parser')
data = {"source": "Hentai2Read", "url": target_url}
# Title
title_tag = soup.select_one('h3.block-title a')
if title_tag: data['title'] = title_tag.text.strip()
# Info list
info_items = soup.select('ul.list-simple-mini li')
for item in info_items:
text = item.text.strip()
if "Author" in text:
data['author'] = [a.strip() for a in text.replace("Author", "").strip(" :" ).split(',')]
elif "Artist" in text:
data['artist'] = [a.strip() for a in text.replace("Artist", "").strip(" :" ).split(',')]
elif "Parody" in text:
data['parody'] = [a.strip() for a in text.replace("Parody", "").strip(" :" ).split(',')]
elif "Storyline" in text or "Content" in text: # Tags
data['tags'] = [a.text.strip() for a in item.select('a')]
elif "Release" in text:
year_match = re.search(r'\\d{4}', text)
if year_match: data['year'] = year_match.group(0)
return data
except Exception as e:
comics_logger.error(f"Hentai2Read lookup failed: {e}")
return None
def _lookup_mangaupdates_metadata(self, title):
"""
Queries MangaUpdates API for metadata (good for Manhwa/Webtoons).
"""
try:
url = "https://api.mangaupdates.com/v1/series/search"
# Replace underscores with spaces for better search
clean_title = title.replace('_', ' ').strip()
comics_logger.info(f"Searching MangaUpdates for: {clean_title}")
response = self.scraper.post(url, json={"search": clean_title}, timeout=10)
if response.status_code == 200:
data = response.json()
if data.get('results'):
series = data['results'][0]['record']
return {
"title": series.get('title'),
"year": series.get('year'),
"url": series.get('url'),
"id": series.get('series_id'),
"tags": [g.get('genre') for g in series.get('genres', [])] if series.get('genres') else [],
"source": "MangaUpdates"
}
except Exception as e:
comics_logger.error(f"MangaUpdates lookup failed: {e}")
return None
def _process_item(self, samba_manager, item, is_archive=False):
item_path = item['path']
item_name = item['name']
name_part, artist_part, doujin_id = self._get_series_info(item_name)
api_data = None
# 1. Try Mangadex (Standard/Manhwa)
if not api_data:
api_data = self._lookup_mangadex_metadata(name_part)
# 2. Try nhentai (Doujinshi)
if not api_data:
# Need to re-add "source": "nhentai" to the existing method logic or wrapper
data = self._lookup_doujin_metadata(name_part, artist_part, doujin_id)
if data:
data['source'] = "nhentai"
api_data = data
# 3. Try Hentai2Read (Fallback Doujin)
if not api_data:
api_data = self._lookup_hentai2read_metadata(name_part)
# 4. Try MangaUpdates (Final Fallback)
if not api_data:
api_data = self._lookup_mangaupdates_metadata(name_part)
final_series_name = name_part
if api_data:
if api_data.get('title'):
title_clean = re.sub(r'[<>:"/\\|?*]', '', api_data['title']).strip()
final_series_name = title_clean.replace(' ', '_')
if api_data.get('year'): final_series_name = f"{final_series_name}_({api_data['year']})"
elif api_data.get('id'): final_series_name = f"{final_series_name}_({api_data['id']})"
final_series_name = final_series_name.replace(' ', '_')
series_folder_path = f"{self.comics_root}/{final_series_name}"
try:
samba_manager.create_directory(series_folder_path)
except OperationFailure: pass
if item_name == final_series_name: return
comics_logger.info(f"Processing: {item_name} -> {final_series_name}")
with tempfile.TemporaryDirectory() as temp_dir:
extraction_path = os.path.join(temp_dir, "extracted")
os.makedirs(extraction_path, exist_ok=True)
if is_archive:
local_archive = os.path.join(temp_dir, item_name)
with open(local_archive, 'wb') as f: samba_manager.download_file(item_path, f)
try: self._extract_archive(local_archive, extraction_path)
except Exception as e:
comics_logger.error(f"Failed to extract {item_name}: {e}")
return
else: self._download_directory(samba_manager, item_path, extraction_path)
has_images = False
for root, _, files in os.walk(extraction_path):
if any(f.lower().endswith(('.jpg', '.jpeg', '.png', '.webp')) for f in files):
has_images = True
break
if has_images:
cbz_name = f"{final_series_name}.cbz"
temp_cbz = os.path.join(temp_dir, cbz_name)
self._create_cbz(extraction_path, temp_cbz)
dest_cbz_path = f"{series_folder_path}/{cbz_name}"
with open(temp_cbz, 'rb') as f:
res = samba_manager.upload_file(dest_cbz_path, f)
if isinstance(res, dict) and "error" in res: raise Exception(f"Upload failed: {res['error']}")
meta_content = self._create_metadata_file(final_series_name, artist=artist_part, api_data=api_data)
meta_path = f"{series_folder_path}/{final_series_name}_info.txt"
with tempfile.NamedTemporaryFile(mode='w+', delete=False) as tmp_meta:
tmp_meta.write(meta_content)
tmp_meta.flush()
tmp_meta.seek(0)
with open(tmp_meta.name, 'rb') as f: samba_manager.upload_file(meta_path, f)
os.unlink(tmp_meta.name)
if is_archive: samba_manager.delete_file(item_path)
else: samba_manager.delete_directory_recursive(item_path)
else: comics_logger.warning(f"No images found in {item_name}, skipping.")
def organize_comics(self, samba_manager: SambaManager):
comics_logger.info("Starting comics organization process...")
comics_root = "/comics/toberead"
comics_dest_root = "/comics/manga"
comics_logger.info(f"Starting organization in {self.comics_root}...")
self._recursive_scan_and_process(samba_manager, self.comics_root)
def _recursive_scan_and_process(self, samba_manager, current_path):
try:
items_in_toberead = samba_manager.list_path(comics_root)
if "error" in items_in_toberead:
raise Exception(f"Failed to list files in 'toberead': {items_in_toberead['error']}")
items = samba_manager.list_path(current_path)
if isinstance(items, dict) and "error" in items: return
for item in items:
if item['name'] in ['.', '..']: continue
if item['is_directory']:
sub_items = samba_manager.list_path(item['path'])
if any(sub['name'].lower().endswith(('.jpg', '.jpeg', '.png', '.webp')) for sub in sub_items):
self._process_item(samba_manager, item, is_archive=False)
else: self._recursive_scan_and_process(samba_manager, item['path'])
elif item['name'].lower().endswith(('.zip', '.cbz')):
self._process_item(samba_manager, item, is_archive=True)
except Exception as e: comics_logger.error(f"Error scanning {current_path}: {e}")
series_chapters = defaultdict(list)
for item in items_in_toberead:
if not item["is_directory"]:
series_name = self._get_series_name(item["name"])
series_chapters[series_name].append(item)
def get_pending_comics(self, samba_manager: SambaManager):
try:
items = samba_manager.list_path(self.comics_root)
if isinstance(items, dict) and "error" in items: return []
return [{"name": i['name'], "path": i['path']} for i in items if i['is_directory'] and i['name'] not in ['.', '..']]
except Exception: return []
def move_series(self, src_samba: SambaManager, series_names, dest_share=None, dest_path=None):
"""Moves folders from isolation/staging to library share."""
share = dest_share if dest_share else self.library_share_default
path = dest_path if dest_path else self.library_path_default
# Connect to destination share
dest_samba = SambaManager(src_samba.server_ip, share, src_samba.username, src_samba.password)
results = {"success": [], "failed": []}
for series in series_names:
src_folder = f"{self.comics_root}/{series}"
target_folder = f"{path}/{series}"
comics_logger.info(f"Found {len(series_chapters)} series to process.")
try:
# Ensure target directory exists
dest_samba.create_directory(target_folder)
# List files in source (on isolation share)
items = src_samba.list_path(src_folder)
for item in items:
if item['name'] in ['.', '..']: continue
# Cross-share move: Download -> Upload -> Delete
with tempfile.TemporaryDirectory() as temp_dir:
local_file = os.path.join(temp_dir, item['name'])
with open(local_file, 'wb') as f: src_samba.download_file(item['path'], f)
with open(local_file, 'rb') as f: dest_samba.upload_file(f"{target_folder}/{item['name']}", f)
src_samba.delete_file(item['path'])
src_samba.delete_directory(src_folder)
results["success"].append(series)
comics_logger.info(f"Moved {series} to {share}:{target_folder}")
except Exception as e:
results["failed"].append({"name": series, "error": str(e)})
comics_logger.error(f"Failed to move {series}: {e}")
dest_samba.close()
return results
for series_name, chapters in series_chapters.items():
comics_logger.info(f"Processing series: {series_name}")
final_series_folder_path = f"{comics_dest_root}/{series_name}".replace("\\", "/")
def update_existing_metadata(self, samba_manager, target_path="/comics/manga", force=False):
"""
Scans an existing library directory for series folders and updates/creates metadata files.
"""
comics_logger.info(f"Starting metadata update in {target_path} (Force: {force})...")
try:
items = samba_manager.list_path(target_path)
if isinstance(items, dict) and "error" in items:
comics_logger.error(f"Error listing {target_path}: {items['error']}")
return
for item in items:
if item['name'] in ['.', '..']: continue
if not item['is_directory']: continue
series_name = item['name']
series_path = item['path']
# Check for existing metadata
meta_filename = f"{series_name}_info.txt"
meta_path = f"{series_path}/{meta_filename}"
# Check if meta exists
series_contents = samba_manager.list_path(series_path)
has_meta = False
if isinstance(series_contents, list):
for sub in series_contents:
if sub['name'] == meta_filename:
has_meta = True
break
if has_meta and not force:
continue
comics_logger.info(f"Updating metadata for: {series_name}")
# Parse series info from FOLDER NAME
clean_name = series_name.replace('_', ' ')
clean_name = re.sub(r'\s*\\(\\d+\\)$', '', clean_name).strip()
artist = "Unknown"
artist_match = re.match(r'^\\[(.*?)\\]', clean_name)
if artist_match:
artist = artist_match.group(1)
clean_name = clean_name[artist_match.end():].strip()
doujin_id = None
id_match = re.search(r'\\(\\d{5,7}\\)$', series_name)
if id_match:
doujin_id = id_match.group(1)
# 1. Try Mangadex
api_data = self._lookup_mangadex_metadata(clean_name)
# 2. Try nhentai
if not api_data:
data = self._lookup_doujin_metadata(clean_name, artist, doujin_id)
if data:
data['source'] = "nhentai"
api_data = data
# 3. Try Hentai2Read
if not api_data:
api_data = self._lookup_hentai2read_metadata(clean_name)
# 4. Try MangaUpdates
if not api_data:
api_data = self._lookup_mangaupdates_metadata(clean_name)
meta_content = self._create_metadata_file(clean_name, artist=artist, api_data=api_data)
with tempfile.NamedTemporaryFile(mode='w+', delete=False) as tmp_meta:
tmp_meta.write(meta_content)
tmp_meta.flush()
tmp_meta.seek(0)
with open(tmp_meta.name, 'rb') as f:
samba_manager.upload_file(meta_path, f)
os.unlink(tmp_meta.name)
except Exception as e:
comics_logger.error(f"Error updating metadata in {target_path}: {e}")
def _collect_all_folders(self, samba_manager, path):
folders = []
try:
items = samba_manager.list_path(path)
for item in items:
if item['name'] in ['.', '..']: continue
if item['is_directory']:
folders.append({'name': item['name'], 'path': item['path']})
# Recurse
folders.extend(self._collect_all_folders(samba_manager, item['path']))
except Exception as e:
comics_logger.error(f"Error listing path {path}: {e}")
return folders
def find_similar_folders(self, samba_manager: SambaManager, root_path="/comics/manga", threshold=0.9):
"""
Scans for folders with similar names.
"""
comics_logger.info(f"Scanning for duplicate folders in {root_path}...")
all_dirs = self._collect_all_folders(samba_manager, root_path)
comics_logger.info(f"Found {len(all_dirs)} directories. Comparing...")
groups = []
processed_indices = set()
for i in range(len(all_dirs)):
if i in processed_indices: continue
current_group = [all_dirs[i]]
for j in range(i + 1, len(all_dirs)):
if j in processed_indices: continue
name1 = all_dirs[i]['name'].lower().replace('_', ' ')
name2 = all_dirs[j]['name'].lower().replace('_', ' ')
ratio = difflib.SequenceMatcher(None, name1, name2).ratio()
if ratio >= threshold:
current_group.append(all_dirs[j])
processed_indices.add(j)
if len(current_group) > 1:
groups.append({
"name": all_dirs[i]['name'],
"folders": current_group
})
processed_indices.add(i)
return groups
def delete_folder(self, samba_manager: SambaManager, folder_path):
"""
Deletes a specific folder.
"""
try:
comics_logger.info(f"Deleting duplicate folder: {folder_path}")
samba_manager.delete_directory_recursive(folder_path)
return {"success": True}
except Exception as e:
comics_logger.error(f"Failed to delete folder {folder_path}: {e}")
return {"error": str(e)}
def _get_artist_from_info(self, samba_manager, series_path, series_name):
info_filename = f"{series_name}_info.txt"
info_path = f"{series_path}/{info_filename}"
try:
with tempfile.NamedTemporaryFile(mode='w+b', delete=False) as tmp:
samba_manager.download_file(info_path, tmp)
tmp.seek(0)
content = tmp.read().decode('utf-8', errors='ignore')
# Parse content
artist = "Unknown"
author = "Unknown"
for line in content.splitlines():
if line.startswith("Artist:"):
val = line.split(":", 1)[1].strip()
if val and val.lower() != "unknown":
artist = val.split(',')[0].strip() # Take first artist if multiple
elif line.startswith("Author:"):
val = line.split(":", 1)[1].strip()
if val and val.lower() != "unknown":
author = val.split(',')[0].strip()
if artist != "Unknown": return artist
if author != "Unknown": return author
except Exception as e:
# File might not exist or other error
pass
finally:
if 'tmp' in locals() and os.path.exists(tmp.name):
os.unlink(tmp.name)
return "_Unknown"
def sort_by_artist(self, samba_manager, root_path="/comics/manga"):
comics_logger.info(f"Sorting by Artist in {root_path}...")
try:
items = samba_manager.list_path(root_path)
if isinstance(items, dict) and "error" in items:
comics_logger.error(f"Error listing {root_path}: {items['error']}")
return
for item in items:
if item['name'] in ['.', '..', '_Unknown']: continue
if not item['is_directory']: continue
series_name = item['name']
series_path = item['path']
# Check if this is a Series Folder
# Criteria: Contains .cbz, .zip, or _info.txt
try:
sub_items = samba_manager.list_path(series_path)
if isinstance(sub_items, dict) and "error" in sub_items: continue
is_series = False
for sub in sub_items:
if sub['name'].lower().endswith(('.cbz', '.zip', '_info.txt')):
is_series = True
break
if not is_series:
comics_logger.info(f"Skipping potential Artist folder or empty folder: {series_name}")
continue
except Exception:
continue
# Attempt to get artist
artist = self._get_artist_from_info(samba_manager, series_path, series_name)
# Sanitize artist name for folder
clean_artist = re.sub(r'[<>:"/\\|?*]', '', artist).strip().replace(' ', '_')
if not clean_artist: clean_artist = "_Unknown"
# Target path: /comics/manga/Artist/Series
artist_folder = f"{root_path}/{clean_artist}"
target_path = f"{artist_folder}/{series_name}"
# Skip if already in place
if series_name == clean_artist:
continue
# Check if we are moving into itself (e.g. Root/Artist -> Root/Artist/Artist)
# This happens if 'SeriesName' == 'ArtistName' and it was already sorted?
# But we checked is_series. An Artist folder usually doesn't have cbz inside directly.
comics_logger.info(f"Moving '{series_name}' to Artist folder '{clean_artist}'")
try:
samba_manager.create_directory(final_series_folder_path)
except OperationFailure as e:
if "FILE_OBJECT_NAME_COLLISION" not in str(e):
comics_logger.error(f"Could not create directory {final_series_folder_path}: {e}")
for chapter_item in chapters:
original_path = chapter_item["path"].replace("\\", "/")
original_filename = chapter_item["name"]
path_to_move = ""
filename_to_move = ""
if original_filename.lower().endswith('.zip'):
cbz_filename = os.path.splitext(original_filename)[0] + '.cbz'
temp_cbz_path = f"{comics_root}/{cbz_filename}".replace("\\", "/")
try:
samba_manager.rename_file(original_path, temp_cbz_path)
path_to_move = temp_cbz_path
filename_to_move = cbz_filename
comics_logger.info(f"Renamed {original_filename} to {cbz_filename}")
except OperationFailure as e:
comics_logger.error(f"Failed to rename ZIP {original_filename}: {e}")
continue
elif original_filename.lower().endswith('.cbz'):
path_to_move = original_path
filename_to_move = original_filename
else:
continue
final_cbz_path = f"{final_series_folder_path}/{filename_to_move}".replace("\\", "/")
# Create Artist folder
try:
samba_manager.rename_file(path_to_move, final_cbz_path)
comics_logger.info(f"Moved {filename_to_move} to {final_series_folder_path}")
except OperationFailure as e:
comics_logger.error(f"Failed to move {filename_to_move} to {final_series_folder_path}: {e}")
samba_manager.create_directory(artist_folder)
except OperationFailure: pass # Exists
# Move Series folder
samba_manager.rename_file(series_path, target_path)
except Exception as e:
comics_logger.error(f"Failed to move {series_name}: {e}")
except Exception as e:
comics_logger.error(f"An error occurred during comics organization: {e}", exc_info=True)
comics_logger.info("Comics organization process finished.")
return
comics_logger.error(f"Sort by artist failed: {e}")
+44 -4
View File
@@ -2,6 +2,8 @@ from pydantic_settings import BaseSettings, SettingsConfigDict
import subprocess
import logging
import shutil
import json
import os
logger = logging.getLogger(__name__)
@@ -14,6 +16,8 @@ logging.basicConfig(
]
)
SETTINGS_FILE = "resources/config/app_settings.json"
def check_gpu_support():
try:
logger.debug("Checking ffmpeg for CUDA support...")
@@ -29,12 +33,48 @@ def check_gpu_support():
return False
class Settings(BaseSettings):
samba_server_ip: str
samba_username: str
samba_password: str
samba_server_ip: str = "127.0.0.1"
samba_username: str = "guest"
samba_password: str = ""
comicvine_api_key: str | None = None
qbittorrent_url: str | None = None
qbittorrent_username: str | None = None
qbittorrent_password: str | None = None
gpu_enabled: bool = check_gpu_support()
# Defaults
default_comics_path: str = "/comics/manga"
default_videos_path: str = "/videos"
model_config = SettingsConfigDict(env_file=".env")
# Stash Integration
stash_enabled: bool = False
stash_db_path: str = "stash.sqlite"
stash_remote_db_path: str = "/appdata/stashapp/config/stash-go.sqlite"
stash_generated_path: str = "/media/stashapp/generated"
stash_remote_base: str = "/media/videos"
stash_container_base: str = "/data"
stash_share: str = "main"
model_config = SettingsConfigDict(env_file=".env", extra="ignore")
def load_from_json(self):
if os.path.exists(SETTINGS_FILE):
try:
with open(SETTINGS_FILE, 'r') as f:
data = json.load(f)
for key, value in data.items():
if hasattr(self, key):
setattr(self, key, value)
except Exception as e:
logger.error(f"Failed to load settings: {e}")
def save_to_json(self):
try:
os.makedirs(os.path.dirname(SETTINGS_FILE), exist_ok=True)
with open(SETTINGS_FILE, 'w') as f:
json.dump(self.model_dump(), f, indent=4)
except Exception as e:
logger.error(f"Failed to save settings: {e}")
settings = Settings()
settings.load_from_json()
@@ -2,54 +2,96 @@ import logging
import os
import subprocess
import tempfile
import json
import time
import itertools
from PIL import Image
import imagehash
from sqlalchemy.orm import Session
from .samba_manager import SambaManager
from .config import Settings
from .stash_service import StashService
from . import models
import itertools
import logging
import time
logger = logging.getLogger(__name__)
class CancellationException(Exception):
pass
class CancellableWriter:
def __init__(self, file_obj, check_cancel_func):
self.file_obj = file_obj
self.check_cancel_func = check_cancel_func
def write(self, data):
if self.check_cancel_func():
raise CancellationException("Scan canceled by user")
return self.file_obj.write(data)
def close(self):
return self.file_obj.close()
def flush(self):
return self.file_obj.flush()
def tell(self):
return self.file_obj.tell()
def seek(self, offset, whence=0):
return self.file_obj.seek(offset, whence)
class DuplicatesManager:
def __init__(self, settings: Settings, db: Session):
self.settings = settings
self.db = db
self.videos_root = "/videos"
self.stash_service = StashService(settings)
self.exclusions_file = "resources/config/exclusions.json"
self._load_exclusions()
def _load_exclusions(self):
if os.path.exists(self.exclusions_file):
try:
with open(self.exclusions_file, 'r') as f:
self.exclusions = json.load(f)
except:
self.exclusions = []
else:
self.exclusions = []
def _save_exclusions(self):
os.makedirs(os.path.dirname(self.exclusions_file), exist_ok=True)
with open(self.exclusions_file, 'w') as f:
json.dump(self.exclusions, f)
def add_exclusion(self, path):
if path not in self.exclusions:
self.exclusions.append(path)
self._save_exclusions()
def remove_exclusion(self, path):
if path in self.exclusions:
self.exclusions.remove(path)
self._save_exclusions()
def _is_excluded(self, path):
for excl in self.exclusions:
if path.startswith(excl):
return True
return False
def _get_video_duration(self, filepath):
logger.debug(f"Running ffprobe for duration of {filepath}")
try:
command = [
"ffprobe",
"-v",
"error",
"-show_entries",
"format=duration",
"-of",
"default=noprint_wrappers=1:nokey=1",
filepath,
"ffprobe", "-v", "error", "-show_entries", "format=duration",
"-of", "default=noprint_wrappers=1:nokey=1", filepath
]
logger.debug(f"ffprobe command: {' '.join(command)}")
result = subprocess.run(
command,
capture_output=True,
text=True,
check=True,
)
logger.debug(f"ffprobe stdout: {result.stdout.strip()}")
logger.debug(f"ffprobe stderr: {result.stderr.strip()}")
result = subprocess.run(command, capture_output=True, text=True, check=True)
return float(result.stdout)
except (subprocess.CalledProcessError, FileNotFoundError) as e:
except Exception as e:
logger.error(f"ffprobe failed for {filepath}: {e}")
return None
def _get_frame_hash(self, filepath):
logger.debug(f"Running ffmpeg for frame hash of {filepath}")
def _get_frame_hash(self, filepath, algorithm='phash'):
tmp_frame_path = ""
try:
with tempfile.NamedTemporaryFile(suffix=".jpg", delete=False) as tmp_frame:
@@ -57,264 +99,390 @@ class DuplicatesManager:
command = ["ffmpeg"]
if self.settings.gpu_enabled:
command.extend(["-hwaccel", "cuda"])
command.extend([
"-i",
filepath,
"-ss",
"00:00:10",
"-vframes",
"1",
"-y",
tmp_frame_path,
])
logger.debug(f"ffmpeg command: {' '.join(command)}")
result = subprocess.run(
command,
capture_output=True,
check=True,
)
logger.debug(f"ffmpeg stdout: {result.stdout.strip()}")
logger.debug(f"ffmpeg stderr: {result.stderr.strip()}")
# Extract frame at 10s or 10%? Fixed 10s for now.
command.extend(["-i", filepath, "-ss", "00:00:10", "-vframes", "1", "-y", tmp_frame_path])
subprocess.run(command, capture_output=True, check=True)
if os.path.exists(tmp_frame_path):
logger.debug(f"Temporary frame file exists: {tmp_frame_path}, size: {os.path.getsize(tmp_frame_path)} bytes")
phash = imagehash.phash(Image.open(tmp_frame_path))
return str(phash)
else:
logger.warning(f"Temporary frame file was not created: {tmp_frame_path}")
return None
except (subprocess.CalledProcessError, FileNotFoundError) as e:
logger.error(f"ffmpeg failed for {filepath}: {e}")
return None
except Image.UnidentifiedImageError as e:
logger.error(f"PIL.UnidentifiedImageError for {filepath} with temp file {tmp_frame_path}: {e}")
return None
if os.path.exists(tmp_frame_path) and os.path.getsize(tmp_frame_path) > 0:
img = Image.open(tmp_frame_path)
if algorithm == 'ahash': h = imagehash.average_hash(img)
elif algorithm == 'dhash': h = imagehash.dhash(img)
else: h = imagehash.phash(img)
return str(h)
except Exception as e:
logger.error(f"Hash generation failed for {filepath}: {e}")
finally:
if os.path.exists(tmp_frame_path):
os.remove(tmp_frame_path)
return None
def _is_video_file(self, filename):
video_extensions = ['.mp4', '.mkv', '.avi', '.mov', '.wmv', '.flv', '.webm']
return any(filename.lower().endswith(ext) for ext in video_extensions)
def generate_contact_sheet(self, video_path):
"""
Generates a 3x3 contact sheet for the video.
Returns (relative_path, phash_str)
"""
try:
duration = self._get_video_duration(video_path)
if not duration or duration < 10: return None, None
# Extract 9 frames at intervals
interval = duration / 10
timestamps = [interval * i for i in range(1, 10)]
# We use a temp dir to store frames, then stitch
# ffmpeg tile filter is good but seeking is faster for sparse frames on large files?
# Actually, `ffmpeg -i ... -vf fps=... tile=...` reads the whole file which is slow over network/SMB.
# Best to seek.
# Since we have the file locally in tmp_path (downloaded), seeking is fast.
frames = []
with tempfile.TemporaryDirectory() as temp_frames_dir:
for idx, ts in enumerate(timestamps):
out_frame = os.path.join(temp_frames_dir, f"frame_{idx}.jpg")
# fast seek
subprocess.run(
["ffmpeg", "-ss", str(ts), "-i", video_path, "-vframes", "1", "-q:v", "5", "-vf", "scale=320:-1", "-y", out_frame],
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=False
)
if os.path.exists(out_frame):
frames.append(Image.open(out_frame))
if len(frames) < 4: return None, None # Need at least some frames
# Stitch 3x3 (or adaptive)
# Create blank image
w, h = frames[0].size
grid_w = w * 3
grid_h = h * 3
contact_sheet = Image.new('RGB', (grid_w, grid_h))
for idx, frame in enumerate(frames):
if idx >= 9: break
x = (idx % 3) * w
y = (idx // 3) * h
contact_sheet.paste(frame, (x, y))
# Save
cache_dir = "resources/cache/thumbnails"
os.makedirs(cache_dir, exist_ok=True)
# Use hash of path to ensure uniqueness/retrievability
filename_hash = imagehash.hex_to_hash(os.path.basename(video_path)) # Just use random or md5
import hashlib
file_hash = hashlib.md5(video_path.encode()).hexdigest()
out_name = f"{file_hash}.jpg"
out_path = os.path.join(cache_dir, out_name)
contact_sheet.save(out_path, "JPEG", quality=80)
# Calculate Hash of the SHEET
sheet_hash = imagehash.phash(contact_sheet)
return out_name, str(sheet_hash)
except Exception as e:
logger.error(f"Contact sheet generation failed: {e}")
return None, None
def _process_video_file(self, samba_manager, filepath, filename, size, algorithm='phash', scan_type='fast', log_func=None, cancel_check_func=None):
if cancel_check_func and cancel_check_func(): return
if self._is_excluded(filepath):
return
def _process_video_file(self, samba_manager: SambaManager, filepath, filename, size):
logger.info(f"Processing video: {filepath} ({filename})")
existing_video = self.db.query(models.VideoFile).filter_by(filepath=filepath).first()
# If in scene mode, check if we already have the scene data
if existing_video:
if existing_video.size == size:
logger.info(f"Skipping already processed and unaltered video: {filepath}")
return
if scan_type == 'scene' and not existing_video.contact_sheet_path:
if log_func: log_func(f"Updating {filename} with contact sheet")
# Continue to processing
elif existing_video.size == size:
return # Skip if unchanged
else:
logger.info(f"File {filepath} has altered size ({existing_video.size} -> {size}). Re-processing.")
self.db.delete(existing_video)
self.db.commit()
existing_video = None
if log_func: log_func(f"Processing: {filename} (Mode: {scan_type})")
# --- Stash Integration ---
if self.settings.stash_enabled:
if cancel_check_func and cancel_check_func(): return
stash_phash, stash_oshash, scene_id, stash_duration = self.stash_service.get_file_metadata(filepath)
if stash_phash:
if log_func: log_func(f"Found Stash metadata for {filename}")
sheet_path = None
if scan_type == 'scene' and stash_oshash:
if cancel_check_func and cancel_check_func(): return
remote_sprite = self.stash_service.get_sprite_path(stash_oshash)
local_sheet_name = f"stash_{stash_oshash}.jpg"
local_sheet_path = os.path.join("resources/cache/thumbnails", local_sheet_name)
if not os.path.exists(local_sheet_path):
try:
# Ensure directory exists
os.makedirs(os.path.dirname(local_sheet_path), exist_ok=True)
# Use stash_share for sprite download
stash_samba = samba_manager
if samba_manager.share_name != self.settings.stash_share:
stash_samba = SambaManager(
self.settings.samba_server_ip,
self.settings.stash_share,
self.settings.samba_username,
self.settings.samba_password
)
try:
with open(local_sheet_path, "wb") as f:
stash_samba.download_file(remote_sprite, f)
sheet_path = local_sheet_name
finally:
if stash_samba != samba_manager:
stash_samba.close()
except Exception as e:
if log_func: log_func(f"Failed to download Stash sprite for {filename}: {e}")
else:
sheet_path = local_sheet_name
# Save to DB
video = models.VideoFile(
filepath=filepath, filename=filename, size=size,
duration=stash_duration or 0, phash=stash_phash,
contact_sheet_path=sheet_path,
scene_phash=None
)
self.db.add(video)
self.db.commit()
return
# --- End Stash Integration ---
if cancel_check_func and cancel_check_func(): return
# Download to temp
with tempfile.NamedTemporaryFile(delete=False) as tmp_video:
try:
samba_manager.download_file(filepath, tmp_video)
if log_func: log_func(f"Downloading {filename} ({size/1024/1024:.2f} MB)...")
# Wrap for cancellation during download
writer = tmp_video
if cancel_check_func:
writer = CancellableWriter(tmp_video, cancel_check_func)
samba_manager.download_file(filepath, writer)
tmp_path = tmp_video.name
except CancellationException:
if log_func: log_func(f"Download aborted for {filename}")
tmp_video.close()
os.remove(tmp_video.name)
return
except Exception as e:
logger.error(f"Failed to download {filepath}: {e}")
if log_func: log_func(f"Download failed: {filepath} - {e}")
tmp_video.close()
os.remove(tmp_video.name)
return
duration = self._get_video_duration(tmp_video.name)
phash = self._get_frame_hash(tmp_video.name)
os.remove(tmp_video.name)
if duration is not None and phash is not None:
video_file = models.VideoFile(
filepath=filepath,
filename=filename,
size=size,
duration=duration,
phash=phash,
)
self.db.add(video_file)
self.db.commit()
logger.info(f"Processed video: {filepath}")
else:
logger.warning(f"Could not get duration or hash for {filepath}")
def scan_videos(self, samba_manager: SambaManager):
state_file = "logs/duplicates_scan.state"
progress_file = "logs/duplicates_scan.progress"
pause_file = "logs/duplicates_scan.pause"
dirs_to_scan = []
try:
if os.path.exists(progress_file):
with open(progress_file, 'r') as f:
last_processed_dir = f.read().strip()
logger.info(f"Resuming scan from last in-progress directory: {last_processed_dir}")
dirs_to_scan.append(last_processed_dir)
if os.path.exists(state_file):
with open(state_file, 'r') as f:
dirs_to_scan.extend([line.strip() for line in f if line.strip()])
logger.info(f"Loaded {len(dirs_to_scan)} directories from state file.")
if cancel_check_func and cancel_check_func(): return
if not dirs_to_scan:
dirs_to_scan = [self.videos_root]
duration = self._get_video_duration(tmp_path)
phash = self._get_frame_hash(tmp_path, algorithm)
if cancel_check_func and cancel_check_func(): return
logger.info(f"Starting scan for videos in {self.videos_root} on share 'isolation'")
sheet_path = None
scene_hash = None
if scan_type == 'scene':
sheet_path, scene_hash = self.generate_contact_sheet(tmp_path)
if duration is not None: # phash might be None if image generation failed
if not existing_video:
video = models.VideoFile(
filepath=filepath, filename=filename, size=size,
duration=duration, phash=phash or "",
contact_sheet_path=sheet_path,
scene_phash=scene_hash
)
self.db.add(video)
else:
# Update existing
existing_video.duration = duration
existing_video.phash = phash or ""
if sheet_path: existing_video.contact_sheet_path = sheet_path
if scene_hash: existing_video.scene_phash = scene_hash
self.db.commit()
finally:
if os.path.exists(tmp_path):
try: os.remove(tmp_path)
except: pass
def scan_videos(self, samba_manager, root_paths=["/videos"], algorithm='phash', scan_type='fast'):
state_file = "logs/duplicates_scan.state"
progress_file = "logs/duplicates_scan.json"
cancel_file = "logs/duplicates_scan.cancel"
log_file = "logs/duplicates_scan.log"
# Helper to log to file and console
def log(msg):
try:
with open(log_file, "a") as f:
f.write(f"{msg}\n")
except: pass
logger.info(msg)
is_cancelled = False
# Helper to check cancellation
def check_cancel():
nonlocal is_cancelled
if is_cancelled: return True
if os.path.exists(cancel_file):
log("Scan canceled by user.")
try:
os.remove(cancel_file)
except OSError:
pass
with open(progress_file, 'w') as f:
json.dump({"status": "canceled", "processed": processed_files}, f)
is_cancelled = True
return True
return False
# Clear log file
with open(log_file, "w") as f:
f.write("Scan started...\n")
# Ensure cancel file is gone before starting
if os.path.exists(cancel_file):
try: os.remove(cancel_file)
except OSError: pass
dirs_to_scan = list(root_paths)
processed_files = 0
try:
while dirs_to_scan:
while os.path.exists(pause_file):
logger.info("Scan is paused. Waiting for resume signal...")
time.sleep(5)
if check_cancel(): return
current_path = dirs_to_scan.pop(0)
with open(progress_file, 'w') as f:
f.write(current_path)
json.dump({"status": "scanning", "current": current_path, "processed": processed_files}, f)
logger.info(f"Scanning directory: {current_path}")
files_and_dirs = samba_manager.list_path(current_path)
if "error" in files_and_dirs:
logger.error(f"Failed to list path {current_path}: {files_and_dirs['error']}")
try:
log(f"Scanning directory: {current_path}")
items = samba_manager.list_path(current_path)
except Exception as e:
log(f"Error listing {current_path}: {e}")
continue
subdirs = []
for item in files_and_dirs:
while os.path.exists(pause_file):
logger.info("Scan is paused. Waiting for resume signal...")
time.sleep(5)
for item in items:
if check_cancel(): return
if item["is_directory"]:
subdirs.append(item["path"])
elif self._is_video_file(item["name"]):
self._process_video_file(samba_manager, item["path"], item["name"], item["size"])
if item['name'] in ['.', '..']: continue
if item['is_directory']:
if not self._is_excluded(item['path']):
dirs_to_scan.append(item['path'])
elif any(item['name'].lower().endswith(ext) for ext in ['.mp4', '.mkv', '.avi', '.mov', '.wmv']):
self._process_video_file(
samba_manager, item['path'], item['name'], item['size'],
algorithm, scan_type,
log_func=log, cancel_check_func=check_cancel
)
processed_files += 1
with open(progress_file, 'w') as f:
json.dump({"status": "scanning", "current": item['path'], "processed": processed_files}, f)
log("Scan completed.")
with open(progress_file, 'w') as f:
json.dump({"status": "completed", "processed": processed_files}, f)
dirs_to_scan = subdirs + dirs_to_scan
if os.path.exists(progress_file):
os.remove(progress_file)
with open(state_file, 'w') as f:
for d in dirs_to_scan:
f.write(d + '\n')
if os.path.exists(state_file):
os.remove(state_file)
logger.info("Video scan complete.")
return {"status": "Scan complete"}
except Exception as e:
logger.error(f"An error occurred during video scan: {e}", exc_info=True)
return {"status": "Scan failed", "error": str(e)}
log(f"Scan failed: {e}")
with open(progress_file, 'w') as f:
json.dump({"status": "failed", "error": str(e)}, f)
def _hamming_distance(self, s1, s2):
return sum(c1 != c2 for c1, c2 in zip(s1, s2))
def _calculate_similarity(self, file1: models.VideoFile, file2: models.VideoFile):
size_similarity = 1 - (abs(file1.size - file2.size) / max(file1.size, file2.size))
duration_similarity = 1 - (abs(file1.duration - file2.duration) / max(file1.duration, file2.duration))
hash_similarity = 1 - (self._hamming_distance(file1.phash, file2.phash) / len(file1.phash))
return (size_similarity * 0.2) + (duration_similarity * 0.3) + (hash_similarity * 0.5)
def find_duplicates(self, threshold=0.95):
def find_duplicates(self, threshold=0.95, method='fast'):
report = models.DuplicateReport(status="running")
self.db.add(report)
self.db.commit()
videos = self.db.query(models.VideoFile).all()
groups = []
processed_videos = set()
processed_ids = set()
for video1, video2 in itertools.combinations(videos, 2):
if video1.id in processed_videos or video2.id in processed_videos:
continue
score = self._calculate_similarity(video1, video2)
if score >= threshold:
existing_group = None
for group in groups:
if video1.id in group["video_ids"] or video2.id in group["video_ids"]:
existing_group = group
break
for i in range(len(videos)):
if videos[i].id in processed_ids: continue
group = [videos[i]]
scores = []
for j in range(i + 1, len(videos)):
if videos[j].id in processed_ids: continue
if existing_group:
existing_group["video_ids"].add(video1.id)
existing_group["video_ids"].add(video2.id)
existing_group["scores"].append(score)
v1, v2 = videos[i], videos[j]
score = 0
if method == 'scene' and v1.scene_phash and v2.scene_phash:
# Compare Scene Hashes
dist = imagehash.hex_to_hash(v1.scene_phash) - imagehash.hex_to_hash(v2.scene_phash)
score = max(0, 1.0 - (dist / 64.0)) # 64 is typical max distance for 8x8 hash
else:
groups.append({"video_ids": {video1.id, video2.id}, "scores": [score]})
# Standard Comparison
dist = imagehash.hex_to_hash(v1.phash) - imagehash.hex_to_hash(v2.phash) if v1.phash and v2.phash else 64
hash_sim = max(0, 1.0 - (dist / 64.0))
dur_sim = 1.0 - (abs(v1.duration - v2.duration) / max(v1.duration, v2.duration)) if max(v1.duration, v2.duration) > 0 else 1.0
size_sim = 1.0 - (abs(v1.size - v2.size) / max(v1.size, v2.size)) if max(v1.size, v2.size) > 0 else 1.0
score = (hash_sim * 0.6) + (dur_sim * 0.3) + (size_sim * 0.1)
processed_videos.add(video1.id)
processed_videos.add(video2.id)
for group_data in groups:
avg_score = sum(group_data["scores"]) / len(group_data["scores"])
db_group = models.DuplicateFileGroup(report_id=report.id, score=avg_score)
self.db.add(db_group)
self.db.commit()
for video_id in group_data["video_ids"]:
db_file = models.DuplicateFile(group_id=db_group.id, video_file_id=video_id)
self.db.add(db_file)
if score >= threshold:
group.append(v2)
scores.append(score)
processed_ids.add(v2.id)
if len(group) > 1:
processed_ids.add(videos[i].id)
avg_score = sum(scores) / len(scores)
db_group = models.DuplicateFileGroup(report_id=report.id, score=avg_score)
self.db.add(db_group)
self.db.commit()
for v in group:
db_file = models.DuplicateFile(group_id=db_group.id, video_file_id=v.id)
self.db.add(db_file)
report.status = "completed"
self.db.commit()
return {"report_id": report.id, "status": "completed"}
return {"report_id": report.id}
def get_reports(self):
return self.db.query(models.DuplicateReport).all()
return self.db.query(models.DuplicateReport).order_by(models.DuplicateReport.created_at.desc()).all()
def get_duplicate_report(self, report_id: int):
report = self.db.query(models.DuplicateReport).filter(models.DuplicateReport.id == report_id).first()
if not report:
return {"error": "Report not found"}
groups = []
for group in report.groups:
files = []
for duplicate_file in group.files:
files.append(duplicate_file.video_file)
groups.append({
"group_id": group.id,
"score": group.score,
"files": files,
})
def get_report(self, report_id):
report = self.db.query(models.DuplicateReport).filter_by(id=report_id).first()
if not report: return None
return {
"report_id": report.id,
"created_at": report.created_at,
res = {
"id": report.id,
"status": report.status,
"groups": groups,
"date": report.created_at,
"groups": []
}
def delete_files(self, filepaths: list[str]):
samba_manager = SambaManager(
self.settings.samba_server_ip,
"isolation", # Assuming all duplicates are in the isolation share
self.settings.samba_username,
self.settings.samba_password,
)
try:
for filepath in filepaths:
# Delete from Samba
samba_manager.delete_file(filepath)
# Delete from database
video_file = self.db.query(models.VideoFile).filter_by(filepath=filepath).first()
if video_file:
# Delete associations in DuplicateFile
self.db.query(models.DuplicateFile).filter_by(video_file_id=video_file.id).delete()
self.db.delete(video_file)
self.db.commit()
return {"status": "success"}
except Exception as e:
self.db.rollback()
logger.error(f"Error deleting files: {e}", exc_info=True)
return {"status": "error", "message": str(e)}
finally:
samba_manager.close()
for g in report.groups:
files = [{
"id": f.video_file.id,
"path": f.video_file.filepath,
"size": f.video_file.size,
"duration": f.video_file.duration,
"contact_sheet": f.video_file.contact_sheet_path
} for f in g.files]
res["groups"].append({"id": g.id, "score": g.score, "files": files})
return res
+33 -8
View File
@@ -1,20 +1,39 @@
from fastapi import FastAPI, Depends
import os
import json
from fastapi import FastAPI
from fastapi.middleware.cors import CORSMiddleware
from sqlalchemy.orm import Session
from contextlib import asynccontextmanager
from . import models
from .database import SessionLocal, engine, get_db
from .routers import samba, comics, duplicates, downloader
from .routers import samba, comics, duplicates, downloader, qbittorrent, system, tasks, settings, stash, scheduler as scheduler_router
from .scheduler import start_scheduler, scheduler
models.Base.metadata.create_all(bind=engine)
app = FastAPI()
def reset_stuck_scans():
progress_file = "logs/duplicates_scan.json"
if os.path.exists(progress_file):
try:
with open(progress_file, 'r') as f:
data = json.load(f)
if data.get("status") == "scanning":
with open(progress_file, 'w') as f:
json.dump({"status": "failed", "error": "Scan interrupted by server restart"}, f)
except Exception:
pass
origins = [
"http://localhost:5173", # Assuming frontend runs on this port during development
"http://127.0.0.1:5173",
# Add your production frontend URL(s) here when deploying
]
@asynccontextmanager
async def lifespan(app: FastAPI):
reset_stuck_scans()
start_scheduler()
yield
scheduler.shutdown()
app = FastAPI(lifespan=lifespan)
origins = ["*"]
app.add_middleware(
CORSMiddleware,
@@ -28,6 +47,12 @@ app.include_router(samba.router, prefix="/samba", tags=["samba"])
app.include_router(comics.router, prefix="/comics", tags=["comics"])
app.include_router(duplicates.router, prefix="/duplicates", tags=["duplicates"])
app.include_router(downloader.router, prefix="/downloader", tags=["downloader"])
app.include_router(qbittorrent.router, prefix="/qbittorrent", tags=["qbittorrent"])
app.include_router(system.router, prefix="/system", tags=["system"])
app.include_router(tasks.router, prefix="/tasks", tags=["tasks"])
app.include_router(settings.router, prefix="/settings", tags=["settings"])
app.include_router(stash.router, prefix="/stash", tags=["stash"])
app.include_router(scheduler_router.router, prefix="/scheduler", tags=["scheduler"])
+14
View File
@@ -22,6 +22,8 @@ class VideoFile(Base):
size = Column(Integer)
duration = Column(Float)
phash = Column(String)
contact_sheet_path = Column(String, nullable=True)
scene_phash = Column(String, nullable=True)
class DuplicateFileGroup(Base):
@@ -42,3 +44,15 @@ class DuplicateFile(Base):
video_file_id = Column(Integer, ForeignKey("video_files.id"))
group = relationship("DuplicateFileGroup", back_populates="files")
video_file = relationship("VideoFile")
class TaskHistory(Base):
__tablename__ = "task_history"
id = Column(Integer, primary_key=True, index=True)
task_id = Column(String, index=True) # Huey Task ID
name = Column(String)
status = Column(String) # running, success, failed
start_time = Column(DateTime, default=datetime.datetime.utcnow)
end_time = Column(DateTime, nullable=True)
details = Column(String, nullable=True)
@@ -0,0 +1,70 @@
import qbittorrentapi
import logging
from .config import settings
logger = logging.getLogger(__name__)
class QbittorrentManager:
def __init__(self):
self.url = settings.qbittorrent_url
self.username = settings.qbittorrent_username
self.password = settings.qbittorrent_password
self.client = None
def connect(self):
if not self.url:
raise Exception("qBittorrent URL not configured")
try:
self.client = qbittorrentapi.Client(
host=self.url,
username=self.username,
password=self.password,
VERIFY_WEBUI_CERTIFICATE=False
)
self.client.auth_log_in()
except qbittorrentapi.LoginFailed as e:
logger.error(f"qBittorrent Login Failed: {e}")
raise Exception("qBittorrent Login Failed")
except Exception as e:
logger.error(f"qBittorrent Connection Failed: {e}")
raise
def get_torrents(self):
self.connect()
# Return list of dicts with relevant info
torrents = self.client.torrents_info()
return [
{
"hash": t.hash,
"name": t.name,
"state": t.state,
"save_path": t.save_path,
"progress": t.progress,
"size": t.size,
"dlspeed": t.dlspeed,
"upspeed": t.upspeed,
"eta": t.eta
}
for t in torrents
]
def pause_torrents(self, hashes):
self.connect()
self.client.torrents_pause(torrent_hashes=hashes)
def resume_torrents(self, hashes):
self.connect()
self.client.torrents_resume(torrent_hashes=hashes)
def delete_torrents(self, hashes, delete_files=False):
self.connect()
self.client.torrents_delete(torrent_hashes=hashes, delete_files=delete_files)
def recheck_torrents(self, hashes):
self.connect()
self.client.torrents_recheck(torrent_hashes=hashes)
def set_location(self, hashes, location):
self.connect()
self.client.torrents_set_location(location=location, torrent_hashes=hashes)
+103 -28
View File
@@ -1,40 +1,115 @@
from fastapi import APIRouter, BackgroundTasks
from fastapi import APIRouter, HTTPException, WebSocket, WebSocketDisconnect
from pydantic import BaseModel
from typing import List, Optional
import asyncio
import os
from ..comics_manager import ComicsManager
from ..config import settings
from ..samba_manager import SambaManager
from ..tasks import task_organize_comics, task_move_comics, task_update_metadata, task_sort_by_artist
router = APIRouter()
def run_organize_comics_background():
app_settings = settings
manager = ComicsManager()
samba_manager = SambaManager(
app_settings.samba_server_ip,
"isolation",
app_settings.samba_username,
app_settings.samba_password,
class MoveRequest(BaseModel):
series_names: List[str]
destination_share: Optional[str] = "comics"
destination_path: Optional[str] = "/manga"
class MetadataUpdateRequest(BaseModel):
target_path: Optional[str] = "/comics/manga"
force_update: Optional[bool] = False
class FolderScanRequest(BaseModel):
root_path: Optional[str] = "/comics/manga"
threshold: float = 0.9
class DeleteFolderRequest(BaseModel):
folder_path: str
class SortRequest(BaseModel):
root_path: Optional[str] = "/comics/manga"
def get_samba_manager(share="isolation"):
return SambaManager(
settings.samba_server_ip,
share,
settings.samba_username,
settings.samba_password,
)
manager.organize_comics(samba_manager)
samba_manager.close()
@router.post("/organize")
def organize_comics(background_tasks: BackgroundTasks):
background_tasks.add_task(run_organize_comics_background)
return {"message": "Comics organization started in the background."}
def organize_comics():
task = task_organize_comics()
return {"message": f"Organization queued (Task ID: {task.id}). Check logs."}
def run_cleanup_toberead_background():
app_settings = settings
@router.get("/pending")
def get_pending_comics():
manager = ComicsManager()
samba_manager = SambaManager(
app_settings.samba_server_ip,
"isolation",
app_settings.samba_username,
app_settings.samba_password,
)
manager.cleanup_toberead(samba_manager)
samba_manager.close()
samba_manager = get_samba_manager("isolation")
try:
return manager.get_pending_comics(samba_manager)
finally:
samba_manager.close()
@router.delete("/cleanup_toberead")
def cleanup_toberead(background_tasks: BackgroundTasks):
background_tasks.add_task(run_cleanup_toberead_background)
return {"message": "Comics cleanup started in the background."}
@router.post("/move")
def move_comics(request: MoveRequest):
task = task_move_comics(request.series_names, request.destination_share, request.destination_path)
return {"message": f"Move queued (Task ID: {task.id}). Check logs."}
@router.post("/update_metadata")
def update_metadata(request: MetadataUpdateRequest):
task = task_update_metadata(request.target_path, request.force_update)
return {"message": f"Metadata update queued (Task ID: {task.id}). Check logs."}
@router.post("/folders/scan")
def scan_folders(request: FolderScanRequest):
# This remains synchronous to return results immediately to UI
manager = ComicsManager()
samba_manager = get_samba_manager("isolation")
try:
groups = manager.find_similar_folders(samba_manager, request.root_path, request.threshold)
return groups
finally:
samba_manager.close()
@router.post("/folders/delete")
def delete_folder(request: DeleteFolderRequest):
manager = ComicsManager()
samba_manager = get_samba_manager("isolation")
try:
return manager.delete_folder(samba_manager, request.folder_path)
finally:
samba_manager.close()
@router.post("/sort_by_artist")
def sort_by_artist(request: SortRequest):
task = task_sort_by_artist(request.root_path)
return {"message": f"Sort by Artist queued (Task ID: {task.id}). Check logs."}
@router.websocket("/ws/logs")
async def websocket_endpoint(websocket: WebSocket):
await websocket.accept()
file_path = "logs/comics_organization.log"
# Ensure file exists
if not os.path.exists(file_path):
os.makedirs(os.path.dirname(file_path), exist_ok=True)
with open(file_path, "w") as f: f.write("")
try:
with open(file_path, "r") as f:
# Send last 20 lines for context
lines = f.readlines()
for line in lines[-20:]:
await websocket.send_text(line)
# Follow the file
f.seek(0, 2)
while True:
line = f.readline()
if line:
await websocket.send_text(line)
else:
await asyncio.sleep(0.1)
except WebSocketDisconnect:
pass
@@ -1,19 +1,30 @@
from fastapi import APIRouter, Depends, BackgroundTasks
from fastapi import APIRouter, Depends, BackgroundTasks, HTTPException
from fastapi.responses import FileResponse
from sqlalchemy.orm import Session
from pydantic import BaseModel
from ..duplicates_manager import DuplicatesManager
from ..config import settings, Settings
from ..database import get_db, SessionLocal
from ..samba_manager import SambaManager
from .. import models
import os
import json
from typing import List
router = APIRouter()
class DeleteFilesRequest(BaseModel):
filepaths: List[str]
class ScanConfig(BaseModel):
paths: List[str] = ["/videos"]
algorithm: str = "phash"
scan_type: str = "fast" # fast or scene
def run_scan_videos_background():
class DeleteRequest(BaseModel):
ids: List[int]
class ExclusionRequest(BaseModel):
path: str
def run_scan_videos_background(paths, algorithm, scan_type):
db = SessionLocal()
app_settings = Settings()
manager = DuplicatesManager(app_settings, db)
@@ -24,42 +35,72 @@ def run_scan_videos_background():
app_settings.samba_password,
)
try:
manager.scan_videos(samba_manager)
manager.scan_videos(samba_manager, paths, algorithm, scan_type)
finally:
samba_manager.close()
db.close()
@router.post("/scan")
def scan_videos(background_tasks: BackgroundTasks):
background_tasks.add_task(run_scan_videos_background)
return {"message": "Video scan started in the background."}
@router.post("/scan/start")
def start_scan(config: ScanConfig, background_tasks: BackgroundTasks):
background_tasks.add_task(run_scan_videos_background, config.paths, config.algorithm, config.scan_type)
return {"message": "Scan started"}
@router.post("/scan/pause")
def pause_scan():
with open("logs/duplicates_scan.pause", 'w') as f:
pass
return {"message": "Video scan paused."}
@router.post("/scan/cancel")
def cancel_scan():
try:
os.makedirs("logs", exist_ok=True)
with open("logs/duplicates_scan.cancel", 'w') as f:
f.write("cancel")
except Exception as e:
raise HTTPException(status_code=500, detail=f"Failed to cancel scan: {e}")
return {"message": "Cancellation requested"}
@router.post("/scan/resume")
def resume_scan():
if os.path.exists("logs/duplicates_scan.pause"):
os.remove("logs/duplicates_scan.pause")
return {"message": "Video scan resumed."}
@router.get("/scan/log")
def get_scan_log(limit: int = 100):
log_file = "logs/duplicates_scan.log"
if not os.path.exists(log_file):
return {"lines": []}
try:
# Simple read and tail (not efficient for huge files but fine here)
with open(log_file, "r", encoding="utf-8", errors="replace") as f:
lines = f.readlines()
return {"lines": lines[-limit:]}
except Exception as e:
return {"lines": [f"Error reading log: {e}"]}
@router.get("/scan/status")
def get_scan_status():
if os.path.exists("logs/duplicates_scan.pause"):
return {"status": "Paused"}
if os.path.exists("logs/duplicates_scan.progress") or os.path.exists("logs/duplicates_scan.state"):
return {"status": "Scanning"}
return {"status": "Idle"}
@router.get("/scan/progress")
def get_scan_progress():
path = "logs/duplicates_scan.json"
if os.path.exists(path):
try:
with open(path, 'r') as f:
return json.load(f)
except:
return {"status": "error", "message": "Read failed"}
return {"status": "idle"}
@router.get("/exclusions")
def get_exclusions(db: Session = Depends(get_db)):
manager = DuplicatesManager(settings, db)
return manager.exclusions
@router.post("/exclusions")
def add_exclusion(req: ExclusionRequest, db: Session = Depends(get_db)):
manager = DuplicatesManager(settings, db)
manager.add_exclusion(req.path)
return {"message": "Added"}
@router.delete("/exclusions")
def remove_exclusion(req: ExclusionRequest, db: Session = Depends(get_db)):
manager = DuplicatesManager(settings, db)
manager.remove_exclusion(req.path)
return {"message": "Removed"}
@router.post("/find")
def find_duplicates(threshold: float = 0.95, db: Session = Depends(get_db)):
def find_duplicates(threshold: float = 0.95, method: str = "fast", db: Session = Depends(get_db)):
manager = DuplicatesManager(settings, db)
result = manager.find_duplicates(threshold)
return result
return manager.find_duplicates(threshold, method)
@router.get("/reports")
def get_reports(db: Session = Depends(get_db)):
@@ -69,9 +110,24 @@ def get_reports(db: Session = Depends(get_db)):
@router.get("/report/{report_id}")
def get_report(report_id: int, db: Session = Depends(get_db)):
manager = DuplicatesManager(settings, db)
return manager.get_duplicate_report(report_id)
res = manager.get_report(report_id)
if not res: raise HTTPException(status_code=404)
return res
@router.delete("/files")
def delete_files(request: DeleteFilesRequest, db: Session = Depends(get_db)):
def delete_files(request: DeleteRequest, db: Session = Depends(get_db)):
manager = DuplicatesManager(settings, db)
return manager.delete_files(request.filepaths)
deleted = manager.delete_files(request.ids)
return {"deleted_ids": deleted}
@router.get("/thumbnails/{video_id}")
def get_thumbnail(video_id: int, db: Session = Depends(get_db)):
video = db.query(models.VideoFile).get(video_id)
if not video or not video.contact_sheet_path:
raise HTTPException(status_code=404, detail="Thumbnail not found")
path = os.path.join("resources/cache/thumbnails", video.contact_sheet_path)
if not os.path.exists(path):
raise HTTPException(status_code=404, detail="Thumbnail file missing")
return FileResponse(path)
@@ -0,0 +1,63 @@
from fastapi import APIRouter, HTTPException
from pydantic import BaseModel
from typing import List
from ..qbittorrent_manager import QbittorrentManager
router = APIRouter()
manager = QbittorrentManager()
class TorrentActionRequest(BaseModel):
hashes: List[str]
class DeleteRequest(TorrentActionRequest):
delete_files: bool = False
class LocationRequest(TorrentActionRequest):
location: str
@router.get("/")
def list_torrents():
try:
return manager.get_torrents()
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
@router.post("/pause")
def pause_torrents(request: TorrentActionRequest):
try:
manager.pause_torrents(request.hashes)
return {"status": "success"}
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
@router.post("/resume")
def resume_torrents(request: TorrentActionRequest):
try:
manager.resume_torrents(request.hashes)
return {"status": "success"}
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
@router.post("/delete")
def delete_torrents(request: DeleteRequest):
try:
manager.delete_torrents(request.hashes, request.delete_files)
return {"status": "success"}
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
@router.post("/recheck")
def recheck_torrents(request: TorrentActionRequest):
try:
manager.recheck_torrents(request.hashes)
return {"status": "success"}
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
@router.post("/location")
def set_location(request: LocationRequest):
try:
manager.set_location(request.hashes, request.location)
return {"status": "success"}
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
@@ -215,3 +215,16 @@ def rename_file(share_name: str, request: RenameRequest):
result = manager.rename_file(request.old_path, request.new_path)
manager.close()
return result
@router.get("/search/{share_name}")
def search_files(share_name: str, query: str, path: str = "/"):
manager = SambaManager(
settings.samba_server_ip,
share_name,
username=settings.samba_username,
password=settings.samba_password,
)
try:
return manager.search_files(query, path)
finally:
manager.close()
@@ -0,0 +1,34 @@
from fastapi import APIRouter, HTTPException
from pydantic import BaseModel
from typing import List, Optional, Dict, Any
from ..scheduler import add_job, get_jobs, remove_job
router = APIRouter()
class JobCreate(BaseModel):
task_name: str
cron_expression: str
args: Optional[List[Any]] = []
kwargs: Optional[Dict[str, Any]] = {}
@router.get("/")
def list_jobs():
return get_jobs()
@router.post("/")
def create_job(job: JobCreate):
try:
job_id = add_job(job.task_name, job.cron_expression, job.args, job.kwargs)
return {"id": job_id, "message": "Job scheduled"}
except ValueError as e:
raise HTTPException(status_code=400, detail=str(e))
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
@router.delete("/{job_id}")
def delete_job(job_id: str):
try:
remove_job(job_id)
return {"message": "Job deleted"}
except Exception:
raise HTTPException(status_code=404, detail="Job not found")
@@ -0,0 +1,34 @@
from fastapi import APIRouter
from pydantic import BaseModel
from typing import Optional
from ..config import settings
router = APIRouter()
class SettingsUpdate(BaseModel):
samba_server_ip: str
samba_username: str
samba_password: str
qbittorrent_url: Optional[str]
qbittorrent_username: Optional[str]
qbittorrent_password: Optional[str]
default_comics_path: str
default_videos_path: str
stash_enabled: Optional[bool]
stash_db_path: Optional[str]
stash_generated_path: Optional[str]
stash_remote_base: Optional[str]
stash_container_base: Optional[str]
stash_share: Optional[str]
@router.get("/")
def get_settings():
return settings.model_dump()
@router.post("/")
def update_settings(update: SettingsUpdate):
for key, value in update.model_dump().items():
if hasattr(settings, key):
setattr(settings, key, value)
settings.save_to_json()
return {"message": "Settings updated"}
@@ -0,0 +1,44 @@
from fastapi import APIRouter, Depends, BackgroundTasks, HTTPException
from sqlalchemy.orm import Session
from ..config import settings, Settings
from ..samba_manager import SambaManager
from ..database import get_db
import os
import logging
router = APIRouter()
logger = logging.getLogger(__name__)
def sync_stash_db_task():
samba_manager = SambaManager(
settings.samba_server_ip,
settings.stash_share,
settings.samba_username,
settings.samba_password,
)
try:
logger.info(f"Syncing Stash DB from {settings.stash_remote_db_path}")
with open(settings.stash_db_path, "wb") as f:
samba_manager.download_file(settings.stash_remote_db_path, f)
logger.info("Stash DB synced successfully")
except Exception as e:
logger.error(f"Stash DB sync failed: {e}")
finally:
samba_manager.close()
@router.post("/sync")
def sync_stash(background_tasks: BackgroundTasks):
background_tasks.add_task(sync_stash_db_task)
return {"message": "Stash sync started"}
@router.get("/status")
def get_stash_status():
exists = os.path.exists(settings.stash_db_path)
size = os.path.getsize(settings.stash_db_path) if exists else 0
mtime = os.path.getmtime(settings.stash_db_path) if exists else 0
return {
"enabled": settings.stash_enabled,
"db_exists": exists,
"db_size": size,
"db_mtime": mtime
}
@@ -0,0 +1,60 @@
from fastapi import APIRouter
import psutil
import time
import os
router = APIRouter()
def get_size(bytes, suffix="B"):
"""
Scale bytes to its proper format
e.g:
1253656 => '1.20MB'
1253656678 => '1.17GB'
"""
factor = 1024
for unit in ["", "K", "M", "G", "T", "P"]:
if bytes < factor:
return f"{bytes:.2f}{unit}{suffix}"
bytes /= factor
@router.get("/stats")
def get_system_stats():
# CPU
cpu_usage = psutil.cpu_percent(interval=1)
# Memory
svmem = psutil.virtual_memory()
memory_stats = {
"total": get_size(svmem.total),
"available": get_size(svmem.available),
"percent": svmem.percent,
"used": get_size(svmem.used)
}
# Disk (Root)
try:
partition = psutil.disk_usage("/")
disk_stats = {
"total": get_size(partition.total),
"free": get_size(partition.free),
"percent": partition.percent,
"used": get_size(partition.used)
}
except Exception:
disk_stats = {"error": "Unavailable"}
# Uptime
boot_time = psutil.boot_time()
uptime_seconds = time.time() - boot_time
uptime_string = time.strftime("%H:%M:%S", time.gmtime(uptime_seconds))
if uptime_seconds > 86400:
days = int(uptime_seconds // 86400)
uptime_string = f"{days} days, " + uptime_string
return {
"cpu": cpu_usage,
"memory": memory_stats,
"disk": disk_stats,
"uptime": uptime_string
}
@@ -0,0 +1,10 @@
from fastapi import APIRouter, Depends
from sqlalchemy.orm import Session
from ..database import get_db
from ..models import TaskHistory
router = APIRouter()
@router.get("/")
def get_tasks(limit: int = 50, db: Session = Depends(get_db)):
return db.query(TaskHistory).order_by(TaskHistory.start_time.desc()).limit(limit).all()
@@ -35,13 +35,54 @@ class SambaManager:
"name": f.filename,
"is_directory": f.isDirectory,
"size": f.file_size,
"path": full_item_path
"path": full_item_path,
"last_modified": f.last_write_time
})
return items
except Exception as e:
logger.error(f"Error listing path '{current_dir_path}' on share '{self.share_name}': {e}", exc_info=True)
return []
def search_files(self, query, start_path="/"):
"""
Recursively searches for files matching query (partial, case-insensitive).
"""
matches = []
try:
stack = [start_path]
while stack:
current_path = stack.pop()
try:
files = self.conn.listPath(self.share_name, current_path)
except Exception:
continue # Skip unreadable dirs
for f in files:
if f.filename in ['.', '..']: continue
normalized_current_path = current_path
if normalized_current_path != '/' and not normalized_current_path.endswith('/'):
normalized_current_path += '/'
full_item_path = f"{normalized_current_path}{f.filename}".replace("\\", "/")
if query.lower() in f.filename.lower():
matches.append({
"name": f.filename,
"is_directory": f.isDirectory,
"size": f.file_size,
"path": full_item_path,
"last_modified": f.last_write_time
})
if f.isDirectory:
stack.append(full_item_path)
except Exception as e:
logger.error(f"Error searching path '{start_path}': {e}")
return matches
def delete_file(self, path):
try:
@@ -63,6 +104,8 @@ class SambaManager:
self.conn.retrieveFile(self.share_name, path, file_obj)
return {"success": True}
except Exception as e:
if "Scan canceled by user" in str(e):
raise e
return {"error": str(e)}
def download_file_range(self, path, file_obj, offset, max_length):
@@ -0,0 +1,68 @@
from apscheduler.schedulers.background import BackgroundScheduler
from apscheduler.jobstores.sqlalchemy import SQLAlchemyJobStore
from .database import engine
from .tasks import task_organize_comics, task_update_metadata, task_sort_by_artist
import logging
logger = logging.getLogger(__name__)
# Map readable names to Huey task functions
TASK_MAP = {
"organize_comics": task_organize_comics,
"update_metadata": task_update_metadata,
"sort_by_artist": task_sort_by_artist
}
jobstores = {
'default': SQLAlchemyJobStore(engine=engine)
}
scheduler = BackgroundScheduler(jobstores=jobstores)
def start_scheduler():
try:
scheduler.start()
logger.info("Scheduler started.")
except Exception as e:
logger.error(f"Failed to start scheduler: {e}")
def add_job(task_name, cron_str, args=None, kwargs=None):
if task_name not in TASK_MAP:
raise ValueError(f"Unknown task: {task_name}")
func = TASK_MAP[task_name]
# Parse cron string (e.g. "* * * * *") -> minute, hour, day, month, day_of_week
# Simple implementation: expect "min hour day month day_of_week"
try:
parts = cron_str.split()
if len(parts) != 5:
raise ValueError("Invalid cron format. Expected 5 fields.")
trigger_args = {
'minute': parts[0],
'hour': parts[1],
'day': parts[2],
'month': parts[3],
'day_of_week': parts[4]
}
job = scheduler.add_job(func, 'cron', args=args, kwargs=kwargs, **trigger_args)
return job.id
except Exception as e:
logger.error(f"Failed to add job: {e}")
raise e
def get_jobs():
jobs = []
for job in scheduler.get_jobs():
jobs.append({
"id": job.id,
"name": job.name,
"next_run": str(job.next_run_time),
"trigger": str(job.trigger)
})
return jobs
def remove_job(job_id):
scheduler.remove_job(job_id)
@@ -0,0 +1,88 @@
import sqlite3
import os
import logging
from .config import Settings
logger = logging.getLogger(__name__)
class StashService:
def __init__(self, settings: Settings):
self.settings = settings
self.db_path = settings.stash_db_path
def get_db_connection(self):
if not os.path.exists(self.db_path):
logger.warning(f"Stash database not found at {self.db_path}")
return None
return sqlite3.connect(self.db_path)
def translate_to_stash_path(self, local_path):
"""
Translates a local/SMB path to Stash internal container path.
Example: /media/videos/Girl/Scene.mp4 -> /data/Girl/Scene.mp4
"""
remote_base = self.settings.stash_remote_base.rstrip('/')
container_base = self.settings.stash_container_base.rstrip('/')
if local_path.startswith(remote_base):
return local_path.replace(remote_base, container_base, 1)
# If it doesn't start with remote_base, maybe it's already relative or formatted differently?
# Stash also uses basenames in 'files' table.
return local_path
def get_file_metadata(self, local_path):
"""
Returns (phash, oshash, scene_id, duration) from Stash DB for a given file.
"""
stash_path = self.translate_to_stash_path(local_path)
conn = self.get_db_connection()
if not conn:
return None, None, None, None
try:
cursor = conn.cursor()
cursor.execute("SELECT id FROM files WHERE path = ?", (stash_path,))
row = cursor.fetchone()
if not row:
return None, None, None, None
file_id = row[0]
# Get fingerprints
cursor.execute("SELECT type, fingerprint FROM files_fingerprints WHERE file_id = ?", (file_id,))
fingerprints = cursor.fetchall()
phash = None
oshash = None
for f_type, f_val in fingerprints:
if f_type == 'phash':
phash = hex(int(f_val) & 0xffffffffffffffff)[2:].zfill(16)
elif f_type == 'oshash':
oshash = f_val
# Get scene_id
cursor.execute("SELECT scene_id FROM scenes_files WHERE file_id = ?", (file_id,))
scene_row = cursor.fetchone()
scene_id = scene_row[0] if scene_row else None
# Get duration
cursor.execute("SELECT duration FROM video_files WHERE file_id = ?", (file_id,))
dur_row = cursor.fetchone()
duration = dur_row[0] if dur_row else None
return phash, oshash, scene_id, duration
except Exception as e:
logger.error(f"Failed to query Stash metadata for {local_path}: {e}")
return None, None, None, None
finally:
conn.close()
def get_sprite_path(self, oshash):
"""
Returns the remote SMB path for the sprite.
Example: /media/stashapp/generated/vtt/{oshash}_sprite.jpg
"""
if not oshash: return None
return f"{self.settings.stash_generated_path}/vtt/{oshash}_sprite.jpg"
@@ -0,0 +1,30 @@
from .database import SessionLocal
from .models import TaskHistory
import datetime
def record_task_start(task_id, name, details=None):
db = SessionLocal()
try:
task = TaskHistory(task_id=str(task_id), name=name, status="running", details=str(details) if details else None)
db.add(task)
db.commit()
except Exception as e:
print(f"Error recording task start: {e}")
finally:
db.close()
def record_task_end(task_id, status, details=None):
db = SessionLocal()
try:
task = db.query(TaskHistory).filter_by(task_id=str(task_id)).first()
if task:
task.status = status
task.end_time = datetime.datetime.utcnow()
if details:
# Append details if existing? Or overwrite? Overwrite for now or simpler append
task.details = str(details)
db.commit()
except Exception as e:
print(f"Error recording task end: {e}")
finally:
db.close()
+83
View File
@@ -0,0 +1,83 @@
from huey import SqliteHuey
import os
import logging
from .comics_manager import ComicsManager
from .samba_manager import SambaManager
from .config import settings
from .task_tracker import record_task_start, record_task_end
# Ensure dir exists
os.makedirs("resources/config", exist_ok=True)
# Configure Huey with SQLite backend for persistence
huey = SqliteHuey(filename="resources/config/tasks.db")
logger = logging.getLogger('huey')
def get_samba_manager(share="isolation"):
return SambaManager(
settings.samba_server_ip,
share,
settings.samba_username,
settings.samba_password,
)
@huey.task(context=True)
def task_organize_comics(task=None):
record_task_start(task.id, "Organize Comics")
logger.info(f"Task {task.id}: Organize Comics started")
manager = ComicsManager()
samba_manager = get_samba_manager("isolation")
try:
manager.organize_comics(samba_manager)
record_task_end(task.id, "success")
except Exception as e:
logger.error(f"Task Organize Comics failed: {e}")
record_task_end(task.id, "failed", str(e))
finally:
samba_manager.close()
@huey.task(context=True)
def task_move_comics(series_names, share, path, task=None):
record_task_start(task.id, "Move Comics", f"Count: {len(series_names)} -> {share}:{path}")
logger.info(f"Task {task.id}: Move Comics started")
manager = ComicsManager()
samba_manager = get_samba_manager("isolation")
try:
manager.move_series(samba_manager, series_names, share, path)
record_task_end(task.id, "success")
except Exception as e:
logger.error(f"Task Move Comics failed: {e}")
record_task_end(task.id, "failed", str(e))
finally:
samba_manager.close()
@huey.task(context=True)
def task_update_metadata(target_path, force, task=None):
record_task_start(task.id, "Update Metadata", f"Path: {target_path}")
logger.info(f"Task {task.id}: Update Metadata started")
manager = ComicsManager()
samba_manager = get_samba_manager("isolation")
try:
manager.update_existing_metadata(samba_manager, target_path, force)
record_task_end(task.id, "success")
except Exception as e:
logger.error(f"Task Update Metadata failed: {e}")
record_task_end(task.id, "failed", str(e))
finally:
samba_manager.close()
@huey.task(context=True)
def task_sort_by_artist(root_path, task=None):
record_task_start(task.id, "Sort by Artist", f"Path: {root_path}")
logger.info(f"Task {task.id}: Sort by Artist started")
manager = ComicsManager()
samba_manager = get_samba_manager("isolation")
try:
manager.sort_by_artist(samba_manager, root_path)
record_task_end(task.id, "success")
except Exception as e:
logger.error(f"Task Sort by Artist failed: {e}")
record_task_end(task.id, "failed", str(e))
finally:
samba_manager.close()
File diff suppressed because it is too large Load Diff
@@ -0,0 +1 @@
{"status": "scanning", "current": "/videos/jav/incest/HUNTC/hhd800.com@HUNTC-328.mp4", "processed": 4309}
File diff suppressed because it is too large Load Diff
+19
View File
@@ -0,0 +1,19 @@
from app.database import engine
from sqlalchemy import text
def migrate():
with engine.connect() as conn:
try:
conn.execute(text("ALTER TABLE video_files ADD COLUMN contact_sheet_path VARCHAR"))
print("Added contact_sheet_path")
except Exception as e:
print(f"Skipped contact_sheet_path (probably exists): {e}")
try:
conn.execute(text("ALTER TABLE video_files ADD COLUMN scene_phash VARCHAR"))
print("Added scene_phash")
except Exception as e:
print(f"Skipped scene_phash (probably exists): {e}")
if __name__ == "__main__":
migrate()
+3 -1
View File
@@ -9,4 +9,6 @@ pydantic-settings
requests
pillow
beautifulsoup4
cloudscraper
cloudscraper
huey
qbittorrent-api