feat: Add Performer Merge Studio, Missing Metadata Radar, Tag Normalizer & Post-Move Pipeline Automation

This commit is contained in:
david
2026-08-18 10:33:57 -04:00
parent 1a0e7f619a
commit baf3a927a4
2 changed files with 837 additions and 0 deletions
+298
View File
@@ -801,8 +801,14 @@ class RobustProxyHandler(http.server.SimpleHTTPRequestHandler):
self.handle_stash_duplicates()
elif self.path.startswith('/api/stash/plugins'):
self.handle_stash_plugins_list()
elif self.path.startswith('/api/stash/performers/duplicates'):
self.handle_stash_performer_duplicates()
elif self.path.startswith('/api/stash/performers/missing-images'):
self.handle_stash_missing_images()
elif self.path.startswith('/api/stash/scenes/missing-metadata'):
self.handle_stash_missing_metadata()
elif self.path.startswith('/api/stash/tags/duplicates'):
self.handle_stash_tag_duplicates()
# SMB API
elif self.path.startswith('/api/smb/status'):
self.handle_smb_status()
@@ -866,12 +872,18 @@ class RobustProxyHandler(http.server.SimpleHTTPRequestHandler):
self.handle_stash_run_plugin_task()
elif self.path == '/api/stash/scenes/destroy':
self.handle_stash_scene_destroy()
elif self.path == '/api/stash/scenes/batch-autotag':
self.handle_stash_batch_autotag()
elif self.path == '/api/stash/performers/find-images':
self.handle_stash_find_images()
elif self.path == '/api/stash/performers/commit-image':
self.handle_stash_commit_image()
elif self.path == '/api/stash/performers/batch-commit-images':
self.handle_stash_batch_commit_images()
elif self.path == '/api/stash/performers/merge':
self.handle_stash_performer_merge()
elif self.path == '/api/stash/tags/merge':
self.handle_stash_tag_merge()
# SMB API Moves, Deletes & Cleanups
elif self.path == '/api/smb/move':
self.handle_smb_move()
@@ -2314,6 +2326,292 @@ class RobustProxyHandler(http.server.SimpleHTTPRequestHandler):
except Exception as e:
self.send_json_response(500, {"error": str(e)})
# ==================== Performer Deduplication & Merge Handlers ====================
def handle_stash_performer_duplicates(self):
parsed = urllib.parse.urlparse(self.path)
params = urllib.parse.parse_qs(parsed.query)
stash_url = params.get('stash_url', [DEFAULT_STASH_URL])[0]
query = '''
{
findPerformers(filter: { per_page: -1 }) {
count
performers {
id
name
gender
scene_count
image_path
alias_list
stash_ids {
endpoint
stash_id
}
}
}
}
'''
try:
res = execute_stash_graphql(query, stash_url, timeout=30)
performers = res.get('data', {}).get('findPerformers', {}).get('performers') or []
# Group by normalized clean name
clusters = {}
for p in performers:
raw_name = p.get('name') or ''
# Normalize: remove dots, underscores, dashes, trailing roman numerals/numbers
norm = re.sub(r'\s*\([ivx0-9]+\)\s*$', '', raw_name, flags=re.IGNORECASE)
norm = re.sub(r'[\._\-]+', ' ', norm).strip().lower()
norm_compact = re.sub(r'[^a-z0-9]', '', norm)
if not norm_compact:
continue
if norm_compact not in clusters:
clusters[norm_compact] = []
clusters[norm_compact].append(p)
# Filter only clusters with > 1 performer
duplicate_clusters = []
for k, group in clusters.items():
if len(group) > 1:
# Sort so performer with most scenes / image is primary
sorted_group = sorted(group, key=lambda x: (
bool(x.get('image_path')),
x.get('scene_count') or 0,
len(x.get('stash_ids') or [])
), reverse=True)
duplicate_clusters.append({
"normalized_name": k,
"primary": sorted_group[0],
"candidates": sorted_group[1:],
"all": sorted_group,
"count": len(sorted_group)
})
self.send_json_response(200, {
"clusters": duplicate_clusters,
"cluster_count": len(duplicate_clusters),
"total_performers": len(performers)
})
except Exception as e:
self.send_json_response(500, {"error": str(e), "clusters": []})
def handle_stash_performer_merge(self):
content_length = int(self.headers.get('Content-Length', 0))
post_data = self.rfile.read(content_length)
try:
payload = json.loads(post_data.decode('utf-8'))
stash_url = payload.get('stash_url') or DEFAULT_STASH_URL
source_ids = [str(sid) for sid in payload.get('source_ids', [])]
dest_id = str(payload.get('destination_id'))
if not source_ids or not dest_id:
self.send_json_response(400, {"error": "Missing source_ids or destination_id"})
return
mutation = '''
mutation PerformerMerge($input: PerformerMergeInput!) {
performerMerge(input: $input) {
id
name
scene_count
}
}
'''
variables = {
"input": {
"source": source_ids,
"destination": dest_id
}
}
res = execute_stash_graphql(mutation, stash_url, variables=variables, timeout=20)
self.send_json_response(200, {"success": True, "merged_performer": res.get('data', {}).get('performerMerge')})
except Exception as e:
self.send_json_response(500, {"error": str(e)})
# ==================== Missing Metadata Radar Handlers ====================
def handle_stash_missing_metadata(self):
parsed = urllib.parse.urlparse(self.path)
params = urllib.parse.parse_qs(parsed.query)
stash_url = params.get('stash_url', [DEFAULT_STASH_URL])[0]
missing_type = params.get('type', ['studio'])[0]
page = int(params.get('page', [1])[0])
per_page = int(params.get('per_page', [24])[0])
query = '''
query FindMissingMetadataScenes($scene_filter: SceneFilterType, $filter: FindFilterType) {
findScenes(scene_filter: $scene_filter, filter: $filter) {
count
scenes {
id
title
date
rating100
paths {
screenshot
preview
}
files {
id
path
size
duration
video_codec
width
height
}
performers {
id
name
image_path
}
studio {
id
name
}
tags {
id
name
}
}
}
}
'''
try:
variables = {
"scene_filter": {"is_missing": missing_type},
"filter": {"per_page": per_page, "page": page}
}
res = execute_stash_graphql(query, stash_url, variables=variables, timeout=20)
data = res.get('data', {}).get('findScenes') or {}
self.send_json_response(200, {
"type": missing_type,
"count": data.get('count', 0),
"scenes": data.get('scenes', []),
"page": page,
"per_page": per_page
})
except Exception as e:
self.send_json_response(500, {"error": str(e), "scenes": [], "count": 0})
def handle_stash_batch_autotag(self):
content_length = int(self.headers.get('Content-Length', 0))
post_data = self.rfile.read(content_length)
try:
payload = json.loads(post_data.decode('utf-8'))
stash_url = payload.get('stash_url') or DEFAULT_STASH_URL
paths = payload.get('paths', [])
tag_performers = bool(payload.get('performers', True))
tag_studios = bool(payload.get('studios', True))
tag_tags = bool(payload.get('tags', True))
mutation = '''
mutation AutoTagScenes($input: AutoTagMetadataInput!) {
metadataAutoTag(input: $input)
}
'''
variables = {
"input": {
"paths": paths if paths else [],
"performers": tag_performers,
"studios": tag_studios,
"tags": tag_tags
}
}
res = execute_stash_graphql(mutation, stash_url, variables=variables, timeout=20)
self.send_json_response(200, {"success": True, "job_id": res.get('data', {}).get('metadataAutoTag')})
except Exception as e:
self.send_json_response(500, {"error": str(e)})
# ==================== Tag Normalizer & Deduplication Handlers ====================
def handle_stash_tag_duplicates(self):
parsed = urllib.parse.urlparse(self.path)
params = urllib.parse.parse_qs(parsed.query)
stash_url = params.get('stash_url', [DEFAULT_STASH_URL])[0]
query = '''
{
findTags(filter: { per_page: -1 }) {
count
tags {
id
name
scene_count
image_path
aliases
}
}
}
'''
try:
res = execute_stash_graphql(query, stash_url, timeout=20)
tags = res.get('data', {}).get('findTags', {}).get('tags') or []
clusters = {}
for t in tags:
raw_name = t.get('name') or ''
norm = re.sub(r'[\._\-]+', ' ', raw_name).strip().lower()
norm_compact = re.sub(r'[^a-z0-9]', '', norm)
if not norm_compact:
continue
if norm_compact not in clusters:
clusters[norm_compact] = []
clusters[norm_compact].append(t)
duplicate_clusters = []
for k, group in clusters.items():
if len(group) > 1:
sorted_group = sorted(group, key=lambda x: (x.get('scene_count') or 0), reverse=True)
duplicate_clusters.append({
"normalized_name": k,
"primary": sorted_group[0],
"candidates": sorted_group[1:],
"all": sorted_group,
"count": len(sorted_group)
})
self.send_json_response(200, {
"clusters": duplicate_clusters,
"cluster_count": len(duplicate_clusters),
"total_tags": len(tags)
})
except Exception as e:
self.send_json_response(500, {"error": str(e), "clusters": []})
def handle_stash_tag_merge(self):
content_length = int(self.headers.get('Content-Length', 0))
post_data = self.rfile.read(content_length)
try:
payload = json.loads(post_data.decode('utf-8'))
stash_url = payload.get('stash_url') or DEFAULT_STASH_URL
source_ids = [str(sid) for sid in payload.get('source_ids', [])]
dest_id = str(payload.get('destination_id'))
if not source_ids or not dest_id:
self.send_json_response(400, {"error": "Missing source_ids or destination_id"})
return
mutation = '''
mutation TagsMerge($input: TagsMergeInput!) {
tagsMerge(input: $input) {
id
name
scene_count
}
}
'''
variables = {
"input": {
"source": source_ids,
"destination": dest_id
}
}
res = execute_stash_graphql(mutation, stash_url, variables=variables, timeout=20)
self.send_json_response(200, {"success": True, "merged_tag": res.get('data', {}).get('tagsMerge')})
except Exception as e:
self.send_json_response(500, {"error": str(e)})
# ==================== SMB Maintenance & Cleanup Handlers ====================
def handle_smb_scan_cleanup(self):
parsed = urllib.parse.urlparse(self.path)