Implement distributed cluster upscaling, custom output dirs, date-based naming, active preemption, and SMB mappings

- Add Output Destination folder selector with file browser and fallback to outputs/
- Apply YYYY-MM-DD start-date suffix naming rule to upscaled outputs
- Implement active job preemption (demoting active running job back to queue) via reorder
- Implement Coordinator / Worker roles with automatic pull-based chunk processing
- Support Shared Storage mode and HTTP-based frame upload/download fallback
- Add SMB mapped share configuration and background CIFS mount support
- Update index.html and app.js with the new Settings sections and modals
This commit is contained in:
2026-06-29 13:13:08 -04:00
parent f662c5806f
commit f610c86f8b
4 changed files with 1228 additions and 46 deletions
+108 -5
View File
@@ -7,6 +7,9 @@ import time
import threading
from typing import Dict, Any, Callable
dist_chunks: Dict[str, list] = {}
dist_chunks_lock = threading.Lock()
_ffmpeg_filters_cache = {}
def has_ffmpeg_filter(filter_name: str) -> bool:
global _ffmpeg_filters_cache
@@ -39,9 +42,12 @@ class UpscaleJob:
denoise: bool = False, sharpen: bool = False, interpolation: bool = False,
webhook_url: str = None, transcode_format: str = "mp4", is_preview: bool = False,
ai_face_restoration: bool = False, ai_rife_interpolation: bool = False,
ai_audio_denoise: bool = False, temp_dir: str = None):
ai_audio_denoise: bool = False, temp_dir: str = None,
output_dir: str = None, source_filename: str = None):
self.job_id = job_id
self.temp_dir = temp_dir
self.output_dir = output_dir
self.source_filename = source_filename
self.video_path = video_path
self.model = model
self.scale = scale
@@ -404,8 +410,45 @@ def run_upscale_pipeline(job: UpscaleJob, on_progress_update: Callable[[str, Dic
job.update_status("upscaling", progress=20, current_frame=actual_total - remaining_inputs)
on_progress_update(job.job_id, {"status": "upscaling", "progress": 20, "current_frame": actual_total - remaining_inputs, "total_frames": actual_total})
# Check if role in settings is "coordinator"
settings = {}
main_mod = sys.modules.get("app.main")
if main_mod and hasattr(main_mod, "load_settings"):
try:
settings = main_mod.load_settings()
except Exception:
pass
if not settings:
settings_path = os.path.join(BASE_DIR, "settings.json")
if os.path.exists(settings_path):
try:
with open(settings_path, "r") as f:
settings = json.load(f)
except Exception:
pass
is_coordinator = settings.get("role") == "coordinator"
skip_local_upscale = False
if is_coordinator:
skip_local_upscale = True
files_to_upscale = sorted([f for f in os.listdir(input_frames_dir) if f.startswith("frame_")])
chunk_size = 50
chunks_list = []
for idx_chunk, i in enumerate(range(0, len(files_to_upscale), chunk_size)):
chunk_files = files_to_upscale[i : i + chunk_size]
chunks_list.append({
"chunk_id": f"{job.job_id}_{idx_chunk}",
"files": chunk_files,
"status": "pending",
"worker_url": None,
"updated_at": time.time()
})
with dist_chunks_lock:
dist_chunks[job.job_id] = chunks_list
print(f"[{time.strftime('%Y-%m-%d %H:%M:%S')}] [Job {job.job_id}] Coordinator mode active. Initialized {len(chunks_list)} chunks.")
current_tile_size = job.tile_size
while True:
while not skip_local_upscale:
# Launch Real-ESRGAN on directory
upscale_cmd = [
BIN_PATH,
@@ -509,6 +552,57 @@ def run_upscale_pipeline(job: UpscaleJob, on_progress_update: Callable[[str, Dic
else:
break
# If coordinator, wait for all chunks to be completed
if is_coordinator:
print(f"[{time.strftime('%Y-%m-%d %H:%M:%S')}] [Job {job.job_id}] Coordinator waiting for all chunks to complete...")
while True:
if job._is_cancelled:
return
if getattr(job, "_is_paused", False) or job.status == "paused":
return
with dist_chunks_lock:
chunks = dist_chunks.get(job.job_id, [])
if not chunks:
break
all_done = all(c["status"] == "completed" for c in chunks)
completed_count = sum(1 for c in chunks if c["status"] == "completed")
total_chunks = len(chunks)
upscale_progress = 20.0
if total_chunks > 0:
upscale_progress += (completed_count / total_chunks) * 60.0
processed_files = len(os.listdir(output_frames_dir))
job.update_status(
"upscaling",
progress=upscale_progress,
current_frame=processed_files,
eta=f"Waiting for workers... Chunks: {completed_count}/{total_chunks}"
)
on_progress_update(job.job_id, {
"status": "upscaling",
"progress": upscale_progress,
"current_frame": processed_files,
"total_frames": actual_total,
"eta": f"Workers processing chunks: {completed_count}/{total_chunks}"
})
if all_done:
break
# Timeout check: reset chunk if assigned but no update in 60s
with dist_chunks_lock:
for c in chunks:
if c["status"] == "assigned" and time.time() - c["updated_at"] > 60:
print(f"Chunk {c['chunk_id']} timed out. Requeuing.")
c["status"] = "pending"
c["worker_url"] = None
c["updated_at"] = time.time()
time.sleep(1.0)
# Final validation of upscale output
processed_files = len(os.listdir(output_frames_dir))
job.update_status("upscaling", progress=80.0, current_frame=processed_files)
@@ -599,9 +693,18 @@ def run_upscale_pipeline(job: UpscaleJob, on_progress_update: Callable[[str, Dic
job.update_status("assembling", progress=85.0)
on_progress_update(job.job_id, {"status": "assembling", "progress": 85.0})
# Get source filename basename, append with _upscaled_YYYY-MM-DD
source_file = job.source_filename if getattr(job, "source_filename", None) else job.video_path
base_name = os.path.basename(source_file)
name_without_ext, _ = os.path.splitext(base_name)
date_str = time.strftime("%Y-%m-%d")
transcode_fmt = getattr(job, "transcode_format", "mp4")
out_filename = f"upscaled_{job.job_id}.{transcode_fmt}"
out_filepath = os.path.join(OUTPUT_DIR, out_filename)
out_filename = f"{name_without_ext}_upscaled_{date_str}.{transcode_fmt}"
output_dir = job.output_dir if getattr(job, "output_dir", None) else OUTPUT_DIR
os.makedirs(output_dir, exist_ok=True)
out_filepath = os.path.join(output_dir, out_filename)
job.output_file = out_filepath
# Choose codecs based on format
@@ -697,7 +800,7 @@ def run_upscale_pipeline(job: UpscaleJob, on_progress_update: Callable[[str, Dic
except Exception as e:
import traceback
traceback.print_exc()
if not job._is_cancelled and not getattr(job, "_is_paused", False) and job.status != "paused":
if not job._is_cancelled and not getattr(job, "_is_paused", False) and job.status not in ["paused", "queued"] and "paused" not in str(e).lower():
job.update_status("failed", error=str(e))
on_progress_update(job.job_id, {"status": "failed", "error": str(e)})
print(f"[{time.strftime('%Y-%m-%d %H:%M:%S')}] [Job {job.job_id}] pipeline failed. Error: {e}")