Implement distributed cluster upscaling, custom output dirs, date-based naming, active preemption, and SMB mappings
- Add Output Destination folder selector with file browser and fallback to outputs/ - Apply YYYY-MM-DD start-date suffix naming rule to upscaled outputs - Implement active job preemption (demoting active running job back to queue) via reorder - Implement Coordinator / Worker roles with automatic pull-based chunk processing - Support Shared Storage mode and HTTP-based frame upload/download fallback - Add SMB mapped share configuration and background CIFS mount support - Update index.html and app.js with the new Settings sections and modals
This commit is contained in:
+108
-5
@@ -7,6 +7,9 @@ import time
|
||||
import threading
|
||||
from typing import Dict, Any, Callable
|
||||
|
||||
dist_chunks: Dict[str, list] = {}
|
||||
dist_chunks_lock = threading.Lock()
|
||||
|
||||
_ffmpeg_filters_cache = {}
|
||||
def has_ffmpeg_filter(filter_name: str) -> bool:
|
||||
global _ffmpeg_filters_cache
|
||||
@@ -39,9 +42,12 @@ class UpscaleJob:
|
||||
denoise: bool = False, sharpen: bool = False, interpolation: bool = False,
|
||||
webhook_url: str = None, transcode_format: str = "mp4", is_preview: bool = False,
|
||||
ai_face_restoration: bool = False, ai_rife_interpolation: bool = False,
|
||||
ai_audio_denoise: bool = False, temp_dir: str = None):
|
||||
ai_audio_denoise: bool = False, temp_dir: str = None,
|
||||
output_dir: str = None, source_filename: str = None):
|
||||
self.job_id = job_id
|
||||
self.temp_dir = temp_dir
|
||||
self.output_dir = output_dir
|
||||
self.source_filename = source_filename
|
||||
self.video_path = video_path
|
||||
self.model = model
|
||||
self.scale = scale
|
||||
@@ -404,8 +410,45 @@ def run_upscale_pipeline(job: UpscaleJob, on_progress_update: Callable[[str, Dic
|
||||
job.update_status("upscaling", progress=20, current_frame=actual_total - remaining_inputs)
|
||||
on_progress_update(job.job_id, {"status": "upscaling", "progress": 20, "current_frame": actual_total - remaining_inputs, "total_frames": actual_total})
|
||||
|
||||
# Check if role in settings is "coordinator"
|
||||
settings = {}
|
||||
main_mod = sys.modules.get("app.main")
|
||||
if main_mod and hasattr(main_mod, "load_settings"):
|
||||
try:
|
||||
settings = main_mod.load_settings()
|
||||
except Exception:
|
||||
pass
|
||||
if not settings:
|
||||
settings_path = os.path.join(BASE_DIR, "settings.json")
|
||||
if os.path.exists(settings_path):
|
||||
try:
|
||||
with open(settings_path, "r") as f:
|
||||
settings = json.load(f)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
is_coordinator = settings.get("role") == "coordinator"
|
||||
skip_local_upscale = False
|
||||
if is_coordinator:
|
||||
skip_local_upscale = True
|
||||
files_to_upscale = sorted([f for f in os.listdir(input_frames_dir) if f.startswith("frame_")])
|
||||
chunk_size = 50
|
||||
chunks_list = []
|
||||
for idx_chunk, i in enumerate(range(0, len(files_to_upscale), chunk_size)):
|
||||
chunk_files = files_to_upscale[i : i + chunk_size]
|
||||
chunks_list.append({
|
||||
"chunk_id": f"{job.job_id}_{idx_chunk}",
|
||||
"files": chunk_files,
|
||||
"status": "pending",
|
||||
"worker_url": None,
|
||||
"updated_at": time.time()
|
||||
})
|
||||
with dist_chunks_lock:
|
||||
dist_chunks[job.job_id] = chunks_list
|
||||
print(f"[{time.strftime('%Y-%m-%d %H:%M:%S')}] [Job {job.job_id}] Coordinator mode active. Initialized {len(chunks_list)} chunks.")
|
||||
|
||||
current_tile_size = job.tile_size
|
||||
while True:
|
||||
while not skip_local_upscale:
|
||||
# Launch Real-ESRGAN on directory
|
||||
upscale_cmd = [
|
||||
BIN_PATH,
|
||||
@@ -509,6 +552,57 @@ def run_upscale_pipeline(job: UpscaleJob, on_progress_update: Callable[[str, Dic
|
||||
else:
|
||||
break
|
||||
|
||||
# If coordinator, wait for all chunks to be completed
|
||||
if is_coordinator:
|
||||
print(f"[{time.strftime('%Y-%m-%d %H:%M:%S')}] [Job {job.job_id}] Coordinator waiting for all chunks to complete...")
|
||||
while True:
|
||||
if job._is_cancelled:
|
||||
return
|
||||
if getattr(job, "_is_paused", False) or job.status == "paused":
|
||||
return
|
||||
|
||||
with dist_chunks_lock:
|
||||
chunks = dist_chunks.get(job.job_id, [])
|
||||
if not chunks:
|
||||
break
|
||||
|
||||
all_done = all(c["status"] == "completed" for c in chunks)
|
||||
completed_count = sum(1 for c in chunks if c["status"] == "completed")
|
||||
total_chunks = len(chunks)
|
||||
|
||||
upscale_progress = 20.0
|
||||
if total_chunks > 0:
|
||||
upscale_progress += (completed_count / total_chunks) * 60.0
|
||||
|
||||
processed_files = len(os.listdir(output_frames_dir))
|
||||
job.update_status(
|
||||
"upscaling",
|
||||
progress=upscale_progress,
|
||||
current_frame=processed_files,
|
||||
eta=f"Waiting for workers... Chunks: {completed_count}/{total_chunks}"
|
||||
)
|
||||
on_progress_update(job.job_id, {
|
||||
"status": "upscaling",
|
||||
"progress": upscale_progress,
|
||||
"current_frame": processed_files,
|
||||
"total_frames": actual_total,
|
||||
"eta": f"Workers processing chunks: {completed_count}/{total_chunks}"
|
||||
})
|
||||
|
||||
if all_done:
|
||||
break
|
||||
|
||||
# Timeout check: reset chunk if assigned but no update in 60s
|
||||
with dist_chunks_lock:
|
||||
for c in chunks:
|
||||
if c["status"] == "assigned" and time.time() - c["updated_at"] > 60:
|
||||
print(f"Chunk {c['chunk_id']} timed out. Requeuing.")
|
||||
c["status"] = "pending"
|
||||
c["worker_url"] = None
|
||||
c["updated_at"] = time.time()
|
||||
|
||||
time.sleep(1.0)
|
||||
|
||||
# Final validation of upscale output
|
||||
processed_files = len(os.listdir(output_frames_dir))
|
||||
job.update_status("upscaling", progress=80.0, current_frame=processed_files)
|
||||
@@ -599,9 +693,18 @@ def run_upscale_pipeline(job: UpscaleJob, on_progress_update: Callable[[str, Dic
|
||||
job.update_status("assembling", progress=85.0)
|
||||
on_progress_update(job.job_id, {"status": "assembling", "progress": 85.0})
|
||||
|
||||
# Get source filename basename, append with _upscaled_YYYY-MM-DD
|
||||
source_file = job.source_filename if getattr(job, "source_filename", None) else job.video_path
|
||||
base_name = os.path.basename(source_file)
|
||||
name_without_ext, _ = os.path.splitext(base_name)
|
||||
|
||||
date_str = time.strftime("%Y-%m-%d")
|
||||
transcode_fmt = getattr(job, "transcode_format", "mp4")
|
||||
out_filename = f"upscaled_{job.job_id}.{transcode_fmt}"
|
||||
out_filepath = os.path.join(OUTPUT_DIR, out_filename)
|
||||
out_filename = f"{name_without_ext}_upscaled_{date_str}.{transcode_fmt}"
|
||||
|
||||
output_dir = job.output_dir if getattr(job, "output_dir", None) else OUTPUT_DIR
|
||||
os.makedirs(output_dir, exist_ok=True)
|
||||
out_filepath = os.path.join(output_dir, out_filename)
|
||||
job.output_file = out_filepath
|
||||
|
||||
# Choose codecs based on format
|
||||
@@ -697,7 +800,7 @@ def run_upscale_pipeline(job: UpscaleJob, on_progress_update: Callable[[str, Dic
|
||||
except Exception as e:
|
||||
import traceback
|
||||
traceback.print_exc()
|
||||
if not job._is_cancelled and not getattr(job, "_is_paused", False) and job.status != "paused":
|
||||
if not job._is_cancelled and not getattr(job, "_is_paused", False) and job.status not in ["paused", "queued"] and "paused" not in str(e).lower():
|
||||
job.update_status("failed", error=str(e))
|
||||
on_progress_update(job.job_id, {"status": "failed", "error": str(e)})
|
||||
print(f"[{time.strftime('%Y-%m-%d %H:%M:%S')}] [Job {job.job_id}] pipeline failed. Error: {e}")
|
||||
|
||||
Reference in New Issue
Block a user