feat: implement job resumption, custom queue reordering, and auto-venv start script setup

This commit is contained in:
2026-06-24 13:37:17 -04:00
parent 2ceb948b98
commit 212291ea94
6 changed files with 569 additions and 140 deletions
+154 -5
View File
@@ -27,21 +27,97 @@ app.add_middleware(
)
# In-memory databases
class CustomJobQueue:
def __init__(self):
self.queue = []
self.lock = threading.Lock()
self.condition = threading.Condition(self.lock)
def put(self, job_id: str):
with self.lock:
if job_id not in self.queue:
self.queue.append(job_id)
self.condition.notify()
def get(self) -> str:
with self.lock:
while not self.queue:
self.condition.wait()
return self.queue.pop(0)
def remove(self, job_id: str) -> bool:
with self.lock:
if job_id in self.queue:
self.queue.remove(job_id)
return True
return False
def get_all(self) -> List[str]:
with self.lock:
return list(self.queue)
def reorder(self, job_ids: List[str]):
with self.lock:
valid_ids = [jid for jid in job_ids if jid in self.queue]
missing_ids = [jid for jid in self.queue if jid not in valid_ids]
self.queue = valid_ids + missing_ids
def task_done(self):
pass
def empty(self) -> bool:
with self.lock:
return len(self.queue) == 0
def qsize(self) -> int:
with self.lock:
return len(self.queue)
jobs_db: Dict[str, upscaler.UpscaleJob] = {}
ws_connections: Dict[str, List[WebSocket]] = {}
preview_db: Dict[str, Dict[str, str]] = {} # preview_id -> {orig, upscaled}
# FIFO queue for upscaling jobs to prevent GPU memory overload
job_queue = queue.Queue()
# Custom thread-safe queue for upscaling jobs to support reordering & cancellation
job_queue = CustomJobQueue()
queue_lock = threading.Lock()
current_running_job_id = None
main_loop = None
JOBS_FILE = os.path.join(upscaler.BASE_DIR, "jobs.json")
def load_jobs_db():
global jobs_db
if os.path.exists(JOBS_FILE):
try:
with open(JOBS_FILE, "r") as f:
data = json.load(f)
for job_id, job_data in data.items():
job = upscaler.UpscaleJob.from_dict(job_data)
# Automatically put queued items back in the queue
if job.status == "queued":
job_queue.put(job_id)
# Mark active items as interrupted so they can be resumed
elif job.status in ["analyzing", "extracting", "upscaling", "assembling"]:
job.status = "interrupted"
job.eta = "Interrupted"
jobs_db[job_id] = job
except Exception as e:
print(f"Error loading jobs database: {e}")
def save_jobs_db():
try:
with open(JOBS_FILE, "w") as f:
data = {job_id: job.to_dict() for job_id, job in jobs_db.items()}
json.dump(data, f, indent=4)
except Exception as e:
print(f"Error saving jobs database: {e}")
@app.on_event("startup")
def startup_event():
global main_loop
main_loop = asyncio.get_event_loop()
load_jobs_db()
global_webhook_url = None
@@ -64,6 +140,7 @@ def send_webhook_notification(url: str, payload: dict):
# Broadcast updates to websockets and webhooks
def broadcast_progress(job_id: str, data: dict):
save_jobs_db()
job = jobs_db.get(job_id)
if job:
data["is_preview"] = getattr(job, "is_preview", False)
@@ -344,6 +421,7 @@ def start_upscale(req: StartUpscaleRequest):
jobs_db[job_id] = job
job_queue.put(job_id)
save_jobs_db()
# Broadcast initial queued progress
broadcast_progress(job_id, {
@@ -386,6 +464,9 @@ def cancel_job(job_id: str):
if not job:
raise HTTPException(status_code=404, detail="Job not found.")
# Remove from queue if it was queued
job_queue.remove(job_id)
job.cancel()
# Broadcast cancellation status
broadcast_progress(job_id, {
@@ -395,6 +476,7 @@ def cancel_job(job_id: str):
"total_frames": job.total_frames,
"eta": "N/A"
})
save_jobs_db()
return {"job_id": job_id, "status": "cancelled"}
@app.post("/api/preview/generate")
@@ -486,7 +568,22 @@ def safe_delete_file(file_path: str):
@app.get("/api/jobs")
def list_jobs():
"""List details of all submitted jobs"""
"""List details of all submitted jobs in queue-sorted order"""
active_id = current_running_job_id
queued_ids = job_queue.get_all()
# Sort active first, then queued in order, then history by start time descending
def get_sort_key(job):
if job.job_id == active_id:
return (0, 0)
elif job.job_id in queued_ids:
return (1, queued_ids.index(job.job_id))
else:
t = job.start_time if job.start_time is not None else 0
return (2, -t)
sorted_jobs = sorted(jobs_db.values(), key=get_sort_key)
return [
{
"job_id": job.job_id,
@@ -500,9 +597,10 @@ def list_jobs():
"scale": job.scale,
"output_file": os.path.basename(job.output_file) if job.output_file else None,
"video_path": job.video_path,
"is_preview": getattr(job, "is_preview", False)
"is_preview": getattr(job, "is_preview", False),
"queue_position": queued_ids.index(job.job_id) if job.job_id in queued_ids else -1 if job.job_id == active_id else None
}
for job in jobs_db.values()
for job in sorted_jobs
]
@app.delete("/api/jobs/{job_id}")
@@ -512,6 +610,9 @@ def delete_job(job_id: str):
if not job:
raise HTTPException(status_code=404, detail="Job not found.")
# Remove from queue if it is queued
job_queue.remove(job_id)
# Safely delete original preview video if present
for ext in [".mp4", ".mkv", ".avi", ".mov", ".webm"]:
orig_prev_path = os.path.join(upscaler.OUTPUT_DIR, f"original_{job_id}{ext}")
@@ -534,6 +635,8 @@ def delete_job(job_id: str):
# Delete from in-memory db
if job_id in jobs_db:
del jobs_db[job_id]
save_jobs_db()
return {"job_id": job_id, "status": "purged"}
@@ -562,11 +665,57 @@ def purge_all_jobs():
# Reset in-memory database
jobs_db.clear()
# Re-initialize custom queue
global job_queue
job_queue = CustomJobQueue()
# Reset upload metadata file
save_upload_metadata({})
save_jobs_db()
return {"status": "all purged"}
class ReorderQueueRequest(BaseModel):
job_ids: List[str]
@app.post("/api/queue/reorder")
def reorder_queue(req: ReorderQueueRequest):
"""Reorder the job queue"""
job_queue.reorder(req.job_ids)
save_jobs_db()
return {"status": "success", "queue": job_queue.get_all()}
@app.get("/api/queue")
def get_queue():
"""Get the current job queue order"""
return {"queue": job_queue.get_all()}
@app.post("/api/upscale/resume/{job_id}")
def resume_job(job_id: str):
"""Resume an interrupted/failed upscale job"""
job = jobs_db.get(job_id)
if not job:
raise HTTPException(status_code=404, detail="Job not found.")
# Re-queue the job
job.status = "queued"
job.error = None
job.eta = "Queued for resume..."
job_queue.put(job_id)
save_jobs_db()
broadcast_progress(job_id, {
"status": "queued",
"progress": job.progress,
"current_frame": job.current_frame,
"total_frames": job.total_frames,
"eta": "Queued for resume..."
})
return {"job_id": job_id, "status": "queued"}
# Websocket endpoint for real-time progress updates
@app.websocket("/ws/progress/{job_id}")
async def websocket_progress(websocket: WebSocket, job_id: str):
+177 -117
View File
@@ -78,6 +78,31 @@ class UpscaleJob:
self._is_cancelled = False
self._lock = threading.Lock()
def to_dict(self) -> dict:
"""Serialize job attributes, excluding internal thread/process resources."""
return {k: v for k, v in self.__dict__.items() if not k.startswith('_')}
@classmethod
def from_dict(cls, data: dict) -> 'UpscaleJob':
"""Deserialize job from dictionary, reconstructing internal locks and processes."""
job = cls(
job_id=data.get('job_id'),
video_path=data.get('video_path'),
model=data.get('model'),
scale=data.get('scale', 4),
tile_size=data.get('tile_size', 256),
preserve_audio=data.get('preserve_audio', True),
webhook_url=data.get('webhook_url'),
transcode_format=data.get('transcode_format', 'mp4'),
is_preview=data.get('is_preview', False)
)
for k, v in data.items():
setattr(job, k, v)
job._processes = []
job._is_cancelled = False
job._lock = threading.Lock()
return job
def update_status(self, status: str, progress: float = None, current_frame: int = None, eta: str = None, error: str = None):
with self._lock:
self.status = status
@@ -256,133 +281,168 @@ def run_upscale_pipeline(job: UpscaleJob, on_progress_update: Callable[[str, Dic
except Exception as cut_err:
print(f"Error cutting original preview video: {cut_err}")
# Step 1: Extract Frames
job.update_status("extracting", progress=10)
on_progress_update(job.job_id, {"status": "extracting", "progress": 10})
# Step 1: Extract Frames (Support Skipping on Resume)
skip_extraction = False
if os.path.exists(input_frames_dir):
extracted_files = sorted([f for f in os.listdir(input_frames_dir) if f.startswith("frame_")])
if len(extracted_files) > 0:
skip_extraction = True
print(f"Job {job.job_id}: Found existing input frames ({len(extracted_files)} frames). Skipping extraction step.")
job.total_frames = len(extracted_files)
# High quality JPG frames to balance disk usage and speed
extract_cmd = ["ffmpeg", "-y"]
if job.ss is not None:
extract_cmd.extend(["-ss", str(job.ss)])
if job.t is not None:
extract_cmd.extend(["-t", str(job.t)])
extract_cmd.extend(["-i", job.video_path])
# Apply unsharp pre-filter if enabled
if getattr(job, "unsharp", False):
extract_cmd.extend(["-vf", "unsharp"])
if not skip_extraction:
job.update_status("extracting", progress=10)
on_progress_update(job.job_id, {"status": "extracting", "progress": 10})
extract_cmd.extend([
"-q:v", "2",
os.path.join(input_frames_dir, "frame_%08d.jpg")
])
p_extract = job.run_command(extract_cmd)
stdout, stderr = p_extract.communicate()
job.cleanup_process(p_extract)
if p_extract.returncode != 0:
raise RuntimeError(f"FFmpeg frame extraction failed: {stderr}")
# High quality JPG frames to balance disk usage and speed
extract_cmd = ["ffmpeg", "-y"]
if job.ss is not None:
extract_cmd.extend(["-ss", str(job.ss)])
if job.t is not None:
extract_cmd.extend(["-t", str(job.t)])
extract_cmd.extend(["-i", job.video_path])
# Count actual frames extracted
extracted_files = sorted([f for f in os.listdir(input_frames_dir) if f.startswith("frame_")])
actual_total = len(extracted_files)
if actual_total == 0:
raise RuntimeError("No frames extracted from video")
job.total_frames = actual_total
# Step 2: Upscale Frames
job.update_status("upscaling", progress=20, current_frame=0)
on_progress_update(job.job_id, {"status": "upscaling", "progress": 20, "current_frame": 0, "total_frames": actual_total})
current_tile_size = job.tile_size
while True:
# Launch Real-ESRGAN on directory
upscale_cmd = [
BIN_PATH,
"-i", input_frames_dir,
"-o", output_frames_dir,
"-n", job.model,
"-s", str(job.scale),
"-t", str(current_tile_size),
"-f", "jpg"
]
if getattr(job, "gpu_ids", None) is not None:
upscale_cmd.extend(["-g", str(job.gpu_ids)])
if getattr(job, "tta", False):
upscale_cmd.append("-x")
# Apply unsharp pre-filter if enabled
if getattr(job, "unsharp", False):
extract_cmd.extend(["-vf", "unsharp"])
upscale_start_time = time.time()
p_upscale = job.run_command(upscale_cmd)
extract_cmd.extend([
"-q:v", "2",
os.path.join(input_frames_dir, "frame_%08d.jpg")
])
# Monitor thread for output files
while p_upscale.poll() is None:
p_extract = job.run_command(extract_cmd)
stdout, stderr = p_extract.communicate()
job.cleanup_process(p_extract)
if p_extract.returncode != 0:
raise RuntimeError(f"FFmpeg frame extraction failed: {stderr}")
# Count actual frames extracted
extracted_files = sorted([f for f in os.listdir(input_frames_dir) if f.startswith("frame_")])
actual_total = len(extracted_files)
if actual_total == 0:
raise RuntimeError("No frames extracted from video")
job.total_frames = actual_total
else:
actual_total = job.total_frames
# Step 2: Upscale Frames (Support Resuming by Skipping already upscaled frames)
if os.path.exists(output_frames_dir):
output_files = os.listdir(output_frames_dir)
skipped_frames = 0
for f in output_files:
if f.startswith("frame_") and f.endswith(".jpg"):
out_path = os.path.join(output_frames_dir, f)
if os.path.exists(out_path) and os.path.getsize(out_path) > 0:
in_path = os.path.join(input_frames_dir, f)
if os.path.exists(in_path):
try:
os.remove(in_path)
skipped_frames += 1
except Exception as ex:
print(f"Error removing resumed frame {in_path}: {ex}")
if skipped_frames > 0:
print(f"Job {job.job_id}: Skipping {skipped_frames} already upscaled frames.")
remaining_inputs = len(os.listdir(input_frames_dir)) if os.path.exists(input_frames_dir) else 0
if remaining_inputs == 0:
print(f"Job {job.job_id}: All frames already upscaled. Skipping upscaling step.")
job.update_status("upscaling", progress=80.0, current_frame=actual_total)
on_progress_update(job.job_id, {"status": "upscaling", "progress": 80.0, "current_frame": actual_total, "total_frames": actual_total})
else:
job.update_status("upscaling", progress=20, current_frame=actual_total - remaining_inputs)
on_progress_update(job.job_id, {"status": "upscaling", "progress": 20, "current_frame": actual_total - remaining_inputs, "total_frames": actual_total})
current_tile_size = job.tile_size
while True:
# Launch Real-ESRGAN on directory
upscale_cmd = [
BIN_PATH,
"-i", input_frames_dir,
"-o", output_frames_dir,
"-n", job.model,
"-s", str(job.scale),
"-t", str(current_tile_size),
"-f", "jpg"
]
if getattr(job, "gpu_ids", None) is not None:
upscale_cmd.extend(["-g", str(job.gpu_ids)])
if getattr(job, "tta", False):
upscale_cmd.append("-x")
upscale_start_time = time.time()
p_upscale = job.run_command(upscale_cmd)
# Monitor thread for output files
while p_upscale.poll() is None:
if job._is_cancelled:
return
processed_files = len(os.listdir(output_frames_dir))
progress_pct = 20.0 + (float(processed_files) / actual_total) * 60.0 # upscaling is 20% to 80%
# Estimate ETA
elapsed = time.time() - upscale_start_time
this_run_processed = processed_files - (actual_total - remaining_inputs)
if this_run_processed > 0:
sec_per_frame = elapsed / this_run_processed
rem_frames = actual_total - processed_files
eta_sec = rem_frames * sec_per_frame
# Format ETA
if eta_sec > 60:
eta_str = f"{int(eta_sec // 60)}m {int(eta_sec % 60)}s"
else:
eta_str = f"{int(eta_sec)}s"
else:
eta_str = "Calculating..."
job.update_status("upscaling", progress=progress_pct, current_frame=processed_files, eta=eta_str)
on_progress_update(job.job_id, {
"status": "upscaling",
"progress": progress_pct,
"current_frame": processed_files,
"total_frames": actual_total,
"eta": eta_str
})
time.sleep(0.5)
stdout, stderr = p_upscale.communicate()
job.cleanup_process(p_upscale)
if job._is_cancelled:
return
processed_files = len(os.listdir(output_frames_dir))
progress_pct = 20.0 + (float(processed_files) / actual_total) * 60.0 # upscaling is 20% to 80%
# Estimate ETA
elapsed = time.time() - upscale_start_time
if processed_files > 0:
sec_per_frame = elapsed / processed_files
rem_frames = actual_total - processed_files
eta_sec = rem_frames * sec_per_frame
if p_upscale.returncode != 0:
err_msg = (stdout or "") + "\n" + (stderr or "")
is_alloc_error = any(x in err_msg.lower() for x in ["vkallocatememory", "out of memory", "allocation", "vram", "failed to allocate"])
# Format ETA
if eta_sec > 60:
eta_str = f"{int(eta_sec // 60)}m {int(eta_sec % 60)}s"
else:
eta_str = f"{int(eta_sec)}s"
if is_alloc_error:
if current_tile_size <= 0:
next_tile_size = 256
else:
next_tile_size = current_tile_size // 2
if next_tile_size >= 32:
print(f"Job {job.job_id}: Real-ESRGAN failed with VRAM allocation error. Retrying with tile size halved from {current_tile_size} to {next_tile_size}.")
current_tile_size = next_tile_size
# Clean up only output frames that we attempted to upscale in this run
for filename in os.listdir(input_frames_dir):
out_path = os.path.join(output_frames_dir, filename)
if os.path.exists(out_path):
try:
os.unlink(out_path)
except Exception:
pass
continue
raise RuntimeError(f"Real-ESRGAN failed with exit code {p_upscale.returncode}: {err_msg}")
else:
eta_str = "Calculating..."
job.update_status("upscaling", progress=progress_pct, current_frame=processed_files, eta=eta_str)
on_progress_update(job.job_id, {
"status": "upscaling",
"progress": progress_pct,
"current_frame": processed_files,
"total_frames": actual_total,
"eta": eta_str
})
time.sleep(0.5)
stdout, stderr = p_upscale.communicate()
job.cleanup_process(p_upscale)
if job._is_cancelled:
return
if p_upscale.returncode != 0:
err_msg = (stdout or "") + "\n" + (stderr or "")
is_alloc_error = any(x in err_msg.lower() for x in ["vkallocatememory", "out of memory", "allocation", "vram", "failed to allocate"])
if is_alloc_error:
if current_tile_size <= 0:
next_tile_size = 256
else:
next_tile_size = current_tile_size // 2
if next_tile_size >= 32:
print(f"Job {job.job_id}: Real-ESRGAN failed with VRAM allocation error. Retrying with tile size halved from {current_tile_size} to {next_tile_size}.")
current_tile_size = next_tile_size
# Clean up output frames directory before retrying
for filename in os.listdir(output_frames_dir):
file_path = os.path.join(output_frames_dir, filename)
try:
if os.path.isfile(file_path) or os.path.islink(file_path):
os.unlink(file_path)
elif os.path.isdir(file_path):
shutil.rmtree(file_path)
except Exception as cleanup_err:
print(f"Error cleaning file {file_path}: {cleanup_err}")
continue
raise RuntimeError(f"Real-ESRGAN failed with exit code {p_upscale.returncode}: {err_msg}")
else:
break
break
# Final validation of upscale output
processed_files = len(os.listdir(output_frames_dir))