This commit is contained in:
2026-01-12 13:42:39 -05:00
3 changed files with 33 additions and 9 deletions
@@ -1,5 +1,7 @@
import os import os
import sys import sys
import time
from datetime import datetime
from google import genai from google import genai
from google.genai import types from google.genai import types
from tenacity import retry, stop_after_attempt, wait_exponential, retry_if_exception_type from tenacity import retry, stop_after_attempt, wait_exponential, retry_if_exception_type
@@ -15,36 +17,58 @@ from utils import LANGUAGE_MAP
def translate_via_ollama(source_srt_content, target_language="English", model="llama3"): def translate_via_ollama(source_srt_content, target_language="English", model="llama3"):
""" """
Translates SRT content using a local Ollama model (Line-by-Line for progress). Translates SRT content using a local Ollama model (Line-by-Line for progress).
Includes retries and debug logging.
""" """
debug_log_path = "ollama_debug.log"
try: try:
subs = pysubs2.SSAFile.from_string(source_srt_content) subs = pysubs2.SSAFile.from_string(source_srt_content)
# Using tqdm for progress bar # Using tqdm for progress bar
for line in tqdm(subs, desc=" Ollama Progress", unit="line"): # dynamic_ncols=True helps it resize properly. leave=True ensures it stays after completion.
for line in tqdm(subs, desc=" Ollama Progress", unit="line", dynamic_ncols=True, leave=True):
text = line.text.strip() text = line.text.strip()
# Skip empty, numeric-only, or extremely short non-word text # Skip empty, numeric-only, or extremely short non-word text
if not text or text.isdigit() or len(text) < 2: if not text or text.isdigit() or len(text) < 2:
continue continue
if text: prompt = (
prompt = ( f"Translate this subtitle text to {target_language}. Output ONLY the translation.\n"
f"Translate this subtitle text to {target_language}. Output ONLY the translation.\n" f"Text: {text}"
f"Text: {text}" )
)
# Retry loop for stability
max_retries = 3
for attempt in range(max_retries):
try: try:
response = ollama.chat(model=model, messages=[{'role': 'user', 'content': prompt}]) response = ollama.chat(model=model, messages=[{'role': 'user', 'content': prompt}])
# Check for "model not found" or other soft errors in response if API wraps them
# Usually ollama library raises ResponseError for 404
translated_text = response['message']['content'].strip() translated_text = response['message']['content'].strip()
if translated_text: if translated_text:
line.text = translated_text line.text = translated_text
break # Success, exit retry loop
except Exception as e: except Exception as e:
# Silent fail on line, logs would be too spammy in progress bar # Log error details
pass with open(debug_log_path, "a") as log:
log.write(f"[{datetime.now()}] Error on line '{text}': {str(e)}\n")
if attempt < max_retries - 1:
time.sleep(2) # Wait before retry
else:
# If all retries fail, keep original text or empty?
# Keeping original might be safer than silence, or just skip.
# For now, we skip updating 'line.text' so it stays as source language (better than corruption)
pass
return subs.to_string(format_="srt") return subs.to_string(format_="srt")
except Exception as e: except Exception as e:
print(f" [Local LLM] Error: {e}") print(f" [Local LLM] Critical Error: {e}")
return None return None
def translate_fallback_mymemory(source_srt_content, target_language="en"): def translate_fallback_mymemory(source_srt_content, target_language="en"):