diff --git a/video_transcription/ai_transcriber_v2/__pycache__/translator.cpython-313.pyc b/video_transcription/ai_transcriber_v2/__pycache__/translator.cpython-313.pyc index 0862884..190b863 100644 Binary files a/video_transcription/ai_transcriber_v2/__pycache__/translator.cpython-313.pyc and b/video_transcription/ai_transcriber_v2/__pycache__/translator.cpython-313.pyc differ diff --git a/video_transcription/ai_transcriber_v2/__pycache__/utils.cpython-313.pyc b/video_transcription/ai_transcriber_v2/__pycache__/utils.cpython-313.pyc index e9125e2..8750af8 100644 Binary files a/video_transcription/ai_transcriber_v2/__pycache__/utils.cpython-313.pyc and b/video_transcription/ai_transcriber_v2/__pycache__/utils.cpython-313.pyc differ diff --git a/video_transcription/ai_transcriber_v2/translator.py b/video_transcription/ai_transcriber_v2/translator.py index 8fbbff1..db38c65 100644 --- a/video_transcription/ai_transcriber_v2/translator.py +++ b/video_transcription/ai_transcriber_v2/translator.py @@ -1,5 +1,7 @@ import os import sys +import time +from datetime import datetime from google import genai from google.genai import types from tenacity import retry, stop_after_attempt, wait_exponential, retry_if_exception_type @@ -15,36 +17,58 @@ from utils import LANGUAGE_MAP def translate_via_ollama(source_srt_content, target_language="English", model="llama3"): """ Translates SRT content using a local Ollama model (Line-by-Line for progress). + Includes retries and debug logging. """ + debug_log_path = "ollama_debug.log" + try: subs = pysubs2.SSAFile.from_string(source_srt_content) # Using tqdm for progress bar - for line in tqdm(subs, desc=" Ollama Progress", unit="line"): + # dynamic_ncols=True helps it resize properly. leave=True ensures it stays after completion. + for line in tqdm(subs, desc=" Ollama Progress", unit="line", dynamic_ncols=True, leave=True): text = line.text.strip() # Skip empty, numeric-only, or extremely short non-word text if not text or text.isdigit() or len(text) < 2: continue - if text: - prompt = ( - f"Translate this subtitle text to {target_language}. Output ONLY the translation.\n" - f"Text: {text}" - ) + prompt = ( + f"Translate this subtitle text to {target_language}. Output ONLY the translation.\n" + f"Text: {text}" + ) + + # Retry loop for stability + max_retries = 3 + for attempt in range(max_retries): try: response = ollama.chat(model=model, messages=[{'role': 'user', 'content': prompt}]) + + # Check for "model not found" or other soft errors in response if API wraps them + # Usually ollama library raises ResponseError for 404 + translated_text = response['message']['content'].strip() if translated_text: line.text = translated_text + break # Success, exit retry loop + except Exception as e: - # Silent fail on line, logs would be too spammy in progress bar - pass + # Log error details + with open(debug_log_path, "a") as log: + log.write(f"[{datetime.now()}] Error on line '{text}': {str(e)}\n") + + if attempt < max_retries - 1: + time.sleep(2) # Wait before retry + else: + # If all retries fail, keep original text or empty? + # Keeping original might be safer than silence, or just skip. + # For now, we skip updating 'line.text' so it stays as source language (better than corruption) + pass return subs.to_string(format_="srt") except Exception as e: - print(f" [Local LLM] Error: {e}") + print(f" [Local LLM] Critical Error: {e}") return None def translate_fallback_mymemory(source_srt_content, target_language="en"):