#!/usr/bin/env python3 import os import sys import argparse import subprocess from dotenv import load_dotenv # Load config script_dir = os.path.dirname(os.path.abspath(__file__)) env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe')) if os.path.exists(env_path): load_dotenv(env_path) else: load_dotenv() # Add ai_transcriber to path so we can import modules sys.path.append(os.path.join(script_dir, 'ai_transcriber')) from ai_transcriber.translator import translate_srt from ai_transcriber.utils import validate_and_repair_srt import pysubs2 from deep_translator import GoogleTranslator from datetime import datetime def translate_fallback_free(source_srt_path, output_srt_path, target_lang="en"): """ Fallback translation using deep-translator (free Google Translate). Parses SRT, translates text line-by-line, and saves new SRT. """ # ... (rest of translate_fallback_free is same) ... # ... (rest of re_embed_subtitles is same) ... def process_recovery(folder_path, target_lang="English"): print(f"Scanning {folder_path} for incomplete translations...") # Define log file recovery_log_file = os.path.join(folder_path, "recovery_status.log") print(f"Logging actions to: {recovery_log_file}") count_fixed = 0 count_skipped = 0 video_extensions = ('.mp4', '.mkv', '.mov', '.avi') for root, dirs, files in os.walk(folder_path): for file in files: # We are looking for the SOURCE SRT files primarily # Exclude our output files to avoid loops if file.endswith(".srt") and \ not file.endswith(f".{target_lang}.srt") and \ not file.endswith(f".{target_lang}.deep_translate.srt"): source_srt_path = os.path.join(root, file) base_name = os.path.splitext(file)[0] # Check if Translation exists (Standard or Fallback) path_gemini = os.path.join(root, f"{base_name}.{target_lang}.srt") path_deep = os.path.join(root, f"{base_name}.{target_lang}.deep_translate.srt") if os.path.exists(path_gemini) or os.path.exists(path_deep): continue print(f"\nFound untranslated transcript: {file}") # Try to translate with open(source_srt_path, "r", encoding="utf-8") as f: content = f.read() new_srt_content = translate_srt(content, target_language=target_lang) success = False method_used = "None" final_srt_path = path_gemini # Default if success if new_srt_content: # Save Gemini translation with open(path_gemini, "w", encoding="utf-8") as f: f.write(new_srt_content) success = True method_used = "Gemini" final_srt_path = path_gemini else: print("❌ Gemini API failed. Attempting Free Fallback...") lang_map = { "English": "en", "French": "fr", "Spanish": "es", "German": "de", "Italian": "it", "Portuguese": "pt", "Russian": "ru", "Japanese": "ja", "Chinese": "zh-CN" } target_code = lang_map.get(target_lang, "en") if translate_fallback_free(source_srt_path, path_deep, target_lang=target_code): success = True method_used = "DeepTranslate" final_srt_path = path_deep else: print("❌ All translation methods failed. Skipping.") continue if success: # Log result with open(recovery_log_file, "a", encoding="utf-8") as log: log.write(f"{datetime.now().isoformat()} | {method_used} | {file} -> {os.path.basename(final_srt_path)}\n") validate_and_repair_srt(final_srt_path) # Now, find the video file to update # Case 1: Original name video_candidates = [ os.path.join(root, base_name + ".mp4"), os.path.join(root, base_name + ".mkv"), # Case 2: .subbed name (if original deleted) os.path.join(root, base_name + ".subbed.mp4"), ] found_video = None for v in video_candidates: if os.path.exists(v): found_video = v break if found_video: print(f"Found video to fix: {found_video}") if re_embed_subtitles(found_video, final_srt_path): count_fixed += 1 else: print("Warning: Could not find a corresponding video file to embed into.") print(f"\nRecovery Complete. Fixed {count_fixed} files.") if __name__ == "__main__": if len(sys.argv) < 2: print("Usage: ./recover_and_fix.py [target_lang]") sys.exit(1) folder = sys.argv[1] lang = sys.argv[2] if len(sys.argv) > 2 else "English" process_recovery(folder, lang)