Files
personal_development/video_transcription/recover_and_fix_v2.py
T
2026-01-11 15:10:56 -05:00

238 lines
9.4 KiB
Python
Executable File

#!/usr/bin/env python3
import os
import sys
import argparse
import subprocess
from dotenv import load_dotenv
from datetime import datetime
import pysubs2
from deep_translator import GoogleTranslator
# Load config
script_dir = os.path.dirname(os.path.abspath(__file__))
env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe'))
if os.path.exists(env_path):
load_dotenv(env_path)
else:
load_dotenv()
# Add ai_transcriber_v2 to path so we can import modules
sys.path.append(os.path.join(script_dir, 'ai_transcriber_v2'))
from ai_transcriber_v2.translator import translate_srt
from ai_transcriber_v2.utils import validate_and_repair_srt
def translate_fallback_free(source_srt_path, output_srt_path, target_lang="en"):
"""
Fallback translation using deep-translator (free Google Translate).
Parses SRT, translates text line-by-line, and saves new SRT.
"""
print(f" [Free Fallback] Translating {source_srt_path}...")
subs = None
encodings_to_try = ['utf-8', 'shift_jis', 'euc_jp', 'latin-1', 'cp1252', 'utf-16']
for enc in encodings_to_try:
try:
subs = pysubs2.load(source_srt_path, encoding=enc)
break
except Exception:
continue
if subs is None:
print(f" [Free Fallback] Critical Error: Could not decode file with standard encodings.")
return False
try:
translator = GoogleTranslator(source='auto', target=target_lang)
for line in subs:
text = line.text.strip()
if text:
# Sanity check: Skip lines that are too long (likely garbage/corruption)
if len(text) > 4000:
print(f" Warning: Skipping line with excessive length ({len(text)} chars). Likely corrupted.")
continue
try:
original_text = text.replace(r"\N", " ")
translated_text = translator.translate(original_text)
if translated_text:
line.text = translated_text
except Exception as e:
print(f" Warning: Failed to translate line: {e}")
subs.save(output_srt_path)
print(f" [Free Fallback] Saved to {output_srt_path}")
return True
except Exception as e:
print(f" [Free Fallback] Critical Error: {e}")
return False
def re_embed_subtitles(video_path, srt_path, output_path=None):
"""
Re-embeds subtitles into an EXISTING video file, replacing the old tracks.
Uses robust flags to handle bad metadata.
"""
if not os.path.exists(video_path) or not os.path.exists(srt_path):
print("Error: Video or SRT file not found.")
return False
temp_output = video_path + ".temp.mp4"
print(f"Re-embedding subtitles into: {video_path}...")
sub_codec = "mov_text" if video_path.lower().endswith(".mp4") else "srt"
command = [
"ffmpeg",
"-ignore_editlist", "1",
"-i", video_path,
"-i", srt_path,
"-map", "0:v",
"-map", "0:a",
"-map", "1:0",
"-c", "copy",
"-c:s", sub_codec,
"-disposition:s:0", "default",
"-metadata:s:s:0", "language=eng",
"-metadata:s:s:0", "title=English (AI Translated)",
"-max_interleave_delta", "0",
"-avoid_negative_ts", "make_zero",
"-y",
"-v", "error",
temp_output
]
try:
subprocess.run(command, check=True)
os.replace(temp_output, video_path)
print(f"✅ Fixed: {video_path}")
return True
except subprocess.CalledProcessError as e:
print(f"Error re-embedding: {e}")
if os.path.exists(temp_output):
os.remove(temp_output)
return False
def process_recovery(folder_path, target_lang="English", prefer_deep=False):
print(f"Scanning {folder_path} for incomplete translations (V2)...")
if prefer_deep:
print("Preference: DeepTranslate (Google Translate Free) > Gemini")
else:
print("Preference: Gemini (API) > DeepTranslate")
recovery_log_file = os.path.join(folder_path, "recovery_status.log")
print(f"Logging actions to: {recovery_log_file}")
count_fixed = 0
video_extensions = ('.mp4', '.mkv', '.mov', '.avi')
for root, dirs, files in os.walk(folder_path):
for file in files:
if file.endswith(".srt") and \
not file.endswith(f".{target_lang}.srt") and \
not file.endswith(f".{target_lang}.deep_translate.srt"):
source_srt_path = os.path.join(root, file)
base_name = os.path.splitext(file)[0]
path_gemini = os.path.join(root, f"{base_name}.{target_lang}.srt")
path_deep = os.path.join(root, f"{base_name}.{target_lang}.deep_translate.srt")
if os.path.exists(path_gemini) or os.path.exists(path_deep):
continue
print(f"\nFound untranslated transcript: {file}")
content = None
# extended list to include common Japanese encodings
encodings_to_try = ['utf-8', 'shift_jis', 'euc_jp', 'latin-1', 'cp1252', 'utf-16']
for enc in encodings_to_try:
try:
with open(source_srt_path, "r", encoding=enc) as f:
content = f.read()
break # Success
except UnicodeDecodeError:
continue
if content is None:
print(f"❌ Error: Could not decode {file} with any standard encoding. Skipping.")
continue
# Helpers for translation attempts
def try_gemini():
res = translate_srt(content, target_language=target_lang)
if res:
with open(path_gemini, "w", encoding="utf-8") as f:
f.write(res)
return True, path_gemini, "Gemini"
return False, None, None
def try_deep():
lang_map = {
"English": "en", "French": "fr", "Spanish": "es",
"German": "de", "Italian": "it", "Portuguese": "pt",
"Russian": "ru", "Japanese": "ja", "Chinese": "zh-CN"
}
target_code = lang_map.get(target_lang, "en")
if translate_fallback_free(source_srt_path, path_deep, target_lang=target_code):
return True, path_deep, "DeepTranslate"
return False, None, None
success = False
method_used = "None"
final_srt_path = None
if prefer_deep:
# 1. Try DeepTranslate
success, final_srt_path, method_used = try_deep()
if not success:
print("❌ DeepTranslate failed. Attempting Gemini fallback...")
success, final_srt_path, method_used = try_gemini()
else:
# 1. Try Gemini
success, final_srt_path, method_used = try_gemini()
if not success:
print("❌ Gemini API failed. Attempting Free Fallback...")
success, final_srt_path, method_used = try_deep()
if success and final_srt_path:
# Log result
with open(recovery_log_file, "a", encoding="utf-8") as log:
log.write(f"{datetime.now().isoformat()} | {method_used} | {file} -> {os.path.basename(final_srt_path)}\n")
validate_and_repair_srt(final_srt_path)
video_candidates = [
os.path.join(root, base_name + ".mp4"),
os.path.join(root, base_name + ".mkv"),
os.path.join(root, base_name + ".subbed.mp4"),
]
found_video = None
for v in video_candidates:
if os.path.exists(v):
found_video = v
break
if found_video:
print(f"Found video to fix: {found_video}")
if re_embed_subtitles(found_video, final_srt_path):
count_fixed += 1
else:
print("Warning: Could not find a corresponding video file to embed into.")
else:
print("❌ All translation methods failed. Skipping.")
print(f"\nRecovery Complete. Fixed {count_fixed} files.")
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Recover and Fix Translations (V2)")
parser.add_argument("folder", help="Path to the folder to scan")
parser.add_argument("lang", nargs="?", default="English", help="Target language (default: English)")
parser.add_argument("--prefer-deep", action="store_true", help="Prefer DeepTranslate (Free) over Gemini API")
args = parser.parse_args()
process_recovery(args.folder, args.lang, args.prefer_deep)