Added revisions to the translation app
This commit is contained in:
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+69
@@ -0,0 +1,69 @@
|
||||
#!/bin/bash
|
||||
|
||||
# Configuration
|
||||
MOUNT_POINT="/mnt/truenas_isolation"
|
||||
SHARE="//truenas.local/isolation"
|
||||
|
||||
echo "--- SMB Mount Tool (V1) ---"
|
||||
|
||||
# Determine privilege escalation method
|
||||
PRIV_CMD=""
|
||||
if [ "$EUID" -eq 0 ]; then
|
||||
echo "Running as root."
|
||||
else
|
||||
if command -v sudo &> /dev/null; then
|
||||
PRIV_CMD="sudo"
|
||||
elif command -v flatpak-spawn &> /dev/null; then
|
||||
echo "Detected Flatpak environment. Attempting to use host permissions via sudo..."
|
||||
# We need to run sudo ON THE HOST.
|
||||
# flatpak-spawn --host runs as the current user on the host.
|
||||
# So we run 'sudo' inside that host shell.
|
||||
PRIV_CMD="flatpak-spawn --host sudo"
|
||||
# Note: This requires the flatpak to have permission to talk to the host
|
||||
else
|
||||
echo "❌ Error: This script requires root privileges to mount drives."
|
||||
echo " 'sudo' was not found."
|
||||
echo " Please run this script as root: su -c ./mount_truenas.sh"
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
# 1. Create mount point if it doesn't exist
|
||||
if [ ! -d "$MOUNT_POINT" ]; then
|
||||
echo "Creating directory $MOUNT_POINT..."
|
||||
# We try to create it. If it fails (e.g. inside read-only flatpak mount namespace), warn user.
|
||||
$PRIV_CMD mkdir -p "$MOUNT_POINT"
|
||||
if [ $? -ne 0 ]; then
|
||||
echo "Error creating directory. If you are in a Flatpak, you might not have access to host /mnt."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
# 2. Get Credentials
|
||||
read -p "Enter SMB Username [guest]: " SMB_USER
|
||||
SMB_USER=${SMB_USER:-guest}
|
||||
|
||||
# 3. Mount
|
||||
echo "Mounting $SHARE to $MOUNT_POINT..."
|
||||
|
||||
# IMPORTANT: We force the mount to be owned by the current user (UID 1000 usually)
|
||||
# This fixes "Permission Denied" errors when writing to the share.
|
||||
# We also set file_mode/dir_mode to 0777 as a fallback to ensure full access.
|
||||
MOUNT_OPTS="vers=3.0,uid=$(id -u),gid=$(id -g),file_mode=0777,dir_mode=0777,noperm"
|
||||
|
||||
if [ "$SMB_USER" == "guest" ]; then
|
||||
$PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o "guest,$MOUNT_OPTS"
|
||||
else
|
||||
# This will prompt for the SMB password
|
||||
$PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o "username=$SMB_USER,$MOUNT_OPTS"
|
||||
fi
|
||||
|
||||
# 4. Check result
|
||||
if [ $? -eq 0 ]; then
|
||||
echo "✅ Success! Share is now available at $MOUNT_POINT"
|
||||
echo "Files are now owned by $(id -un):$(id -gn) with full write access."
|
||||
echo "The mapping will disappear automatically after you reboot."
|
||||
else
|
||||
echo "❌ Error: Failed to mount the share."
|
||||
echo "Ensure 'cifs-utils' is installed and the server is reachable."
|
||||
fi
|
||||
+3
-6
@@ -7,17 +7,14 @@ from dotenv import load_dotenv
|
||||
|
||||
# Load config
|
||||
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe'))
|
||||
env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe'))
|
||||
if os.path.exists(env_path):
|
||||
load_dotenv(env_path)
|
||||
else:
|
||||
load_dotenv()
|
||||
|
||||
# Add ai_transcriber to path so we can import modules
|
||||
sys.path.append(os.path.join(script_dir, 'ai_transcriber'))
|
||||
|
||||
from ai_transcriber.translator import translate_srt
|
||||
from ai_transcriber.utils import validate_and_repair_srt
|
||||
from translator import translate_srt
|
||||
from utils import validate_and_repair_srt
|
||||
import pysubs2
|
||||
from deep_translator import GoogleTranslator
|
||||
from datetime import datetime
|
||||
+3
-3
@@ -35,7 +35,7 @@ def main():
|
||||
try:
|
||||
from dotenv import load_dotenv
|
||||
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe'))
|
||||
env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe'))
|
||||
if os.path.exists(env_path):
|
||||
load_dotenv(env_path)
|
||||
except ImportError:
|
||||
@@ -107,9 +107,9 @@ def main():
|
||||
print("Using HF_TOKEN from environment.")
|
||||
|
||||
# 5. Build Command
|
||||
# script is in ai_transcriber/main.py relative to this script
|
||||
# script is in main.py relative to this script
|
||||
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
main_script = os.path.join(script_dir, "ai_transcriber", "main.py")
|
||||
main_script = os.path.join(script_dir, "main.py")
|
||||
|
||||
cmd = [sys.executable, main_script]
|
||||
# Add all inputs
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -1,6 +1,7 @@
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
from utils import verify_file_not_empty
|
||||
|
||||
def extract_audio(video_path, output_path=None):
|
||||
"""
|
||||
@@ -22,7 +23,7 @@ def extract_audio(video_path, output_path=None):
|
||||
output_path = f"{base_name}.wav"
|
||||
|
||||
# Check if output file already exists to avoid redundant processing
|
||||
if os.path.exists(output_path):
|
||||
if verify_file_not_empty(output_path):
|
||||
print(f"Audio file already exists: {output_path}")
|
||||
return output_path
|
||||
|
||||
@@ -43,11 +44,18 @@ def extract_audio(video_path, output_path=None):
|
||||
|
||||
try:
|
||||
subprocess.run(command, check=True)
|
||||
|
||||
if not verify_file_not_empty(output_path):
|
||||
raise Exception("FFmpeg command succeeded but output file is empty or missing.")
|
||||
|
||||
print(f"Audio extracted to: {output_path}")
|
||||
return output_path
|
||||
except subprocess.CalledProcessError as e:
|
||||
print(f"Error extracting audio: {e}")
|
||||
sys.exit(1)
|
||||
except Exception as e:
|
||||
print(f"Error: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
def embed_subtitles(video_path, srt_path, output_path=None):
|
||||
"""
|
||||
@@ -80,6 +88,7 @@ def embed_subtitles(video_path, srt_path, output_path=None):
|
||||
|
||||
command = [
|
||||
"ffmpeg",
|
||||
"-ignore_editlist", "1",
|
||||
"-i", video_path,
|
||||
"-i", srt_path,
|
||||
"-map", "0:v",
|
||||
@@ -90,6 +99,8 @@ def embed_subtitles(video_path, srt_path, output_path=None):
|
||||
"-disposition:s:0", "default",
|
||||
"-metadata:s:s:0", "language=eng",
|
||||
"-metadata:s:s:0", "title=English (AI Translated)",
|
||||
"-max_interleave_delta", "0",
|
||||
"-avoid_negative_ts", "make_zero",
|
||||
"-y",
|
||||
"-v", "error",
|
||||
output_path
|
||||
@@ -97,6 +108,11 @@ def embed_subtitles(video_path, srt_path, output_path=None):
|
||||
|
||||
try:
|
||||
subprocess.run(command, check=True)
|
||||
if not verify_file_not_empty(output_path):
|
||||
raise Exception("FFmpeg command succeeded but output video is empty or missing.")
|
||||
|
||||
print(f"Subtitles embedded successfully: {output_path} (Set as primary)")
|
||||
except subprocess.CalledProcessError as e:
|
||||
print(f"Error embedding subtitles: {e}")
|
||||
except Exception as e:
|
||||
print(f"Error embedding subtitles: {e}")
|
||||
|
||||
@@ -0,0 +1,57 @@
|
||||
#!/bin/bash
|
||||
|
||||
# install_local_llm.sh
|
||||
# Installs Ollama and a translation-capable model on Linux (Bazzite/Fedora/Debian compatible)
|
||||
|
||||
set -e
|
||||
|
||||
echo "================================================="
|
||||
echo " Local LLM Setup for AI Transcriber (Ollama)"
|
||||
echo "================================================="
|
||||
|
||||
# 1. Check if Ollama is already installed
|
||||
if command -v ollama &> /dev/null; then
|
||||
echo "✅ Ollama is already installed."
|
||||
else
|
||||
echo "⬇️ Installing Ollama..."
|
||||
# Standard Ollama install script (Works on Bazzite/Silverblue as /usr/local is writable)
|
||||
curl -fsSL https://ollama.com/install.sh | sh
|
||||
fi
|
||||
|
||||
# 2. Check GPU availability for Ollama
|
||||
echo "-------------------------------------------------"
|
||||
if command -v nvidia-smi &> /dev/null; then
|
||||
echo "✅ Nvidia GPU detected. Ollama should run efficiently."
|
||||
else
|
||||
echo "⚠️ Nvidia GPU not found (or drivers missing)."
|
||||
echo " Ollama will run on CPU, which might be slow for translation."
|
||||
fi
|
||||
echo "-------------------------------------------------"
|
||||
|
||||
# 3. Start Ollama Server (Background)
|
||||
# In some dev containers, systemd isn't available, so we try to start it manually if not running.
|
||||
if ! pgrep -x "ollama" > /dev/null; then
|
||||
echo "🚀 Starting Ollama server in the background..."
|
||||
nohup ollama serve > ollama.log 2>&1 &
|
||||
PID=$!
|
||||
echo " (PID: $PID) - Waiting 5 seconds for initialization..."
|
||||
sleep 5
|
||||
else
|
||||
echo "✅ Ollama server is already running."
|
||||
fi
|
||||
|
||||
# 4. Pull a Model
|
||||
# 'llama3' (8B) is a great balance of speed and quality for translation.
|
||||
# 'gemma:7b' is also good.
|
||||
MODEL="llama3"
|
||||
|
||||
echo "⬇️ Pulling model: $MODEL (This may take a few minutes)..."
|
||||
ollama pull $MODEL
|
||||
|
||||
echo "-------------------------------------------------"
|
||||
echo "✅ Installation Complete!"
|
||||
echo ""
|
||||
echo "You can test it manually with: ollama run $MODEL 'Translate this to Spanish: Hello World'"
|
||||
echo ""
|
||||
echo "The AI Transcriber scripts will now detect and use this as a fallback."
|
||||
echo "================================================="
|
||||
@@ -19,9 +19,9 @@ else:
|
||||
load_dotenv()
|
||||
|
||||
from extractor import extract_audio, embed_subtitles
|
||||
from transcriber import transcribe_audio, save_as_srt
|
||||
from translator import translate_srt, translate_fallback_free
|
||||
from utils import validate_and_repair_srt
|
||||
from transcriber import transcribe_audio, save_as_srt, load_whisper_model
|
||||
from translator import translate_with_auto_fallback
|
||||
from utils import validate_and_repair_srt, check_srt_duration_match, GracefulKiller, ensure_ollama_running, check_service_availability, check_path_permissions
|
||||
from diarizer import diarize_audio, merge_diarization_with_transcript
|
||||
import tracker
|
||||
from tracker import JobStatus
|
||||
@@ -51,7 +51,7 @@ def save_srt_with_speakers(segments, output_path):
|
||||
f.write(f"{text}\n\n")
|
||||
print(f"SRT saved to: {output_path}")
|
||||
|
||||
def process_file(file_path, args, source_lang=None):
|
||||
def process_file(file_path, args, source_lang=None, loaded_model=None, service_status=None):
|
||||
tracker.logger.info(f"=== Processing: {file_path} ===")
|
||||
|
||||
# Initialize Job
|
||||
@@ -81,7 +81,8 @@ def process_file(file_path, args, source_lang=None):
|
||||
with open(transcript_file, "r", encoding="utf-8") as f:
|
||||
srt_content = f.read()
|
||||
else:
|
||||
result = transcribe_audio(audio_path, model_size=args.model, language=source_lang)
|
||||
# Use loaded_model if available
|
||||
result = transcribe_audio(audio_path, model_size=args.model, language=source_lang, loaded_model=loaded_model)
|
||||
segments = result["segments"]
|
||||
|
||||
if args.diarize:
|
||||
@@ -110,66 +111,75 @@ def process_file(file_path, args, source_lang=None):
|
||||
|
||||
base_translated = os.path.splitext(file_path)[0] + f".{args.lang}.srt"
|
||||
deep_translated = os.path.splitext(file_path)[0] + f".{args.lang}.deep_translate.srt"
|
||||
local_translated = os.path.splitext(file_path)[0] + f".{args.lang}.local_llm.srt"
|
||||
|
||||
translated_file = base_translated # Default
|
||||
# Determine output path logic
|
||||
target_path_gemini = base_translated
|
||||
target_path_deep = deep_translated
|
||||
target_path_local = local_translated
|
||||
|
||||
translated_file = None
|
||||
translation_success = False
|
||||
method_used = "None"
|
||||
|
||||
if (os.path.exists(base_translated) or os.path.exists(deep_translated)) and not args.force:
|
||||
if os.path.exists(deep_translated):
|
||||
# Check existing
|
||||
if (os.path.exists(base_translated) or os.path.exists(deep_translated) or os.path.exists(local_translated)) and not args.force:
|
||||
if os.path.exists(local_translated):
|
||||
translated_file = local_translated
|
||||
method_used = "Local LLM (Existing)"
|
||||
elif os.path.exists(deep_translated):
|
||||
translated_file = deep_translated
|
||||
method_used = "DeepTranslate (Existing)"
|
||||
else:
|
||||
translated_file = base_translated
|
||||
method_used = "Gemini (Existing)"
|
||||
|
||||
tracker.logger.info(f"Translation exists: {translated_file} ({method_used}). Skipping translation.")
|
||||
final_srt_path = translated_file
|
||||
translation_success = True
|
||||
else:
|
||||
if srt_content:
|
||||
# Helper functions
|
||||
def try_gemini():
|
||||
res = translate_srt(srt_content, target_language=args.lang)
|
||||
if res:
|
||||
with open(base_translated, "w", encoding="utf-8") as f:
|
||||
f.write(res)
|
||||
return True, base_translated, "Gemini"
|
||||
return False, None, None
|
||||
|
||||
def try_deep():
|
||||
lang_map = {
|
||||
"English": "en", "French": "fr", "Spanish": "es", "German": "de",
|
||||
"Italian": "it", "Portuguese": "pt", "Russian": "ru",
|
||||
"Japanese": "ja", "Chinese": "zh-CN"
|
||||
}
|
||||
target_code = lang_map.get(args.lang, "en")
|
||||
res = translate_fallback_free(srt_content, target_language=target_code)
|
||||
if res:
|
||||
with open(deep_translated, "w", encoding="utf-8") as f:
|
||||
f.write(res)
|
||||
return True, deep_translated, "DeepTranslate"
|
||||
return False, None, None
|
||||
|
||||
success = False
|
||||
res_content, method = translate_with_auto_fallback(
|
||||
srt_content,
|
||||
target_language=args.lang,
|
||||
prefer_deep=args.prefer_deep,
|
||||
prefer_local=args.prefer_local,
|
||||
available_services=service_status
|
||||
)
|
||||
|
||||
if args.prefer_deep:
|
||||
success, path, method = try_deep()
|
||||
if not success:
|
||||
tracker.logger.info("DeepTranslate failed. Attempting Gemini...")
|
||||
success, path, method = try_gemini()
|
||||
if res_content:
|
||||
# Save based on method used
|
||||
if "DeepTranslate" in method:
|
||||
save_path = target_path_deep
|
||||
elif "Local LLM" in method:
|
||||
save_path = target_path_local
|
||||
else:
|
||||
save_path = target_path_gemini
|
||||
|
||||
with open(save_path, "w", encoding="utf-8") as f:
|
||||
f.write(res_content)
|
||||
|
||||
tracker.logger.info(f"Translation saved to: {save_path} ({method})")
|
||||
validate_and_repair_srt(save_path)
|
||||
|
||||
# Duration Check
|
||||
is_valid_duration, msg = check_srt_duration_match(transcript_file, save_path)
|
||||
if is_valid_duration:
|
||||
tracker.logger.info(f"Validation: {msg}")
|
||||
final_srt_path = save_path
|
||||
translation_success = True
|
||||
method_used = method
|
||||
else:
|
||||
tracker.logger.error(f"VALIDATION FAILED: {msg}")
|
||||
tracker.logger.error("Marking translation as failed due to incomplete coverage.")
|
||||
|
||||
redo_file = os.path.join(os.path.dirname(file_path), "redo_queue.txt")
|
||||
with open(redo_file, "a", encoding="utf-8") as rf:
|
||||
rf.write(f"{file_path} | {msg}\n")
|
||||
|
||||
translation_success = False
|
||||
else:
|
||||
success, path, method = try_gemini()
|
||||
if not success:
|
||||
tracker.logger.warning("Gemini failed. Attempting DeepTranslate...")
|
||||
success, path, method = try_deep()
|
||||
|
||||
if success:
|
||||
tracker.logger.info(f"Translation saved to: {path} ({method})")
|
||||
validate_and_repair_srt(path)
|
||||
final_srt_path = path
|
||||
translation_success = True
|
||||
method_used = method
|
||||
else:
|
||||
tracker.logger.error("TRANSLATION FAILED.")
|
||||
tracker.logger.error("TRANSLATION FAILED (All methods attempted).")
|
||||
tracker.update_step(file_path, "step_translate", "failed")
|
||||
translation_success = False
|
||||
|
||||
@@ -237,6 +247,7 @@ def main():
|
||||
parser.add_argument("--delete-source", action="store_true", help="Delete original file after embedding")
|
||||
parser.add_argument("--retry-failed", action="store_true", help="Retry FAILED jobs from DB")
|
||||
parser.add_argument("--prefer-deep", action="store_true", help="Prefer DeepTranslate (Free) over Gemini")
|
||||
parser.add_argument("--prefer-local", action="store_true", help="Prefer Local LLM (Ollama) over cloud APIs")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
@@ -257,9 +268,13 @@ def main():
|
||||
user_input = input("Enter source language (e.g. 'French'). Enter for Auto: ").strip()
|
||||
source_lang = user_input if user_input else None
|
||||
|
||||
# Load model for retries too
|
||||
loaded_model = load_whisper_model(args.model)
|
||||
service_status = check_service_availability()
|
||||
|
||||
for file_path in failed_files:
|
||||
if os.path.exists(file_path):
|
||||
process_file(file_path, args, source_lang)
|
||||
process_file(file_path, args, source_lang, loaded_model=loaded_model, service_status=service_status)
|
||||
else:
|
||||
print(f"Skipping missing file: {file_path}")
|
||||
return
|
||||
@@ -274,22 +289,65 @@ def main():
|
||||
source_lang = user_input if user_input else None
|
||||
print(f"Selected: {source_lang if source_lang else 'Auto-detect'}")
|
||||
|
||||
# --- Ensure Ollama is Running ---
|
||||
ensure_ollama_running()
|
||||
# --------------------------------
|
||||
|
||||
# --- Check Service Health ---
|
||||
service_status = check_service_availability()
|
||||
# ----------------------------
|
||||
|
||||
# --- Check Path Permissions ---
|
||||
valid_inputs = []
|
||||
print("Checking Input Permissions...")
|
||||
for inp in args.inputs:
|
||||
ok, msg = check_path_permissions(inp)
|
||||
print(msg)
|
||||
if ok:
|
||||
valid_inputs.append(inp)
|
||||
|
||||
if not valid_inputs:
|
||||
print("\n❌ Error: No valid inputs with read/write permissions found. Exiting.")
|
||||
return
|
||||
# ------------------------------
|
||||
|
||||
# --- Load Model Once ---
|
||||
loaded_model = load_whisper_model(args.model)
|
||||
# -----------------------
|
||||
|
||||
# Initialize Graceful Exit Handler
|
||||
killer = GracefulKiller()
|
||||
|
||||
video_extensions = ('.mp4', '.mkv', '.mov', '.avi', '.webm', '.flv', '.wmv', '.m4v')
|
||||
|
||||
for input_path in args.inputs:
|
||||
for input_path in valid_inputs:
|
||||
if killer.kill_now:
|
||||
break
|
||||
|
||||
if os.path.isfile(input_path):
|
||||
process_file(input_path, args, source_lang)
|
||||
process_file(input_path, args, source_lang, loaded_model=loaded_model, service_status=service_status)
|
||||
elif os.path.isdir(input_path):
|
||||
found = False
|
||||
for root, dirs, files in os.walk(input_path):
|
||||
if killer.kill_now:
|
||||
break
|
||||
|
||||
for file in files:
|
||||
if killer.kill_now:
|
||||
break
|
||||
|
||||
if file.lower().endswith(video_extensions):
|
||||
found = True
|
||||
process_file(os.path.join(root, file), args, source_lang)
|
||||
process_file(os.path.join(root, file), args, source_lang, loaded_model=loaded_model, service_status=service_status)
|
||||
if not found:
|
||||
print(f"No video files found in {input_path}")
|
||||
else:
|
||||
print(f"Error: Invalid input path '{input_path}'")
|
||||
|
||||
if killer.kill_now:
|
||||
print("\n🛑 Process stopped by user. Progress saved in database.")
|
||||
else:
|
||||
print("\n✅ All jobs finished.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+9
-3
@@ -4,7 +4,7 @@
|
||||
MOUNT_POINT="/mnt/truenas_isolation"
|
||||
SHARE="//truenas.local/isolation"
|
||||
|
||||
echo "--- SMB Mount Tool ---"
|
||||
echo "--- SMB Mount Tool (V2) ---"
|
||||
|
||||
# Determine privilege escalation method
|
||||
PRIV_CMD=""
|
||||
@@ -46,16 +46,22 @@ SMB_USER=${SMB_USER:-guest}
|
||||
# 3. Mount
|
||||
echo "Mounting $SHARE to $MOUNT_POINT..."
|
||||
|
||||
# IMPORTANT: We force the mount to be owned by the current user (UID 1000 usually)
|
||||
# This fixes "Permission Denied" errors when writing to the share.
|
||||
# We also set file_mode/dir_mode to 0777 as a fallback to ensure full access.
|
||||
MOUNT_OPTS="vers=3.0,uid=$(id -u),gid=$(id -g),file_mode=0777,dir_mode=0777,noperm"
|
||||
|
||||
if [ "$SMB_USER" == "guest" ]; then
|
||||
$PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o guest,vers=3.0
|
||||
$PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o "guest,$MOUNT_OPTS"
|
||||
else
|
||||
# This will prompt for the SMB password
|
||||
$PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o username="$SMB_USER",vers=3.0
|
||||
$PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o "username=$SMB_USER,$MOUNT_OPTS"
|
||||
fi
|
||||
|
||||
# 4. Check result
|
||||
if [ $? -eq 0 ]; then
|
||||
echo "✅ Success! Share is now available at $MOUNT_POINT"
|
||||
echo "Files are now owned by $(id -un):$(id -gn) with full write access."
|
||||
echo "The mapping will disappear automatically after you reboot."
|
||||
else
|
||||
echo "❌ Error: Failed to mount the share."
|
||||
+254
@@ -0,0 +1,254 @@
|
||||
#!/usr/bin/env python3
|
||||
import os
|
||||
import sys
|
||||
import argparse
|
||||
import subprocess
|
||||
from dotenv import load_dotenv
|
||||
from datetime import datetime
|
||||
import pysubs2
|
||||
|
||||
# Load config
|
||||
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe'))
|
||||
if os.path.exists(env_path):
|
||||
load_dotenv(env_path)
|
||||
else:
|
||||
load_dotenv()
|
||||
|
||||
from translator import translate_with_auto_fallback
|
||||
from utils import validate_and_repair_srt, check_srt_duration_match, GracefulKiller, ensure_ollama_running, detect_file_encoding, check_service_availability, check_path_permissions
|
||||
from extractor import embed_subtitles
|
||||
|
||||
def process_recovery(folder_path, target_lang="English", prefer_deep=False, prefer_local=False):
|
||||
print(f"Scanning {folder_path} for incomplete translations (V2)...")
|
||||
|
||||
# Check Permissions
|
||||
perm_ok, perm_msg = check_path_permissions(folder_path)
|
||||
print(perm_msg)
|
||||
if not perm_ok:
|
||||
print("Aborting due to permission errors.")
|
||||
return
|
||||
|
||||
if prefer_local:
|
||||
print("Preference: Local LLM (Ollama) > Gemini/Deep")
|
||||
elif prefer_deep:
|
||||
print("Preference: DeepTranslate (Google Translate Free) > Gemini")
|
||||
else:
|
||||
print("Preference: Gemini (API) > DeepTranslate")
|
||||
|
||||
recovery_log_file = os.path.join(folder_path, "recovery_status.log")
|
||||
print(f"Logging actions to: {recovery_log_file}")
|
||||
|
||||
# Ensure Ollama is ready
|
||||
ensure_ollama_running()
|
||||
|
||||
# Check Service Health
|
||||
service_status = check_service_availability()
|
||||
|
||||
# Initialize Graceful Exit
|
||||
killer = GracefulKiller()
|
||||
|
||||
count_fixed = 0
|
||||
video_extensions = ('.mp4', '.mkv', '.mov', '.avi')
|
||||
|
||||
for root, dirs, files in os.walk(folder_path):
|
||||
if killer.kill_now:
|
||||
break
|
||||
|
||||
for file in files:
|
||||
if killer.kill_now:
|
||||
break
|
||||
|
||||
if file.endswith(".srt") and \
|
||||
not file.endswith(f".{target_lang}.srt") and \
|
||||
not file.endswith(f".{target_lang}.deep_translate.srt") and \
|
||||
not file.endswith(f".{target_lang}.local_llm.srt"):
|
||||
|
||||
source_srt_path = os.path.join(root, file)
|
||||
base_name = os.path.splitext(file)[0]
|
||||
|
||||
path_gemini = os.path.join(root, f"{base_name}.{target_lang}.srt")
|
||||
path_deep = os.path.join(root, f"{base_name}.{target_lang}.deep_translate.srt")
|
||||
path_local = os.path.join(root, f"{base_name}.{target_lang}.local_llm.srt")
|
||||
|
||||
needs_translation = False
|
||||
existing_translation_path = None
|
||||
|
||||
# Check if translation exists
|
||||
if os.path.exists(path_gemini):
|
||||
existing_translation_path = path_gemini
|
||||
elif os.path.exists(path_deep):
|
||||
existing_translation_path = path_deep
|
||||
elif os.path.exists(path_local):
|
||||
existing_translation_path = path_local
|
||||
|
||||
if existing_translation_path:
|
||||
# Validate duration
|
||||
is_valid, msg = check_srt_duration_match(source_srt_path, existing_translation_path)
|
||||
if not is_valid:
|
||||
print(f"\n⚠️ Found partial/broken translation: {existing_translation_path}")
|
||||
print(f" Reason: {msg}")
|
||||
print(" -> Queueing for re-translation...")
|
||||
needs_translation = True
|
||||
else:
|
||||
# Missing translation
|
||||
print(f"\nFound untranslated transcript: {file}")
|
||||
needs_translation = True
|
||||
|
||||
if not needs_translation:
|
||||
continue
|
||||
|
||||
# --- Proceed with Translation ---
|
||||
|
||||
content = None
|
||||
|
||||
# 1. Try automatic detection
|
||||
detected_enc = detect_file_encoding(source_srt_path)
|
||||
try:
|
||||
with open(source_srt_path, "r", encoding=detected_enc) as f:
|
||||
content = f.read()
|
||||
except Exception:
|
||||
# 2. Fallback to brute force if chardet was wrong
|
||||
encodings_to_try = ['utf-8', 'shift_jis', 'euc_jp', 'latin-1', 'cp1252', 'utf-16']
|
||||
for enc in encodings_to_try:
|
||||
try:
|
||||
with open(source_srt_path, "r", encoding=enc) as f:
|
||||
content = f.read()
|
||||
break # Success
|
||||
except UnicodeDecodeError:
|
||||
continue
|
||||
|
||||
if content is None:
|
||||
print(f"❌ Error: Could not decode {file}. Skipping.")
|
||||
continue
|
||||
|
||||
# Use shared translation logic
|
||||
res_content, method_used = translate_with_auto_fallback(
|
||||
content,
|
||||
target_language=target_lang,
|
||||
prefer_deep=prefer_deep,
|
||||
prefer_local=prefer_local,
|
||||
available_services=service_status
|
||||
)
|
||||
|
||||
final_srt_path = None
|
||||
|
||||
# --- Helper to save result ---
|
||||
def save_translation(text, method):
|
||||
path = None
|
||||
if "DeepTranslate" in method:
|
||||
path = path_deep
|
||||
elif "Local LLM" in method:
|
||||
path = path_local
|
||||
else:
|
||||
path = path_gemini
|
||||
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
f.write(text)
|
||||
return path
|
||||
|
||||
if res_content:
|
||||
final_srt_path = save_translation(res_content, method_used)
|
||||
|
||||
if final_srt_path:
|
||||
# Validate the NEW translation immediately
|
||||
is_valid_new, msg_new = check_srt_duration_match(source_srt_path, final_srt_path)
|
||||
|
||||
if not is_valid_new:
|
||||
print(f"❌ New translation ({method_used}) failed validation: {msg_new}")
|
||||
|
||||
# --- RETRY WITH LOCAL LLM ---
|
||||
# Only retry if we haven't already used Local LLM and it is available
|
||||
if "Local LLM" not in method_used and service_status.get("Ollama", False):
|
||||
print(" -> Retrying with Local LLM (Ollama) as fallback strategy...")
|
||||
|
||||
# Force try Ollama
|
||||
from translator import translate_via_ollama
|
||||
retry_content = translate_via_ollama(content, target_language=target_lang)
|
||||
|
||||
if retry_content:
|
||||
retry_path = path_local
|
||||
with open(retry_path, "w", encoding="utf-8") as f:
|
||||
f.write(retry_content)
|
||||
|
||||
# Validate Retry
|
||||
valid_retry, msg_retry = check_srt_duration_match(source_srt_path, retry_path)
|
||||
if valid_retry:
|
||||
print(f" ✅ Local LLM Retry Succeeded! Using: {os.path.basename(retry_path)}")
|
||||
# Rename/Cleanup the previous failed attempt
|
||||
invalid_path = final_srt_path + ".invalid"
|
||||
os.replace(final_srt_path, invalid_path)
|
||||
|
||||
final_srt_path = retry_path
|
||||
method_used = "Local LLM (Retry)"
|
||||
is_valid_new = True # Mark as valid so we proceed to embedding
|
||||
else:
|
||||
print(f" ❌ Local LLM Retry also failed validation: {msg_retry}")
|
||||
# Cleanup retry attempt
|
||||
os.replace(retry_path, retry_path + ".invalid")
|
||||
|
||||
if not is_valid_new:
|
||||
# Rename the invalid file so it doesn't sit there as a "fake" good translation
|
||||
invalid_path = final_srt_path + ".invalid"
|
||||
if os.path.exists(final_srt_path):
|
||||
os.replace(final_srt_path, invalid_path)
|
||||
print(f" -> Moved failed attempt to: {os.path.basename(invalid_path)}")
|
||||
|
||||
with open(recovery_log_file, "a", encoding="utf-8") as log:
|
||||
log.write(f"{datetime.now().isoformat()} | {method_used} | FAILED_VALIDATION | {file}\n")
|
||||
continue
|
||||
|
||||
# Log result
|
||||
with open(recovery_log_file, "a", encoding="utf-8") as log:
|
||||
log.write(f"{datetime.now().isoformat()} | {method_used} | FIXED | {file} -> {os.path.basename(final_srt_path)}\n")
|
||||
|
||||
validate_and_repair_srt(final_srt_path)
|
||||
|
||||
video_candidates = [
|
||||
os.path.join(root, base_name + ".mp4"),
|
||||
os.path.join(root, base_name + ".mkv"),
|
||||
os.path.join(root, base_name + ".subbed.mp4"),
|
||||
]
|
||||
|
||||
found_video = None
|
||||
for v in video_candidates:
|
||||
if os.path.exists(v):
|
||||
found_video = v
|
||||
break
|
||||
|
||||
if found_video:
|
||||
print(f"Found video to fix: {found_video}")
|
||||
temp_video_out = found_video + ".temp_fix.mp4"
|
||||
try:
|
||||
embed_subtitles(found_video, final_srt_path, output_path=temp_video_out)
|
||||
os.replace(temp_video_out, found_video)
|
||||
print(f"✅ Fixed: {found_video}")
|
||||
count_fixed += 1
|
||||
except Exception as e:
|
||||
print(f"Error re-embedding: {e}")
|
||||
if os.path.exists(temp_video_out):
|
||||
os.remove(temp_video_out)
|
||||
else:
|
||||
print("Warning: Could not find a corresponding video file to embed into.")
|
||||
else:
|
||||
print("❌ All translation methods failed. Skipping.")
|
||||
|
||||
if killer.kill_now:
|
||||
print("\n🛑 Recovery process stopped by user.")
|
||||
else:
|
||||
print(f"\nRecovery Complete. Fixed {count_fixed} files.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description="Recover and Fix Translations (V2)")
|
||||
parser.add_argument("folders", nargs='+', help="One or more paths to folders to scan")
|
||||
parser.add_argument("--lang", default="English", help="Target language (default: English)")
|
||||
parser.add_argument("--prefer-deep", action="store_true", help="Prefer DeepTranslate (Free) over Gemini API")
|
||||
parser.add_argument("--prefer-local", action="store_true", help="Prefer Local LLM (Ollama) over cloud APIs")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
for folder in args.folders:
|
||||
if os.path.exists(folder):
|
||||
process_recovery(folder, args.lang, args.prefer_deep, args.prefer_local)
|
||||
else:
|
||||
print(f"Error: Folder '{folder}' does not exist. Skipping.")
|
||||
@@ -6,3 +6,7 @@ numpy
|
||||
tenacity
|
||||
pysubs2
|
||||
pyannote.audio
|
||||
deep-translator
|
||||
ollama
|
||||
chardet
|
||||
tqdm
|
||||
|
||||
@@ -36,7 +36,7 @@ def main():
|
||||
try:
|
||||
from dotenv import load_dotenv
|
||||
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe'))
|
||||
env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe'))
|
||||
if os.path.exists(env_path):
|
||||
load_dotenv(env_path)
|
||||
except ImportError:
|
||||
@@ -59,7 +59,7 @@ def main():
|
||||
continue
|
||||
|
||||
# Clean up input
|
||||
input_path = input_path.strip("\'"")
|
||||
input_path = input_path.strip('"\'')
|
||||
input_path = input_path.replace(r'\ ', ' ')
|
||||
|
||||
# Expand user (~) and resolve absolute path
|
||||
@@ -98,6 +98,7 @@ def main():
|
||||
do_delete_source = get_yes_no("Delete original source files after embedding?", default="n")
|
||||
|
||||
do_prefer_deep = get_yes_no("Prefer DeepTranslate (Free) over Gemini API?", default="n")
|
||||
do_prefer_local = get_yes_no("Prefer Local LLM (Ollama) over all cloud options?", default="n")
|
||||
|
||||
hf_token = None
|
||||
if do_diarize:
|
||||
@@ -110,7 +111,7 @@ def main():
|
||||
# 5. Build Command
|
||||
# Point to v2 main script
|
||||
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
main_script = os.path.join(script_dir, "ai_transcriber_v2", "main.py")
|
||||
main_script = os.path.join(script_dir, "main.py")
|
||||
|
||||
cmd = [sys.executable, main_script]
|
||||
cmd.extend(input_paths)
|
||||
@@ -138,6 +139,9 @@ def main():
|
||||
if do_prefer_deep:
|
||||
cmd.append("--prefer-deep")
|
||||
|
||||
if do_prefer_local:
|
||||
cmd.append("--prefer-local")
|
||||
|
||||
# 6. Confirmation and Execution
|
||||
clear_screen()
|
||||
print_header()
|
||||
@@ -154,6 +158,7 @@ def main():
|
||||
print(f"Delete Src: {do_delete_source}")
|
||||
print(f"Diarization: {do_diarize}")
|
||||
print(f"Prefer Deep: {do_prefer_deep}")
|
||||
print(f"Prefer Local: {do_prefer_local}")
|
||||
print("-" * 30)
|
||||
|
||||
if not get_yes_no("Run this job now?", default="y"):
|
||||
+8
-3
@@ -36,7 +36,7 @@ def main():
|
||||
try:
|
||||
from dotenv import load_dotenv
|
||||
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe'))
|
||||
env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe'))
|
||||
if os.path.exists(env_path):
|
||||
load_dotenv(env_path)
|
||||
except ImportError:
|
||||
@@ -59,7 +59,7 @@ def main():
|
||||
continue
|
||||
|
||||
# Clean up input
|
||||
input_path = input_path.strip("\'"")
|
||||
input_path = input_path.strip('"\'')
|
||||
input_path = input_path.replace(r'\ ', ' ')
|
||||
|
||||
# Expand user (~) and resolve absolute path
|
||||
@@ -98,6 +98,7 @@ def main():
|
||||
do_delete_source = get_yes_no("Delete original source files after embedding?", default="n")
|
||||
|
||||
do_prefer_deep = get_yes_no("Prefer DeepTranslate (Free) over Gemini API?", default="n")
|
||||
do_prefer_local = get_yes_no("Prefer Local LLM (Ollama) over all cloud options?", default="n")
|
||||
|
||||
hf_token = None
|
||||
if do_diarize:
|
||||
@@ -110,7 +111,7 @@ def main():
|
||||
# 5. Build Command
|
||||
# Point to v2 main script
|
||||
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
main_script = os.path.join(script_dir, "ai_transcriber_v2", "main.py")
|
||||
main_script = os.path.join(script_dir, "main.py")
|
||||
|
||||
cmd = [sys.executable, main_script]
|
||||
cmd.extend(input_paths)
|
||||
@@ -138,6 +139,9 @@ def main():
|
||||
if do_prefer_deep:
|
||||
cmd.append("--prefer-deep")
|
||||
|
||||
if do_prefer_local:
|
||||
cmd.append("--prefer-local")
|
||||
|
||||
# 6. Confirmation and Execution
|
||||
clear_screen()
|
||||
print_header()
|
||||
@@ -154,6 +158,7 @@ def main():
|
||||
print(f"Delete Src: {do_delete_source}")
|
||||
print(f"Diarization: {do_diarize}")
|
||||
print(f"Prefer Deep: {do_prefer_deep}")
|
||||
print(f"Prefer Local: {do_prefer_local}")
|
||||
print("-" * 30)
|
||||
|
||||
if not get_yes_no("Run this job now?", default="y"):
|
||||
@@ -115,7 +115,28 @@ def save_as_srt(result, output_path):
|
||||
f.write(f"{text}\n\n")
|
||||
print(f"SRT saved to: {output_path}")
|
||||
|
||||
def transcribe_audio(audio_path, model_size="auto", language=None):
|
||||
def load_whisper_model(model_size="auto"):
|
||||
"""
|
||||
Loads and returns the Whisper model.
|
||||
"""
|
||||
check_gpu_health()
|
||||
|
||||
if model_size == "auto":
|
||||
model_size = get_optimal_model_size()
|
||||
print(f"Auto-selected model: '{model_size}'")
|
||||
|
||||
print(f"Loading Whisper model ('{model_size}')...")
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
print(f"Using device: {device}")
|
||||
|
||||
try:
|
||||
model = whisper.load_model(model_size, device=device)
|
||||
return model
|
||||
except Exception as e:
|
||||
print(f"Error loading model: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
def transcribe_audio(audio_path, model_size="auto", language=None, loaded_model=None):
|
||||
"""
|
||||
Transcribes an audio file using OpenAI's Whisper model.
|
||||
|
||||
@@ -123,6 +144,7 @@ def transcribe_audio(audio_path, model_size="auto", language=None):
|
||||
audio_path (str): Path to the input audio file.
|
||||
model_size (str): Size of the Whisper model to use. If "auto", selects based on VRAM.
|
||||
language (str, optional): Language code (e.g., "en", "fr", "es"). If None, auto-detects.
|
||||
loaded_model (object, optional): Pre-loaded Whisper model object.
|
||||
|
||||
Returns:
|
||||
dict: The full transcription result containing segments and text.
|
||||
@@ -130,37 +152,14 @@ def transcribe_audio(audio_path, model_size="auto", language=None):
|
||||
if not os.path.exists(audio_path):
|
||||
raise FileNotFoundError(f"Audio file not found: {audio_path}")
|
||||
|
||||
# Run health check once
|
||||
check_gpu_health()
|
||||
|
||||
# Determine model size if auto
|
||||
if model_size == "auto":
|
||||
model_size = get_optimal_model_size()
|
||||
print(f"Auto-selected model: '{model_size}'")
|
||||
|
||||
print(f"Loading Whisper model ('{model_size}')...")
|
||||
|
||||
# Check for GPU availability
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
print(f"Using device: {device}")
|
||||
|
||||
try:
|
||||
model = whisper.load_model(model_size, device=device)
|
||||
except RuntimeError as e:
|
||||
if "out of memory" in str(e).lower():
|
||||
print("Error: GPU Out of Memory. Try using a smaller model size.")
|
||||
else:
|
||||
print(f"Error loading model: {e}")
|
||||
sys.exit(1)
|
||||
except Exception as e:
|
||||
print(f"Error loading model: {e}")
|
||||
sys.exit(1)
|
||||
model = loaded_model
|
||||
if model is None:
|
||||
model = load_whisper_model(model_size)
|
||||
|
||||
print(f"Transcribing {audio_path}...")
|
||||
try:
|
||||
# fp16=False is needed for CPU, but we can let whisper handle defaults usually.
|
||||
# language=None allows auto-detection.
|
||||
result = model.transcribe(audio_path, language=language)
|
||||
# Enable verbose=True to show progress in terminal
|
||||
result = model.transcribe(audio_path, language=language, verbose=True)
|
||||
print("Transcription complete.")
|
||||
return result
|
||||
except Exception as e:
|
||||
|
||||
@@ -4,11 +4,71 @@ from google import genai
|
||||
from google.genai import types
|
||||
from tenacity import retry, stop_after_attempt, wait_exponential, retry_if_exception_type
|
||||
import pysubs2
|
||||
from deep_translator import GoogleTranslator
|
||||
from deep_translator import GoogleTranslator, MyMemoryTranslator
|
||||
import ollama
|
||||
from tqdm import tqdm
|
||||
|
||||
# Define a retry decorator
|
||||
# ... (retry_policy remains)
|
||||
|
||||
def translate_via_ollama(source_srt_content, target_language="English", model="llama3"):
|
||||
"""
|
||||
Translates SRT content using a local Ollama model (Line-by-Line for progress).
|
||||
"""
|
||||
try:
|
||||
subs = pysubs2.SSAFile.from_string(source_srt_content)
|
||||
|
||||
# Using tqdm for progress bar
|
||||
for line in tqdm(subs, desc=" Ollama Progress", unit="line"):
|
||||
text = line.text.strip()
|
||||
if text:
|
||||
prompt = (
|
||||
f"Translate this subtitle text to {target_language}. Output ONLY the translation.\n"
|
||||
f"Text: {text}"
|
||||
)
|
||||
try:
|
||||
response = ollama.chat(model=model, messages=[{'role': 'user', 'content': prompt}])
|
||||
translated_text = response['message']['content'].strip()
|
||||
if translated_text:
|
||||
line.text = translated_text
|
||||
except Exception as e:
|
||||
# Silent fail on line, logs would be too spammy in progress bar
|
||||
pass
|
||||
|
||||
return subs.to_string(format_="srt")
|
||||
|
||||
except Exception as e:
|
||||
print(f" [Local LLM] Error: {e}")
|
||||
return None
|
||||
|
||||
def translate_fallback_mymemory(source_srt_content, target_language="en"):
|
||||
"""
|
||||
Fallback translation using MyMemory (via deep-translator).
|
||||
Limit: 1000 words/day roughly for anonymous usage. Good last resort.
|
||||
"""
|
||||
try:
|
||||
subs = pysubs2.SSAFile.from_string(source_srt_content)
|
||||
# MyMemory uses ISO 639-1 usually
|
||||
translator = MyMemoryTranslator(source='auto', target=target_language)
|
||||
|
||||
for line in tqdm(subs, desc=" MyMemory Progress", unit="line"):
|
||||
text = line.text.strip()
|
||||
if text:
|
||||
if len(text) > 500: # MyMemory has stricter limits often
|
||||
continue
|
||||
try:
|
||||
original_text = text.replace(r"\N", " ")
|
||||
translated_text = translator.translate(original_text)
|
||||
if translated_text:
|
||||
line.text = translated_text
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return subs.to_string(format_="srt")
|
||||
except Exception as e:
|
||||
print(f" [MyMemory Fallback] Critical Error: {e}")
|
||||
return None
|
||||
|
||||
def translate_fallback_free(source_srt_content, target_language="en"):
|
||||
"""
|
||||
Fallback translation using deep-translator (free Google Translate).
|
||||
@@ -20,19 +80,17 @@ def translate_fallback_free(source_srt_content, target_language="en"):
|
||||
Returns:
|
||||
str: Translated SRT content, or None if failed.
|
||||
"""
|
||||
print(f" [Free Fallback] Translating via Google Translate (deep-translator)...")
|
||||
try:
|
||||
# Load from string
|
||||
subs = pysubs2.SSAFile.from_string(source_srt_content)
|
||||
translator = GoogleTranslator(source='auto', target=target_language)
|
||||
|
||||
# Simple line-by-line translation
|
||||
for line in subs:
|
||||
for line in tqdm(subs, desc=" DeepTranslate Progress", unit="line"):
|
||||
text = line.text.strip()
|
||||
if text:
|
||||
# Sanity check: Skip lines that are too long
|
||||
if len(text) > 4000:
|
||||
print(f" Warning: Skipping line with excessive length ({len(text)} chars).")
|
||||
continue
|
||||
|
||||
try:
|
||||
@@ -41,8 +99,8 @@ def translate_fallback_free(source_srt_content, target_language="en"):
|
||||
translated_text = translator.translate(original_text)
|
||||
if translated_text:
|
||||
line.text = translated_text
|
||||
except Exception as e:
|
||||
print(f" Warning: Failed to translate line: {e}")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Return as string
|
||||
return subs.to_string(format_="srt")
|
||||
@@ -53,25 +111,26 @@ def translate_fallback_free(source_srt_content, target_language="en"):
|
||||
# Define a retry decorator
|
||||
# Waits 2^x * 1 seconds between retries (1s, 2s, 4s...)
|
||||
# Stop after 15 attempts
|
||||
# before_sleep logic can print a simple message
|
||||
def log_retry_attempt(retry_state):
|
||||
if retry_state.attempt_number > 1:
|
||||
print(f" [Gemini] Rate limit hit. Retrying in {retry_state.next_action.sleep}s...", end='\r')
|
||||
|
||||
retry_policy = retry(
|
||||
stop=stop_after_attempt(15),
|
||||
wait=wait_exponential(multiplier=1, min=2, max=60),
|
||||
retry=retry_if_exception_type(Exception),
|
||||
reraise=True
|
||||
reraise=True,
|
||||
before_sleep=log_retry_attempt
|
||||
)
|
||||
|
||||
@retry_policy
|
||||
def _generate_with_retry(client, model_name, prompt):
|
||||
"""Internal function to wrap the API call with retry logic."""
|
||||
try:
|
||||
return client.models.generate_content(
|
||||
model=model_name,
|
||||
contents=prompt
|
||||
)
|
||||
except Exception as e:
|
||||
if "429" in str(e) or "Resource has been exhausted" in str(e):
|
||||
print(f" [Rate Limit Hit] Waiting for quota reset... ({e})")
|
||||
raise e
|
||||
return client.models.generate_content(
|
||||
model=model_name,
|
||||
contents=prompt
|
||||
)
|
||||
|
||||
def get_best_available_model(client):
|
||||
"""
|
||||
@@ -126,7 +185,6 @@ def translate_srt(srt_content, target_language="English", api_key=None):
|
||||
# Automatically select the best model
|
||||
# Note: v2 SDK might use 'gemini-1.5-flash' directly without 'models/' prefix usually
|
||||
model_name = "gemini-2.0-flash"
|
||||
print(f"Using Gemini Model (v2): {model_name}")
|
||||
|
||||
prompt = (
|
||||
"You are a professional subtitle translator. Your task is to translate the following SRT subtitle file "
|
||||
@@ -140,11 +198,9 @@ def translate_srt(srt_content, target_language="English", api_key=None):
|
||||
f"{srt_content}"
|
||||
)
|
||||
|
||||
print(f"Translating subtitles to {target_language} (with retries)...")
|
||||
try:
|
||||
# Call the retried internal function
|
||||
response = _generate_with_retry(client, model_name, prompt)
|
||||
print("Translation complete.")
|
||||
|
||||
# Cleanup: sometimes models wrap output in ```srt ... ``` or ``` ... ```
|
||||
cleaned_text = response.text.strip()
|
||||
@@ -171,3 +227,84 @@ def translate_srt(srt_content, target_language="English", api_key=None):
|
||||
print(f"Fallback failed: {inner_e}")
|
||||
|
||||
return None
|
||||
|
||||
def translate_with_auto_fallback(srt_content, target_language="English", prefer_deep=False, prefer_local=False, available_services=None):
|
||||
"""
|
||||
Attempts to translate SRT content using Gemini, DeepTranslate, and Local LLM with fallback logic.
|
||||
|
||||
Args:
|
||||
srt_content (str): The source SRT content.
|
||||
target_language (str): Target language name (e.g., "English", "French").
|
||||
prefer_deep (bool): If True, try DeepTranslate first (among cloud services).
|
||||
prefer_local (bool): If True, try Local LLM (Ollama) first.
|
||||
available_services (dict, optional): Result of check_service_availability().
|
||||
|
||||
Returns:
|
||||
tuple: (translated_content, method_name) or (None, None) if all failed.
|
||||
"""
|
||||
|
||||
# Map full language name to code for DeepTranslate
|
||||
lang_map = {
|
||||
"English": "en", "French": "fr", "Spanish": "es", "German": "de",
|
||||
"Italian": "it", "Portuguese": "pt", "Russian": "ru",
|
||||
"Japanese": "ja", "Chinese": "zh-CN"
|
||||
}
|
||||
target_code = lang_map.get(target_language, "en")
|
||||
|
||||
# Determine which services to even try
|
||||
def is_ok(name):
|
||||
if available_services is None: return True
|
||||
return available_services.get(name, True)
|
||||
|
||||
def try_gemini():
|
||||
if not is_ok("Gemini"): return None, None
|
||||
res = translate_srt(srt_content, target_language=target_language)
|
||||
if res: return res, "Gemini"
|
||||
return None, None
|
||||
|
||||
def try_deep():
|
||||
if not is_ok("DeepTranslate"): return None, None
|
||||
res = translate_fallback_free(srt_content, target_language=target_code)
|
||||
if res: return res, "DeepTranslate"
|
||||
return None, None
|
||||
|
||||
def try_ollama():
|
||||
if not is_ok("Ollama"): return None, None
|
||||
res = translate_via_ollama(srt_content, target_language=target_language)
|
||||
if res: return res, "Local LLM (Ollama)"
|
||||
return None, None
|
||||
|
||||
def try_mymemory():
|
||||
res = translate_fallback_mymemory(srt_content, target_language=target_code)
|
||||
if res: return res, "MyMemory"
|
||||
return None, None
|
||||
|
||||
# Logic flow
|
||||
attempts = []
|
||||
|
||||
if prefer_local:
|
||||
attempts.append(try_ollama)
|
||||
if prefer_deep:
|
||||
attempts.extend([try_deep, try_gemini])
|
||||
else:
|
||||
attempts.extend([try_gemini, try_deep])
|
||||
else:
|
||||
if prefer_deep:
|
||||
attempts.extend([try_deep, try_gemini])
|
||||
else:
|
||||
attempts.extend([try_gemini, try_deep])
|
||||
attempts.append(try_ollama)
|
||||
|
||||
# Final last resort
|
||||
attempts.append(try_mymemory)
|
||||
|
||||
# Execute attempts
|
||||
for i, method_func in enumerate(attempts):
|
||||
if i > 0:
|
||||
print(f" Attempt {i} failed. Trying next fallback...")
|
||||
|
||||
content, method = method_func()
|
||||
if content:
|
||||
return content, method
|
||||
|
||||
return None, None
|
||||
|
||||
@@ -1,5 +1,220 @@
|
||||
import pysubs2
|
||||
import os
|
||||
import signal
|
||||
import sys
|
||||
import subprocess
|
||||
import time
|
||||
import socket
|
||||
import shutil
|
||||
import chardet
|
||||
from tqdm import tqdm
|
||||
|
||||
class GracefulKiller:
|
||||
"""
|
||||
Handles SIGINT (Ctrl+C) and SIGTERM signals.
|
||||
Allows the application to finish the current task before exiting.
|
||||
"""
|
||||
kill_now = False
|
||||
|
||||
def __init__(self):
|
||||
signal.signal(signal.SIGINT, self.exit_gracefully)
|
||||
signal.signal(signal.SIGTERM, self.exit_gracefully)
|
||||
|
||||
def exit_gracefully(self, signum, frame):
|
||||
if not self.kill_now:
|
||||
self.kill_now = True
|
||||
print("\n\n[STOP REQUESTED] The script will exit after the current file finishes processing.")
|
||||
print("Press Ctrl+C again to force quit immediately (not recommended).\n")
|
||||
else:
|
||||
print("\n[FORCE QUIT] Exiting immediately...")
|
||||
sys.exit(1)
|
||||
|
||||
def is_port_open(host, port):
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
||||
s.settimeout(1)
|
||||
return s.connect_ex((host, port)) == 0
|
||||
|
||||
def detect_file_encoding(file_path):
|
||||
"""
|
||||
Robustly detects the encoding of a file using chardet.
|
||||
Returns 'utf-8' if detection fails or confidence is low, as a safe default.
|
||||
"""
|
||||
try:
|
||||
with open(file_path, 'rb') as f:
|
||||
raw_data = f.read(10000) # Read first 10KB
|
||||
result = chardet.detect(raw_data)
|
||||
|
||||
encoding = result['encoding']
|
||||
confidence = result['confidence']
|
||||
|
||||
if encoding and confidence > 0.7:
|
||||
# Shift-JIS is often detected as other Japanese variants, which is fine,
|
||||
# but sometimes we want to be specific. Chardet is usually good.
|
||||
return encoding
|
||||
return 'utf-8'
|
||||
except Exception:
|
||||
return 'utf-8'
|
||||
|
||||
def verify_file_not_empty(file_path):
|
||||
"""
|
||||
Checks if a file exists and is larger than 0 bytes.
|
||||
"""
|
||||
if os.path.exists(file_path) and os.path.getsize(file_path) > 0:
|
||||
return True
|
||||
return False
|
||||
|
||||
def ensure_ollama_running(model_name="llama3"):
|
||||
"""
|
||||
Checks if Ollama is running. If not, attempts to start it.
|
||||
Supports Flatpak by escaping to host via flatpak-spawn.
|
||||
"""
|
||||
in_flatpak = os.path.exists("/.flatpak-info")
|
||||
|
||||
def run_cmd(cmd_list, capture=False):
|
||||
if in_flatpak:
|
||||
full_cmd = ["flatpak-spawn", "--host"] + cmd_list
|
||||
else:
|
||||
full_cmd = cmd_list
|
||||
|
||||
try:
|
||||
if capture:
|
||||
return subprocess.run(full_cmd, capture_output=True, text=True)
|
||||
else:
|
||||
# For serve, we use Popen
|
||||
return subprocess.Popen(
|
||||
full_cmd,
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
start_new_session=True
|
||||
)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
# 1. Start Server if port is closed
|
||||
if not is_port_open("127.0.0.1", 11434):
|
||||
print("Starting Ollama server (Local LLM)...")
|
||||
run_cmd(["ollama", "serve"])
|
||||
|
||||
print(" Waiting for Ollama to initialize...", end="", flush=True)
|
||||
for _ in range(10):
|
||||
if is_port_open("127.0.0.1", 11434):
|
||||
print(" Done.")
|
||||
break
|
||||
time.sleep(1)
|
||||
print(".", end="", flush=True)
|
||||
else:
|
||||
print("\n Warning: Ollama server failed to start or binary not found.")
|
||||
return False
|
||||
|
||||
# 2. Check Model Presence
|
||||
try:
|
||||
result = run_cmd(["ollama", "list"], capture=True)
|
||||
if result and result.returncode == 0:
|
||||
if model_name not in result.stdout:
|
||||
print(f" Model '{model_name}' not found. Pulling now (this may take a while)...")
|
||||
# Pulling can take a long time, so we don't capture but we want to wait
|
||||
pull_cmd = ["flatpak-spawn", "--host", "ollama", "pull", model_name] if in_flatpak else ["ollama", "pull", model_name]
|
||||
subprocess.run(pull_cmd, check=True)
|
||||
print(" Model pulled successfully.")
|
||||
else:
|
||||
# If we can't run list, but port is open, we assume it's okay and let the library handle it
|
||||
pass
|
||||
except Exception as e:
|
||||
print(f" Warning: Could not verify/pull Ollama model: {e}")
|
||||
|
||||
return True
|
||||
|
||||
def check_service_availability(prefer_deep=False):
|
||||
"""
|
||||
Checks availability of configured translation services by performing tiny tests.
|
||||
Returns a dictionary of status.
|
||||
"""
|
||||
status = {
|
||||
"Gemini": False,
|
||||
"DeepTranslate": False,
|
||||
"Ollama": False
|
||||
}
|
||||
|
||||
print("Checking Services...")
|
||||
|
||||
# 1. Check Gemini (Real Test)
|
||||
gemini_key = os.getenv("GEMINI_API_KEY")
|
||||
if gemini_key:
|
||||
try:
|
||||
# We import here to avoid global import issues if dependencies are missing
|
||||
from google import genai
|
||||
client = genai.Client(api_key=gemini_key)
|
||||
# Try a very cheap call
|
||||
client.models.generate_content(
|
||||
model="gemini-2.0-flash",
|
||||
contents="Hi"
|
||||
)
|
||||
status["Gemini"] = True
|
||||
except Exception as e:
|
||||
# Check for rate limit in string representation
|
||||
if "429" in str(e) or "RESOURCE_EXHAUSTED" in str(e):
|
||||
# It is technically 'configured' but currently useless
|
||||
status["Gemini"] = False
|
||||
else:
|
||||
status["Gemini"] = False
|
||||
|
||||
# 2. Check DeepTranslate (Real Test)
|
||||
try:
|
||||
from deep_translator import GoogleTranslator
|
||||
GoogleTranslator(source='auto', target='en').translate("hola")
|
||||
status["DeepTranslate"] = True
|
||||
except Exception:
|
||||
status["DeepTranslate"] = False
|
||||
|
||||
# 3. Check Ollama
|
||||
if is_port_open("127.0.0.1", 11434):
|
||||
status["Ollama"] = True
|
||||
|
||||
# Print Report
|
||||
gemini_msg = "[READY]" if status['Gemini'] else "[UNAVAILABLE] (Rate Limited or Key Invalid)"
|
||||
if not gemini_key: gemini_msg = "[UNAVAILABLE] (Key missing)"
|
||||
|
||||
print(f"1. Gemini API: {gemini_msg}")
|
||||
print(f"2. DeepTranslate: {'[READY]' if status['DeepTranslate'] else '[UNAVAILABLE] (Network/Block)'}")
|
||||
print(f"3. Local Ollama: {'[READY]' if status['Ollama'] else '[OFFLINE]'}")
|
||||
|
||||
return status
|
||||
|
||||
def check_path_permissions(directory_path):
|
||||
"""
|
||||
Checks if the script has read and write permissions for the given directory.
|
||||
Returns: (bool, message)
|
||||
"""
|
||||
if not os.path.exists(directory_path):
|
||||
return False, f"Path not found: {directory_path}"
|
||||
|
||||
# If it's a file, check parent directory
|
||||
if os.path.isfile(directory_path):
|
||||
directory_path = os.path.dirname(directory_path)
|
||||
|
||||
test_file = os.path.join(directory_path, ".perm_test_tmp")
|
||||
|
||||
try:
|
||||
# Test Write
|
||||
with open(test_file, "w") as f:
|
||||
f.write("test")
|
||||
|
||||
# Test Read
|
||||
with open(test_file, "r") as f:
|
||||
content = f.read()
|
||||
|
||||
# Cleanup
|
||||
os.remove(test_file)
|
||||
|
||||
if content == "test":
|
||||
return True, f" [Permissions] Read/Write OK: {directory_path}"
|
||||
else:
|
||||
return False, f" [Permissions] Read check failed (content mismatch): {directory_path}"
|
||||
|
||||
except PermissionError:
|
||||
return False, f" ❌ [Permissions] DENIED: Cannot write to {directory_path}. Check ownership/mount options."
|
||||
except Exception as e:
|
||||
return False, f" ❌ [Permissions] Error checking {directory_path}: {e}"
|
||||
|
||||
def validate_and_repair_srt(srt_path):
|
||||
"""
|
||||
@@ -26,3 +241,53 @@ def validate_and_repair_srt(srt_path):
|
||||
except Exception as e:
|
||||
print(f"Warning: SRT validation failed: {e}")
|
||||
return False
|
||||
|
||||
def check_srt_duration_match(source_srt_path, target_srt_path, tolerance_seconds=30.0, tolerance_percent=0.10):
|
||||
"""
|
||||
Compares the duration of two SRT files to ensure they cover roughly the same timeframe.
|
||||
Useful for detecting partial translations.
|
||||
|
||||
Args:
|
||||
source_srt_path (str): Path to the original language SRT.
|
||||
target_srt_path (str): Path to the translated SRT.
|
||||
tolerance_seconds (float): Max allowed difference in seconds.
|
||||
tolerance_percent (float): Max allowed difference as a percentage of source duration.
|
||||
|
||||
Returns:
|
||||
tuple: (bool, str) -> (passed, message)
|
||||
"""
|
||||
if not os.path.exists(source_srt_path) or not os.path.exists(target_srt_path):
|
||||
return False, "One or both SRT files missing."
|
||||
|
||||
try:
|
||||
source_subs = pysubs2.load(source_srt_path)
|
||||
target_subs = pysubs2.load(target_srt_path)
|
||||
except Exception as e:
|
||||
return False, f"Error parsing SRTs: {e}"
|
||||
|
||||
if not source_subs:
|
||||
return False, "Source SRT is empty."
|
||||
if not target_subs:
|
||||
return False, "Target SRT is empty."
|
||||
|
||||
# Get the end timestamp of the last event in each file (in milliseconds)
|
||||
source_end = source_subs[-1].end
|
||||
target_end = target_subs[-1].end
|
||||
|
||||
# Convert to seconds
|
||||
source_duration = source_end / 1000.0
|
||||
target_duration = target_end / 1000.0
|
||||
|
||||
diff = abs(source_duration - target_duration)
|
||||
|
||||
# Check absolute difference
|
||||
if diff > tolerance_seconds:
|
||||
# Also check percentage (for very long videos, 30s might be negligible)
|
||||
if source_duration > 0 and (diff / source_duration) > tolerance_percent:
|
||||
return False, f"Duration mismatch: Source={source_duration:.1f}s, Target={target_duration:.1f}s (Diff={diff:.1f}s)"
|
||||
|
||||
# For short videos, if percentage is high, fail
|
||||
if source_duration < 300 and (diff / source_duration) > 0.20:
|
||||
return False, f"Duration mismatch (short video): Source={source_duration:.1f}s, Target={target_duration:.1f}s"
|
||||
|
||||
return True, f"Duration match verified (Diff={diff:.1f}s)"
|
||||
|
||||
@@ -1,254 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
import os
|
||||
import sys
|
||||
import argparse
|
||||
import subprocess
|
||||
from dotenv import load_dotenv
|
||||
from datetime import datetime
|
||||
import pysubs2
|
||||
from deep_translator import GoogleTranslator
|
||||
|
||||
# Load config
|
||||
script_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe'))
|
||||
if os.path.exists(env_path):
|
||||
load_dotenv(env_path)
|
||||
else:
|
||||
load_dotenv()
|
||||
|
||||
# Add ai_transcriber_v2 to path so we can import modules
|
||||
sys.path.append(os.path.join(script_dir, 'ai_transcriber_v2'))
|
||||
|
||||
from ai_transcriber_v2.translator import translate_srt
|
||||
from ai_transcriber_v2.utils import validate_and_repair_srt
|
||||
|
||||
def translate_fallback_free(source_srt_path, output_srt_path, target_lang="en"):
|
||||
"""
|
||||
Fallback translation using deep-translator (free Google Translate).
|
||||
Parses SRT, translates text line-by-line, and saves new SRT.
|
||||
"""
|
||||
print(f" [Free Fallback] Translating {source_srt_path}...")
|
||||
|
||||
subs = None
|
||||
encodings_to_try = ['utf-8', 'shift_jis', 'euc_jp', 'latin-1', 'cp1252', 'utf-16']
|
||||
|
||||
for enc in encodings_to_try:
|
||||
try:
|
||||
subs = pysubs2.load(source_srt_path, encoding=enc)
|
||||
break
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
if subs is None:
|
||||
print(f" [Free Fallback] Critical Error: Could not decode file with standard encodings.")
|
||||
return False
|
||||
|
||||
try:
|
||||
translator = GoogleTranslator(source='auto', target=target_lang)
|
||||
|
||||
for line in subs:
|
||||
text = line.text.strip()
|
||||
if text:
|
||||
# Sanity check: Skip lines that are too long (likely garbage/corruption)
|
||||
if len(text) > 4000:
|
||||
print(f" Warning: Skipping line with excessive length ({len(text)} chars). Likely corrupted.")
|
||||
continue
|
||||
|
||||
try:
|
||||
original_text = text.replace(r"\N", " ")
|
||||
translated_text = translator.translate(original_text)
|
||||
if translated_text:
|
||||
line.text = translated_text
|
||||
except Exception as e:
|
||||
print(f" Warning: Failed to translate line: {e}")
|
||||
|
||||
subs.save(output_srt_path)
|
||||
print(f" [Free Fallback] Saved to {output_srt_path}")
|
||||
return True
|
||||
except Exception as e:
|
||||
print(f" [Free Fallback] Critical Error: {e}")
|
||||
return False
|
||||
|
||||
def re_embed_subtitles(video_path, srt_path, output_path=None):
|
||||
"""
|
||||
Re-embeds subtitles into an EXISTING video file, replacing the old tracks.
|
||||
Uses robust flags to handle bad metadata.
|
||||
"""
|
||||
if not os.path.exists(video_path) or not os.path.exists(srt_path):
|
||||
print("Error: Video or SRT file not found.")
|
||||
return False
|
||||
|
||||
temp_output = video_path + ".temp.mp4"
|
||||
print(f"Re-embedding subtitles into: {video_path}...")
|
||||
|
||||
sub_codec = "mov_text" if video_path.lower().endswith(".mp4") else "srt"
|
||||
|
||||
command = [
|
||||
"ffmpeg",
|
||||
"-ignore_editlist", "1",
|
||||
"-i", video_path,
|
||||
"-i", srt_path,
|
||||
"-map", "0:v",
|
||||
"-map", "0:a",
|
||||
"-map", "1:0",
|
||||
"-c", "copy",
|
||||
"-c:s", sub_codec,
|
||||
"-disposition:s:0", "default",
|
||||
"-metadata:s:s:0", "language=eng",
|
||||
"-metadata:s:s:0", "title=English (AI Translated)",
|
||||
"-max_interleave_delta", "0",
|
||||
"-avoid_negative_ts", "make_zero",
|
||||
"-y",
|
||||
"-v", "error",
|
||||
temp_output
|
||||
]
|
||||
|
||||
try:
|
||||
subprocess.run(command, check=True)
|
||||
os.replace(temp_output, video_path)
|
||||
print(f"✅ Fixed: {video_path}")
|
||||
return True
|
||||
except subprocess.CalledProcessError as e:
|
||||
print(f"Error re-embedding: {e}")
|
||||
if os.path.exists(temp_output):
|
||||
os.remove(temp_output)
|
||||
return False
|
||||
|
||||
def process_recovery(folder_path, target_lang="English", prefer_deep=False):
|
||||
print(f"Scanning {folder_path} for incomplete translations (V2)...")
|
||||
if prefer_deep:
|
||||
print("Preference: DeepTranslate (Google Translate Free) > Gemini")
|
||||
else:
|
||||
print("Preference: Gemini (API) > DeepTranslate")
|
||||
|
||||
recovery_log_file = os.path.join(folder_path, "recovery_status.log")
|
||||
print(f"Logging actions to: {recovery_log_file}")
|
||||
|
||||
count_fixed = 0
|
||||
video_extensions = ('.mp4', '.mkv', '.mov', '.avi')
|
||||
|
||||
for root, dirs, files in os.walk(folder_path):
|
||||
for file in files:
|
||||
if file.endswith(".srt") and \
|
||||
not file.endswith(f".{target_lang}.srt") and \
|
||||
not file.endswith(f".{target_lang}.deep_translate.srt"):
|
||||
|
||||
source_srt_path = os.path.join(root, file)
|
||||
base_name = os.path.splitext(file)[0]
|
||||
|
||||
path_gemini = os.path.join(root, f"{base_name}.{target_lang}.srt")
|
||||
path_deep = os.path.join(root, f"{base_name}.{target_lang}.deep_translate.srt")
|
||||
|
||||
if os.path.exists(path_gemini) or os.path.exists(path_deep):
|
||||
continue
|
||||
|
||||
print(f"\nFound untranslated transcript: {file}")
|
||||
|
||||
content = None
|
||||
# extended list to include common Japanese encodings
|
||||
encodings_to_try = ['utf-8', 'shift_jis', 'euc_jp', 'latin-1', 'cp1252', 'utf-16']
|
||||
|
||||
for enc in encodings_to_try:
|
||||
try:
|
||||
with open(source_srt_path, "r", encoding=enc) as f:
|
||||
content = f.read()
|
||||
break # Success
|
||||
except UnicodeDecodeError:
|
||||
continue
|
||||
|
||||
if content is None:
|
||||
print(f"❌ Error: Could not decode {file} with any standard encoding. Skipping.")
|
||||
continue
|
||||
|
||||
# Helpers for translation attempts
|
||||
def try_gemini():
|
||||
res = translate_srt(content, target_language=target_lang)
|
||||
if res:
|
||||
with open(path_gemini, "w", encoding="utf-8") as f:
|
||||
f.write(res)
|
||||
return True, path_gemini, "Gemini"
|
||||
return False, None, None
|
||||
|
||||
def try_deep():
|
||||
lang_map = {
|
||||
"English": "en", "French": "fr", "Spanish": "es",
|
||||
"German": "de", "Italian": "it", "Portuguese": "pt",
|
||||
"Russian": "ru", "Japanese": "ja", "Chinese": "zh-CN"
|
||||
}
|
||||
target_code = lang_map.get(target_lang, "en")
|
||||
if translate_fallback_free(source_srt_path, path_deep, target_lang=target_code):
|
||||
return True, path_deep, "DeepTranslate"
|
||||
return False, None, None
|
||||
|
||||
success = False
|
||||
method_used = "None"
|
||||
final_srt_path = None
|
||||
|
||||
if prefer_deep:
|
||||
# 1. Try DeepTranslate
|
||||
success, final_srt_path, method_used = try_deep()
|
||||
if not success:
|
||||
print("❌ DeepTranslate failed. Attempting Gemini fallback...")
|
||||
success, final_srt_path, method_used = try_gemini()
|
||||
else:
|
||||
# 1. Try Gemini
|
||||
success, final_srt_path, method_used = try_gemini()
|
||||
if not success:
|
||||
print("❌ Gemini API failed. Attempting Free Fallback...")
|
||||
success, final_srt_path, method_used = try_deep()
|
||||
|
||||
if success and final_srt_path:
|
||||
# Log result
|
||||
with open(recovery_log_file, "a", encoding="utf-8") as log:
|
||||
log.write(f"{datetime.now().isoformat()} | {method_used} | {file} -> {os.path.basename(final_srt_path)}\n")
|
||||
|
||||
validate_and_repair_srt(final_srt_path)
|
||||
|
||||
video_candidates = [
|
||||
os.path.join(root, base_name + ".mp4"),
|
||||
os.path.join(root, base_name + ".mkv"),
|
||||
os.path.join(root, base_name + ".subbed.mp4"),
|
||||
]
|
||||
|
||||
found_video = None
|
||||
for v in video_candidates:
|
||||
if os.path.exists(v):
|
||||
found_video = v
|
||||
break
|
||||
|
||||
if found_video:
|
||||
print(f"Found video to fix: {found_video}")
|
||||
if re_embed_subtitles(found_video, final_srt_path):
|
||||
count_fixed += 1
|
||||
else:
|
||||
print("Warning: Could not find a corresponding video file to embed into.")
|
||||
else:
|
||||
print("❌ All translation methods failed. Skipping.")
|
||||
|
||||
print(f"\nRecovery Complete. Fixed {count_fixed} files.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
parser = argparse.ArgumentParser(description="Recover and Fix Translations (V2)")
|
||||
|
||||
parser.add_argument("folders", nargs='+', help="One or more paths to folders to scan")
|
||||
|
||||
parser.add_argument("--lang", default="English", help="Target language (default: English)")
|
||||
|
||||
parser.add_argument("--prefer-deep", action="store_true", help="Prefer DeepTranslate (Free) over Gemini API")
|
||||
|
||||
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
|
||||
|
||||
for folder in args.folders:
|
||||
|
||||
if os.path.exists(folder):
|
||||
|
||||
process_recovery(folder, args.lang, args.prefer_deep)
|
||||
|
||||
else:
|
||||
|
||||
print(f"Error: Folder '{folder}' does not exist. Skipping.")
|
||||
Reference in New Issue
Block a user