diff --git a/video_transcription/ai_transcriber/__pycache__/__init__.cpython-313.pyc b/video_transcription/ai_transcriber/__pycache__/__init__.cpython-313.pyc deleted file mode 100644 index a5a3b71..0000000 Binary files a/video_transcription/ai_transcriber/__pycache__/__init__.cpython-313.pyc and /dev/null differ diff --git a/video_transcription/ai_transcriber/__pycache__/diarizer.cpython-313.pyc b/video_transcription/ai_transcriber/__pycache__/diarizer.cpython-313.pyc deleted file mode 100644 index e6bb14d..0000000 Binary files a/video_transcription/ai_transcriber/__pycache__/diarizer.cpython-313.pyc and /dev/null differ diff --git a/video_transcription/ai_transcriber/__pycache__/extractor.cpython-313.pyc b/video_transcription/ai_transcriber/__pycache__/extractor.cpython-313.pyc deleted file mode 100644 index cdcfd4b..0000000 Binary files a/video_transcription/ai_transcriber/__pycache__/extractor.cpython-313.pyc and /dev/null differ diff --git a/video_transcription/ai_transcriber/__pycache__/transcriber.cpython-313.pyc b/video_transcription/ai_transcriber/__pycache__/transcriber.cpython-313.pyc deleted file mode 100644 index 9ac7e8d..0000000 Binary files a/video_transcription/ai_transcriber/__pycache__/transcriber.cpython-313.pyc and /dev/null differ diff --git a/video_transcription/ai_transcriber/__pycache__/translator.cpython-313.pyc b/video_transcription/ai_transcriber/__pycache__/translator.cpython-313.pyc deleted file mode 100644 index 5fa0c24..0000000 Binary files a/video_transcription/ai_transcriber/__pycache__/translator.cpython-313.pyc and /dev/null differ diff --git a/video_transcription/ai_transcriber/__pycache__/utils.cpython-313.pyc b/video_transcription/ai_transcriber/__pycache__/utils.cpython-313.pyc deleted file mode 100644 index c82c193..0000000 Binary files a/video_transcription/ai_transcriber/__pycache__/utils.cpython-313.pyc and /dev/null differ diff --git a/video_transcription/ai_transcriber/README.md b/video_transcription/ai_transcriber_v1/README.md similarity index 100% rename from video_transcription/ai_transcriber/README.md rename to video_transcription/ai_transcriber_v1/README.md diff --git a/video_transcription/ai_transcriber/__init__.py b/video_transcription/ai_transcriber_v1/__init__.py similarity index 100% rename from video_transcription/ai_transcriber/__init__.py rename to video_transcription/ai_transcriber_v1/__init__.py diff --git a/video_transcription/diagnostic_scan.py b/video_transcription/ai_transcriber_v1/diagnostic_scan.py similarity index 100% rename from video_transcription/diagnostic_scan.py rename to video_transcription/ai_transcriber_v1/diagnostic_scan.py diff --git a/video_transcription/ai_transcriber/diarizer.py b/video_transcription/ai_transcriber_v1/diarizer.py similarity index 100% rename from video_transcription/ai_transcriber/diarizer.py rename to video_transcription/ai_transcriber_v1/diarizer.py diff --git a/video_transcription/extract_audio.sh b/video_transcription/ai_transcriber_v1/extract_audio.sh similarity index 100% rename from video_transcription/extract_audio.sh rename to video_transcription/ai_transcriber_v1/extract_audio.sh diff --git a/video_transcription/ai_transcriber/extractor.py b/video_transcription/ai_transcriber_v1/extractor.py similarity index 100% rename from video_transcription/ai_transcriber/extractor.py rename to video_transcription/ai_transcriber_v1/extractor.py diff --git a/video_transcription/ai_transcriber/main.py b/video_transcription/ai_transcriber_v1/main.py similarity index 100% rename from video_transcription/ai_transcriber/main.py rename to video_transcription/ai_transcriber_v1/main.py diff --git a/video_transcription/ai_transcriber_v1/mount_truenas.sh b/video_transcription/ai_transcriber_v1/mount_truenas.sh new file mode 100755 index 0000000..e4b5ebf --- /dev/null +++ b/video_transcription/ai_transcriber_v1/mount_truenas.sh @@ -0,0 +1,69 @@ +#!/bin/bash + +# Configuration +MOUNT_POINT="/mnt/truenas_isolation" +SHARE="//truenas.local/isolation" + +echo "--- SMB Mount Tool (V1) ---" + +# Determine privilege escalation method +PRIV_CMD="" +if [ "$EUID" -eq 0 ]; then + echo "Running as root." +else + if command -v sudo &> /dev/null; then + PRIV_CMD="sudo" + elif command -v flatpak-spawn &> /dev/null; then + echo "Detected Flatpak environment. Attempting to use host permissions via sudo..." + # We need to run sudo ON THE HOST. + # flatpak-spawn --host runs as the current user on the host. + # So we run 'sudo' inside that host shell. + PRIV_CMD="flatpak-spawn --host sudo" + # Note: This requires the flatpak to have permission to talk to the host + else + echo "❌ Error: This script requires root privileges to mount drives." + echo " 'sudo' was not found." + echo " Please run this script as root: su -c ./mount_truenas.sh" + exit 1 + fi +fi + +# 1. Create mount point if it doesn't exist +if [ ! -d "$MOUNT_POINT" ]; then + echo "Creating directory $MOUNT_POINT..." + # We try to create it. If it fails (e.g. inside read-only flatpak mount namespace), warn user. + $PRIV_CMD mkdir -p "$MOUNT_POINT" + if [ $? -ne 0 ]; then + echo "Error creating directory. If you are in a Flatpak, you might not have access to host /mnt." + exit 1 + fi +fi + +# 2. Get Credentials +read -p "Enter SMB Username [guest]: " SMB_USER +SMB_USER=${SMB_USER:-guest} + +# 3. Mount +echo "Mounting $SHARE to $MOUNT_POINT..." + +# IMPORTANT: We force the mount to be owned by the current user (UID 1000 usually) +# This fixes "Permission Denied" errors when writing to the share. +# We also set file_mode/dir_mode to 0777 as a fallback to ensure full access. +MOUNT_OPTS="vers=3.0,uid=$(id -u),gid=$(id -g),file_mode=0777,dir_mode=0777,noperm" + +if [ "$SMB_USER" == "guest" ]; then + $PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o "guest,$MOUNT_OPTS" +else + # This will prompt for the SMB password + $PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o "username=$SMB_USER,$MOUNT_OPTS" +fi + +# 4. Check result +if [ $? -eq 0 ]; then + echo "✅ Success! Share is now available at $MOUNT_POINT" + echo "Files are now owned by $(id -un):$(id -gn) with full write access." + echo "The mapping will disappear automatically after you reboot." +else + echo "❌ Error: Failed to mount the share." + echo "Ensure 'cifs-utils' is installed and the server is reachable." +fi \ No newline at end of file diff --git a/video_transcription/recover_and_fix.py b/video_transcription/ai_transcriber_v1/recover_and_fix.py similarity index 94% rename from video_transcription/recover_and_fix.py rename to video_transcription/ai_transcriber_v1/recover_and_fix.py index 1214a8b..d05fb45 100755 --- a/video_transcription/recover_and_fix.py +++ b/video_transcription/ai_transcriber_v1/recover_and_fix.py @@ -7,17 +7,14 @@ from dotenv import load_dotenv # Load config script_dir = os.path.dirname(os.path.abspath(__file__)) -env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe')) +env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe')) if os.path.exists(env_path): load_dotenv(env_path) else: load_dotenv() -# Add ai_transcriber to path so we can import modules -sys.path.append(os.path.join(script_dir, 'ai_transcriber')) - -from ai_transcriber.translator import translate_srt -from ai_transcriber.utils import validate_and_repair_srt +from translator import translate_srt +from utils import validate_and_repair_srt import pysubs2 from deep_translator import GoogleTranslator from datetime import datetime diff --git a/video_transcription/ai_transcriber/requirements.txt b/video_transcription/ai_transcriber_v1/requirements.txt similarity index 100% rename from video_transcription/ai_transcriber/requirements.txt rename to video_transcription/ai_transcriber_v1/requirements.txt diff --git a/video_transcription/run_wizard.py b/video_transcription/ai_transcriber_v1/run_wizard.py similarity index 97% rename from video_transcription/run_wizard.py rename to video_transcription/ai_transcriber_v1/run_wizard.py index b2752c2..7630472 100755 --- a/video_transcription/run_wizard.py +++ b/video_transcription/ai_transcriber_v1/run_wizard.py @@ -35,7 +35,7 @@ def main(): try: from dotenv import load_dotenv script_dir = os.path.dirname(os.path.abspath(__file__)) - env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe')) + env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe')) if os.path.exists(env_path): load_dotenv(env_path) except ImportError: @@ -107,9 +107,9 @@ def main(): print("Using HF_TOKEN from environment.") # 5. Build Command - # script is in ai_transcriber/main.py relative to this script + # script is in main.py relative to this script script_dir = os.path.dirname(os.path.abspath(__file__)) - main_script = os.path.join(script_dir, "ai_transcriber", "main.py") + main_script = os.path.join(script_dir, "main.py") cmd = [sys.executable, main_script] # Add all inputs diff --git a/video_transcription/ai_transcriber/tracker.py b/video_transcription/ai_transcriber_v1/tracker.py similarity index 100% rename from video_transcription/ai_transcriber/tracker.py rename to video_transcription/ai_transcriber_v1/tracker.py diff --git a/video_transcription/ai_transcriber/transcriber.py b/video_transcription/ai_transcriber_v1/transcriber.py similarity index 100% rename from video_transcription/ai_transcriber/transcriber.py rename to video_transcription/ai_transcriber_v1/transcriber.py diff --git a/video_transcription/ai_transcriber/translator.py b/video_transcription/ai_transcriber_v1/translator.py similarity index 100% rename from video_transcription/ai_transcriber/translator.py rename to video_transcription/ai_transcriber_v1/translator.py diff --git a/video_transcription/ai_transcriber/utils.py b/video_transcription/ai_transcriber_v1/utils.py similarity index 100% rename from video_transcription/ai_transcriber/utils.py rename to video_transcription/ai_transcriber_v1/utils.py diff --git a/video_transcription/ai_transcriber_v2/__pycache__/__init__.cpython-313.pyc b/video_transcription/ai_transcriber_v2/__pycache__/__init__.cpython-313.pyc deleted file mode 100644 index c7756f3..0000000 Binary files a/video_transcription/ai_transcriber_v2/__pycache__/__init__.cpython-313.pyc and /dev/null differ diff --git a/video_transcription/ai_transcriber_v2/__pycache__/diarizer.cpython-313.pyc b/video_transcription/ai_transcriber_v2/__pycache__/diarizer.cpython-313.pyc deleted file mode 100644 index e6bb14d..0000000 Binary files a/video_transcription/ai_transcriber_v2/__pycache__/diarizer.cpython-313.pyc and /dev/null differ diff --git a/video_transcription/ai_transcriber_v2/__pycache__/extractor.cpython-313.pyc b/video_transcription/ai_transcriber_v2/__pycache__/extractor.cpython-313.pyc index cdcfd4b..958760b 100644 Binary files a/video_transcription/ai_transcriber_v2/__pycache__/extractor.cpython-313.pyc and b/video_transcription/ai_transcriber_v2/__pycache__/extractor.cpython-313.pyc differ diff --git a/video_transcription/ai_transcriber_v2/__pycache__/main.cpython-313.pyc b/video_transcription/ai_transcriber_v2/__pycache__/main.cpython-313.pyc new file mode 100644 index 0000000..bb494c4 Binary files /dev/null and b/video_transcription/ai_transcriber_v2/__pycache__/main.cpython-313.pyc differ diff --git a/video_transcription/ai_transcriber_v2/__pycache__/recover_and_fix_v2.cpython-313.pyc b/video_transcription/ai_transcriber_v2/__pycache__/recover_and_fix_v2.cpython-313.pyc new file mode 100644 index 0000000..5b16c0a Binary files /dev/null and b/video_transcription/ai_transcriber_v2/__pycache__/recover_and_fix_v2.cpython-313.pyc differ diff --git a/video_transcription/ai_transcriber_v2/__pycache__/run_v2.cpython-313.pyc b/video_transcription/ai_transcriber_v2/__pycache__/run_v2.cpython-313.pyc new file mode 100644 index 0000000..799fb5b Binary files /dev/null and b/video_transcription/ai_transcriber_v2/__pycache__/run_v2.cpython-313.pyc differ diff --git a/video_transcription/ai_transcriber_v2/__pycache__/run_wizard_v2.cpython-313.pyc b/video_transcription/ai_transcriber_v2/__pycache__/run_wizard_v2.cpython-313.pyc new file mode 100644 index 0000000..ebc1575 Binary files /dev/null and b/video_transcription/ai_transcriber_v2/__pycache__/run_wizard_v2.cpython-313.pyc differ diff --git a/video_transcription/ai_transcriber_v2/__pycache__/transcriber.cpython-313.pyc b/video_transcription/ai_transcriber_v2/__pycache__/transcriber.cpython-313.pyc deleted file mode 100644 index 9ac7e8d..0000000 Binary files a/video_transcription/ai_transcriber_v2/__pycache__/transcriber.cpython-313.pyc and /dev/null differ diff --git a/video_transcription/ai_transcriber_v2/__pycache__/translator.cpython-313.pyc b/video_transcription/ai_transcriber_v2/__pycache__/translator.cpython-313.pyc index 5a97144..0940c92 100644 Binary files a/video_transcription/ai_transcriber_v2/__pycache__/translator.cpython-313.pyc and b/video_transcription/ai_transcriber_v2/__pycache__/translator.cpython-313.pyc differ diff --git a/video_transcription/ai_transcriber_v2/__pycache__/utils.cpython-313.pyc b/video_transcription/ai_transcriber_v2/__pycache__/utils.cpython-313.pyc index 6eb6afc..e9125e2 100644 Binary files a/video_transcription/ai_transcriber_v2/__pycache__/utils.cpython-313.pyc and b/video_transcription/ai_transcriber_v2/__pycache__/utils.cpython-313.pyc differ diff --git a/video_transcription/ai_transcriber_v2/extractor.py b/video_transcription/ai_transcriber_v2/extractor.py index 986e38b..00fb93f 100644 --- a/video_transcription/ai_transcriber_v2/extractor.py +++ b/video_transcription/ai_transcriber_v2/extractor.py @@ -1,6 +1,7 @@ import os import subprocess import sys +from utils import verify_file_not_empty def extract_audio(video_path, output_path=None): """ @@ -22,7 +23,7 @@ def extract_audio(video_path, output_path=None): output_path = f"{base_name}.wav" # Check if output file already exists to avoid redundant processing - if os.path.exists(output_path): + if verify_file_not_empty(output_path): print(f"Audio file already exists: {output_path}") return output_path @@ -43,11 +44,18 @@ def extract_audio(video_path, output_path=None): try: subprocess.run(command, check=True) + + if not verify_file_not_empty(output_path): + raise Exception("FFmpeg command succeeded but output file is empty or missing.") + print(f"Audio extracted to: {output_path}") return output_path except subprocess.CalledProcessError as e: print(f"Error extracting audio: {e}") sys.exit(1) + except Exception as e: + print(f"Error: {e}") + sys.exit(1) def embed_subtitles(video_path, srt_path, output_path=None): """ @@ -80,6 +88,7 @@ def embed_subtitles(video_path, srt_path, output_path=None): command = [ "ffmpeg", + "-ignore_editlist", "1", "-i", video_path, "-i", srt_path, "-map", "0:v", @@ -90,6 +99,8 @@ def embed_subtitles(video_path, srt_path, output_path=None): "-disposition:s:0", "default", "-metadata:s:s:0", "language=eng", "-metadata:s:s:0", "title=English (AI Translated)", + "-max_interleave_delta", "0", + "-avoid_negative_ts", "make_zero", "-y", "-v", "error", output_path @@ -97,6 +108,11 @@ def embed_subtitles(video_path, srt_path, output_path=None): try: subprocess.run(command, check=True) + if not verify_file_not_empty(output_path): + raise Exception("FFmpeg command succeeded but output video is empty or missing.") + print(f"Subtitles embedded successfully: {output_path} (Set as primary)") except subprocess.CalledProcessError as e: print(f"Error embedding subtitles: {e}") + except Exception as e: + print(f"Error embedding subtitles: {e}") diff --git a/video_transcription/ai_transcriber_v2/install_local_llm.sh b/video_transcription/ai_transcriber_v2/install_local_llm.sh new file mode 100755 index 0000000..4890d11 --- /dev/null +++ b/video_transcription/ai_transcriber_v2/install_local_llm.sh @@ -0,0 +1,57 @@ +#!/bin/bash + +# install_local_llm.sh +# Installs Ollama and a translation-capable model on Linux (Bazzite/Fedora/Debian compatible) + +set -e + +echo "=================================================" +echo " Local LLM Setup for AI Transcriber (Ollama)" +echo "=================================================" + +# 1. Check if Ollama is already installed +if command -v ollama &> /dev/null; then + echo "✅ Ollama is already installed." +else + echo "⬇️ Installing Ollama..." + # Standard Ollama install script (Works on Bazzite/Silverblue as /usr/local is writable) + curl -fsSL https://ollama.com/install.sh | sh +fi + +# 2. Check GPU availability for Ollama +echo "-------------------------------------------------" +if command -v nvidia-smi &> /dev/null; then + echo "✅ Nvidia GPU detected. Ollama should run efficiently." +else + echo "⚠️ Nvidia GPU not found (or drivers missing)." + echo " Ollama will run on CPU, which might be slow for translation." +fi +echo "-------------------------------------------------" + +# 3. Start Ollama Server (Background) +# In some dev containers, systemd isn't available, so we try to start it manually if not running. +if ! pgrep -x "ollama" > /dev/null; then + echo "🚀 Starting Ollama server in the background..." + nohup ollama serve > ollama.log 2>&1 & + PID=$! + echo " (PID: $PID) - Waiting 5 seconds for initialization..." + sleep 5 +else + echo "✅ Ollama server is already running." +fi + +# 4. Pull a Model +# 'llama3' (8B) is a great balance of speed and quality for translation. +# 'gemma:7b' is also good. +MODEL="llama3" + +echo "⬇️ Pulling model: $MODEL (This may take a few minutes)..." +ollama pull $MODEL + +echo "-------------------------------------------------" +echo "✅ Installation Complete!" +echo "" +echo "You can test it manually with: ollama run $MODEL 'Translate this to Spanish: Hello World'" +echo "" +echo "The AI Transcriber scripts will now detect and use this as a fallback." +echo "=================================================" diff --git a/video_transcription/ai_transcriber_v2/main.py b/video_transcription/ai_transcriber_v2/main.py index 9fa6a5d..64b65bc 100644 --- a/video_transcription/ai_transcriber_v2/main.py +++ b/video_transcription/ai_transcriber_v2/main.py @@ -19,9 +19,9 @@ else: load_dotenv() from extractor import extract_audio, embed_subtitles -from transcriber import transcribe_audio, save_as_srt -from translator import translate_srt, translate_fallback_free -from utils import validate_and_repair_srt +from transcriber import transcribe_audio, save_as_srt, load_whisper_model +from translator import translate_with_auto_fallback +from utils import validate_and_repair_srt, check_srt_duration_match, GracefulKiller, ensure_ollama_running, check_service_availability, check_path_permissions from diarizer import diarize_audio, merge_diarization_with_transcript import tracker from tracker import JobStatus @@ -51,7 +51,7 @@ def save_srt_with_speakers(segments, output_path): f.write(f"{text}\n\n") print(f"SRT saved to: {output_path}") -def process_file(file_path, args, source_lang=None): +def process_file(file_path, args, source_lang=None, loaded_model=None, service_status=None): tracker.logger.info(f"=== Processing: {file_path} ===") # Initialize Job @@ -81,7 +81,8 @@ def process_file(file_path, args, source_lang=None): with open(transcript_file, "r", encoding="utf-8") as f: srt_content = f.read() else: - result = transcribe_audio(audio_path, model_size=args.model, language=source_lang) + # Use loaded_model if available + result = transcribe_audio(audio_path, model_size=args.model, language=source_lang, loaded_model=loaded_model) segments = result["segments"] if args.diarize: @@ -110,66 +111,75 @@ def process_file(file_path, args, source_lang=None): base_translated = os.path.splitext(file_path)[0] + f".{args.lang}.srt" deep_translated = os.path.splitext(file_path)[0] + f".{args.lang}.deep_translate.srt" + local_translated = os.path.splitext(file_path)[0] + f".{args.lang}.local_llm.srt" - translated_file = base_translated # Default + # Determine output path logic + target_path_gemini = base_translated + target_path_deep = deep_translated + target_path_local = local_translated + + translated_file = None translation_success = False method_used = "None" - if (os.path.exists(base_translated) or os.path.exists(deep_translated)) and not args.force: - if os.path.exists(deep_translated): + # Check existing + if (os.path.exists(base_translated) or os.path.exists(deep_translated) or os.path.exists(local_translated)) and not args.force: + if os.path.exists(local_translated): + translated_file = local_translated + method_used = "Local LLM (Existing)" + elif os.path.exists(deep_translated): translated_file = deep_translated method_used = "DeepTranslate (Existing)" else: + translated_file = base_translated method_used = "Gemini (Existing)" + tracker.logger.info(f"Translation exists: {translated_file} ({method_used}). Skipping translation.") final_srt_path = translated_file translation_success = True else: if srt_content: - # Helper functions - def try_gemini(): - res = translate_srt(srt_content, target_language=args.lang) - if res: - with open(base_translated, "w", encoding="utf-8") as f: - f.write(res) - return True, base_translated, "Gemini" - return False, None, None - - def try_deep(): - lang_map = { - "English": "en", "French": "fr", "Spanish": "es", "German": "de", - "Italian": "it", "Portuguese": "pt", "Russian": "ru", - "Japanese": "ja", "Chinese": "zh-CN" - } - target_code = lang_map.get(args.lang, "en") - res = translate_fallback_free(srt_content, target_language=target_code) - if res: - with open(deep_translated, "w", encoding="utf-8") as f: - f.write(res) - return True, deep_translated, "DeepTranslate" - return False, None, None - - success = False + res_content, method = translate_with_auto_fallback( + srt_content, + target_language=args.lang, + prefer_deep=args.prefer_deep, + prefer_local=args.prefer_local, + available_services=service_status + ) - if args.prefer_deep: - success, path, method = try_deep() - if not success: - tracker.logger.info("DeepTranslate failed. Attempting Gemini...") - success, path, method = try_gemini() + if res_content: + # Save based on method used + if "DeepTranslate" in method: + save_path = target_path_deep + elif "Local LLM" in method: + save_path = target_path_local + else: + save_path = target_path_gemini + + with open(save_path, "w", encoding="utf-8") as f: + f.write(res_content) + + tracker.logger.info(f"Translation saved to: {save_path} ({method})") + validate_and_repair_srt(save_path) + + # Duration Check + is_valid_duration, msg = check_srt_duration_match(transcript_file, save_path) + if is_valid_duration: + tracker.logger.info(f"Validation: {msg}") + final_srt_path = save_path + translation_success = True + method_used = method + else: + tracker.logger.error(f"VALIDATION FAILED: {msg}") + tracker.logger.error("Marking translation as failed due to incomplete coverage.") + + redo_file = os.path.join(os.path.dirname(file_path), "redo_queue.txt") + with open(redo_file, "a", encoding="utf-8") as rf: + rf.write(f"{file_path} | {msg}\n") + + translation_success = False else: - success, path, method = try_gemini() - if not success: - tracker.logger.warning("Gemini failed. Attempting DeepTranslate...") - success, path, method = try_deep() - - if success: - tracker.logger.info(f"Translation saved to: {path} ({method})") - validate_and_repair_srt(path) - final_srt_path = path - translation_success = True - method_used = method - else: - tracker.logger.error("TRANSLATION FAILED.") + tracker.logger.error("TRANSLATION FAILED (All methods attempted).") tracker.update_step(file_path, "step_translate", "failed") translation_success = False @@ -237,6 +247,7 @@ def main(): parser.add_argument("--delete-source", action="store_true", help="Delete original file after embedding") parser.add_argument("--retry-failed", action="store_true", help="Retry FAILED jobs from DB") parser.add_argument("--prefer-deep", action="store_true", help="Prefer DeepTranslate (Free) over Gemini") + parser.add_argument("--prefer-local", action="store_true", help="Prefer Local LLM (Ollama) over cloud APIs") args = parser.parse_args() @@ -257,9 +268,13 @@ def main(): user_input = input("Enter source language (e.g. 'French'). Enter for Auto: ").strip() source_lang = user_input if user_input else None + # Load model for retries too + loaded_model = load_whisper_model(args.model) + service_status = check_service_availability() + for file_path in failed_files: if os.path.exists(file_path): - process_file(file_path, args, source_lang) + process_file(file_path, args, source_lang, loaded_model=loaded_model, service_status=service_status) else: print(f"Skipping missing file: {file_path}") return @@ -274,22 +289,65 @@ def main(): source_lang = user_input if user_input else None print(f"Selected: {source_lang if source_lang else 'Auto-detect'}") + # --- Ensure Ollama is Running --- + ensure_ollama_running() + # -------------------------------- + + # --- Check Service Health --- + service_status = check_service_availability() + # ---------------------------- + + # --- Check Path Permissions --- + valid_inputs = [] + print("Checking Input Permissions...") + for inp in args.inputs: + ok, msg = check_path_permissions(inp) + print(msg) + if ok: + valid_inputs.append(inp) + + if not valid_inputs: + print("\n❌ Error: No valid inputs with read/write permissions found. Exiting.") + return + # ------------------------------ + + # --- Load Model Once --- + loaded_model = load_whisper_model(args.model) + # ----------------------- + + # Initialize Graceful Exit Handler + killer = GracefulKiller() + video_extensions = ('.mp4', '.mkv', '.mov', '.avi', '.webm', '.flv', '.wmv', '.m4v') - for input_path in args.inputs: + for input_path in valid_inputs: + if killer.kill_now: + break + if os.path.isfile(input_path): - process_file(input_path, args, source_lang) + process_file(input_path, args, source_lang, loaded_model=loaded_model, service_status=service_status) elif os.path.isdir(input_path): found = False for root, dirs, files in os.walk(input_path): + if killer.kill_now: + break + for file in files: + if killer.kill_now: + break + if file.lower().endswith(video_extensions): found = True - process_file(os.path.join(root, file), args, source_lang) + process_file(os.path.join(root, file), args, source_lang, loaded_model=loaded_model, service_status=service_status) if not found: print(f"No video files found in {input_path}") else: print(f"Error: Invalid input path '{input_path}'") + + if killer.kill_now: + print("\n🛑 Process stopped by user. Progress saved in database.") + else: + print("\n✅ All jobs finished.") if __name__ == "__main__": main() \ No newline at end of file diff --git a/video_transcription/mount_truenas.sh b/video_transcription/ai_transcriber_v2/mount_truenas.sh similarity index 76% rename from video_transcription/mount_truenas.sh rename to video_transcription/ai_transcriber_v2/mount_truenas.sh index 34c9991..e634deb 100755 --- a/video_transcription/mount_truenas.sh +++ b/video_transcription/ai_transcriber_v2/mount_truenas.sh @@ -4,7 +4,7 @@ MOUNT_POINT="/mnt/truenas_isolation" SHARE="//truenas.local/isolation" -echo "--- SMB Mount Tool ---" +echo "--- SMB Mount Tool (V2) ---" # Determine privilege escalation method PRIV_CMD="" @@ -46,16 +46,22 @@ SMB_USER=${SMB_USER:-guest} # 3. Mount echo "Mounting $SHARE to $MOUNT_POINT..." +# IMPORTANT: We force the mount to be owned by the current user (UID 1000 usually) +# This fixes "Permission Denied" errors when writing to the share. +# We also set file_mode/dir_mode to 0777 as a fallback to ensure full access. +MOUNT_OPTS="vers=3.0,uid=$(id -u),gid=$(id -g),file_mode=0777,dir_mode=0777,noperm" + if [ "$SMB_USER" == "guest" ]; then - $PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o guest,vers=3.0 + $PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o "guest,$MOUNT_OPTS" else # This will prompt for the SMB password - $PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o username="$SMB_USER",vers=3.0 + $PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o "username=$SMB_USER,$MOUNT_OPTS" fi # 4. Check result if [ $? -eq 0 ]; then echo "✅ Success! Share is now available at $MOUNT_POINT" + echo "Files are now owned by $(id -un):$(id -gn) with full write access." echo "The mapping will disappear automatically after you reboot." else echo "❌ Error: Failed to mount the share." diff --git a/video_transcription/ai_transcriber_v2/recover_and_fix_v2.py b/video_transcription/ai_transcriber_v2/recover_and_fix_v2.py new file mode 100755 index 0000000..1ef30c3 --- /dev/null +++ b/video_transcription/ai_transcriber_v2/recover_and_fix_v2.py @@ -0,0 +1,254 @@ +#!/usr/bin/env python3 +import os +import sys +import argparse +import subprocess +from dotenv import load_dotenv +from datetime import datetime +import pysubs2 + +# Load config +script_dir = os.path.dirname(os.path.abspath(__file__)) +env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe')) +if os.path.exists(env_path): + load_dotenv(env_path) +else: + load_dotenv() + +from translator import translate_with_auto_fallback +from utils import validate_and_repair_srt, check_srt_duration_match, GracefulKiller, ensure_ollama_running, detect_file_encoding, check_service_availability, check_path_permissions +from extractor import embed_subtitles + +def process_recovery(folder_path, target_lang="English", prefer_deep=False, prefer_local=False): + print(f"Scanning {folder_path} for incomplete translations (V2)...") + + # Check Permissions + perm_ok, perm_msg = check_path_permissions(folder_path) + print(perm_msg) + if not perm_ok: + print("Aborting due to permission errors.") + return + + if prefer_local: + print("Preference: Local LLM (Ollama) > Gemini/Deep") + elif prefer_deep: + print("Preference: DeepTranslate (Google Translate Free) > Gemini") + else: + print("Preference: Gemini (API) > DeepTranslate") + + recovery_log_file = os.path.join(folder_path, "recovery_status.log") + print(f"Logging actions to: {recovery_log_file}") + + # Ensure Ollama is ready + ensure_ollama_running() + + # Check Service Health + service_status = check_service_availability() + + # Initialize Graceful Exit + killer = GracefulKiller() + + count_fixed = 0 + video_extensions = ('.mp4', '.mkv', '.mov', '.avi') + + for root, dirs, files in os.walk(folder_path): + if killer.kill_now: + break + + for file in files: + if killer.kill_now: + break + + if file.endswith(".srt") and \ + not file.endswith(f".{target_lang}.srt") and \ + not file.endswith(f".{target_lang}.deep_translate.srt") and \ + not file.endswith(f".{target_lang}.local_llm.srt"): + + source_srt_path = os.path.join(root, file) + base_name = os.path.splitext(file)[0] + + path_gemini = os.path.join(root, f"{base_name}.{target_lang}.srt") + path_deep = os.path.join(root, f"{base_name}.{target_lang}.deep_translate.srt") + path_local = os.path.join(root, f"{base_name}.{target_lang}.local_llm.srt") + + needs_translation = False + existing_translation_path = None + + # Check if translation exists + if os.path.exists(path_gemini): + existing_translation_path = path_gemini + elif os.path.exists(path_deep): + existing_translation_path = path_deep + elif os.path.exists(path_local): + existing_translation_path = path_local + + if existing_translation_path: + # Validate duration + is_valid, msg = check_srt_duration_match(source_srt_path, existing_translation_path) + if not is_valid: + print(f"\n⚠️ Found partial/broken translation: {existing_translation_path}") + print(f" Reason: {msg}") + print(" -> Queueing for re-translation...") + needs_translation = True + else: + # Missing translation + print(f"\nFound untranslated transcript: {file}") + needs_translation = True + + if not needs_translation: + continue + + # --- Proceed with Translation --- + + content = None + + # 1. Try automatic detection + detected_enc = detect_file_encoding(source_srt_path) + try: + with open(source_srt_path, "r", encoding=detected_enc) as f: + content = f.read() + except Exception: + # 2. Fallback to brute force if chardet was wrong + encodings_to_try = ['utf-8', 'shift_jis', 'euc_jp', 'latin-1', 'cp1252', 'utf-16'] + for enc in encodings_to_try: + try: + with open(source_srt_path, "r", encoding=enc) as f: + content = f.read() + break # Success + except UnicodeDecodeError: + continue + + if content is None: + print(f"❌ Error: Could not decode {file}. Skipping.") + continue + + # Use shared translation logic + res_content, method_used = translate_with_auto_fallback( + content, + target_language=target_lang, + prefer_deep=prefer_deep, + prefer_local=prefer_local, + available_services=service_status + ) + + final_srt_path = None + + # --- Helper to save result --- + def save_translation(text, method): + path = None + if "DeepTranslate" in method: + path = path_deep + elif "Local LLM" in method: + path = path_local + else: + path = path_gemini + + with open(path, "w", encoding="utf-8") as f: + f.write(text) + return path + + if res_content: + final_srt_path = save_translation(res_content, method_used) + + if final_srt_path: + # Validate the NEW translation immediately + is_valid_new, msg_new = check_srt_duration_match(source_srt_path, final_srt_path) + + if not is_valid_new: + print(f"❌ New translation ({method_used}) failed validation: {msg_new}") + + # --- RETRY WITH LOCAL LLM --- + # Only retry if we haven't already used Local LLM and it is available + if "Local LLM" not in method_used and service_status.get("Ollama", False): + print(" -> Retrying with Local LLM (Ollama) as fallback strategy...") + + # Force try Ollama + from translator import translate_via_ollama + retry_content = translate_via_ollama(content, target_language=target_lang) + + if retry_content: + retry_path = path_local + with open(retry_path, "w", encoding="utf-8") as f: + f.write(retry_content) + + # Validate Retry + valid_retry, msg_retry = check_srt_duration_match(source_srt_path, retry_path) + if valid_retry: + print(f" ✅ Local LLM Retry Succeeded! Using: {os.path.basename(retry_path)}") + # Rename/Cleanup the previous failed attempt + invalid_path = final_srt_path + ".invalid" + os.replace(final_srt_path, invalid_path) + + final_srt_path = retry_path + method_used = "Local LLM (Retry)" + is_valid_new = True # Mark as valid so we proceed to embedding + else: + print(f" ❌ Local LLM Retry also failed validation: {msg_retry}") + # Cleanup retry attempt + os.replace(retry_path, retry_path + ".invalid") + + if not is_valid_new: + # Rename the invalid file so it doesn't sit there as a "fake" good translation + invalid_path = final_srt_path + ".invalid" + if os.path.exists(final_srt_path): + os.replace(final_srt_path, invalid_path) + print(f" -> Moved failed attempt to: {os.path.basename(invalid_path)}") + + with open(recovery_log_file, "a", encoding="utf-8") as log: + log.write(f"{datetime.now().isoformat()} | {method_used} | FAILED_VALIDATION | {file}\n") + continue + + # Log result + with open(recovery_log_file, "a", encoding="utf-8") as log: + log.write(f"{datetime.now().isoformat()} | {method_used} | FIXED | {file} -> {os.path.basename(final_srt_path)}\n") + + validate_and_repair_srt(final_srt_path) + + video_candidates = [ + os.path.join(root, base_name + ".mp4"), + os.path.join(root, base_name + ".mkv"), + os.path.join(root, base_name + ".subbed.mp4"), + ] + + found_video = None + for v in video_candidates: + if os.path.exists(v): + found_video = v + break + + if found_video: + print(f"Found video to fix: {found_video}") + temp_video_out = found_video + ".temp_fix.mp4" + try: + embed_subtitles(found_video, final_srt_path, output_path=temp_video_out) + os.replace(temp_video_out, found_video) + print(f"✅ Fixed: {found_video}") + count_fixed += 1 + except Exception as e: + print(f"Error re-embedding: {e}") + if os.path.exists(temp_video_out): + os.remove(temp_video_out) + else: + print("Warning: Could not find a corresponding video file to embed into.") + else: + print("❌ All translation methods failed. Skipping.") + + if killer.kill_now: + print("\n🛑 Recovery process stopped by user.") + else: + print(f"\nRecovery Complete. Fixed {count_fixed} files.") + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description="Recover and Fix Translations (V2)") + parser.add_argument("folders", nargs='+', help="One or more paths to folders to scan") + parser.add_argument("--lang", default="English", help="Target language (default: English)") + parser.add_argument("--prefer-deep", action="store_true", help="Prefer DeepTranslate (Free) over Gemini API") + parser.add_argument("--prefer-local", action="store_true", help="Prefer Local LLM (Ollama) over cloud APIs") + + args = parser.parse_args() + + for folder in args.folders: + if os.path.exists(folder): + process_recovery(folder, args.lang, args.prefer_deep, args.prefer_local) + else: + print(f"Error: Folder '{folder}' does not exist. Skipping.") \ No newline at end of file diff --git a/video_transcription/ai_transcriber_v2/requirements.txt b/video_transcription/ai_transcriber_v2/requirements.txt index 9856a52..30881b2 100644 --- a/video_transcription/ai_transcriber_v2/requirements.txt +++ b/video_transcription/ai_transcriber_v2/requirements.txt @@ -6,3 +6,7 @@ numpy tenacity pysubs2 pyannote.audio +deep-translator +ollama +chardet +tqdm diff --git a/video_transcription/run_v2.bat b/video_transcription/ai_transcriber_v2/run_v2.bat similarity index 100% rename from video_transcription/run_v2.bat rename to video_transcription/ai_transcriber_v2/run_v2.bat diff --git a/video_transcription/run_v2.py b/video_transcription/ai_transcriber_v2/run_v2.py similarity index 94% rename from video_transcription/run_v2.py rename to video_transcription/ai_transcriber_v2/run_v2.py index 560d011..46b4f67 100755 --- a/video_transcription/run_v2.py +++ b/video_transcription/ai_transcriber_v2/run_v2.py @@ -36,7 +36,7 @@ def main(): try: from dotenv import load_dotenv script_dir = os.path.dirname(os.path.abspath(__file__)) - env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe')) + env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe')) if os.path.exists(env_path): load_dotenv(env_path) except ImportError: @@ -59,7 +59,7 @@ def main(): continue # Clean up input - input_path = input_path.strip("\'"") + input_path = input_path.strip('"\'') input_path = input_path.replace(r'\ ', ' ') # Expand user (~) and resolve absolute path @@ -98,6 +98,7 @@ def main(): do_delete_source = get_yes_no("Delete original source files after embedding?", default="n") do_prefer_deep = get_yes_no("Prefer DeepTranslate (Free) over Gemini API?", default="n") + do_prefer_local = get_yes_no("Prefer Local LLM (Ollama) over all cloud options?", default="n") hf_token = None if do_diarize: @@ -110,7 +111,7 @@ def main(): # 5. Build Command # Point to v2 main script script_dir = os.path.dirname(os.path.abspath(__file__)) - main_script = os.path.join(script_dir, "ai_transcriber_v2", "main.py") + main_script = os.path.join(script_dir, "main.py") cmd = [sys.executable, main_script] cmd.extend(input_paths) @@ -138,6 +139,9 @@ def main(): if do_prefer_deep: cmd.append("--prefer-deep") + if do_prefer_local: + cmd.append("--prefer-local") + # 6. Confirmation and Execution clear_screen() print_header() @@ -154,6 +158,7 @@ def main(): print(f"Delete Src: {do_delete_source}") print(f"Diarization: {do_diarize}") print(f"Prefer Deep: {do_prefer_deep}") + print(f"Prefer Local: {do_prefer_local}") print("-" * 30) if not get_yes_no("Run this job now?", default="y"): diff --git a/video_transcription/run_wizard_v2.py b/video_transcription/ai_transcriber_v2/run_wizard_v2.py similarity index 94% rename from video_transcription/run_wizard_v2.py rename to video_transcription/ai_transcriber_v2/run_wizard_v2.py index 560d011..46b4f67 100755 --- a/video_transcription/run_wizard_v2.py +++ b/video_transcription/ai_transcriber_v2/run_wizard_v2.py @@ -36,7 +36,7 @@ def main(): try: from dotenv import load_dotenv script_dir = os.path.dirname(os.path.abspath(__file__)) - env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe')) + env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe')) if os.path.exists(env_path): load_dotenv(env_path) except ImportError: @@ -59,7 +59,7 @@ def main(): continue # Clean up input - input_path = input_path.strip("\'"") + input_path = input_path.strip('"\'') input_path = input_path.replace(r'\ ', ' ') # Expand user (~) and resolve absolute path @@ -98,6 +98,7 @@ def main(): do_delete_source = get_yes_no("Delete original source files after embedding?", default="n") do_prefer_deep = get_yes_no("Prefer DeepTranslate (Free) over Gemini API?", default="n") + do_prefer_local = get_yes_no("Prefer Local LLM (Ollama) over all cloud options?", default="n") hf_token = None if do_diarize: @@ -110,7 +111,7 @@ def main(): # 5. Build Command # Point to v2 main script script_dir = os.path.dirname(os.path.abspath(__file__)) - main_script = os.path.join(script_dir, "ai_transcriber_v2", "main.py") + main_script = os.path.join(script_dir, "main.py") cmd = [sys.executable, main_script] cmd.extend(input_paths) @@ -138,6 +139,9 @@ def main(): if do_prefer_deep: cmd.append("--prefer-deep") + if do_prefer_local: + cmd.append("--prefer-local") + # 6. Confirmation and Execution clear_screen() print_header() @@ -154,6 +158,7 @@ def main(): print(f"Delete Src: {do_delete_source}") print(f"Diarization: {do_diarize}") print(f"Prefer Deep: {do_prefer_deep}") + print(f"Prefer Local: {do_prefer_local}") print("-" * 30) if not get_yes_no("Run this job now?", default="y"): diff --git a/video_transcription/ai_transcriber_v2/transcriber.py b/video_transcription/ai_transcriber_v2/transcriber.py index f202022..2a3148f 100644 --- a/video_transcription/ai_transcriber_v2/transcriber.py +++ b/video_transcription/ai_transcriber_v2/transcriber.py @@ -115,7 +115,28 @@ def save_as_srt(result, output_path): f.write(f"{text}\n\n") print(f"SRT saved to: {output_path}") -def transcribe_audio(audio_path, model_size="auto", language=None): +def load_whisper_model(model_size="auto"): + """ + Loads and returns the Whisper model. + """ + check_gpu_health() + + if model_size == "auto": + model_size = get_optimal_model_size() + print(f"Auto-selected model: '{model_size}'") + + print(f"Loading Whisper model ('{model_size}')...") + device = "cuda" if torch.cuda.is_available() else "cpu" + print(f"Using device: {device}") + + try: + model = whisper.load_model(model_size, device=device) + return model + except Exception as e: + print(f"Error loading model: {e}") + sys.exit(1) + +def transcribe_audio(audio_path, model_size="auto", language=None, loaded_model=None): """ Transcribes an audio file using OpenAI's Whisper model. @@ -123,6 +144,7 @@ def transcribe_audio(audio_path, model_size="auto", language=None): audio_path (str): Path to the input audio file. model_size (str): Size of the Whisper model to use. If "auto", selects based on VRAM. language (str, optional): Language code (e.g., "en", "fr", "es"). If None, auto-detects. + loaded_model (object, optional): Pre-loaded Whisper model object. Returns: dict: The full transcription result containing segments and text. @@ -130,37 +152,14 @@ def transcribe_audio(audio_path, model_size="auto", language=None): if not os.path.exists(audio_path): raise FileNotFoundError(f"Audio file not found: {audio_path}") - # Run health check once - check_gpu_health() - - # Determine model size if auto - if model_size == "auto": - model_size = get_optimal_model_size() - print(f"Auto-selected model: '{model_size}'") - - print(f"Loading Whisper model ('{model_size}')...") - - # Check for GPU availability - device = "cuda" if torch.cuda.is_available() else "cpu" - print(f"Using device: {device}") - - try: - model = whisper.load_model(model_size, device=device) - except RuntimeError as e: - if "out of memory" in str(e).lower(): - print("Error: GPU Out of Memory. Try using a smaller model size.") - else: - print(f"Error loading model: {e}") - sys.exit(1) - except Exception as e: - print(f"Error loading model: {e}") - sys.exit(1) + model = loaded_model + if model is None: + model = load_whisper_model(model_size) print(f"Transcribing {audio_path}...") try: - # fp16=False is needed for CPU, but we can let whisper handle defaults usually. - # language=None allows auto-detection. - result = model.transcribe(audio_path, language=language) + # Enable verbose=True to show progress in terminal + result = model.transcribe(audio_path, language=language, verbose=True) print("Transcription complete.") return result except Exception as e: diff --git a/video_transcription/ai_transcriber_v2/translator.py b/video_transcription/ai_transcriber_v2/translator.py index b548a88..0419891 100644 --- a/video_transcription/ai_transcriber_v2/translator.py +++ b/video_transcription/ai_transcriber_v2/translator.py @@ -4,11 +4,71 @@ from google import genai from google.genai import types from tenacity import retry, stop_after_attempt, wait_exponential, retry_if_exception_type import pysubs2 -from deep_translator import GoogleTranslator +from deep_translator import GoogleTranslator, MyMemoryTranslator +import ollama +from tqdm import tqdm # Define a retry decorator # ... (retry_policy remains) +def translate_via_ollama(source_srt_content, target_language="English", model="llama3"): + """ + Translates SRT content using a local Ollama model (Line-by-Line for progress). + """ + try: + subs = pysubs2.SSAFile.from_string(source_srt_content) + + # Using tqdm for progress bar + for line in tqdm(subs, desc=" Ollama Progress", unit="line"): + text = line.text.strip() + if text: + prompt = ( + f"Translate this subtitle text to {target_language}. Output ONLY the translation.\n" + f"Text: {text}" + ) + try: + response = ollama.chat(model=model, messages=[{'role': 'user', 'content': prompt}]) + translated_text = response['message']['content'].strip() + if translated_text: + line.text = translated_text + except Exception as e: + # Silent fail on line, logs would be too spammy in progress bar + pass + + return subs.to_string(format_="srt") + + except Exception as e: + print(f" [Local LLM] Error: {e}") + return None + +def translate_fallback_mymemory(source_srt_content, target_language="en"): + """ + Fallback translation using MyMemory (via deep-translator). + Limit: 1000 words/day roughly for anonymous usage. Good last resort. + """ + try: + subs = pysubs2.SSAFile.from_string(source_srt_content) + # MyMemory uses ISO 639-1 usually + translator = MyMemoryTranslator(source='auto', target=target_language) + + for line in tqdm(subs, desc=" MyMemory Progress", unit="line"): + text = line.text.strip() + if text: + if len(text) > 500: # MyMemory has stricter limits often + continue + try: + original_text = text.replace(r"\N", " ") + translated_text = translator.translate(original_text) + if translated_text: + line.text = translated_text + except Exception: + pass + + return subs.to_string(format_="srt") + except Exception as e: + print(f" [MyMemory Fallback] Critical Error: {e}") + return None + def translate_fallback_free(source_srt_content, target_language="en"): """ Fallback translation using deep-translator (free Google Translate). @@ -20,19 +80,17 @@ def translate_fallback_free(source_srt_content, target_language="en"): Returns: str: Translated SRT content, or None if failed. """ - print(f" [Free Fallback] Translating via Google Translate (deep-translator)...") try: # Load from string subs = pysubs2.SSAFile.from_string(source_srt_content) translator = GoogleTranslator(source='auto', target=target_language) # Simple line-by-line translation - for line in subs: + for line in tqdm(subs, desc=" DeepTranslate Progress", unit="line"): text = line.text.strip() if text: # Sanity check: Skip lines that are too long if len(text) > 4000: - print(f" Warning: Skipping line with excessive length ({len(text)} chars).") continue try: @@ -41,8 +99,8 @@ def translate_fallback_free(source_srt_content, target_language="en"): translated_text = translator.translate(original_text) if translated_text: line.text = translated_text - except Exception as e: - print(f" Warning: Failed to translate line: {e}") + except Exception: + pass # Return as string return subs.to_string(format_="srt") @@ -53,25 +111,26 @@ def translate_fallback_free(source_srt_content, target_language="en"): # Define a retry decorator # Waits 2^x * 1 seconds between retries (1s, 2s, 4s...) # Stop after 15 attempts +# before_sleep logic can print a simple message +def log_retry_attempt(retry_state): + if retry_state.attempt_number > 1: + print(f" [Gemini] Rate limit hit. Retrying in {retry_state.next_action.sleep}s...", end='\r') + retry_policy = retry( stop=stop_after_attempt(15), wait=wait_exponential(multiplier=1, min=2, max=60), retry=retry_if_exception_type(Exception), - reraise=True + reraise=True, + before_sleep=log_retry_attempt ) @retry_policy def _generate_with_retry(client, model_name, prompt): """Internal function to wrap the API call with retry logic.""" - try: - return client.models.generate_content( - model=model_name, - contents=prompt - ) - except Exception as e: - if "429" in str(e) or "Resource has been exhausted" in str(e): - print(f" [Rate Limit Hit] Waiting for quota reset... ({e})") - raise e + return client.models.generate_content( + model=model_name, + contents=prompt + ) def get_best_available_model(client): """ @@ -126,7 +185,6 @@ def translate_srt(srt_content, target_language="English", api_key=None): # Automatically select the best model # Note: v2 SDK might use 'gemini-1.5-flash' directly without 'models/' prefix usually model_name = "gemini-2.0-flash" - print(f"Using Gemini Model (v2): {model_name}") prompt = ( "You are a professional subtitle translator. Your task is to translate the following SRT subtitle file " @@ -140,11 +198,9 @@ def translate_srt(srt_content, target_language="English", api_key=None): f"{srt_content}" ) - print(f"Translating subtitles to {target_language} (with retries)...") try: # Call the retried internal function response = _generate_with_retry(client, model_name, prompt) - print("Translation complete.") # Cleanup: sometimes models wrap output in ```srt ... ``` or ``` ... ``` cleaned_text = response.text.strip() @@ -171,3 +227,84 @@ def translate_srt(srt_content, target_language="English", api_key=None): print(f"Fallback failed: {inner_e}") return None + +def translate_with_auto_fallback(srt_content, target_language="English", prefer_deep=False, prefer_local=False, available_services=None): + """ + Attempts to translate SRT content using Gemini, DeepTranslate, and Local LLM with fallback logic. + + Args: + srt_content (str): The source SRT content. + target_language (str): Target language name (e.g., "English", "French"). + prefer_deep (bool): If True, try DeepTranslate first (among cloud services). + prefer_local (bool): If True, try Local LLM (Ollama) first. + available_services (dict, optional): Result of check_service_availability(). + + Returns: + tuple: (translated_content, method_name) or (None, None) if all failed. + """ + + # Map full language name to code for DeepTranslate + lang_map = { + "English": "en", "French": "fr", "Spanish": "es", "German": "de", + "Italian": "it", "Portuguese": "pt", "Russian": "ru", + "Japanese": "ja", "Chinese": "zh-CN" + } + target_code = lang_map.get(target_language, "en") + + # Determine which services to even try + def is_ok(name): + if available_services is None: return True + return available_services.get(name, True) + + def try_gemini(): + if not is_ok("Gemini"): return None, None + res = translate_srt(srt_content, target_language=target_language) + if res: return res, "Gemini" + return None, None + + def try_deep(): + if not is_ok("DeepTranslate"): return None, None + res = translate_fallback_free(srt_content, target_language=target_code) + if res: return res, "DeepTranslate" + return None, None + + def try_ollama(): + if not is_ok("Ollama"): return None, None + res = translate_via_ollama(srt_content, target_language=target_language) + if res: return res, "Local LLM (Ollama)" + return None, None + + def try_mymemory(): + res = translate_fallback_mymemory(srt_content, target_language=target_code) + if res: return res, "MyMemory" + return None, None + + # Logic flow + attempts = [] + + if prefer_local: + attempts.append(try_ollama) + if prefer_deep: + attempts.extend([try_deep, try_gemini]) + else: + attempts.extend([try_gemini, try_deep]) + else: + if prefer_deep: + attempts.extend([try_deep, try_gemini]) + else: + attempts.extend([try_gemini, try_deep]) + attempts.append(try_ollama) + + # Final last resort + attempts.append(try_mymemory) + + # Execute attempts + for i, method_func in enumerate(attempts): + if i > 0: + print(f" Attempt {i} failed. Trying next fallback...") + + content, method = method_func() + if content: + return content, method + + return None, None diff --git a/video_transcription/ai_transcriber_v2/utils.py b/video_transcription/ai_transcriber_v2/utils.py index 8c452bf..f28e092 100644 --- a/video_transcription/ai_transcriber_v2/utils.py +++ b/video_transcription/ai_transcriber_v2/utils.py @@ -1,5 +1,220 @@ import pysubs2 import os +import signal +import sys +import subprocess +import time +import socket +import shutil +import chardet +from tqdm import tqdm + +class GracefulKiller: + """ + Handles SIGINT (Ctrl+C) and SIGTERM signals. + Allows the application to finish the current task before exiting. + """ + kill_now = False + + def __init__(self): + signal.signal(signal.SIGINT, self.exit_gracefully) + signal.signal(signal.SIGTERM, self.exit_gracefully) + + def exit_gracefully(self, signum, frame): + if not self.kill_now: + self.kill_now = True + print("\n\n[STOP REQUESTED] The script will exit after the current file finishes processing.") + print("Press Ctrl+C again to force quit immediately (not recommended).\n") + else: + print("\n[FORCE QUIT] Exiting immediately...") + sys.exit(1) + +def is_port_open(host, port): + with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s: + s.settimeout(1) + return s.connect_ex((host, port)) == 0 + +def detect_file_encoding(file_path): + """ + Robustly detects the encoding of a file using chardet. + Returns 'utf-8' if detection fails or confidence is low, as a safe default. + """ + try: + with open(file_path, 'rb') as f: + raw_data = f.read(10000) # Read first 10KB + result = chardet.detect(raw_data) + + encoding = result['encoding'] + confidence = result['confidence'] + + if encoding and confidence > 0.7: + # Shift-JIS is often detected as other Japanese variants, which is fine, + # but sometimes we want to be specific. Chardet is usually good. + return encoding + return 'utf-8' + except Exception: + return 'utf-8' + +def verify_file_not_empty(file_path): + """ + Checks if a file exists and is larger than 0 bytes. + """ + if os.path.exists(file_path) and os.path.getsize(file_path) > 0: + return True + return False + +def ensure_ollama_running(model_name="llama3"): + """ + Checks if Ollama is running. If not, attempts to start it. + Supports Flatpak by escaping to host via flatpak-spawn. + """ + in_flatpak = os.path.exists("/.flatpak-info") + + def run_cmd(cmd_list, capture=False): + if in_flatpak: + full_cmd = ["flatpak-spawn", "--host"] + cmd_list + else: + full_cmd = cmd_list + + try: + if capture: + return subprocess.run(full_cmd, capture_output=True, text=True) + else: + # For serve, we use Popen + return subprocess.Popen( + full_cmd, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + start_new_session=True + ) + except Exception: + return None + + # 1. Start Server if port is closed + if not is_port_open("127.0.0.1", 11434): + print("Starting Ollama server (Local LLM)...") + run_cmd(["ollama", "serve"]) + + print(" Waiting for Ollama to initialize...", end="", flush=True) + for _ in range(10): + if is_port_open("127.0.0.1", 11434): + print(" Done.") + break + time.sleep(1) + print(".", end="", flush=True) + else: + print("\n Warning: Ollama server failed to start or binary not found.") + return False + + # 2. Check Model Presence + try: + result = run_cmd(["ollama", "list"], capture=True) + if result and result.returncode == 0: + if model_name not in result.stdout: + print(f" Model '{model_name}' not found. Pulling now (this may take a while)...") + # Pulling can take a long time, so we don't capture but we want to wait + pull_cmd = ["flatpak-spawn", "--host", "ollama", "pull", model_name] if in_flatpak else ["ollama", "pull", model_name] + subprocess.run(pull_cmd, check=True) + print(" Model pulled successfully.") + else: + # If we can't run list, but port is open, we assume it's okay and let the library handle it + pass + except Exception as e: + print(f" Warning: Could not verify/pull Ollama model: {e}") + + return True + +def check_service_availability(prefer_deep=False): + """ + Checks availability of configured translation services by performing tiny tests. + Returns a dictionary of status. + """ + status = { + "Gemini": False, + "DeepTranslate": False, + "Ollama": False + } + + print("Checking Services...") + + # 1. Check Gemini (Real Test) + gemini_key = os.getenv("GEMINI_API_KEY") + if gemini_key: + try: + # We import here to avoid global import issues if dependencies are missing + from google import genai + client = genai.Client(api_key=gemini_key) + # Try a very cheap call + client.models.generate_content( + model="gemini-2.0-flash", + contents="Hi" + ) + status["Gemini"] = True + except Exception as e: + # Check for rate limit in string representation + if "429" in str(e) or "RESOURCE_EXHAUSTED" in str(e): + # It is technically 'configured' but currently useless + status["Gemini"] = False + else: + status["Gemini"] = False + + # 2. Check DeepTranslate (Real Test) + try: + from deep_translator import GoogleTranslator + GoogleTranslator(source='auto', target='en').translate("hola") + status["DeepTranslate"] = True + except Exception: + status["DeepTranslate"] = False + + # 3. Check Ollama + if is_port_open("127.0.0.1", 11434): + status["Ollama"] = True + + # Print Report + gemini_msg = "[READY]" if status['Gemini'] else "[UNAVAILABLE] (Rate Limited or Key Invalid)" + if not gemini_key: gemini_msg = "[UNAVAILABLE] (Key missing)" + + print(f"1. Gemini API: {gemini_msg}") + print(f"2. DeepTranslate: {'[READY]' if status['DeepTranslate'] else '[UNAVAILABLE] (Network/Block)'}") + print(f"3. Local Ollama: {'[READY]' if status['Ollama'] else '[OFFLINE]'}") + + return status + +def check_path_permissions(directory_path): + """ + Checks if the script has read and write permissions for the given directory. + Returns: (bool, message) + """ + if not os.path.exists(directory_path): + return False, f"Path not found: {directory_path}" + + # If it's a file, check parent directory + if os.path.isfile(directory_path): + directory_path = os.path.dirname(directory_path) + + test_file = os.path.join(directory_path, ".perm_test_tmp") + + try: + # Test Write + with open(test_file, "w") as f: + f.write("test") + + # Test Read + with open(test_file, "r") as f: + content = f.read() + + # Cleanup + os.remove(test_file) + + if content == "test": + return True, f" [Permissions] Read/Write OK: {directory_path}" + else: + return False, f" [Permissions] Read check failed (content mismatch): {directory_path}" + + except PermissionError: + return False, f" ❌ [Permissions] DENIED: Cannot write to {directory_path}. Check ownership/mount options." + except Exception as e: + return False, f" ❌ [Permissions] Error checking {directory_path}: {e}" def validate_and_repair_srt(srt_path): """ @@ -26,3 +241,53 @@ def validate_and_repair_srt(srt_path): except Exception as e: print(f"Warning: SRT validation failed: {e}") return False + +def check_srt_duration_match(source_srt_path, target_srt_path, tolerance_seconds=30.0, tolerance_percent=0.10): + """ + Compares the duration of two SRT files to ensure they cover roughly the same timeframe. + Useful for detecting partial translations. + + Args: + source_srt_path (str): Path to the original language SRT. + target_srt_path (str): Path to the translated SRT. + tolerance_seconds (float): Max allowed difference in seconds. + tolerance_percent (float): Max allowed difference as a percentage of source duration. + + Returns: + tuple: (bool, str) -> (passed, message) + """ + if not os.path.exists(source_srt_path) or not os.path.exists(target_srt_path): + return False, "One or both SRT files missing." + + try: + source_subs = pysubs2.load(source_srt_path) + target_subs = pysubs2.load(target_srt_path) + except Exception as e: + return False, f"Error parsing SRTs: {e}" + + if not source_subs: + return False, "Source SRT is empty." + if not target_subs: + return False, "Target SRT is empty." + + # Get the end timestamp of the last event in each file (in milliseconds) + source_end = source_subs[-1].end + target_end = target_subs[-1].end + + # Convert to seconds + source_duration = source_end / 1000.0 + target_duration = target_end / 1000.0 + + diff = abs(source_duration - target_duration) + + # Check absolute difference + if diff > tolerance_seconds: + # Also check percentage (for very long videos, 30s might be negligible) + if source_duration > 0 and (diff / source_duration) > tolerance_percent: + return False, f"Duration mismatch: Source={source_duration:.1f}s, Target={target_duration:.1f}s (Diff={diff:.1f}s)" + + # For short videos, if percentage is high, fail + if source_duration < 300 and (diff / source_duration) > 0.20: + return False, f"Duration mismatch (short video): Source={source_duration:.1f}s, Target={target_duration:.1f}s" + + return True, f"Duration match verified (Diff={diff:.1f}s)" diff --git a/video_transcription/recover_and_fix_v2.py b/video_transcription/recover_and_fix_v2.py deleted file mode 100755 index ee89e4f..0000000 --- a/video_transcription/recover_and_fix_v2.py +++ /dev/null @@ -1,254 +0,0 @@ -#!/usr/bin/env python3 -import os -import sys -import argparse -import subprocess -from dotenv import load_dotenv -from datetime import datetime -import pysubs2 -from deep_translator import GoogleTranslator - -# Load config -script_dir = os.path.dirname(os.path.abspath(__file__)) -env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe')) -if os.path.exists(env_path): - load_dotenv(env_path) -else: - load_dotenv() - -# Add ai_transcriber_v2 to path so we can import modules -sys.path.append(os.path.join(script_dir, 'ai_transcriber_v2')) - -from ai_transcriber_v2.translator import translate_srt -from ai_transcriber_v2.utils import validate_and_repair_srt - -def translate_fallback_free(source_srt_path, output_srt_path, target_lang="en"): - """ - Fallback translation using deep-translator (free Google Translate). - Parses SRT, translates text line-by-line, and saves new SRT. - """ - print(f" [Free Fallback] Translating {source_srt_path}...") - - subs = None - encodings_to_try = ['utf-8', 'shift_jis', 'euc_jp', 'latin-1', 'cp1252', 'utf-16'] - - for enc in encodings_to_try: - try: - subs = pysubs2.load(source_srt_path, encoding=enc) - break - except Exception: - continue - - if subs is None: - print(f" [Free Fallback] Critical Error: Could not decode file with standard encodings.") - return False - - try: - translator = GoogleTranslator(source='auto', target=target_lang) - - for line in subs: - text = line.text.strip() - if text: - # Sanity check: Skip lines that are too long (likely garbage/corruption) - if len(text) > 4000: - print(f" Warning: Skipping line with excessive length ({len(text)} chars). Likely corrupted.") - continue - - try: - original_text = text.replace(r"\N", " ") - translated_text = translator.translate(original_text) - if translated_text: - line.text = translated_text - except Exception as e: - print(f" Warning: Failed to translate line: {e}") - - subs.save(output_srt_path) - print(f" [Free Fallback] Saved to {output_srt_path}") - return True - except Exception as e: - print(f" [Free Fallback] Critical Error: {e}") - return False - -def re_embed_subtitles(video_path, srt_path, output_path=None): - """ - Re-embeds subtitles into an EXISTING video file, replacing the old tracks. - Uses robust flags to handle bad metadata. - """ - if not os.path.exists(video_path) or not os.path.exists(srt_path): - print("Error: Video or SRT file not found.") - return False - - temp_output = video_path + ".temp.mp4" - print(f"Re-embedding subtitles into: {video_path}...") - - sub_codec = "mov_text" if video_path.lower().endswith(".mp4") else "srt" - - command = [ - "ffmpeg", - "-ignore_editlist", "1", - "-i", video_path, - "-i", srt_path, - "-map", "0:v", - "-map", "0:a", - "-map", "1:0", - "-c", "copy", - "-c:s", sub_codec, - "-disposition:s:0", "default", - "-metadata:s:s:0", "language=eng", - "-metadata:s:s:0", "title=English (AI Translated)", - "-max_interleave_delta", "0", - "-avoid_negative_ts", "make_zero", - "-y", - "-v", "error", - temp_output - ] - - try: - subprocess.run(command, check=True) - os.replace(temp_output, video_path) - print(f"✅ Fixed: {video_path}") - return True - except subprocess.CalledProcessError as e: - print(f"Error re-embedding: {e}") - if os.path.exists(temp_output): - os.remove(temp_output) - return False - -def process_recovery(folder_path, target_lang="English", prefer_deep=False): - print(f"Scanning {folder_path} for incomplete translations (V2)...") - if prefer_deep: - print("Preference: DeepTranslate (Google Translate Free) > Gemini") - else: - print("Preference: Gemini (API) > DeepTranslate") - - recovery_log_file = os.path.join(folder_path, "recovery_status.log") - print(f"Logging actions to: {recovery_log_file}") - - count_fixed = 0 - video_extensions = ('.mp4', '.mkv', '.mov', '.avi') - - for root, dirs, files in os.walk(folder_path): - for file in files: - if file.endswith(".srt") and \ - not file.endswith(f".{target_lang}.srt") and \ - not file.endswith(f".{target_lang}.deep_translate.srt"): - - source_srt_path = os.path.join(root, file) - base_name = os.path.splitext(file)[0] - - path_gemini = os.path.join(root, f"{base_name}.{target_lang}.srt") - path_deep = os.path.join(root, f"{base_name}.{target_lang}.deep_translate.srt") - - if os.path.exists(path_gemini) or os.path.exists(path_deep): - continue - - print(f"\nFound untranslated transcript: {file}") - - content = None - # extended list to include common Japanese encodings - encodings_to_try = ['utf-8', 'shift_jis', 'euc_jp', 'latin-1', 'cp1252', 'utf-16'] - - for enc in encodings_to_try: - try: - with open(source_srt_path, "r", encoding=enc) as f: - content = f.read() - break # Success - except UnicodeDecodeError: - continue - - if content is None: - print(f"❌ Error: Could not decode {file} with any standard encoding. Skipping.") - continue - - # Helpers for translation attempts - def try_gemini(): - res = translate_srt(content, target_language=target_lang) - if res: - with open(path_gemini, "w", encoding="utf-8") as f: - f.write(res) - return True, path_gemini, "Gemini" - return False, None, None - - def try_deep(): - lang_map = { - "English": "en", "French": "fr", "Spanish": "es", - "German": "de", "Italian": "it", "Portuguese": "pt", - "Russian": "ru", "Japanese": "ja", "Chinese": "zh-CN" - } - target_code = lang_map.get(target_lang, "en") - if translate_fallback_free(source_srt_path, path_deep, target_lang=target_code): - return True, path_deep, "DeepTranslate" - return False, None, None - - success = False - method_used = "None" - final_srt_path = None - - if prefer_deep: - # 1. Try DeepTranslate - success, final_srt_path, method_used = try_deep() - if not success: - print("❌ DeepTranslate failed. Attempting Gemini fallback...") - success, final_srt_path, method_used = try_gemini() - else: - # 1. Try Gemini - success, final_srt_path, method_used = try_gemini() - if not success: - print("❌ Gemini API failed. Attempting Free Fallback...") - success, final_srt_path, method_used = try_deep() - - if success and final_srt_path: - # Log result - with open(recovery_log_file, "a", encoding="utf-8") as log: - log.write(f"{datetime.now().isoformat()} | {method_used} | {file} -> {os.path.basename(final_srt_path)}\n") - - validate_and_repair_srt(final_srt_path) - - video_candidates = [ - os.path.join(root, base_name + ".mp4"), - os.path.join(root, base_name + ".mkv"), - os.path.join(root, base_name + ".subbed.mp4"), - ] - - found_video = None - for v in video_candidates: - if os.path.exists(v): - found_video = v - break - - if found_video: - print(f"Found video to fix: {found_video}") - if re_embed_subtitles(found_video, final_srt_path): - count_fixed += 1 - else: - print("Warning: Could not find a corresponding video file to embed into.") - else: - print("❌ All translation methods failed. Skipping.") - - print(f"\nRecovery Complete. Fixed {count_fixed} files.") - -if __name__ == "__main__": - - parser = argparse.ArgumentParser(description="Recover and Fix Translations (V2)") - - parser.add_argument("folders", nargs='+', help="One or more paths to folders to scan") - - parser.add_argument("--lang", default="English", help="Target language (default: English)") - - parser.add_argument("--prefer-deep", action="store_true", help="Prefer DeepTranslate (Free) over Gemini API") - - - - args = parser.parse_args() - - - - for folder in args.folders: - - if os.path.exists(folder): - - process_recovery(folder, args.lang, args.prefer_deep) - - else: - - print(f"Error: Folder '{folder}' does not exist. Skipping.") \ No newline at end of file