diff --git a/video_transcription/ai_transcriber/main.py b/video_transcription/ai_transcriber/main.py index 6709489..e351015 100644 --- a/video_transcription/ai_transcriber/main.py +++ b/video_transcription/ai_transcriber/main.py @@ -4,14 +4,11 @@ import sys from dotenv import load_dotenv # Load environment variables from central .env_files directory -# Path: .../personal_development/video_transcription/ai_transcriber/main.py -# Target: .../personal_development/.env_files/.env.aitranscribe script_dir = os.path.dirname(os.path.abspath(__file__)) env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe')) if os.path.exists(env_path): load_dotenv(env_path) - # print(f"Loaded configuration from: {env_path}") # Optional: Uncomment for debugging else: # Fallback: check local .env local_env = os.path.join(script_dir, '.env') @@ -23,9 +20,11 @@ else: from extractor import extract_audio, embed_subtitles from transcriber import transcribe_audio, save_as_srt -from translator import translate_srt +from translator import translate_srt, translate_fallback_free from utils import validate_and_repair_srt from diarizer import diarize_audio, merge_diarization_with_transcript +import tracker +from tracker import JobStatus def save_srt_with_speakers(segments, output_path): """Helper to save SRT with speaker labels prepended to text.""" @@ -44,7 +43,6 @@ def save_srt_with_speakers(segments, output_path): text = segment["text"].strip() speaker = segment.get("speaker", "") - # Prepend speaker if present and not "Unknown" if speaker and speaker != "Unknown": text = f"[{speaker}]: {text}" @@ -54,464 +52,248 @@ def save_srt_with_speakers(segments, output_path): print(f"SRT saved to: {output_path}") def process_file(file_path, args, source_lang=None): - print(f"\n=== Processing: {file_path} ===") + tracker.logger.info(f"=== Processing: {file_path} ===") - # 1. Extract Audio - audio_path = extract_audio(file_path) + # Initialize Job + job = tracker.get_job(file_path) - # 2. Transcribe (Generate SRT) - transcript_file = os.path.splitext(file_path)[0] + ".srt" - transcript_exists = os.path.exists(transcript_file) and not args.force - - # Variable to hold final SRT path for embedding - final_srt_path = transcript_file + if job.status == JobStatus.COMPLETED and not args.force: + tracker.logger.info("Job already completed. Skipping.") + return - if transcript_exists: - print(f"Transcript exists: {transcript_file}. Skipping transcription.") - with open(transcript_file, "r", encoding="utf-8") as f: - srt_content = f.read() - else: - # Transcribe - result = transcribe_audio(audio_path, model_size=args.model, language=source_lang) - segments = result["segments"] + tracker.update_job_status(file_path, JobStatus.PROCESSING) - # Optional: Diarization - if args.diarize: - hf_token = args.hf_token or os.getenv("HF_TOKEN") - if hf_token: - print("Running Speaker Diarization...") - diar_segments = diarize_audio(audio_path, hf_token=hf_token) - if diar_segments: - segments = merge_diarization_with_transcript(segments, diar_segments) - print("Diarization merged into transcript.") - else: - print("Warning: --diarize requested but HF_TOKEN not provided. Skipping.") + try: + # 1. Extract Audio + tracker.update_step(file_path, "step_extract", "processing") + audio_path = extract_audio(file_path) + tracker.update_step(file_path, "step_extract", "done") + + # 2. Transcribe (Generate SRT) + tracker.update_step(file_path, "step_transcribe", "processing") + transcript_file = os.path.splitext(file_path)[0] + ".srt" + transcript_exists = os.path.exists(transcript_file) and not args.force + + final_srt_path = transcript_file - # Save SRT - # Use simple save if no speakers, or custom if speakers - if args.diarize: - save_srt_with_speakers(segments, transcript_file) + if transcript_exists: + tracker.logger.info(f"Transcript exists: {transcript_file}. Skipping transcription.") + with open(transcript_file, "r", encoding="utf-8") as f: + srt_content = f.read() else: - save_as_srt(result, transcript_file) - - # Validation - validate_and_repair_srt(transcript_file) - - with open(transcript_file, "r", encoding="utf-8") as f: - srt_content = f.read() + result = transcribe_audio(audio_path, model_size=args.model, language=source_lang) + segments = result["segments"] - # 3. Translate (Generate Translated SRT) - translated_file = os.path.splitext(file_path)[0] + f".{args.lang}.srt" - - translation_success = False + if args.diarize: + hf_token = args.hf_token or os.getenv("HF_TOKEN") + if hf_token: + tracker.logger.info("Running Speaker Diarization...") + diar_segments = diarize_audio(audio_path, hf_token=hf_token) + if diar_segments: + segments = merge_diarization_with_transcript(segments, diar_segments) + tracker.logger.info("Diarization merged into transcript.") + else: + tracker.logger.warning("Warning: --diarize requested but HF_TOKEN not provided. Skipping.") - if os.path.exists(translated_file) and not args.force: - print(f"Translation exists: {translated_file}. Skipping translation.") - final_srt_path = translated_file - translation_success = True - else: - # Only translate if there is content - if srt_content: - translated_srt_content = translate_srt(srt_content, target_language=args.lang) - if translated_srt_content: - with open(translated_file, "w", encoding="utf-8") as f: - f.write(translated_srt_content) - print(f"Translation saved to: {translated_file}") - validate_and_repair_srt(translated_file) - final_srt_path = translated_file - translation_success = True + if args.diarize: + save_srt_with_speakers(segments, transcript_file) else: - print("⚠️ TRANSLATION FAILED.") - translation_success = False + save_as_srt(result, transcript_file) + + validate_and_repair_srt(transcript_file) + with open(transcript_file, "r", encoding="utf-8") as f: + srt_content = f.read() + tracker.update_step(file_path, "step_transcribe", "done") - # 4. Embed Subtitles - # SAFETY: If translation was intended but failed, do NOT embed/delete to prevent - # replacing the video with one containing only untranslated subtitles. - should_embed = args.embed - if args.embed and not translation_success: - print("\n❌ SAFETY HALT: Translation failed. Skipping embedding and deletion to preserve original file.") - should_embed = False - - if should_embed: - embed_subtitles(file_path, final_srt_path) - - import argparse - import os - import sys - from dotenv import load_dotenv + # 3. Translate + tracker.update_step(file_path, "step_translate", "processing") - # Load environment variables from central .env_files directory - # Path: .../personal_development/video_transcription/ai_transcriber/main.py - # Target: .../personal_development/.env_files/.env.aitranscribe - script_dir = os.path.dirname(os.path.abspath(__file__)) - env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe')) + base_translated = os.path.splitext(file_path)[0] + f".{args.lang}.srt" + deep_translated = os.path.splitext(file_path)[0] + f".{args.lang}.deep_translate.srt" - if os.path.exists(env_path): - load_dotenv(env_path) - print(f"Loaded configuration from: {env_path}") + translated_file = base_translated # Default + translation_success = False + method_used = "None" + + if (os.path.exists(base_translated) or os.path.exists(deep_translated)) and not args.force: + if os.path.exists(deep_translated): + translated_file = deep_translated + method_used = "DeepTranslate (Existing)" + else: + method_used = "Gemini (Existing)" + tracker.logger.info(f"Translation exists: {translated_file} ({method_used}). Skipping translation.") + final_srt_path = translated_file + translation_success = True else: - # Fallback: check local .env - local_env = os.path.join(script_dir, '.env') - if os.path.exists(local_env): - load_dotenv(local_env) - else: - # Last resort: just try loading generic (cwd) - load_dotenv() - - from extractor import extract_audio, embed_subtitles - - from transcriber import transcribe_audio, save_as_srt - - from translator import translate_srt, translate_fallback_free - - from utils import validate_and_repair_srt - - from diarizer import diarize_audio, merge_diarization_with_transcript - - import tracker - - from tracker import JobStatus - - - - # ... (save_srt_with_speakers remains same) - - - - def process_file(file_path, args, source_lang=None): - - tracker.logger.info(f"=== Processing: {file_path} ===") - - - - # ... (Job init remains same) ... - - # ... (Step 1 Extract remains same) ... - - # ... (Step 2 Transcribe remains same) ... - - - - # 3. Translate (Generate Translated SRT) - - tracker.update_step(file_path, "step_translate", "processing") - + if srt_content: + # Helper functions + def try_gemini(): + res = translate_srt(srt_content, target_language=args.lang) + if res: + with open(base_translated, "w", encoding="utf-8") as f: + f.write(res) + return True, base_translated, "Gemini" + return False, None, None + + def try_deep(): + lang_map = { + "English": "en", "French": "fr", "Spanish": "es", "German": "de", + "Italian": "it", "Portuguese": "pt", "Russian": "ru", + "Japanese": "ja", "Chinese": "zh-CN" + } + target_code = lang_map.get(args.lang, "en") + res = translate_fallback_free(srt_content, target_language=target_code) + if res: + with open(deep_translated, "w", encoding="utf-8") as f: + f.write(res) + return True, deep_translated, "DeepTranslate" + return False, None, None + + success = False - - # Define paths - - base_translated = os.path.splitext(file_path)[0] + f".{args.lang}.srt" - - deep_translated = os.path.splitext(file_path)[0] + f".{args.lang}.deep_translate.srt" - - - - translated_file = base_translated # Default - - - - translation_success = False - - method_used = "None" - - - - if (os.path.exists(base_translated) or os.path.exists(deep_translated)) and not args.force: - - if os.path.exists(deep_translated): - - translated_file = deep_translated - - method_used = "DeepTranslate (Existing)" - - else: - - method_used = "Gemini (Existing)" - - - - tracker.logger.info(f"Translation exists: {translated_file} ({method_used}). Skipping translation.") - - final_srt_path = translated_file - - translation_success = True - + if args.prefer_deep: + success, path, method = try_deep() + if not success: + tracker.logger.info("DeepTranslate failed. Attempting Gemini...") + success, path, method = try_gemini() else: - - # Only translate if there is content - - if srt_content: - - # Attempt 1: Gemini - - translated_srt_content = translate_srt(srt_content, target_language=args.lang) - - - - if translated_srt_content: - - with open(base_translated, "w", encoding="utf-8") as f: - - f.write(translated_srt_content) - - tracker.logger.info(f"Translation saved to: {base_translated} (Gemini)") - - validate_and_repair_srt(base_translated) - - final_srt_path = base_translated - - translation_success = True - - method_used = "Gemini" - - else: - - # Attempt 2: Fallback - - tracker.logger.warning("Gemini translation failed. Attempting Free Fallback...") - - - - lang_map = { - - "English": "en", "French": "fr", "Spanish": "es", - - "German": "de", "Italian": "it", "Portuguese": "pt", - - "Russian": "ru", "Japanese": "ja", "Chinese": "zh-CN" - - } - - target_code = lang_map.get(args.lang, "en") - - - - translated_srt_content = translate_fallback_free(srt_content, target_language=target_code) - - - - if translated_srt_content: - - translated_file = deep_translated - - with open(translated_file, "w", encoding="utf-8") as f: - - f.write(translated_srt_content) - - tracker.logger.info(f"Translation saved to: {translated_file} (DeepTranslate)") - - validate_and_repair_srt(translated_file) - - final_srt_path = translated_file - - translation_success = True - - method_used = "DeepTranslate" - - else: - - tracker.logger.error("TRANSLATION FAILED (Both Gemini and Fallback).") - - tracker.update_step(file_path, "step_translate", "failed") - - translation_success = False - - - - if translation_success: - - tracker.update_step(file_path, "step_translate", "done") - - # Log method to tracker DB if we added a column for it, or just info log - - tracker.logger.info(f"Translation Method: {method_used}") - - - - # 4. Embed Subtitles - - - tracker.update_step(file_path, "step_embed", "processing") - should_embed = args.embed - if args.embed and not translation_success: - tracker.logger.warning("SAFETY HALT: Translation failed. Skipping embedding and deletion to preserve original file.") - should_embed = False - - if should_embed: - embed_subtitles(file_path, final_srt_path) - - # 5. Delete Source File (Optional & Risky) - if args.delete_source: - if args.embed: - # Safety: Ensure the new subbed video exists before deleting the old one - base, ext = os.path.splitext(file_path) - expected_output = f"{base}.subbed{ext}" - - if os.path.exists(expected_output): - try: - os.remove(file_path) - tracker.logger.info(f"SOURCE DELETED: Original file '{file_path}' has been removed.") - except OSError as e: - tracker.logger.error(f"Error: Could not delete source file: {e}") - else: - tracker.logger.error(f"SAFETY ABORT: Source file NOT deleted. Could not find expected output '{expected_output}'.") - else: - tracker.logger.warning("SAFETY ABORT: Source file NOT deleted. You must enable --embed to safely replace the video.") - tracker.update_step(file_path, "step_embed", "done") - - # 5. Cleanup Audio - if args.cleanup: - try: - os.remove(audio_path) - tracker.logger.info(f"Cleanup: Removed temporary audio file {audio_path}") - except OSError as e: - tracker.logger.warning(f"Warning: Could not remove audio file: {e}") - - # Mark Complete - if translation_success: - tracker.update_job_status(file_path, JobStatus.COMPLETED) + success, path, method = try_gemini() + if not success: + tracker.logger.warning("Gemini failed. Attempting DeepTranslate...") + success, path, method = try_deep() + + if success: + tracker.logger.info(f"Translation saved to: {path} ({method})") + validate_and_repair_srt(path) + final_srt_path = path + translation_success = True + method_used = method else: - # If translation failed but we didn't crash, we technically finished the run but result is partial - tracker.update_job_status(file_path, JobStatus.FAILED, error="Translation failed") - - except Exception as e: - tracker.logger.exception(f"Job Failed for {file_path}") - tracker.update_job_status(file_path, JobStatus.FAILED, error=str(e)) - # Don't exit, allow other files to process - return - - def main(): - parser = argparse.ArgumentParser(description="AI Video Transcriber & Translator") - parser.add_argument("input", nargs='?', help="Path to video file or directory") - parser.add_argument("--model", default="auto", choices=["auto", "tiny", "base", "small", "medium", "large"], help="Whisper model size (default: auto)") - parser.add_argument("--lang", default="English", help="Target language for translation (default: English)") - parser.add_argument("--source-lang", help="Source language of the audio (e.g., 'fr', 'es'). If omitted, you will be prompted.") - parser.add_argument("--force", action="store_true", help="Overwrite existing transcript/translation files") - - # New Arguments - parser.add_argument("--cleanup", action="store_true", help="Delete the temporary .wav file after processing") - parser.add_argument("--embed", action="store_true", help="Embed the final subtitles into the video (Soft Subs)") - parser.add_argument("--diarize", action="store_true", help="Enable speaker diarization (requires HF_TOKEN)") - parser.add_argument("--hf-token", help="HuggingFace Token for pyannote.audio (or set HF_TOKEN env var)") - parser.add_argument("--delete-source", action="store_true", help="Delete the original video file AFTER successful embedding") - parser.add_argument("--retry-failed", action="store_true", help="Retry only jobs marked as FAILED in the database") - - args = parser.parse_args() - - if not os.getenv("GEMINI_API_KEY"): - print("Warning: GEMINI_API_KEY environment variable not set. Translation step will fail.") - - # Handling Retry Logic - if args.retry_failed: - print("Retrying failed jobs from database...") - failed_files = tracker.get_failed_jobs() - if not failed_files: - print("No failed jobs found.") - return - - # We need args.source_lang logic here too if needed, but for retries we might assume context - # For simplicity, we'll prompt if missing just like normal run - - # Determine source language (Prompt if missing) - source_lang = args.source_lang - if not source_lang: - print("\n--- Audio Configuration ---") - user_input = input("Enter the source language of the video(s) (e.g., 'French', 'es').\nPress Enter to use Whisper's auto-detection: ").strip() - if user_input: - source_lang = user_input + tracker.logger.error("TRANSLATION FAILED.") + tracker.update_step(file_path, "step_translate", "failed") + translation_success = False + + if translation_success: + tracker.update_step(file_path, "step_translate", "done") + tracker.logger.info(f"Translation Method: {method_used}") + + # 4. Embed Subtitles + tracker.update_step(file_path, "step_embed", "processing") + should_embed = args.embed + if args.embed and not translation_success: + tracker.logger.warning("SAFETY HALT: Translation failed. Skipping embedding/deletion.") + should_embed = False + + if should_embed: + embed_subtitles(file_path, final_srt_path) + + if args.delete_source: + if args.embed: + base, ext = os.path.splitext(file_path) + expected_output = f"{base}.subbed{ext}" + + if os.path.exists(expected_output): + try: + os.remove(file_path) + tracker.logger.info(f"SOURCE DELETED: {file_path}") + except OSError as e: + tracker.logger.error(f"Error deleting source: {e}") else: - source_lang = None # Let Whisper auto-detect - print("Selected: Auto-detect") - - for file_path in failed_files: - if os.path.exists(file_path): - process_file(file_path, args, source_lang) - else: - print(f"Skipping missing file: {file_path}") - return - - # Normal Logic - if not args.input: - parser.print_help() - sys.exit(1) - - # Determine source language (Prompt if missing) - source_lang = args.source_lang - if not source_lang: - print("\n--- Audio Configuration ---") - user_input = input("Enter the source language of the video(s) (e.g., 'French', 'es').\nPress Enter to use Whisper's auto-detection: ").strip() - if user_input: - source_lang = user_input + tracker.logger.error(f"SAFETY ABORT: Output '{expected_output}' not found.") else: - source_lang = None # Let Whisper auto-detect - print("Selected: Auto-detect") - - if os.path.isfile(args.input): - process_file(args.input, args, source_lang) - elif os.path.isdir(args.input): - video_extensions = ('.mp4', '.mkv', '.mov', '.avi', '.webm', '.flv', '.wmv', '.m4v') - found_files = False - for root, dirs, files in os.walk(args.input): - for file in files: - if file.lower().endswith(video_extensions): - found_files = True - file_path = os.path.join(root, file) - process_file(file_path, args, source_lang) - if not found_files: - print(f"No video files found in {args.input}") - else: - print(f"Error: Invalid input path '{args.input}'") - sys.exit(1) - - if __name__ == "__main__": - main() + tracker.logger.warning("SAFETY ABORT: Enable --embed to delete source.") + tracker.update_step(file_path, "step_embed", "done") + + # 5. Cleanup + if args.cleanup: + try: + os.remove(audio_path) + tracker.logger.info(f"Cleanup: Removed {audio_path}") + except OSError as e: + tracker.logger.warning(f"Warning: Could not remove audio: {e}") + + # Mark Complete + if translation_success: + tracker.update_job_status(file_path, JobStatus.COMPLETED) + else: + tracker.update_job_status(file_path, JobStatus.FAILED, error="Translation failed") + + except Exception as e: + tracker.logger.exception(f"Job Failed for {file_path}") + tracker.update_job_status(file_path, JobStatus.FAILED, error=str(e)) + return + def main(): parser = argparse.ArgumentParser(description="AI Video Transcriber & Translator") - parser.add_argument("input", help="Path to video file or directory") + # Change nargs='?' to nargs='*' or '+' to support multiple inputs + parser.add_argument("inputs", nargs='*', help="Path(s) to video file or directory") parser.add_argument("--model", default="auto", choices=["auto", "tiny", "base", "small", "medium", "large"], help="Whisper model size (default: auto)") parser.add_argument("--lang", default="English", help="Target language for translation (default: English)") - parser.add_argument("--source-lang", help="Source language of the audio (e.g., 'fr', 'es'). If omitted, you will be prompted.") - parser.add_argument("--force", action="store_true", help="Overwrite existing transcript/translation files") - - # New Arguments - parser.add_argument("--cleanup", action="store_true", help="Delete the temporary .wav file after processing") - parser.add_argument("--embed", action="store_true", help="Embed the final subtitles into the video (Soft Subs)") - parser.add_argument("--diarize", action="store_true", help="Enable speaker diarization (requires HF_TOKEN)") - parser.add_argument("--hf-token", help="HuggingFace Token for pyannote.audio (or set HF_TOKEN env var)") - parser.add_argument("--delete-source", action="store_true", help="Delete the original video file AFTER successful embedding") + parser.add_argument("--source-lang", help="Source language of audio. If omitted, prompts user.") + parser.add_argument("--force", action="store_true", help="Overwrite existing files") + parser.add_argument("--cleanup", action="store_true", help="Delete temporary .wav file") + parser.add_argument("--embed", action="store_true", help="Embed subtitles (Soft Subs)") + parser.add_argument("--diarize", action="store_true", help="Enable speaker diarization") + parser.add_argument("--hf-token", help="HuggingFace Token") + parser.add_argument("--delete-source", action="store_true", help="Delete original file after embedding") + parser.add_argument("--retry-failed", action="store_true", help="Retry FAILED jobs from DB") + parser.add_argument("--prefer-deep", action="store_true", help="Prefer DeepTranslate (Free) over Gemini") args = parser.parse_args() if not os.getenv("GEMINI_API_KEY"): print("Warning: GEMINI_API_KEY environment variable not set. Translation step will fail.") - - # Determine source language (Prompt if missing) - source_lang = args.source_lang - if not source_lang: - print("\n--- Audio Configuration ---") - user_input = input("Enter the source language of the video(s) (e.g., 'French', 'es').\nPress Enter to use Whisper's auto-detection: ").strip() - if user_input: - source_lang = user_input - else: - source_lang = None # Let Whisper auto-detect - print("Selected: Auto-detect") - if os.path.isfile(args.input): - process_file(args.input, args, source_lang) - elif os.path.isdir(args.input): - video_extensions = ('.mp4', '.mkv', '.mov', '.avi', '.webm', '.flv', '.wmv', '.m4v') - found_files = False - for root, dirs, files in os.walk(args.input): - for file in files: - if file.lower().endswith(video_extensions): - found_files = True - file_path = os.path.join(root, file) - process_file(file_path, args, source_lang) - if not found_files: - print(f"No video files found in {args.input}") - else: - print(f"Error: Invalid input path '{args.input}'") + source_lang = args.source_lang + + # Retry Logic + if args.retry_failed: + print("Retrying failed jobs from database...") + failed_files = tracker.get_failed_jobs() + if not failed_files: + print("No failed jobs found.") + return + + if not source_lang: + print("\n--- Audio Configuration ---") + user_input = input("Enter source language (e.g. 'French'). Enter for Auto: ").strip() + source_lang = user_input if user_input else None + + for file_path in failed_files: + if os.path.exists(file_path): + process_file(file_path, args, source_lang) + else: + print(f"Skipping missing file: {file_path}") + return + + # Normal Logic + if not args.inputs: + parser.print_help() sys.exit(1) + if not source_lang: + print("\n--- Audio Configuration ---") + user_input = input("Enter source language (e.g. 'French'). Enter for Auto: ").strip() + source_lang = user_input if user_input else None + print(f"Selected: {source_lang if source_lang else 'Auto-detect'}") + + # Process all inputs + video_extensions = ('.mp4', '.mkv', '.mov', '.avi', '.webm', '.flv', '.wmv', '.m4v') + + for input_path in args.inputs: + if os.path.isfile(input_path): + process_file(input_path, args, source_lang) + elif os.path.isdir(input_path): + found = False + for root, dirs, files in os.walk(input_path): + for file in files: + if file.lower().endswith(video_extensions): + found = True + process_file(os.path.join(root, file), args, source_lang) + if not found: + print(f"No video files found in {input_path}") + else: + print(f"Error: Invalid input path '{input_path}'") + if __name__ == "__main__": - main() + main() \ No newline at end of file diff --git a/video_transcription/ai_transcriber_v2/main.py b/video_transcription/ai_transcriber_v2/main.py index eb5d5ef..9fa6a5d 100644 --- a/video_transcription/ai_transcriber_v2/main.py +++ b/video_transcription/ai_transcriber_v2/main.py @@ -4,14 +4,11 @@ import sys from dotenv import load_dotenv # Load environment variables from central .env_files directory -# Path: .../personal_development/video_transcription/ai_transcriber/main.py -# Target: .../personal_development/.env_files/.env.aitranscribe script_dir = os.path.dirname(os.path.abspath(__file__)) env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe')) if os.path.exists(env_path): load_dotenv(env_path) - # print(f"Loaded configuration from: {env_path}") # Optional: Uncomment for debugging else: # Fallback: check local .env local_env = os.path.join(script_dir, '.env') @@ -23,9 +20,11 @@ else: from extractor import extract_audio, embed_subtitles from transcriber import transcribe_audio, save_as_srt -from translator import translate_srt +from translator import translate_srt, translate_fallback_free from utils import validate_and_repair_srt from diarizer import diarize_audio, merge_diarization_with_transcript +import tracker +from tracker import JobStatus def save_srt_with_speakers(segments, output_path): """Helper to save SRT with speaker labels prepended to text.""" @@ -44,7 +43,6 @@ def save_srt_with_speakers(segments, output_path): text = segment["text"].strip() speaker = segment.get("speaker", "") - # Prepend speaker if present and not "Unknown" if speaker and speaker != "Unknown": text = f"[{speaker}]: {text}" @@ -54,462 +52,244 @@ def save_srt_with_speakers(segments, output_path): print(f"SRT saved to: {output_path}") def process_file(file_path, args, source_lang=None): - print(f"\n=== Processing: {file_path} ===") + tracker.logger.info(f"=== Processing: {file_path} ===") - # 1. Extract Audio - audio_path = extract_audio(file_path) + # Initialize Job + job = tracker.get_job(file_path) - # 2. Transcribe (Generate SRT) - transcript_file = os.path.splitext(file_path)[0] + ".srt" - transcript_exists = os.path.exists(transcript_file) and not args.force - - # Variable to hold final SRT path for embedding - final_srt_path = transcript_file + if job.status == JobStatus.COMPLETED and not args.force: + tracker.logger.info("Job already completed. Skipping.") + return - if transcript_exists: - print(f"Transcript exists: {transcript_file}. Skipping transcription.") - with open(transcript_file, "r", encoding="utf-8") as f: - srt_content = f.read() - else: - # Transcribe - result = transcribe_audio(audio_path, model_size=args.model, language=source_lang) - segments = result["segments"] + tracker.update_job_status(file_path, JobStatus.PROCESSING) - # Optional: Diarization - if args.diarize: - hf_token = args.hf_token or os.getenv("HF_TOKEN") - if hf_token: - print("Running Speaker Diarization...") - diar_segments = diarize_audio(audio_path, hf_token=hf_token) - if diar_segments: - segments = merge_diarization_with_transcript(segments, diar_segments) - print("Diarization merged into transcript.") - else: - print("Warning: --diarize requested but HF_TOKEN not provided. Skipping.") + try: + # 1. Extract Audio + tracker.update_step(file_path, "step_extract", "processing") + audio_path = extract_audio(file_path) + tracker.update_step(file_path, "step_extract", "done") + + # 2. Transcribe (Generate SRT) + tracker.update_step(file_path, "step_transcribe", "processing") + transcript_file = os.path.splitext(file_path)[0] + ".srt" + transcript_exists = os.path.exists(transcript_file) and not args.force + + final_srt_path = transcript_file - # Save SRT - # Use simple save if no speakers, or custom if speakers - if args.diarize: - save_srt_with_speakers(segments, transcript_file) + if transcript_exists: + tracker.logger.info(f"Transcript exists: {transcript_file}. Skipping transcription.") + with open(transcript_file, "r", encoding="utf-8") as f: + srt_content = f.read() else: - save_as_srt(result, transcript_file) - - # Validation - validate_and_repair_srt(transcript_file) - - with open(transcript_file, "r", encoding="utf-8") as f: - srt_content = f.read() + result = transcribe_audio(audio_path, model_size=args.model, language=source_lang) + segments = result["segments"] - # 3. Translate (Generate Translated SRT) - translated_file = os.path.splitext(file_path)[0] + f".{args.lang}.srt" - - translation_success = False + if args.diarize: + hf_token = args.hf_token or os.getenv("HF_TOKEN") + if hf_token: + tracker.logger.info("Running Speaker Diarization...") + diar_segments = diarize_audio(audio_path, hf_token=hf_token) + if diar_segments: + segments = merge_diarization_with_transcript(segments, diar_segments) + tracker.logger.info("Diarization merged into transcript.") + else: + tracker.logger.warning("Warning: --diarize requested but HF_TOKEN not provided. Skipping.") - if os.path.exists(translated_file) and not args.force: - print(f"Translation exists: {translated_file}. Skipping translation.") - final_srt_path = translated_file - translation_success = True - else: - # Only translate if there is content - if srt_content: - translated_srt_content = translate_srt(srt_content, target_language=args.lang) - if translated_srt_content: - with open(translated_file, "w", encoding="utf-8") as f: - f.write(translated_srt_content) - print(f"Translation saved to: {translated_file}") - validate_and_repair_srt(translated_file) - final_srt_path = translated_file - translation_success = True + if args.diarize: + save_srt_with_speakers(segments, transcript_file) else: - print("⚠️ TRANSLATION FAILED.") - translation_success = False + save_as_srt(result, transcript_file) + + validate_and_repair_srt(transcript_file) + with open(transcript_file, "r", encoding="utf-8") as f: + srt_content = f.read() + tracker.update_step(file_path, "step_transcribe", "done") - # 4. Embed Subtitles - # SAFETY: If translation was intended but failed, do NOT embed/delete to prevent - # replacing the video with one containing only untranslated subtitles. - should_embed = args.embed - if args.embed and not translation_success: - print("\n❌ SAFETY HALT: Translation failed. Skipping embedding and deletion to preserve original file.") - should_embed = False - - if should_embed: - embed_subtitles(file_path, final_srt_path) - - import argparse - import os - import sys - from dotenv import load_dotenv + # 3. Translate + tracker.update_step(file_path, "step_translate", "processing") - # Load environment variables from central .env_files directory - # Path: .../personal_development/video_transcription/ai_transcriber/main.py - # Target: .../personal_development/.env_files/.env.aitranscribe - script_dir = os.path.dirname(os.path.abspath(__file__)) - env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe')) + base_translated = os.path.splitext(file_path)[0] + f".{args.lang}.srt" + deep_translated = os.path.splitext(file_path)[0] + f".{args.lang}.deep_translate.srt" - if os.path.exists(env_path): - load_dotenv(env_path) - print(f"Loaded configuration from: {env_path}") + translated_file = base_translated # Default + translation_success = False + method_used = "None" + + if (os.path.exists(base_translated) or os.path.exists(deep_translated)) and not args.force: + if os.path.exists(deep_translated): + translated_file = deep_translated + method_used = "DeepTranslate (Existing)" + else: + method_used = "Gemini (Existing)" + tracker.logger.info(f"Translation exists: {translated_file} ({method_used}). Skipping translation.") + final_srt_path = translated_file + translation_success = True else: - # Fallback: check local .env - local_env = os.path.join(script_dir, '.env') - if os.path.exists(local_env): - load_dotenv(local_env) - else: - # Last resort: just try loading generic (cwd) - load_dotenv() - - from extractor import extract_audio, embed_subtitles - - from transcriber import transcribe_audio, save_as_srt - - from translator import translate_srt, translate_fallback_free - - from utils import validate_and_repair_srt - - from diarizer import diarize_audio, merge_diarization_with_transcript - - import tracker - - from tracker import JobStatus - - - - # ... (save_srt_with_speakers remains same) - - - - def process_file(file_path, args, source_lang=None): - - tracker.logger.info(f"=== Processing: {file_path} ===") - - - - # ... (Job init remains same) ... - - # ... (Step 1 Extract remains same) ... - - # ... (Step 2 Transcribe remains same) ... - - - - # 3. Translate (Generate Translated SRT) - - tracker.update_step(file_path, "step_translate", "processing") - + if srt_content: + # Helper functions + def try_gemini(): + res = translate_srt(srt_content, target_language=args.lang) + if res: + with open(base_translated, "w", encoding="utf-8") as f: + f.write(res) + return True, base_translated, "Gemini" + return False, None, None + + def try_deep(): + lang_map = { + "English": "en", "French": "fr", "Spanish": "es", "German": "de", + "Italian": "it", "Portuguese": "pt", "Russian": "ru", + "Japanese": "ja", "Chinese": "zh-CN" + } + target_code = lang_map.get(args.lang, "en") + res = translate_fallback_free(srt_content, target_language=target_code) + if res: + with open(deep_translated, "w", encoding="utf-8") as f: + f.write(res) + return True, deep_translated, "DeepTranslate" + return False, None, None + + success = False - - # Define paths - - base_translated = os.path.splitext(file_path)[0] + f".{args.lang}.srt" - - deep_translated = os.path.splitext(file_path)[0] + f".{args.lang}.deep_translate.srt" - - - - translated_file = base_translated # Default - - - - translation_success = False - - method_used = "None" - - - - if (os.path.exists(base_translated) or os.path.exists(deep_translated)) and not args.force: - - if os.path.exists(deep_translated): - - translated_file = deep_translated - - method_used = "DeepTranslate (Existing)" - - else: - - method_used = "Gemini (Existing)" - - - - tracker.logger.info(f"Translation exists: {translated_file} ({method_used}). Skipping translation.") - - final_srt_path = translated_file - - translation_success = True - + if args.prefer_deep: + success, path, method = try_deep() + if not success: + tracker.logger.info("DeepTranslate failed. Attempting Gemini...") + success, path, method = try_gemini() else: - - # Only translate if there is content - - if srt_content: - - # Attempt 1: Gemini - - translated_srt_content = translate_srt(srt_content, target_language=args.lang) - - - - if translated_srt_content: - - with open(base_translated, "w", encoding="utf-8") as f: - - f.write(translated_srt_content) - - tracker.logger.info(f"Translation saved to: {base_translated} (Gemini)") - - validate_and_repair_srt(base_translated) - - final_srt_path = base_translated - - translation_success = True - - method_used = "Gemini" - - else: - - # Attempt 2: Fallback - - tracker.logger.warning("Gemini translation failed. Attempting Free Fallback...") - - - - lang_map = { - - "English": "en", "French": "fr", "Spanish": "es", - - "German": "de", "Italian": "it", "Portuguese": "pt", - - "Russian": "ru", "Japanese": "ja", "Chinese": "zh-CN" - - } - - target_code = lang_map.get(args.lang, "en") - - - - translated_srt_content = translate_fallback_free(srt_content, target_language=target_code) - - - - if translated_srt_content: - - translated_file = deep_translated - - with open(translated_file, "w", encoding="utf-8") as f: - - f.write(translated_srt_content) - - tracker.logger.info(f"Translation saved to: {translated_file} (DeepTranslate)") - - validate_and_repair_srt(translated_file) - - final_srt_path = translated_file - - translation_success = True - - method_used = "DeepTranslate" - - else: - - tracker.logger.error("TRANSLATION FAILED (Both Gemini and Fallback).") - - tracker.update_step(file_path, "step_translate", "failed") - - translation_success = False - - - - if translation_success: - - tracker.update_step(file_path, "step_translate", "done") - - tracker.logger.info(f"Translation Method: {method_used}") - - - - # 4. Embed Subtitles - - - tracker.update_step(file_path, "step_embed", "processing") - should_embed = args.embed - if args.embed and not translation_success: - tracker.logger.warning("SAFETY HALT: Translation failed. Skipping embedding and deletion to preserve original file.") - should_embed = False - - if should_embed: - embed_subtitles(file_path, final_srt_path) - - # 5. Delete Source File (Optional & Risky) - if args.delete_source: - if args.embed: - # Safety: Ensure the new subbed video exists before deleting the old one - base, ext = os.path.splitext(file_path) - expected_output = f"{base}.subbed{ext}" - - if os.path.exists(expected_output): - try: - os.remove(file_path) - tracker.logger.info(f"SOURCE DELETED: Original file '{file_path}' has been removed.") - except OSError as e: - tracker.logger.error(f"Error: Could not delete source file: {e}") - else: - tracker.logger.error(f"SAFETY ABORT: Source file NOT deleted. Could not find expected output '{expected_output}'.") - else: - tracker.logger.warning("SAFETY ABORT: Source file NOT deleted. You must enable --embed to safely replace the video.") - tracker.update_step(file_path, "step_embed", "done") - - # 5. Cleanup Audio - if args.cleanup: - try: - os.remove(audio_path) - tracker.logger.info(f"Cleanup: Removed temporary audio file {audio_path}") - except OSError as e: - tracker.logger.warning(f"Warning: Could not remove audio file: {e}") - - # Mark Complete - if translation_success: - tracker.update_job_status(file_path, JobStatus.COMPLETED) + success, path, method = try_gemini() + if not success: + tracker.logger.warning("Gemini failed. Attempting DeepTranslate...") + success, path, method = try_deep() + + if success: + tracker.logger.info(f"Translation saved to: {path} ({method})") + validate_and_repair_srt(path) + final_srt_path = path + translation_success = True + method_used = method else: - # If translation failed but we didn't crash, we technically finished the run but result is partial - tracker.update_job_status(file_path, JobStatus.FAILED, error="Translation failed") - - except Exception as e: - tracker.logger.exception(f"Job Failed for {file_path}") - tracker.update_job_status(file_path, JobStatus.FAILED, error=str(e)) - # Don't exit, allow other files to process - return - - def main(): - parser = argparse.ArgumentParser(description="AI Video Transcriber & Translator") - parser.add_argument("input", nargs='?', help="Path to video file or directory") - parser.add_argument("--model", default="auto", choices=["auto", "tiny", "base", "small", "medium", "large"], help="Whisper model size (default: auto)") - parser.add_argument("--lang", default="English", help="Target language for translation (default: English)") - parser.add_argument("--source-lang", help="Source language of the audio (e.g., 'fr', 'es'). If omitted, you will be prompted.") - parser.add_argument("--force", action="store_true", help="Overwrite existing transcript/translation files") - - # New Arguments - parser.add_argument("--cleanup", action="store_true", help="Delete the temporary .wav file after processing") - parser.add_argument("--embed", action="store_true", help="Embed the final subtitles into the video (Soft Subs)") - parser.add_argument("--diarize", action="store_true", help="Enable speaker diarization (requires HF_TOKEN)") - parser.add_argument("--hf-token", help="HuggingFace Token for pyannote.audio (or set HF_TOKEN env var)") - parser.add_argument("--delete-source", action="store_true", help="Delete the original video file AFTER successful embedding") - parser.add_argument("--retry-failed", action="store_true", help="Retry only jobs marked as FAILED in the database") - - args = parser.parse_args() - - if not os.getenv("GEMINI_API_KEY"): - print("Warning: GEMINI_API_KEY environment variable not set. Translation step will fail.") - - # Handling Retry Logic - if args.retry_failed: - print("Retrying failed jobs from database...") - failed_files = tracker.get_failed_jobs() - if not failed_files: - print("No failed jobs found.") - return - - # We need args.source_lang logic here too if needed, but for retries we might assume context - # For simplicity, we'll prompt if missing just like normal run - - # Determine source language (Prompt if missing) - source_lang = args.source_lang - if not source_lang: - print("\n--- Audio Configuration ---") - user_input = input("Enter the source language of the video(s) (e.g., 'French', 'es').\nPress Enter to use Whisper's auto-detection: ").strip() - if user_input: - source_lang = user_input + tracker.logger.error("TRANSLATION FAILED.") + tracker.update_step(file_path, "step_translate", "failed") + translation_success = False + + if translation_success: + tracker.update_step(file_path, "step_translate", "done") + tracker.logger.info(f"Translation Method: {method_used}") + + # 4. Embed Subtitles + tracker.update_step(file_path, "step_embed", "processing") + should_embed = args.embed + if args.embed and not translation_success: + tracker.logger.warning("SAFETY HALT: Translation failed. Skipping embedding/deletion.") + should_embed = False + + if should_embed: + embed_subtitles(file_path, final_srt_path) + + if args.delete_source: + if args.embed: + base, ext = os.path.splitext(file_path) + expected_output = f"{base}.subbed{ext}" + + if os.path.exists(expected_output): + try: + os.remove(file_path) + tracker.logger.info(f"SOURCE DELETED: {file_path}") + except OSError as e: + tracker.logger.error(f"Error deleting source: {e}") else: - source_lang = None # Let Whisper auto-detect - print("Selected: Auto-detect") - - for file_path in failed_files: - if os.path.exists(file_path): - process_file(file_path, args, source_lang) - else: - print(f"Skipping missing file: {file_path}") - return - - # Normal Logic - if not args.input: - parser.print_help() - sys.exit(1) - - # Determine source language (Prompt if missing) - source_lang = args.source_lang - if not source_lang: - print("\n--- Audio Configuration ---") - user_input = input("Enter the source language of the video(s) (e.g., 'French', 'es').\nPress Enter to use Whisper's auto-detection: ").strip() - if user_input: - source_lang = user_input + tracker.logger.error(f"SAFETY ABORT: Output '{expected_output}' not found.") else: - source_lang = None # Let Whisper auto-detect - print("Selected: Auto-detect") - - if os.path.isfile(args.input): - process_file(args.input, args, source_lang) - elif os.path.isdir(args.input): - video_extensions = ('.mp4', '.mkv', '.mov', '.avi', '.webm', '.flv', '.wmv', '.m4v') - found_files = False - for root, dirs, files in os.walk(args.input): - for file in files: - if file.lower().endswith(video_extensions): - found_files = True - file_path = os.path.join(root, file) - process_file(file_path, args, source_lang) - if not found_files: - print(f"No video files found in {args.input}") - else: - print(f"Error: Invalid input path '{args.input}'") - sys.exit(1) - - if __name__ == "__main__": - main() + tracker.logger.warning("SAFETY ABORT: Enable --embed to delete source.") + tracker.update_step(file_path, "step_embed", "done") + + # 5. Cleanup + if args.cleanup: + try: + os.remove(audio_path) + tracker.logger.info(f"Cleanup: Removed {audio_path}") + except OSError as e: + tracker.logger.warning(f"Warning: Could not remove audio: {e}") + + # Mark Complete + if translation_success: + tracker.update_job_status(file_path, JobStatus.COMPLETED) + else: + tracker.update_job_status(file_path, JobStatus.FAILED, error="Translation failed") + + except Exception as e: + tracker.logger.exception(f"Job Failed for {file_path}") + tracker.update_job_status(file_path, JobStatus.FAILED, error=str(e)) + return + def main(): parser = argparse.ArgumentParser(description="AI Video Transcriber & Translator") - parser.add_argument("input", help="Path to video file or directory") + parser.add_argument("inputs", nargs='*', help="Path(s) to video file or directory") parser.add_argument("--model", default="auto", choices=["auto", "tiny", "base", "small", "medium", "large"], help="Whisper model size (default: auto)") parser.add_argument("--lang", default="English", help="Target language for translation (default: English)") parser.add_argument("--source-lang", help="Source language of the audio (e.g., 'fr', 'es'). If omitted, you will be prompted.") - parser.add_argument("--force", action="store_true", help="Overwrite existing transcript/translation files") - - # New Arguments - parser.add_argument("--cleanup", action="store_true", help="Delete the temporary .wav file after processing") - parser.add_argument("--embed", action="store_true", help="Embed the final subtitles into the video (Soft Subs)") - parser.add_argument("--diarize", action="store_true", help="Enable speaker diarization (requires HF_TOKEN)") - parser.add_argument("--hf-token", help="HuggingFace Token for pyannote.audio (or set HF_TOKEN env var)") - parser.add_argument("--delete-source", action="store_true", help="Delete the original video file AFTER successful embedding") + parser.add_argument("--force", action="store_true", help="Overwrite existing files") + parser.add_argument("--cleanup", action="store_true", help="Delete temporary .wav file") + parser.add_argument("--embed", action="store_true", help="Embed subtitles (Soft Subs)") + parser.add_argument("--diarize", action="store_true", help="Enable speaker diarization") + parser.add_argument("--hf-token", help="HuggingFace Token") + parser.add_argument("--delete-source", action="store_true", help="Delete original file after embedding") + parser.add_argument("--retry-failed", action="store_true", help="Retry FAILED jobs from DB") + parser.add_argument("--prefer-deep", action="store_true", help="Prefer DeepTranslate (Free) over Gemini") args = parser.parse_args() if not os.getenv("GEMINI_API_KEY"): print("Warning: GEMINI_API_KEY environment variable not set. Translation step will fail.") - - # Determine source language (Prompt if missing) - source_lang = args.source_lang - if not source_lang: - print("\n--- Audio Configuration ---") - user_input = input("Enter the source language of the video(s) (e.g., 'French', 'es').\nPress Enter to use Whisper's auto-detection: ").strip() - if user_input: - source_lang = user_input - else: - source_lang = None # Let Whisper auto-detect - print("Selected: Auto-detect") - if os.path.isfile(args.input): - process_file(args.input, args, source_lang) - elif os.path.isdir(args.input): - video_extensions = ('.mp4', '.mkv', '.mov', '.avi', '.webm', '.flv', '.wmv', '.m4v') - found_files = False - for root, dirs, files in os.walk(args.input): - for file in files: - if file.lower().endswith(video_extensions): - found_files = True - file_path = os.path.join(root, file) - process_file(file_path, args, source_lang) - if not found_files: - print(f"No video files found in {args.input}") - else: - print(f"Error: Invalid input path '{args.input}'") + source_lang = args.source_lang + + if args.retry_failed: + print("Retrying failed jobs from database...") + failed_files = tracker.get_failed_jobs() + if not failed_files: + print("No failed jobs found.") + return + + if not source_lang: + print("\n--- Audio Configuration ---") + user_input = input("Enter source language (e.g. 'French'). Enter for Auto: ").strip() + source_lang = user_input if user_input else None + + for file_path in failed_files: + if os.path.exists(file_path): + process_file(file_path, args, source_lang) + else: + print(f"Skipping missing file: {file_path}") + return + + if not args.inputs: + parser.print_help() sys.exit(1) + if not source_lang: + print("\n--- Audio Configuration ---") + user_input = input("Enter source language (e.g. 'French'). Enter for Auto: ").strip() + source_lang = user_input if user_input else None + print(f"Selected: {source_lang if source_lang else 'Auto-detect'}") + + video_extensions = ('.mp4', '.mkv', '.mov', '.avi', '.webm', '.flv', '.wmv', '.m4v') + + for input_path in args.inputs: + if os.path.isfile(input_path): + process_file(input_path, args, source_lang) + elif os.path.isdir(input_path): + found = False + for root, dirs, files in os.walk(input_path): + for file in files: + if file.lower().endswith(video_extensions): + found = True + process_file(os.path.join(root, file), args, source_lang) + if not found: + print(f"No video files found in {input_path}") + else: + print(f"Error: Invalid input path '{input_path}'") + if __name__ == "__main__": - main() + main() \ No newline at end of file diff --git a/video_transcription/mount_truenas.sh b/video_transcription/mount_truenas.sh new file mode 100755 index 0000000..34c9991 --- /dev/null +++ b/video_transcription/mount_truenas.sh @@ -0,0 +1,63 @@ +#!/bin/bash + +# Configuration +MOUNT_POINT="/mnt/truenas_isolation" +SHARE="//truenas.local/isolation" + +echo "--- SMB Mount Tool ---" + +# Determine privilege escalation method +PRIV_CMD="" +if [ "$EUID" -eq 0 ]; then + echo "Running as root." +else + if command -v sudo &> /dev/null; then + PRIV_CMD="sudo" + elif command -v flatpak-spawn &> /dev/null; then + echo "Detected Flatpak environment. Attempting to use host permissions via sudo..." + # We need to run sudo ON THE HOST. + # flatpak-spawn --host runs as the current user on the host. + # So we run 'sudo' inside that host shell. + PRIV_CMD="flatpak-spawn --host sudo" + # Note: This requires the flatpak to have permission to talk to the host + else + echo "❌ Error: This script requires root privileges to mount drives." + echo " 'sudo' was not found." + echo " Please run this script as root: su -c ./mount_truenas.sh" + exit 1 + fi +fi + +# 1. Create mount point if it doesn't exist +if [ ! -d "$MOUNT_POINT" ]; then + echo "Creating directory $MOUNT_POINT..." + # We try to create it. If it fails (e.g. inside read-only flatpak mount namespace), warn user. + $PRIV_CMD mkdir -p "$MOUNT_POINT" + if [ $? -ne 0 ]; then + echo "Error creating directory. If you are in a Flatpak, you might not have access to host /mnt." + exit 1 + fi +fi + +# 2. Get Credentials +read -p "Enter SMB Username [guest]: " SMB_USER +SMB_USER=${SMB_USER:-guest} + +# 3. Mount +echo "Mounting $SHARE to $MOUNT_POINT..." + +if [ "$SMB_USER" == "guest" ]; then + $PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o guest,vers=3.0 +else + # This will prompt for the SMB password + $PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o username="$SMB_USER",vers=3.0 +fi + +# 4. Check result +if [ $? -eq 0 ]; then + echo "✅ Success! Share is now available at $MOUNT_POINT" + echo "The mapping will disappear automatically after you reboot." +else + echo "❌ Error: Failed to mount the share." + echo "Ensure 'cifs-utils' is installed and the server is reachable." +fi diff --git a/video_transcription/recover_and_fix_v2.py b/video_transcription/recover_and_fix_v2.py index b8ffd18..ee89e4f 100755 --- a/video_transcription/recover_and_fix_v2.py +++ b/video_transcription/recover_and_fix_v2.py @@ -228,11 +228,27 @@ def process_recovery(folder_path, target_lang="English", prefer_deep=False): print(f"\nRecovery Complete. Fixed {count_fixed} files.") if __name__ == "__main__": + parser = argparse.ArgumentParser(description="Recover and Fix Translations (V2)") - parser.add_argument("folder", help="Path to the folder to scan") - parser.add_argument("lang", nargs="?", default="English", help="Target language (default: English)") + + parser.add_argument("folders", nargs='+', help="One or more paths to folders to scan") + + parser.add_argument("--lang", default="English", help="Target language (default: English)") + parser.add_argument("--prefer-deep", action="store_true", help="Prefer DeepTranslate (Free) over Gemini API") + + args = parser.parse_args() + - process_recovery(args.folder, args.lang, args.prefer_deep) \ No newline at end of file + + for folder in args.folders: + + if os.path.exists(folder): + + process_recovery(folder, args.lang, args.prefer_deep) + + else: + + print(f"Error: Folder '{folder}' does not exist. Skipping.") \ No newline at end of file diff --git a/video_transcription/run_v2.py b/video_transcription/run_v2.py index 504cbb0..560d011 100755 --- a/video_transcription/run_v2.py +++ b/video_transcription/run_v2.py @@ -42,22 +42,39 @@ def main(): except ImportError: pass - # 1. Input File/Folder + # 1. Input File/Folder (Multiple) + input_paths = [] while True: - input_path = get_input("Enter the path to the video file or folder") + prompt_text = "Enter a path to a video file or folder" + if input_paths: + prompt_text += " (or press Enter to finish)" + input_path = get_input(prompt_text) + + if not input_path: + if input_paths: + break + else: + print("Error: You must provide at least one path.") + continue + # Clean up input - input_path = input_path.strip("'\"") + input_path = input_path.strip("\'"") input_path = input_path.replace(r'\ ', ' ') # Expand user (~) and resolve absolute path input_path = os.path.abspath(os.path.expanduser(input_path)) if os.path.exists(input_path): - break - print(f"Error: Path '{input_path}' does not exist. Please try again.\n") + input_paths.append(input_path) + print(f"Added: {input_path}") + else: + print(f"Error: Path '{input_path}' does not exist. Please try again.\n") - print(f"Selected: {input_path}\n") + print("\nSelected Inputs:") + for p in input_paths: + print(f" - {p}") + print("") # 2. Languages source_lang = get_input("Source Language (e.g., French, es)", default="auto") @@ -80,6 +97,8 @@ def main(): print(" It will only run if the new subtitled video is successfully created.") do_delete_source = get_yes_no("Delete original source files after embedding?", default="n") + do_prefer_deep = get_yes_no("Prefer DeepTranslate (Free) over Gemini API?", default="n") + hf_token = None if do_diarize: if not os.getenv("HF_TOKEN"): @@ -93,7 +112,8 @@ def main(): script_dir = os.path.dirname(os.path.abspath(__file__)) main_script = os.path.join(script_dir, "ai_transcriber_v2", "main.py") - cmd = [sys.executable, main_script, input_path] + cmd = [sys.executable, main_script] + cmd.extend(input_paths) cmd.extend(["--lang", target_lang]) cmd.extend(["--model", model_size]) @@ -115,12 +135,17 @@ def main(): if hf_token: cmd.extend(["--hf-token", hf_token]) + if do_prefer_deep: + cmd.append("--prefer-deep") + # 6. Confirmation and Execution clear_screen() print_header() print("Configuration Complete!") print("-" * 30) - print(f"Input: {input_path}") + print("Inputs:") + for p in input_paths: + print(f" - {p}") print(f"Source Lang: {source_lang}") print(f"Target Lang: {target_lang}") print(f"Model: {model_size}") @@ -128,6 +153,7 @@ def main(): print(f"Embed Subs: {do_embed}") print(f"Delete Src: {do_delete_source}") print(f"Diarization: {do_diarize}") + print(f"Prefer Deep: {do_prefer_deep}") print("-" * 30) if not get_yes_no("Run this job now?", default="y"): @@ -150,4 +176,4 @@ def main(): print("\nJob interrupted by user.") if __name__ == "__main__": - main() + main() \ No newline at end of file diff --git a/video_transcription/run_wizard.py b/video_transcription/run_wizard.py index 08c8d8c..b2752c2 100755 --- a/video_transcription/run_wizard.py +++ b/video_transcription/run_wizard.py @@ -3,19 +3,6 @@ import os import sys import subprocess import shutil -from pathlib import Path - -# Try to load the .env file so the wizard knows what's already configured -try: - from dotenv import load_dotenv - # Path logic matching main.py - script_dir = os.path.dirname(os.path.abspath(__file__)) - # Expected: .../video_transcription/../.env_files -> .../personal_development/.env_files - env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe')) - if os.path.exists(env_path): - load_dotenv(env_path) -except ImportError: - pass def clear_screen(): os.system('cls' if os.name == 'nt' else 'clear') @@ -43,25 +30,50 @@ def print_header(): def main(): clear_screen() print_header() + + # Try to load the .env file so the wizard knows what's already configured + try: + from dotenv import load_dotenv + script_dir = os.path.dirname(os.path.abspath(__file__)) + env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe')) + if os.path.exists(env_path): + load_dotenv(env_path) + except ImportError: + pass - # 1. Input File/Folder + # 1. Input File/Folder (Multiple) + input_paths = [] while True: - input_path = get_input("Enter the path to the video file or folder") + prompt_text = "Enter a path to a video file or folder" + if input_paths: + prompt_text += " (or press Enter to finish)" - # Clean up input: - # 1. Remove surrounding quotes (common when pasting paths) + input_path = get_input(prompt_text) + + if not input_path: + if input_paths: + break + else: + print("Error: You must provide at least one path.") + continue + + # Clean up input input_path = input_path.strip('"\'') - # 2. Handle escaped spaces (e.g., "My\ Folder" -> "My Folder") input_path = input_path.replace(r'\ ', ' ') # Expand user (~) and resolve absolute path input_path = os.path.abspath(os.path.expanduser(input_path)) if os.path.exists(input_path): - break - print(f"Error: Path '{input_path}' does not exist. Please try again.\n") + input_paths.append(input_path) + print(f"Added: {input_path}") + else: + print(f"Error: Path '{input_path}' does not exist. Please try again.\n") - print(f"Selected: {input_path}\n") + print("\nSelected Inputs:") + for p in input_paths: + print(f" - {p}") + print("") # 2. Languages source_lang = get_input("Source Language (e.g., French, es)", default="auto") @@ -84,12 +96,13 @@ def main(): print(" It will only run if the new subtitled video is successfully created.") do_delete_source = get_yes_no("Delete original source files after embedding?", default="n") + do_prefer_deep = get_yes_no("Prefer DeepTranslate (Free) over Gemini API?", default="n") + hf_token = None if do_diarize: if not os.getenv("HF_TOKEN"): print("\nSpeaker Diarization requires a HuggingFace Token.") hf_token = get_input("Enter your HuggingFace Token (hidden)", default="") - # In a real app we might use getpass, but standard input is fine for this wizard level else: print("Using HF_TOKEN from environment.") @@ -98,7 +111,9 @@ def main(): script_dir = os.path.dirname(os.path.abspath(__file__)) main_script = os.path.join(script_dir, "ai_transcriber", "main.py") - cmd = [sys.executable, main_script, input_path] + cmd = [sys.executable, main_script] + # Add all inputs + cmd.extend(input_paths) cmd.extend(["--lang", target_lang]) cmd.extend(["--model", model_size]) @@ -120,12 +135,17 @@ def main(): if hf_token: cmd.extend(["--hf-token", hf_token]) + if do_prefer_deep: + cmd.append("--prefer-deep") + # 6. Confirmation and Execution clear_screen() print_header() print("Configuration Complete!") print("-" * 30) - print(f"Input: {input_path}") + print("Inputs:") + for p in input_paths: + print(f" - {p}") print(f"Source Lang: {source_lang}") print(f"Target Lang: {target_lang}") print(f"Model: {model_size}") @@ -133,6 +153,7 @@ def main(): print(f"Embed Subs: {do_embed}") print(f"Delete Src: {do_delete_source}") print(f"Diarization: {do_diarize}") + print(f"Prefer Deep: {do_prefer_deep}") print("-" * 30) if not get_yes_no("Run this job now?", default="y"): @@ -155,4 +176,4 @@ def main(): print("\nJob interrupted by user.") if __name__ == "__main__": - main() + main() \ No newline at end of file diff --git a/video_transcription/run_wizard_v2.py b/video_transcription/run_wizard_v2.py new file mode 100755 index 0000000..560d011 --- /dev/null +++ b/video_transcription/run_wizard_v2.py @@ -0,0 +1,179 @@ +#!/usr/bin/env python3 +import os +import sys +import subprocess +import shutil + +def clear_screen(): + os.system('cls' if os.name == 'nt' else 'clear') + +def get_input(prompt, default=None): + """Helper to get input with a default value.""" + if default: + user_input = input(f"{prompt} [{default}]: ").strip() + return user_input if user_input else default + else: + return input(f"{prompt}: ").strip() + +def get_yes_no(prompt, default="y"): + """Helper to get boolean input.""" + display_default = "Y/n" if default.lower() in ["y", "yes"] else "y/N" + choice = get_input(f"{prompt} ({display_default})", default).lower() + return choice in ["y", "yes", "true", "1"] + +def print_header(): + print("==========================================") + print(" AI Video Transcriber & Translator V2") + print(" (Powered by Google GenAI SDK)") + print("==========================================") + print("") + +def main(): + clear_screen() + print_header() + + # Try to load the .env file so the wizard knows what's already configured + try: + from dotenv import load_dotenv + script_dir = os.path.dirname(os.path.abspath(__file__)) + env_path = os.path.abspath(os.path.join(script_dir, '../.env_files/.env.aitranscribe')) + if os.path.exists(env_path): + load_dotenv(env_path) + except ImportError: + pass + + # 1. Input File/Folder (Multiple) + input_paths = [] + while True: + prompt_text = "Enter a path to a video file or folder" + if input_paths: + prompt_text += " (or press Enter to finish)" + + input_path = get_input(prompt_text) + + if not input_path: + if input_paths: + break + else: + print("Error: You must provide at least one path.") + continue + + # Clean up input + input_path = input_path.strip("\'"") + input_path = input_path.replace(r'\ ', ' ') + + # Expand user (~) and resolve absolute path + input_path = os.path.abspath(os.path.expanduser(input_path)) + + if os.path.exists(input_path): + input_paths.append(input_path) + print(f"Added: {input_path}") + else: + print(f"Error: Path '{input_path}' does not exist. Please try again.\n") + + print("\nSelected Inputs:") + for p in input_paths: + print(f" - {p}") + print("") + + # 2. Languages + source_lang = get_input("Source Language (e.g., French, es)", default="auto") + target_lang = get_input("Target Language for translation", default="English") + print("") + + # 3. Model Size + print("Model Size Options: tiny, base, small, medium, large, auto") + model_size = get_input("Whisper Model Size", default="auto") + print("") + + # 4. Features + do_cleanup = get_yes_no("Cleanup temporary audio files after processing?", default="y") + do_embed = get_yes_no("Embed subtitles into the video file (Soft Subs)?", default="y") + do_diarize = get_yes_no("Enable Speaker Diarization (Identify speakers)?", default="n") + + do_delete_source = False + if do_embed: + print("\n⚠️ WARNING: Using this next option will PERMANENTLY DELETE the original video files.") + print(" It will only run if the new subtitled video is successfully created.") + do_delete_source = get_yes_no("Delete original source files after embedding?", default="n") + + do_prefer_deep = get_yes_no("Prefer DeepTranslate (Free) over Gemini API?", default="n") + + hf_token = None + if do_diarize: + if not os.getenv("HF_TOKEN"): + print("\nSpeaker Diarization requires a HuggingFace Token.") + hf_token = get_input("Enter your HuggingFace Token (hidden)", default="") + else: + print("Using HF_TOKEN from environment.") + + # 5. Build Command + # Point to v2 main script + script_dir = os.path.dirname(os.path.abspath(__file__)) + main_script = os.path.join(script_dir, "ai_transcriber_v2", "main.py") + + cmd = [sys.executable, main_script] + cmd.extend(input_paths) + + cmd.extend(["--lang", target_lang]) + cmd.extend(["--model", model_size]) + + if source_lang != "auto": + cmd.extend(["--source-lang", source_lang]) + + if do_cleanup: + cmd.append("--cleanup") + + if do_embed: + cmd.append("--embed") + + if do_delete_source: + cmd.append("--delete-source") + + if do_diarize: + cmd.append("--diarize") + if hf_token: + cmd.extend(["--hf-token", hf_token]) + + if do_prefer_deep: + cmd.append("--prefer-deep") + + # 6. Confirmation and Execution + clear_screen() + print_header() + print("Configuration Complete!") + print("-" * 30) + print("Inputs:") + for p in input_paths: + print(f" - {p}") + print(f"Source Lang: {source_lang}") + print(f"Target Lang: {target_lang}") + print(f"Model: {model_size}") + print(f"Cleanup: {do_cleanup}") + print(f"Embed Subs: {do_embed}") + print(f"Delete Src: {do_delete_source}") + print(f"Diarization: {do_diarize}") + print(f"Prefer Deep: {do_prefer_deep}") + print("-" * 30) + + if not get_yes_no("Run this job now?", default="y"): + print("Aborted.") + sys.exit(0) + + print("\nStarting Job (V2)...") + + try: + # Pass environment variables including HF_TOKEN if set + env = os.environ.copy() + if hf_token: + env["HF_TOKEN"] = hf_token + + subprocess.run(cmd, check=True, env=env) + print("\n✅ Job Complete!") + except subprocess.CalledProcessError as e: + print(f"\n❌ Job Failed with error code {e.returncode}") + except KeyboardInterrupt: + print("\nJob interrupted by user.") + +if __name__ == "__main__": + main() \ No newline at end of file