From 2a7878ddadfaa662f32e5db12074728b7f1b7753 Mon Sep 17 00:00:00 2001 From: David Kifer Date: Mon, 12 Jan 2026 11:47:39 -0500 Subject: [PATCH] adding latest documentation and changes to the script. --- .../ai_transcriber_v1/requirements.txt | 1 + .../__pycache__/tracker.cpython-313.pyc | Bin 0 -> 4699 bytes .../context/CURRENT_STATE.md | 33 ++++++++++++ .../documentation/TECHNICAL_OVERVIEW.md | 31 +++++++++++ .../documentation/USER_GUIDE.md | 51 ++++++++++++++++++ video_transcription/ai_transcriber_v2/main.py | 16 ++++-- .../ai_transcriber_v2/requirements.txt | 1 + .../ai_transcriber_v2/tracker.py | 4 ++ .../ai_transcriber_v2/translator.py | 10 ++-- .../ai_transcriber_v2/utils.py | 7 +++ 10 files changed, 144 insertions(+), 10 deletions(-) create mode 100644 video_transcription/ai_transcriber_v2/__pycache__/tracker.cpython-313.pyc create mode 100644 video_transcription/ai_transcriber_v2/context/CURRENT_STATE.md create mode 100644 video_transcription/ai_transcriber_v2/documentation/TECHNICAL_OVERVIEW.md create mode 100644 video_transcription/ai_transcriber_v2/documentation/USER_GUIDE.md diff --git a/video_transcription/ai_transcriber_v1/requirements.txt b/video_transcription/ai_transcriber_v1/requirements.txt index a9d3c11..b345ce2 100644 --- a/video_transcription/ai_transcriber_v1/requirements.txt +++ b/video_transcription/ai_transcriber_v1/requirements.txt @@ -6,3 +6,4 @@ numpy tenacity pysubs2 pyannote.audio +python-dotenv \ No newline at end of file diff --git a/video_transcription/ai_transcriber_v2/__pycache__/tracker.cpython-313.pyc b/video_transcription/ai_transcriber_v2/__pycache__/tracker.cpython-313.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c43eb17bc0732ebc536164e644917e76e4ee90cf GIT binary patch literal 4699 zcmcgvOK=m(8SZ&#W;CORjUV_4w!t<^VVm$+;_${VEQ3unn1w=BQ)6i?1bUdB5y6~9 zHVL?Pt6(cdMTb2&$;DJs6)HL8nmr{)3sWVzA?zkOabq!;-17H~N3xCC)TS!wlK$@h z?|=IBfB)ZqulfBlg7VvTBRLPPe^SOR;(B9khDGQxQjy9`Aneo(W`JiW*f5LP4l3g& zxG<0Tum^j>0v5s|7Q+&j!d~nR%UHH`_z5NK!@jT|`)!_?2;hL~!NDjK>2Rm5)?FwX zR0UNWgc(4itJ>qXBHeDQTEjO|l#TRMC2p$=9a(QiLawY@tKK7w!v&BkAL5R{u4%yy zQ54zlv^P>%9g6gClsIiw0r`RV<)Gi-&Hx$*x6^RfH^8dCK@N6}B8Ruri?+HW;kKx+ zx-ZYQ3sQs8KxD`J-yh)|8QyJc)_`Vhw7nwji1P6Lj_9rcI>c7y>5R5V zhTk76GWNbK+Ub4;JMhE4JJq^D2JcntTM+J6o79FDbc4Y?Y9sKy>Q12h5=^N1H63a) z#1JX2TZWZP8zGVSW7q&rGct)}#vsB_HkC_fh&Z0HjD&%SpjtSYNf2omv@XFngdfS| z(uBWc+_6F|sfio0l#X>PdE3xtbkiU{(=g3sHj~zG!uNbCn=miYNsT*6)6%jxiO<4% z#*E?QjDfYgLig3~ba(tN-w&F1L*4syGiI;N?C9(0-j_0N8>x&AYjyZE46Y}D3vt9_ zcM@SXi_^MAr0aSno-(j`H*h06qg{vTvUt8fJ`>w`hU*|iYkNWZ7$87h4YMkPIhDn{ z%3+VnV*%VoR0S+0#E?Y1;p~iR=~m8s-3nh3an8uZ>8Z$bI2)teg|;`AP0yta%ZS5@ z^kmA2KY`Qs5=qnO`ZbNnnwHMSb1B;PY1*xvZto1Zpr*|xv1z5i2AM1f#ET=7!{d`< zM7|iA8X8em+V&1jUATCDMln4sTXEA_ z3LvFr?H@qiKg*~Gz$vJl$`7(wv^fblk17mu*lY7LXech6qe`U79&Zz&0SfzV?Eri& z0T;BnnrLl=u4_x#UaERw&bswA$o81^o4AHeT;nEgr>fwlgfHYL93bI!C(KBAz?cne zWez9PI-b{V8uO#z5kfALyp=PEl*y%1`b^3I@QZO{R?nrZ5QAx`3$eJ5jv-|;xjE{L zm>#`twaHL3rpCZ+l(bW_3kdORnnee5poBfNgC0L&lW|P165^c&%xQDFbsc+YuWj4i zv`3&2$(fkmMZ|AloW#+(Me4xq6}(B`Um zB7v$P1^Xbfqn@5I;&=}onj+Cb54}rpuS@6|nx6szz}2_pFZ|sbd?YSjDtf<{Kf7Af z_~`ax{ONSD=2(7w)nE7daH07ZQ%hG%eJ6|X^PkF(uGY5{4lnI3)(_+_tkyLb_AQ@>S?y)S+Yp=%Toc5%MVurR;k z%BuBr<;v+mpy99fxGi@EyFchktM!%|LiM9g^f%R5! zAz0v^OKooi}2g-=pq2b5E^ zb(#{I_7TpsFg`@tvy_Yg33+TJIQ4T=FP7K!=Saz#XiPye#=*?)`_rr!& z)(uPb@kKaNYF2X+$agmBODXt>TNIv4y&p6r3|lD`FMbE2Z4HU0(SPs7*XT8Fmx%q~ zbUCNC?~;mdR{PxJhr0@WyLqOWI&_&e*#Ts-b?96Gp=?(KJbr7r$X)=cf;vDi1CiZi zoz^U*JT%5r_Y8TrwXvP9bU{Ca_EgGtTIWjO9DC9yC1GxCj^05Uy`5y6daH>qffoIn z9|NfxRoU_I#KMU$>Q&O?d#;c;i^X_xiF zX`Y9zKzOvDIRf1meZ!rl?>ce#tTqad4kxjsahC||uSO{it_gXpPUQ9HBsHNOr0KP4 zM^D?|Z2N4<35j&vc@>?+DNv@Fl-UgQ4aYFd3lw;PynjROUm>-G)FM*9LW3`m@P zZC)%R_P+8H<-u4X{tWFYEB=Q^7mhwWzHq!CE*^g>{mJ(S-yZ{|17pPlV}F%a4*Y0& zAhH~}yxjTmigG0{eT~Ets(Xg&=q Secondary -> Local LLM (Ollama) -> MyMemory**. +2. **Flattened Directory Structure**: Removed nested `ai_transcriber_v2/ai_transcriber_v2` folders. All V2 modules are now in the root of `ai_transcriber_v2/`. +3. **Flatpak & Bazzite Compatibility**: + * Added `flatpak-spawn --host` support for `ffmpeg`, `ffprobe`, and `ollama`. + * Updated `mount_truenas.sh` to handle UID/GID mapping for write permissions on remote shares. +4. **Hardware-Accelerated Safety Checks**: + * **Size Validation**: Rejects any remuxed file that drops more than 20% of the original size. + * **Duration Validation**: Rejects any remuxed file where the duration differs by more than 1 second. + * **Zero-Byte Check**: Deletes empty outputs immediately. +5. **Intelligent Skipping**: + * Whisper detects source language; if it matches the target (e.g., English to English), translation is skipped entirely. +6. **Progress Tracking**: + * Integrated `tqdm` progress bars for iterative translation steps. + * Enabled `verbose=True` for Whisper to show live transcription segments. +7. **Stability**: + * Fixed SQLAlchemy `DetachedInstanceError` by expunging objects from sessions in `tracker.py`. + * Added `GracefulKiller` for clean `Ctrl+C` shutdowns (finishes current file, then exits). + +## 🛠 Active Setup +- **Environment**: Bazzite (Linux) running VS Code via Flatpak. +- **Local LLM**: Ollama with `llama3` (or `dolphin-llama3`). +- **Media Tools**: Host-level FFmpeg/FFprobe accessible via `flatpak-spawn`. + +## 📌 Next Steps / Future Ideas +- Consider batching small SRT segments for Gemini to reduce API calls and improve context. +- Add a GUI or Web dashboard for tracking `job_history.db`. +- Implement auto-retry for specific "failed_validation" files with different model parameters. diff --git a/video_transcription/ai_transcriber_v2/documentation/TECHNICAL_OVERVIEW.md b/video_transcription/ai_transcriber_v2/documentation/TECHNICAL_OVERVIEW.md new file mode 100644 index 0000000..c88b7bc --- /dev/null +++ b/video_transcription/ai_transcriber_v2/documentation/TECHNICAL_OVERVIEW.md @@ -0,0 +1,31 @@ +# Technical Overview + +## 🏗 Architecture + +The project is modularized into specialized Python scripts: + +* **`main.py`**: The entry point. Manages the batch processing loop and job tracking. +* **`extractor.py`**: Media handling via FFmpeg. Responsible for audio extraction and subtitle embedding. Contains the "Flatpak Escape" logic. +* **`transcriber.py`**: Integration with `openai-whisper`. Manages GPU health checks and model loading. +* **`translator.py`**: The AI translation engine. Implements the 4-tier fallback logic (Gemini -> Google -> Ollama -> MyMemory). +* **`utils.py`**: Shared utilities for encoding detection (chardet), port checking, and service health monitoring. +* **`tracker.py`**: Persistence layer using SQLite/SQLAlchemy to track job status across runs. + +## 🛡 Safety Mechanisms + +1. **Duration Match Check**: Uses `ffprobe` to ensure the final subbed video length matches the original source. +2. **File Size Sanity**: Rejects any remux operation that results in a file < 80% of the original size (preventing video stream loss). +3. **SRT Health Check**: Compares the last timestamp of the translated SRT against the original transcript to detect partial/truncated translations. +4. **Encoding Detection**: Uses `chardet` to reliably read foreign subtitle files without manual configuration. + +## 🐳 Flatpak / Sandbox Support + +Since this project is designed for Bazzite/Atomic distros, all system-level calls (`ffmpeg`, `ffprobe`, `ollama`) are wrapped in a check that detects the presence of `/.flatpak-info`. If found, it automatically prefixes commands with `flatpak-spawn --host` to utilize system-installed binaries. + +## 💾 Database Schema + +The `job_history.db` tracks: +- `file_path`: Absolute path to source. +- `status`: PENDING, PROCESSING, COMPLETED, FAILED. +- `step_status`: Individual status for Extract, Transcribe, Translate, and Embed steps. +- `error_message`: Captured stack traces for failed jobs. diff --git a/video_transcription/ai_transcriber_v2/documentation/USER_GUIDE.md b/video_transcription/ai_transcriber_v2/documentation/USER_GUIDE.md new file mode 100644 index 0000000..c32af35 --- /dev/null +++ b/video_transcription/ai_transcriber_v2/documentation/USER_GUIDE.md @@ -0,0 +1,51 @@ +# User Guide: AI Video Transcriber & Translator V2 + +This tool automates the process of extracting audio from videos, transcribing it using OpenAI Whisper, translating the text via AI (Gemini, Llama3, or Google), and embedding the results back into the video as soft subtitles. + +## 📋 Prerequisites + +1. **FFmpeg**: Must be installed on your host system. + * On Bazzite: `brew install ffmpeg` +2. **Ollama (Optional but Recommended)**: For private, local translation. + * Run `./ai_transcriber_v2/install_local_llm.sh` + * Pull a model: `ollama pull llama3` +3. **Python Packages**: + ```bash + pip install -r ai_transcriber_v2/requirements.txt + ``` + +## 🚀 How to Run + +### 1. The Easy Way (Wizard) +Perfect for first-time runs or single folders. +```bash +./ai_transcriber_v2/run_wizard_v2.py +``` +Follow the interactive prompts to set your languages, model size, and preferences. + +### 2. The Power Way (CLI) +For advanced automation. +```bash +python3 ai_transcriber_v2/main.py /path/to/videos --lang English --prefer-local --embed --cleanup +``` + +### 3. The Library Fixer (Recovery) +If you have a folder with existing transcripts or partial translations that need fixing: +```bash +./ai_transcriber_v2/recover_and_fix_v2.py /path/to/folder +``` +This script intelligently scans for missing translations or translations that don't match the video duration. + +## 🛠 Features + +* **Prefer Local LLM**: Use `--prefer-local` to prioritize your laptop's GPU (via Ollama) for all translations. +* **Safety First**: The script will NEVER replace your original video if the new one is significantly smaller or has a different duration. +* **Diarization**: Use `--diarize` to identify different speakers (requires HuggingFace token). +* **Graceful Exit**: Press `Ctrl+C` once to stop the script. It will finish the current file and save its progress before closing. + +## 📂 File Naming Convention +- `video.srt`: Original language transcript. +- `video.English.srt`: Gemini translated subtitles. +- `video.English.deep_translate.srt`: Google Translate subtitles. +- `video.English.local_llm.srt`: Ollama translated subtitles. +- `video.subbed.mp4`: The final result with embedded soft-subs. diff --git a/video_transcription/ai_transcriber_v2/main.py b/video_transcription/ai_transcriber_v2/main.py index 3555b8f..435c352 100644 --- a/video_transcription/ai_transcriber_v2/main.py +++ b/video_transcription/ai_transcriber_v2/main.py @@ -21,7 +21,7 @@ else: from extractor import extract_audio, embed_subtitles from transcriber import transcribe_audio, save_as_srt, load_whisper_model from translator import translate_with_auto_fallback -from utils import validate_and_repair_srt, check_srt_duration_match, GracefulKiller, ensure_ollama_running, check_service_availability, check_path_permissions +from utils import validate_and_repair_srt, check_srt_duration_match, GracefulKiller, ensure_ollama_running, check_service_availability, check_path_permissions, LANGUAGE_MAP from diarizer import diarize_audio, merge_diarization_with_transcript import tracker from tracker import JobStatus @@ -75,6 +75,7 @@ def process_file(file_path, args, source_lang=None, loaded_model=None, service_s transcript_exists = os.path.exists(transcript_file) and not args.force final_srt_path = transcript_file + detected_iso = None if transcript_exists: tracker.logger.info(f"Transcript exists: {transcript_file}. Skipping transcription.") @@ -84,6 +85,7 @@ def process_file(file_path, args, source_lang=None, loaded_model=None, service_s # Use loaded_model if available result = transcribe_audio(audio_path, model_size=args.model, language=source_lang, loaded_model=loaded_model) segments = result["segments"] + detected_iso = result.get("language") if args.diarize: hf_token = args.hf_token or os.getenv("HF_TOKEN") @@ -122,8 +124,16 @@ def process_file(file_path, args, source_lang=None, loaded_model=None, service_s translation_success = False method_used = "None" - # Check existing - if (os.path.exists(base_translated) or os.path.exists(deep_translated) or os.path.exists(local_translated)) and not args.force: + # Check if source language matches target language + target_iso = LANGUAGE_MAP.get(args.lang) + if detected_iso and target_iso and detected_iso == target_iso: + tracker.logger.info(f"Source language '{detected_iso}' matches target '{target_iso}'. Skipping translation.") + final_srt_path = transcript_file + translation_success = True + method_used = "Source Match" + + # Check existing (if not already handled by match) + elif (os.path.exists(base_translated) or os.path.exists(deep_translated) or os.path.exists(local_translated)) and not args.force: if os.path.exists(local_translated): translated_file = local_translated method_used = "Local LLM (Existing)" diff --git a/video_transcription/ai_transcriber_v2/requirements.txt b/video_transcription/ai_transcriber_v2/requirements.txt index 30881b2..61e9e9a 100644 --- a/video_transcription/ai_transcriber_v2/requirements.txt +++ b/video_transcription/ai_transcriber_v2/requirements.txt @@ -10,3 +10,4 @@ deep-translator ollama chardet tqdm +python-dotenv \ No newline at end of file diff --git a/video_transcription/ai_transcriber_v2/tracker.py b/video_transcription/ai_transcriber_v2/tracker.py index 234e0e1..504d890 100644 --- a/video_transcription/ai_transcriber_v2/tracker.py +++ b/video_transcription/ai_transcriber_v2/tracker.py @@ -56,6 +56,10 @@ def get_job(file_path): job = Job(file_path=file_path) session.add(job) session.commit() + session.refresh(job) # Ensure we have the ID and defaults + + # Detach from session so we can use it after session.close() + session.expunge(job) session.close() return job diff --git a/video_transcription/ai_transcriber_v2/translator.py b/video_transcription/ai_transcriber_v2/translator.py index 0419891..c3823d3 100644 --- a/video_transcription/ai_transcriber_v2/translator.py +++ b/video_transcription/ai_transcriber_v2/translator.py @@ -7,6 +7,7 @@ import pysubs2 from deep_translator import GoogleTranslator, MyMemoryTranslator import ollama from tqdm import tqdm +from utils import LANGUAGE_MAP # Define a retry decorator # ... (retry_policy remains) @@ -243,13 +244,8 @@ def translate_with_auto_fallback(srt_content, target_language="English", prefer_ tuple: (translated_content, method_name) or (None, None) if all failed. """ - # Map full language name to code for DeepTranslate - lang_map = { - "English": "en", "French": "fr", "Spanish": "es", "German": "de", - "Italian": "it", "Portuguese": "pt", "Russian": "ru", - "Japanese": "ja", "Chinese": "zh-CN" - } - target_code = lang_map.get(target_language, "en") + # Use central language mapping + target_code = LANGUAGE_MAP.get(target_language, "en") # Determine which services to even try def is_ok(name): diff --git a/video_transcription/ai_transcriber_v2/utils.py b/video_transcription/ai_transcriber_v2/utils.py index f28e092..7750f4f 100644 --- a/video_transcription/ai_transcriber_v2/utils.py +++ b/video_transcription/ai_transcriber_v2/utils.py @@ -9,6 +9,13 @@ import shutil import chardet from tqdm import tqdm +# Global Language Mapping +LANGUAGE_MAP = { + "English": "en", "French": "fr", "Spanish": "es", "German": "de", + "Italian": "it", "Portuguese": "pt", "Russian": "ru", + "Japanese": "ja", "Chinese": "zh-CN", "auto": "auto" +} + class GracefulKiller: """ Handles SIGINT (Ctrl+C) and SIGTERM signals.