This commit is contained in:
2026-01-13 15:27:30 +00:00
59 changed files with 12937 additions and 0 deletions
+1
View File
@@ -0,0 +1 @@
.env_files/
@@ -0,0 +1,51 @@
# AI Video Transcriber & Translator
This tool extracts audio from videos, transcribes it using OpenAI's Whisper model, and translates the transcript using Google's Gemini API.
## Setup
1. **Install Dependencies:**
```bash
pip install -r requirements.txt
```
*Note: You need `ffmpeg` installed on your system.*
2. **API Key:**
Set your Gemini API key as an environment variable:
```bash
export GEMINI_API_KEY="your_api_key_here"
```
## Usage
### Quick Start (Wizard)
For a user-friendly, interactive experience, run the wizard script in the root directory:
```bash
./run_wizard.py
```
This will guide you through selecting files, languages, and enabling features like cleanup and embedding.
### Advanced (CLI)
Run the `main.py` script directly:
```bash
python ai_transcriber/main.py <path_to_video_or_folder> [options]
```
### Options:
* `--model`: Whisper model size (`tiny`, `base`, `small`, `medium`, `large`). Default: `base`.
* `--lang`: Target language for translation. Default: `English`.
* `--force`: Overwrite existing transcript/translation files.
### Examples:
**Single File:**
```bash
python main.py ../my_video.mp4
```
**Entire Directory (Recursive):**
```bash
python main.py ../videos_folder/ --lang "Spanish" --model small
```
+61
View File
@@ -0,0 +1,61 @@
#!/usr/bin/env python3
import os
import sys
import argparse
def scan_for_missing_translations(folder_path, target_lang="English"):
print(f"--- Diagnostic Scan: Missing Translations ---")
print(f"Target Language: {target_lang}")
print(f"Scanning: {folder_path}\n")
missing_count = 0
total_srt = 0
# Files that need attention
missing_files = []
for root, dirs, files in os.walk(folder_path):
for file in files:
# Look for source SRT files
# Logic: Ends in .srt, but NOT .Lang.srt and NOT .Lang.deep_translate.srt
if file.endswith(".srt") and \
not file.endswith(f".{target_lang}.srt") and \
not file.endswith(f".{target_lang}.deep_translate.srt"):
total_srt += 1
source_path = os.path.join(root, file)
base_name = os.path.splitext(file)[0]
# Check for existing translations
path_gemini = os.path.join(root, f"{base_name}.{target_lang}.srt")
path_deep = os.path.join(root, f"{base_name}.{target_lang}.deep_translate.srt")
if not os.path.exists(path_gemini) and not os.path.exists(path_deep):
missing_count += 1
missing_files.append(source_path)
if missing_files:
print(f"Found {missing_count} files waiting for translation:\n")
for f in missing_files:
print(f" [MISSING] {f}")
else:
print("✅ All subtitles appear to be translated!")
print("\n" + "="*40)
print(f"Summary:")
print(f"Total Source SRTs Found: {total_srt}")
print(f"Missing Translations: {missing_count}")
print("="*40)
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Diagnostic: Find missing translations")
parser.add_argument("folder", help="Path to the folder to scan")
parser.add_argument("lang", nargs="?", default="English", help="Target language (default: English)")
args = parser.parse_args()
if not os.path.exists(args.folder):
print(f"Error: Folder '{args.folder}' does not exist.")
sys.exit(1)
scan_for_missing_translations(args.folder, args.lang)
@@ -0,0 +1,100 @@
# Note: This feature requires pyannote.audio and a HuggingFace token.
# If these are not present, this module will likely fail or raise errors.
# Due to the complexity and weight of pyannote.audio, this is a placeholder
# for where the logic would sit. Implementing full diarization requires
# downloading models and handling complex segment merging.
import os
import sys
def diarize_audio(audio_path, num_speakers=None, hf_token=None):
"""
Performs speaker diarization using pyannote.audio.
Args:
audio_path (str): Path to audio file.
num_speakers (int, optional): Number of speakers if known.
hf_token (str): HuggingFace Auth Token.
Returns:
list: List of segments with speaker labels [(start, end, speaker), ...].
"""
try:
from pyannote.audio import Pipeline
except ImportError:
print("Error: pyannote.audio not installed. Diarization skipped.")
return None
if not hf_token:
print("Error: HuggingFace Token (HF_TOKEN) not found. Diarization skipped.")
return None
print(f"Loading Diarization Pipeline (pyannote/speaker-diarization-3.1)...")
try:
# Note: 'use_auth_token' was deprecated in favor of 'token' in recent versions
pipeline = Pipeline.from_pretrained(
"pyannote/speaker-diarization-3.1",
token=hf_token
)
# Move to GPU if available
import torch
if torch.cuda.is_available():
pipeline.to(torch.device("cuda"))
print(f"Diarizing {audio_path}...")
diarization = pipeline(audio_path, num_speakers=num_speakers)
results = []
for turn, _, speaker in diarization.itertracks(yield_label=True):
results.append({
"start": turn.start,
"end": turn.end,
"speaker": speaker
})
return results
except Exception as e:
print(f"Error during diarization: {e}")
return None
def merge_diarization_with_transcript(transcript_segments, diarization_segments):
"""
Merges Whisper segments with Diarization speaker labels based on time overlap.
Args:
transcript_segments (list): Whisper segments [{'start': 0.0, 'end': 1.0, 'text': 'Hi'}, ...]
diarization_segments (list): Diarization segments [{'start': 0.1, 'end': 0.9, 'speaker': 'SPEAKER_00'}]
Returns:
list: Enhanced transcript segments with 'speaker' key.
"""
if not diarization_segments:
return transcript_segments
# Simple overlap matching logic
for t_seg in transcript_segments:
# Find diarization segment with max overlap
t_start = t_seg['start']
t_end = t_seg['end']
best_speaker = "Unknown"
max_overlap = 0
for d_seg in diarization_segments:
d_start = d_seg['start']
d_end = d_seg['end']
# Calculate intersection
overlap_start = max(t_start, d_start)
overlap_end = min(t_end, d_end)
overlap_duration = max(0, overlap_end - overlap_start)
if overlap_duration > max_overlap:
max_overlap = overlap_duration
best_speaker = d_seg['speaker']
t_seg['speaker'] = best_speaker
return transcript_segments
+89
View File
@@ -0,0 +1,89 @@
#!/bin/bash
# Function to handle single file conversion
convert_file() {
local INPUT_FILE="$1"
# Check if file exists
if [ ! -f "$INPUT_FILE" ]; then
echo "Warning: File '$INPUT_FILE' not found. Skipping."
return
fi
# Get path and filename info
local DIRNAME=$(dirname -- "$INPUT_FILE")
local FILENAME=$(basename -- "$INPUT_FILE")
local FILENAME_NO_EXT="${FILENAME%.*}"
# Define output filename in the same directory
local OUTPUT_FILE="${DIRNAME}/${FILENAME_NO_EXT}.wav"
# Check if we are trying to convert a wav to wav (avoid redundant work or loops)
if [[ "$INPUT_FILE" == *.wav ]]; then
echo "Skipping .wav file: $INPUT_FILE"
return
fi
echo "Processing '$INPUT_FILE' -> '$OUTPUT_FILE'..."
# Run ffmpeg command
# -y overwrites output files without asking
# -v error -stats reduces output verbosity but keeps progress/errors
ffmpeg -i "$INPUT_FILE" -ar 16000 -ac 1 -c:a pcm_s16le -y "$OUTPUT_FILE" < /dev/null -v error -stats
if [ $? -eq 0 ]; then
echo -e "\nSuccess: '$OUTPUT_FILE' created."
else
echo -e "\nError converting '$INPUT_FILE'."
fi
echo "----------------------------------------"
}
# Check if at least one input is provided
if [ -z "$1" ]; then
echo "Usage: $0 <file_or_folder> [file_or_folder2 ...]"
exit 1
fi
# Loop through all provided arguments
for ARG in "$@"; do
# Check for SMB/Network URIs which standard shell tools don't support directly
if [[ "$ARG" == smb://* ]]; then
echo "Error: Network URI '$ARG' detected."
echo "This script works on filesystem paths. Please mount the network share first."
echo " - Linux (GVFS): Check /run/user/\$UID/gvfs/"
echo " - macOS: Check /Volumes/"
echo " - Windows (WSL): Mount the drive to a letter or /mnt/"
continue
fi
if [ -d "$ARG" ]; then
# It's a directory: find files recursively
echo "Scanning directory '$ARG' for video files..."
# Find common video files (case insensitive)
# Using -print0 and while read loop handles filenames with spaces correctly
find "$ARG" -type f \( \
-iname "*.mp4" -o \
-iname "*.mkv" -o \
-iname "*.mov" -o \
-iname "*.avi" -o \
-iname "*.webm" -o \
-iname "*.flv" -o \
-iname "*.wmv" -o \
-iname "*.m4v" -o \
-iname "*.mpg" -o \
-iname "*.mpeg" -o \
-iname "*.3gp" -o \
-iname "*.ts" \
\) -print0 | while IFS= read -r -d '' FOUND_FILE; do
convert_file "$FOUND_FILE"
done
elif [ -f "$ARG" ]; then
# It's a single file
convert_file "$ARG"
else
echo "Warning: '$ARG' is not a valid file or directory."
fi
done
@@ -0,0 +1,102 @@
import os
import subprocess
import sys
def extract_audio(video_path, output_path=None):
"""
Extracts audio from a video file using ffmpeg.
Args:
video_path (str): Path to the input video file.
output_path (str, optional): Path for the output audio file.
If None, defaults to same name with .wav extension.
Returns:
str: Path to the generated audio file.
"""
if not os.path.exists(video_path):
raise FileNotFoundError(f"Video file not found: {video_path}")
if output_path is None:
base_name = os.path.splitext(video_path)[0]
output_path = f"{base_name}.wav"
# Check if output file already exists to avoid redundant processing
if os.path.exists(output_path):
print(f"Audio file already exists: {output_path}")
return output_path
print(f"Extracting audio from {video_path}...")
# Command matching the user's preferred settings: 16kHz, Mono, PCM s16le
# -y overwrites without asking (though we checked existence above, this is for safety if we force it)
command = [
"ffmpeg",
"-i", video_path,
"-ar", "16000",
"-ac", "1",
"-c:a", "pcm_s16le",
"-y",
"-v", "error", # Less verbose
output_path
]
try:
subprocess.run(command, check=True)
print(f"Audio extracted to: {output_path}")
return output_path
except subprocess.CalledProcessError as e:
print(f"Error extracting audio: {e}")
sys.exit(1)
def embed_subtitles(video_path, srt_path, output_path=None):
"""
Embeds subtitles into the video file (Soft Subs) and sets them as primary.
Args:
video_path (str): Path to the input video.
srt_path (str): Path to the SRT file.
output_path (str, optional): Path for the output video.
"""
if not os.path.exists(video_path) or not os.path.exists(srt_path):
print("Error: Video or SRT file not found for embedding.")
return
if output_path is None:
base, ext = os.path.splitext(video_path)
output_path = f"{base}.subbed{ext}"
print(f"Embedding subtitles into: {output_path}...")
# Determine subtitle codec based on container
sub_codec = "mov_text" if video_path.lower().endswith(".mp4") else "srt"
# Command breakdown:
# -map 0:v -map 0:a -> Keep all video and audio from source
# -map 1:0 -> Add the subtitle from the 2nd input (srt_path)
# -c copy -> Copy video/audio streams (no re-encoding)
# -disposition:s:0 default -> Make the first subtitle track (ours) the default
# -metadata:s:s:0 -> Set metadata for the first subtitle stream
command = [
"ffmpeg",
"-i", video_path,
"-i", srt_path,
"-map", "0:v",
"-map", "0:a",
"-map", "1:0",
"-c", "copy",
"-c:s", sub_codec,
"-disposition:s:0", "default",
"-metadata:s:s:0", "language=eng",
"-metadata:s:s:0", "title=English (AI Translated)",
"-y",
"-v", "error",
output_path
]
try:
subprocess.run(command, check=True)
print(f"Subtitles embedded successfully: {output_path} (Set as primary)")
except subprocess.CalledProcessError as e:
print(f"Error embedding subtitles: {e}")
@@ -0,0 +1,299 @@
import argparse
import os
import sys
from dotenv import load_dotenv
# Load environment variables from central .env_files directory
script_dir = os.path.dirname(os.path.abspath(__file__))
env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe'))
if os.path.exists(env_path):
load_dotenv(env_path)
else:
# Fallback: check local .env
local_env = os.path.join(script_dir, '.env')
if os.path.exists(local_env):
load_dotenv(local_env)
else:
# Last resort: just try loading generic (cwd)
load_dotenv()
from extractor import extract_audio, embed_subtitles
from transcriber import transcribe_audio, save_as_srt
from translator import translate_srt, translate_fallback_free
from utils import validate_and_repair_srt
from diarizer import diarize_audio, merge_diarization_with_transcript
import tracker
from tracker import JobStatus
def save_srt_with_speakers(segments, output_path):
"""Helper to save SRT with speaker labels prepended to text."""
def format_timestamp(seconds: float):
whole_seconds = int(seconds)
milliseconds = int((seconds - whole_seconds) * 1000)
hours = whole_seconds // 3600
minutes = (whole_seconds % 3600) // 60
seconds = whole_seconds % 60
return f"{hours:02d}:{minutes:02d}:{seconds:02d},{milliseconds:03d}"
with open(output_path, "w", encoding="utf-8") as f:
for i, segment in enumerate(segments, start=1):
start = format_timestamp(segment["start"])
end = format_timestamp(segment["end"])
text = segment["text"].strip()
speaker = segment.get("speaker", "")
if speaker and speaker != "Unknown":
text = f"[{speaker}]: {text}"
f.write(f"{i}\n")
f.write(f"{start} --> {end}\n")
f.write(f"{text}\n\n")
print(f"SRT saved to: {output_path}")
def process_file(file_path, args, source_lang=None):
tracker.logger.info(f"=== Processing: {file_path} ===")
# Initialize Job
job = tracker.get_job(file_path)
if job.status == JobStatus.COMPLETED and not args.force:
tracker.logger.info("Job already completed. Skipping.")
return
tracker.update_job_status(file_path, JobStatus.PROCESSING)
try:
# 1. Extract Audio
tracker.update_step(file_path, "step_extract", "processing")
audio_path = extract_audio(file_path)
tracker.update_step(file_path, "step_extract", "done")
# 2. Transcribe (Generate SRT)
tracker.update_step(file_path, "step_transcribe", "processing")
transcript_file = os.path.splitext(file_path)[0] + ".srt"
transcript_exists = os.path.exists(transcript_file) and not args.force
final_srt_path = transcript_file
if transcript_exists:
tracker.logger.info(f"Transcript exists: {transcript_file}. Skipping transcription.")
with open(transcript_file, "r", encoding="utf-8") as f:
srt_content = f.read()
else:
result = transcribe_audio(audio_path, model_size=args.model, language=source_lang)
segments = result["segments"]
if args.diarize:
hf_token = args.hf_token or os.getenv("HF_TOKEN")
if hf_token:
tracker.logger.info("Running Speaker Diarization...")
diar_segments = diarize_audio(audio_path, hf_token=hf_token)
if diar_segments:
segments = merge_diarization_with_transcript(segments, diar_segments)
tracker.logger.info("Diarization merged into transcript.")
else:
tracker.logger.warning("Warning: --diarize requested but HF_TOKEN not provided. Skipping.")
if args.diarize:
save_srt_with_speakers(segments, transcript_file)
else:
save_as_srt(result, transcript_file)
validate_and_repair_srt(transcript_file)
with open(transcript_file, "r", encoding="utf-8") as f:
srt_content = f.read()
tracker.update_step(file_path, "step_transcribe", "done")
# 3. Translate
tracker.update_step(file_path, "step_translate", "processing")
base_translated = os.path.splitext(file_path)[0] + f".{args.lang}.srt"
deep_translated = os.path.splitext(file_path)[0] + f".{args.lang}.deep_translate.srt"
translated_file = base_translated # Default
translation_success = False
method_used = "None"
if (os.path.exists(base_translated) or os.path.exists(deep_translated)) and not args.force:
if os.path.exists(deep_translated):
translated_file = deep_translated
method_used = "DeepTranslate (Existing)"
else:
method_used = "Gemini (Existing)"
tracker.logger.info(f"Translation exists: {translated_file} ({method_used}). Skipping translation.")
final_srt_path = translated_file
translation_success = True
else:
if srt_content:
# Helper functions
def try_gemini():
res = translate_srt(srt_content, target_language=args.lang)
if res:
with open(base_translated, "w", encoding="utf-8") as f:
f.write(res)
return True, base_translated, "Gemini"
return False, None, None
def try_deep():
lang_map = {
"English": "en", "French": "fr", "Spanish": "es", "German": "de",
"Italian": "it", "Portuguese": "pt", "Russian": "ru",
"Japanese": "ja", "Chinese": "zh-CN"
}
target_code = lang_map.get(args.lang, "en")
res = translate_fallback_free(srt_content, target_language=target_code)
if res:
with open(deep_translated, "w", encoding="utf-8") as f:
f.write(res)
return True, deep_translated, "DeepTranslate"
return False, None, None
success = False
if args.prefer_deep:
success, path, method = try_deep()
if not success:
tracker.logger.info("DeepTranslate failed. Attempting Gemini...")
success, path, method = try_gemini()
else:
success, path, method = try_gemini()
if not success:
tracker.logger.warning("Gemini failed. Attempting DeepTranslate...")
success, path, method = try_deep()
if success:
tracker.logger.info(f"Translation saved to: {path} ({method})")
validate_and_repair_srt(path)
final_srt_path = path
translation_success = True
method_used = method
else:
tracker.logger.error("TRANSLATION FAILED.")
tracker.update_step(file_path, "step_translate", "failed")
translation_success = False
if translation_success:
tracker.update_step(file_path, "step_translate", "done")
tracker.logger.info(f"Translation Method: {method_used}")
# 4. Embed Subtitles
tracker.update_step(file_path, "step_embed", "processing")
should_embed = args.embed
if args.embed and not translation_success:
tracker.logger.warning("SAFETY HALT: Translation failed. Skipping embedding/deletion.")
should_embed = False
if should_embed:
embed_subtitles(file_path, final_srt_path)
if args.delete_source:
if args.embed:
base, ext = os.path.splitext(file_path)
expected_output = f"{base}.subbed{ext}"
if os.path.exists(expected_output):
try:
os.remove(file_path)
tracker.logger.info(f"SOURCE DELETED: {file_path}")
except OSError as e:
tracker.logger.error(f"Error deleting source: {e}")
else:
tracker.logger.error(f"SAFETY ABORT: Output '{expected_output}' not found.")
else:
tracker.logger.warning("SAFETY ABORT: Enable --embed to delete source.")
tracker.update_step(file_path, "step_embed", "done")
# 5. Cleanup
if args.cleanup:
try:
os.remove(audio_path)
tracker.logger.info(f"Cleanup: Removed {audio_path}")
except OSError as e:
tracker.logger.warning(f"Warning: Could not remove audio: {e}")
# Mark Complete
if translation_success:
tracker.update_job_status(file_path, JobStatus.COMPLETED)
else:
tracker.update_job_status(file_path, JobStatus.FAILED, error="Translation failed")
except Exception as e:
tracker.logger.exception(f"Job Failed for {file_path}")
tracker.update_job_status(file_path, JobStatus.FAILED, error=str(e))
return
def main():
parser = argparse.ArgumentParser(description="AI Video Transcriber & Translator")
# Change nargs='?' to nargs='*' or '+' to support multiple inputs
parser.add_argument("inputs", nargs='*', help="Path(s) to video file or directory")
parser.add_argument("--model", default="auto", choices=["auto", "tiny", "base", "small", "medium", "large"], help="Whisper model size (default: auto)")
parser.add_argument("--lang", default="English", help="Target language for translation (default: English)")
parser.add_argument("--source-lang", help="Source language of audio. If omitted, prompts user.")
parser.add_argument("--force", action="store_true", help="Overwrite existing files")
parser.add_argument("--cleanup", action="store_true", help="Delete temporary .wav file")
parser.add_argument("--embed", action="store_true", help="Embed subtitles (Soft Subs)")
parser.add_argument("--diarize", action="store_true", help="Enable speaker diarization")
parser.add_argument("--hf-token", help="HuggingFace Token")
parser.add_argument("--delete-source", action="store_true", help="Delete original file after embedding")
parser.add_argument("--retry-failed", action="store_true", help="Retry FAILED jobs from DB")
parser.add_argument("--prefer-deep", action="store_true", help="Prefer DeepTranslate (Free) over Gemini")
args = parser.parse_args()
if not os.getenv("GEMINI_API_KEY"):
print("Warning: GEMINI_API_KEY environment variable not set. Translation step will fail.")
source_lang = args.source_lang
# Retry Logic
if args.retry_failed:
print("Retrying failed jobs from database...")
failed_files = tracker.get_failed_jobs()
if not failed_files:
print("No failed jobs found.")
return
if not source_lang:
print("\n--- Audio Configuration ---")
user_input = input("Enter source language (e.g. 'French'). Enter for Auto: ").strip()
source_lang = user_input if user_input else None
for file_path in failed_files:
if os.path.exists(file_path):
process_file(file_path, args, source_lang)
else:
print(f"Skipping missing file: {file_path}")
return
# Normal Logic
if not args.inputs:
parser.print_help()
sys.exit(1)
if not source_lang:
print("\n--- Audio Configuration ---")
user_input = input("Enter source language (e.g. 'French'). Enter for Auto: ").strip()
source_lang = user_input if user_input else None
print(f"Selected: {source_lang if source_lang else 'Auto-detect'}")
# Process all inputs
video_extensions = ('.mp4', '.mkv', '.mov', '.avi', '.webm', '.flv', '.wmv', '.m4v')
for input_path in args.inputs:
if os.path.isfile(input_path):
process_file(input_path, args, source_lang)
elif os.path.isdir(input_path):
found = False
for root, dirs, files in os.walk(input_path):
for file in files:
if file.lower().endswith(video_extensions):
found = True
process_file(os.path.join(root, file), args, source_lang)
if not found:
print(f"No video files found in {input_path}")
else:
print(f"Error: Invalid input path '{input_path}'")
if __name__ == "__main__":
main()
+69
View File
@@ -0,0 +1,69 @@
#!/bin/bash
# Configuration
MOUNT_POINT="/mnt/truenas_isolation"
SHARE="//truenas.local/isolation"
echo "--- SMB Mount Tool (V1) ---"
# Determine privilege escalation method
PRIV_CMD=""
if [ "$EUID" -eq 0 ]; then
echo "Running as root."
else
if command -v sudo &> /dev/null; then
PRIV_CMD="sudo"
elif command -v flatpak-spawn &> /dev/null; then
echo "Detected Flatpak environment. Attempting to use host permissions via sudo..."
# We need to run sudo ON THE HOST.
# flatpak-spawn --host runs as the current user on the host.
# So we run 'sudo' inside that host shell.
PRIV_CMD="flatpak-spawn --host sudo"
# Note: This requires the flatpak to have permission to talk to the host
else
echo "❌ Error: This script requires root privileges to mount drives."
echo " 'sudo' was not found."
echo " Please run this script as root: su -c ./mount_truenas.sh"
exit 1
fi
fi
# 1. Create mount point if it doesn't exist
if [ ! -d "$MOUNT_POINT" ]; then
echo "Creating directory $MOUNT_POINT..."
# We try to create it. If it fails (e.g. inside read-only flatpak mount namespace), warn user.
$PRIV_CMD mkdir -p "$MOUNT_POINT"
if [ $? -ne 0 ]; then
echo "Error creating directory. If you are in a Flatpak, you might not have access to host /mnt."
exit 1
fi
fi
# 2. Get Credentials
read -p "Enter SMB Username [guest]: " SMB_USER
SMB_USER=${SMB_USER:-guest}
# 3. Mount
echo "Mounting $SHARE to $MOUNT_POINT..."
# IMPORTANT: We force the mount to be owned by the current user (UID 1000 usually)
# This fixes "Permission Denied" errors when writing to the share.
# We also set file_mode/dir_mode to 0777 as a fallback to ensure full access.
MOUNT_OPTS="vers=3.0,uid=$(id -u),gid=$(id -g),file_mode=0777,dir_mode=0777,noperm"
if [ "$SMB_USER" == "guest" ]; then
$PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o "guest,$MOUNT_OPTS"
else
# This will prompt for the SMB password
$PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o "username=$SMB_USER,$MOUNT_OPTS"
fi
# 4. Check result
if [ $? -eq 0 ]; then
echo "✅ Success! Share is now available at $MOUNT_POINT"
echo "Files are now owned by $(id -un):$(id -gn) with full write access."
echo "The mapping will disappear automatically after you reboot."
else
echo "❌ Error: Failed to mount the share."
echo "Ensure 'cifs-utils' is installed and the server is reachable."
fi
+135
View File
@@ -0,0 +1,135 @@
#!/usr/bin/env python3
import os
import sys
import argparse
import subprocess
from dotenv import load_dotenv
# Load config
script_dir = os.path.dirname(os.path.abspath(__file__))
env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe'))
if os.path.exists(env_path):
load_dotenv(env_path)
else:
load_dotenv()
from translator import translate_srt
from utils import validate_and_repair_srt
import pysubs2
from deep_translator import GoogleTranslator
from datetime import datetime
def translate_fallback_free(source_srt_path, output_srt_path, target_lang="en"):
"""
Fallback translation using deep-translator (free Google Translate).
Parses SRT, translates text line-by-line, and saves new SRT.
"""
# ... (rest of translate_fallback_free is same) ...
# ... (rest of re_embed_subtitles is same) ...
def process_recovery(folder_path, target_lang="English"):
print(f"Scanning {folder_path} for incomplete translations...")
# Define log file
recovery_log_file = os.path.join(folder_path, "recovery_status.log")
print(f"Logging actions to: {recovery_log_file}")
count_fixed = 0
count_skipped = 0
video_extensions = ('.mp4', '.mkv', '.mov', '.avi')
for root, dirs, files in os.walk(folder_path):
for file in files:
# We are looking for the SOURCE SRT files primarily
# Exclude our output files to avoid loops
if file.endswith(".srt") and \
not file.endswith(f".{target_lang}.srt") and \
not file.endswith(f".{target_lang}.deep_translate.srt"):
source_srt_path = os.path.join(root, file)
base_name = os.path.splitext(file)[0]
# Check if Translation exists (Standard or Fallback)
path_gemini = os.path.join(root, f"{base_name}.{target_lang}.srt")
path_deep = os.path.join(root, f"{base_name}.{target_lang}.deep_translate.srt")
if os.path.exists(path_gemini) or os.path.exists(path_deep):
continue
print(f"\nFound untranslated transcript: {file}")
# Try to translate
with open(source_srt_path, "r", encoding="utf-8") as f:
content = f.read()
new_srt_content = translate_srt(content, target_language=target_lang)
success = False
method_used = "None"
final_srt_path = path_gemini # Default if success
if new_srt_content:
# Save Gemini translation
with open(path_gemini, "w", encoding="utf-8") as f:
f.write(new_srt_content)
success = True
method_used = "Gemini"
final_srt_path = path_gemini
else:
print("❌ Gemini API failed. Attempting Free Fallback...")
lang_map = {
"English": "en", "French": "fr", "Spanish": "es",
"German": "de", "Italian": "it", "Portuguese": "pt",
"Russian": "ru", "Japanese": "ja", "Chinese": "zh-CN"
}
target_code = lang_map.get(target_lang, "en")
if translate_fallback_free(source_srt_path, path_deep, target_lang=target_code):
success = True
method_used = "DeepTranslate"
final_srt_path = path_deep
else:
print("❌ All translation methods failed. Skipping.")
continue
if success:
# Log result
with open(recovery_log_file, "a", encoding="utf-8") as log:
log.write(f"{datetime.now().isoformat()} | {method_used} | {file} -> {os.path.basename(final_srt_path)}\n")
validate_and_repair_srt(final_srt_path)
# Now, find the video file to update
# Case 1: Original name
video_candidates = [
os.path.join(root, base_name + ".mp4"),
os.path.join(root, base_name + ".mkv"),
# Case 2: .subbed name (if original deleted)
os.path.join(root, base_name + ".subbed.mp4"),
]
found_video = None
for v in video_candidates:
if os.path.exists(v):
found_video = v
break
if found_video:
print(f"Found video to fix: {found_video}")
if re_embed_subtitles(found_video, final_srt_path):
count_fixed += 1
else:
print("Warning: Could not find a corresponding video file to embed into.")
print(f"\nRecovery Complete. Fixed {count_fixed} files.")
if __name__ == "__main__":
if len(sys.argv) < 2:
print("Usage: ./recover_and_fix.py <folder_path> [target_lang]")
sys.exit(1)
folder = sys.argv[1]
lang = sys.argv[2] if len(sys.argv) > 2 else "English"
process_recovery(folder, lang)
@@ -0,0 +1,9 @@
openai-whisper
google-generativeai
ffmpeg-python
torch
numpy
tenacity
pysubs2
pyannote.audio
python-dotenv
+179
View File
@@ -0,0 +1,179 @@
#!/usr/bin/env python3
import os
import sys
import subprocess
import shutil
def clear_screen():
os.system('cls' if os.name == 'nt' else 'clear')
def get_input(prompt, default=None):
"""Helper to get input with a default value."""
if default:
user_input = input(f"{prompt} [{default}]: ").strip()
return user_input if user_input else default
else:
return input(f"{prompt}: ").strip()
def get_yes_no(prompt, default="y"):
"""Helper to get boolean input."""
display_default = "Y/n" if default.lower() in ["y", "yes"] else "y/N"
choice = get_input(f"{prompt} ({display_default})", default).lower()
return choice in ["y", "yes", "true", "1"]
def print_header():
print("==========================================")
print(" AI Video Transcriber & Translator Wizard")
print("==========================================")
print("")
def main():
clear_screen()
print_header()
# Try to load the .env file so the wizard knows what's already configured
try:
from dotenv import load_dotenv
script_dir = os.path.dirname(os.path.abspath(__file__))
env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe'))
if os.path.exists(env_path):
load_dotenv(env_path)
except ImportError:
pass
# 1. Input File/Folder (Multiple)
input_paths = []
while True:
prompt_text = "Enter a path to a video file or folder"
if input_paths:
prompt_text += " (or press Enter to finish)"
input_path = get_input(prompt_text)
if not input_path:
if input_paths:
break
else:
print("Error: You must provide at least one path.")
continue
# Clean up input
input_path = input_path.strip('"\'')
input_path = input_path.replace(r'\ ', ' ')
# Expand user (~) and resolve absolute path
input_path = os.path.abspath(os.path.expanduser(input_path))
if os.path.exists(input_path):
input_paths.append(input_path)
print(f"Added: {input_path}")
else:
print(f"Error: Path '{input_path}' does not exist. Please try again.\n")
print("\nSelected Inputs:")
for p in input_paths:
print(f" - {p}")
print("")
# 2. Languages
source_lang = get_input("Source Language (e.g., French, es)", default="auto")
target_lang = get_input("Target Language for translation", default="English")
print("")
# 3. Model Size
print("Model Size Options: tiny, base, small, medium, large, auto")
model_size = get_input("Whisper Model Size", default="auto")
print("")
# 4. Features
do_cleanup = get_yes_no("Cleanup temporary audio files after processing?", default="y")
do_embed = get_yes_no("Embed subtitles into the video file (Soft Subs)?", default="y")
do_diarize = get_yes_no("Enable Speaker Diarization (Identify speakers)?", default="n")
do_delete_source = False
if do_embed:
print("\n⚠️ WARNING: Using this next option will PERMANENTLY DELETE the original video files.")
print(" It will only run if the new subtitled video is successfully created.")
do_delete_source = get_yes_no("Delete original source files after embedding?", default="n")
do_prefer_deep = get_yes_no("Prefer DeepTranslate (Free) over Gemini API?", default="n")
hf_token = None
if do_diarize:
if not os.getenv("HF_TOKEN"):
print("\nSpeaker Diarization requires a HuggingFace Token.")
hf_token = get_input("Enter your HuggingFace Token (hidden)", default="")
else:
print("Using HF_TOKEN from environment.")
# 5. Build Command
# script is in main.py relative to this script
script_dir = os.path.dirname(os.path.abspath(__file__))
main_script = os.path.join(script_dir, "main.py")
cmd = [sys.executable, main_script]
# Add all inputs
cmd.extend(input_paths)
cmd.extend(["--lang", target_lang])
cmd.extend(["--model", model_size])
if source_lang != "auto":
cmd.extend(["--source-lang", source_lang])
if do_cleanup:
cmd.append("--cleanup")
if do_embed:
cmd.append("--embed")
if do_delete_source:
cmd.append("--delete-source")
if do_diarize:
cmd.append("--diarize")
if hf_token:
cmd.extend(["--hf-token", hf_token])
if do_prefer_deep:
cmd.append("--prefer-deep")
# 6. Confirmation and Execution
clear_screen()
print_header()
print("Configuration Complete!")
print("-" * 30)
print("Inputs:")
for p in input_paths:
print(f" - {p}")
print(f"Source Lang: {source_lang}")
print(f"Target Lang: {target_lang}")
print(f"Model: {model_size}")
print(f"Cleanup: {do_cleanup}")
print(f"Embed Subs: {do_embed}")
print(f"Delete Src: {do_delete_source}")
print(f"Diarization: {do_diarize}")
print(f"Prefer Deep: {do_prefer_deep}")
print("-" * 30)
if not get_yes_no("Run this job now?", default="y"):
print("Aborted.")
sys.exit(0)
print("\nStarting Job...\n")
try:
# Pass environment variables including HF_TOKEN if set
env = os.environ.copy()
if hf_token:
env["HF_TOKEN"] = hf_token
subprocess.run(cmd, check=True, env=env)
print("\n✅ Job Complete!")
except subprocess.CalledProcessError as e:
print(f"\n❌ Job Failed with error code {e.returncode}")
except KeyboardInterrupt:
print("\nJob interrupted by user.")
if __name__ == "__main__":
main()
@@ -0,0 +1,84 @@
import logging
import os
from datetime import datetime
from sqlalchemy import create_engine, Column, Integer, String, DateTime, Enum, Text
from sqlalchemy.orm import declarative_base, sessionmaker
import enum
# Setup Logging
log_dir = "logs"
os.makedirs(log_dir, exist_ok=True)
log_file = os.path.join(log_dir, f"transcriber_{datetime.now().strftime('%Y%m%d')}.log")
logging.basicConfig(
level=logging.INFO,
format='%(asctime)s - %(levelname)s - %(message)s',
handlers=[
logging.FileHandler(log_file),
logging.StreamHandler()
]
)
logger = logging.getLogger(__name__)
# Database Setup
Base = declarative_base()
DB_FILE = "job_history.db"
class JobStatus(enum.Enum):
PENDING = "pending"
PROCESSING = "processing"
COMPLETED = "completed"
FAILED = "failed"
class Job(Base):
__tablename__ = 'jobs'
id = Column(Integer, primary_key=True)
file_path = Column(String, unique=True, nullable=False)
status = Column(Enum(JobStatus), default=JobStatus.PENDING)
error_message = Column(Text, nullable=True)
last_updated = Column(DateTime, default=datetime.utcnow, onupdate=datetime.utcnow)
# Track progress of individual steps
step_extract = Column(String, default="pending") # pending, done, failed
step_transcribe = Column(String, default="pending")
step_translate = Column(String, default="pending")
step_embed = Column(String, default="pending")
engine = create_engine(f'sqlite:///{DB_FILE}')
Base.metadata.create_all(engine)
Session = sessionmaker(bind=engine)
def get_job(file_path):
session = Session()
job = session.query(Job).filter_by(file_path=file_path).first()
if not job:
job = Job(file_path=file_path)
session.add(job)
session.commit()
session.close()
return job
def update_job_status(file_path, status, error=None):
session = Session()
job = session.query(Job).filter_by(file_path=file_path).first()
if job:
job.status = status
if error:
job.error_message = str(error)
session.commit()
session.close()
def update_step(file_path, step_name, status):
session = Session()
job = session.query(Job).filter_by(file_path=file_path).first()
if job:
setattr(job, step_name, status)
session.commit()
session.close()
def get_failed_jobs():
session = Session()
jobs = session.query(Job).filter_by(status=JobStatus.FAILED).all()
session.close()
return [j.file_path for j in jobs]
@@ -0,0 +1,168 @@
import whisper
import os
import sys
import subprocess
import torch
def check_gpu_health():
"""
Performs a robust check for GPU availability and prints detailed troubleshooting
info if issues are detected, specific to Bazzite/VS Code environments.
"""
print("Checking GPU health...")
# 1. Check if the OS/Driver sees the GPU
nvidia_smi_ok = False
try:
subprocess.run(["nvidia-smi"], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
nvidia_smi_ok = True
except (subprocess.CalledProcessError, FileNotFoundError):
nvidia_smi_ok = False
# 2. Check if PyTorch sees the GPU
torch_cuda_ok = torch.cuda.is_available()
if torch_cuda_ok:
print(f"✅ GPU is accessible: {torch.cuda.get_device_name(0)}")
print(f" CUDA Version: {torch.version.cuda}")
return True
# --- Troubleshooting Block ---
print("\n⚠️ WARNING: GPU not detected by PyTorch. Falling back to CPU.")
print(" Transcription will be significantly slower.\n")
print("--- Diagnostic Report ---")
if nvidia_smi_ok:
print("1. [OK] 'nvidia-smi' command works. The system driver is installed and visible.")
print("2. [FAIL] PyTorch cannot see the GPU.")
print(" -> Likely Cause: You might have installed the CPU-only version of PyTorch.")
print(" -> Solution: Reinstall PyTorch with CUDA support:")
print(" pip uninstall torch torchvision torchaudio")
print(" pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu118")
else:
print("1. [FAIL] 'nvidia-smi' command failed or not found.")
print(" -> Likely Cause: Nvidia drivers are missing, or the container/sandbox cannot access the GPU.")
print("\n --- Bazzite / VS Code / Container Specific Checks ---")
print(" a. If you are running inside a dev container (DevBox/Distrobox/Toolbox):")
print(" Ensure the container was created with nvidia support.")
print(" (Bazzite usually handles this for 'distrobox', but check your config).")
print(" b. If you are using VS Code Flatpak:")
print(" Flatpak might be restricting access. Check Flatseal permissions for VS Code.")
print(" c. Driver Check:")
print(" Run 'rpm -qa | grep nvidia' in your host terminal to verify drivers are installed.")
print("-------------------------\n")
return False
def get_vram_gb():
"""Returns the total VRAM in GB of the first CUDA device, or 0 if no CUDA."""
if not torch.cuda.is_available():
return 0
try:
# returns bytes
total_mem = torch.cuda.get_device_properties(0).total_memory
return total_mem / (1024 ** 3)
except Exception:
return 0
def get_optimal_model_size():
"""
Determines the best Whisper model based on available VRAM.
Rough estimates for VRAM usage (fp16):
- large: ~10 GB
- medium: ~5 GB
- small: ~2 GB
- base: ~1 GB
- tiny: ~1 GB
"""
vram = get_vram_gb()
if vram == 0:
return "base"
print(f"Detected GPU with {vram:.2f} GB VRAM.")
if vram >= 11:
return "large"
elif vram >= 6:
return "medium"
elif vram >= 3:
return "small"
else:
return "base"
def format_timestamp(seconds: float):
"""Converts seconds to SRT timestamp format (HH:MM:SS,mmm)."""
whole_seconds = int(seconds)
milliseconds = int((seconds - whole_seconds) * 1000)
hours = whole_seconds // 3600
minutes = (whole_seconds % 3600) // 60
seconds = whole_seconds % 60
return f"{hours:02d}:{minutes:02d}:{seconds:02d},{milliseconds:03d}"
def save_as_srt(result, output_path):
"""Saves the Whisper transcription result as an SRT file."""
with open(output_path, "w", encoding="utf-8") as f:
for i, segment in enumerate(result["segments"], start=1):
start = format_timestamp(segment["start"])
end = format_timestamp(segment["end"])
text = segment["text"].strip()
f.write(f"{i}\n")
f.write(f"{start} --> {end}\n")
f.write(f"{text}\n\n")
print(f"SRT saved to: {output_path}")
def transcribe_audio(audio_path, model_size="auto", language=None):
"""
Transcribes an audio file using OpenAI's Whisper model.
Args:
audio_path (str): Path to the input audio file.
model_size (str): Size of the Whisper model to use. If "auto", selects based on VRAM.
language (str, optional): Language code (e.g., "en", "fr", "es"). If None, auto-detects.
Returns:
dict: The full transcription result containing segments and text.
"""
if not os.path.exists(audio_path):
raise FileNotFoundError(f"Audio file not found: {audio_path}")
# Run health check once
check_gpu_health()
# Determine model size if auto
if model_size == "auto":
model_size = get_optimal_model_size()
print(f"Auto-selected model: '{model_size}'")
print(f"Loading Whisper model ('{model_size}')...")
# Check for GPU availability
device = "cuda" if torch.cuda.is_available() else "cpu"
print(f"Using device: {device}")
try:
model = whisper.load_model(model_size, device=device)
except RuntimeError as e:
if "out of memory" in str(e).lower():
print("Error: GPU Out of Memory. Try using a smaller model size.")
else:
print(f"Error loading model: {e}")
sys.exit(1)
except Exception as e:
print(f"Error loading model: {e}")
sys.exit(1)
print(f"Transcribing {audio_path}...")
try:
# fp16=False is needed for CPU, but we can let whisper handle defaults usually.
# language=None allows auto-detection.
result = model.transcribe(audio_path, language=language)
print("Transcription complete.")
return result
except Exception as e:
print(f"Error during transcription: {e}")
sys.exit(1)
@@ -0,0 +1,180 @@
import os
import sys
import warnings
import pysubs2
from deep_translator import GoogleTranslator
# Suppress warnings from google.generativeai about deprecation
warnings.filterwarnings("ignore", category=FutureWarning, module="google.generativeai")
import google.generativeai as genai
from tenacity import retry, stop_after_attempt, wait_exponential, retry_if_exception_type
# ... (retry_policy and _generate_with_retry remain same)
def translate_fallback_free(source_srt_content, target_language="en"):
"""
Fallback translation using deep-translator (free Google Translate).
Args:
source_srt_content (str): Content of the source SRT file.
target_language (str): Target language code (e.g. 'en', 'fr').
Returns:
str: Translated SRT content, or None if failed.
"""
print(f" [Free Fallback] Translating via Google Translate (deep-translator)...")
try:
# Load from string
subs = pysubs2.SSAFile.from_string(source_srt_content)
translator = GoogleTranslator(source='auto', target=target_language)
# Simple line-by-line translation
for line in subs:
text = line.text.strip()
if text:
# Sanity check: Skip lines that are too long
if len(text) > 4000:
print(f" Warning: Skipping line with excessive length ({len(text)} chars).")
continue
try:
# pysubs2 text can contain \N for newlines.
original_text = text.replace(r"\N", " ")
translated_text = translator.translate(original_text)
if translated_text:
line.text = translated_text
except Exception as e:
print(f" Warning: Failed to translate line: {e}")
# Return as string
return subs.to_string(format_="srt")
except Exception as e:
print(f" [Free Fallback] Critical Error: {e}")
return None
def get_best_available_model():
# ... (rest of file)
# Define a retry decorator
# Waits 2^x * 1 seconds between retries (1s, 2s, 4s, 8s, 16s, 32s...)
# With max=60, it will cap at waiting 60s per try.
# Stop after 15 attempts (approx 15 minutes of trying before giving up)
retry_policy = retry(
stop=stop_after_attempt(15),
wait=wait_exponential(multiplier=1, min=2, max=60),
retry=retry_if_exception_type(Exception),
reraise=True
)
@retry_policy
def _generate_with_retry(model, prompt):
"""Internal function to wrap the API call with retry logic."""
try:
return model.generate_content(prompt)
except Exception as e:
if "429" in str(e) or "Resource has been exhausted" in str(e):
print(f" [Rate Limit Hit] Waiting for quota reset... ({e})")
raise e
def get_best_available_model():
"""
Queries the API to find the best available model for text generation.
Priority: gemini-1.5-flash > gemini-1.5-pro > gemini-pro > any 'generateContent' model
"""
try:
available_models = []
for m in genai.list_models():
if 'generateContent' in m.supported_generation_methods:
available_models.append(m.name)
# Priority list
priorities = [
"models/gemini-1.5-flash",
"models/gemini-1.5-pro",
"models/gemini-pro"
]
# Check for priorities first
for p in priorities:
if p in available_models:
return p
# Fallback: check for aliases without 'models/' prefix just in case
for p in priorities:
short_name = p.replace("models/", "")
# Some libraries might return short names, or custom handling
# But genai.list_models() usually returns 'models/name'
pass
# If priority not found, pick the first available gemini model
for m in available_models:
if "gemini" in m:
return m
if available_models:
return available_models[0]
except Exception as e:
print(f"Warning: Could not list models ({e}). Defaulting to 'gemini-pro'.")
return "gemini-pro"
def translate_srt(srt_content, target_language="English", api_key=None):
"""
Translates SRT subtitle content using the Gemini API, preserving timestamps.
Args:
srt_content (str): The raw text content of the SRT file.
target_language (str): The target language for translation.
api_key (str): Google Gemini API key. If None, checks env var GEMINI_API_KEY.
Returns:
str: The translated SRT content.
"""
if not srt_content:
return ""
key = api_key or os.getenv("GEMINI_API_KEY")
if not key:
print("Error: GEMINI_API_KEY not found. Please set the environment variable or pass the key.")
sys.exit(1)
genai.configure(api_key=key)
# Automatically select the best model
model_name = get_best_available_model()
print(f"Using Gemini Model: {model_name}")
model = genai.GenerativeModel(model_name)
prompt = (
"You are a professional subtitle translator. Your task is to translate the following SRT subtitle file "
f"into {target_language}.\n\n"
"RULES:\n"
"1. PRESERVE the SRT format exactly. Do not modify timestamps (e.g., 00:00:01,000 --> 00:00:04,000) or sequence numbers.\n"
"2. Only translate the dialogue text.\n"
"3. Maintain the original tone and context.\n"
"4. Output ONLY the translated SRT content, no markdown code blocks or explanations.\n\n"
"SRT Content:\n"
f"{srt_content}"
)
print(f"Translating subtitles to {target_language} (with retries)...")
try:
# Call the retried internal function
response = _generate_with_retry(model, prompt)
print("Translation complete.")
# Cleanup: sometimes models wrap output in ```srt ... ``` or ``` ... ```
cleaned_text = response.text.strip()
if cleaned_text.startswith("```"):
# Remove first line (```srt or ```) and last line (```)
lines = cleaned_text.split('\n')
if len(lines) >= 2:
cleaned_text = '\n'.join(lines[1:-1])
return cleaned_text
except Exception as e:
print(f"Error during translation after retries: {e}")
return None
@@ -0,0 +1,28 @@
import pysubs2
import os
def validate_and_repair_srt(srt_path):
"""
Validates an SRT file and attempts to repair it using pysubs2.
Args:
srt_path (str): Path to the SRT file.
Returns:
bool: True if valid/repaired, False if critical error.
"""
if not os.path.exists(srt_path):
return False
print(f"Validating SRT: {srt_path}...")
try:
# Load the subtitle file. pysubs2 parser is robust and handles many errors automatically.
subs = pysubs2.load(srt_path)
# Save it back ensures consistent formatting and fixes minor syntax issues
subs.save(srt_path)
print("SRT validation passed (file re-saved with correct formatting).")
return True
except Exception as e:
print(f"Warning: SRT validation failed: {e}")
return False
@@ -0,0 +1,52 @@
# AI Video Transcriber & Translator
This tool extracts audio from videos, transcribes it using OpenAI's Whisper model, and translates the transcript using Google's Gemini API.
## Setup
1. **Install Dependencies:**
```bash
pip install -r requirements.txt
```
*Note: You need `ffmpeg` installed on your system.*
2. **API Key:**
Set your Gemini API key as an environment variable:
```bash
export GEMINI_API_KEY="your_api_key_here"
```
## Usage
### Linux / Mac
Run the wizard script:
```bash
./run_v2.py
```
### Windows
1. **Install FFmpeg:** Download from [ffmpeg.org](https://ffmpeg.org/download.html) and add the `bin` folder to your System PATH.
2. **Run:** Double-click `run_v2.bat`.
* It will automatically create the virtual environment, install dependencies, and launch the tool.
### Manual CLI
```bash
python ai_transcriber_v2/main.py <path_to_video> [options]
```
### Options:
* `--model`: Whisper model size (`tiny`, `base`, `small`, `medium`, `large`). Default: `base`.
* `--lang`: Target language for translation. Default: `English`.
* `--force`: Overwrite existing transcript/translation files.
### Examples:
**Single File:**
```bash
python main.py ../my_video.mp4
```
**Entire Directory (Recursive):**
```bash
python main.py ../videos_folder/ --lang "Spanish" --model small
```
@@ -0,0 +1,33 @@
# Current State of AI Transcriber V2
**Date:** January 12, 2026
**Version:** 2.0 (Refactored & Robust)
## 🚀 Recently Completed Features
1. **Unified Translation Logic**: All scripts now use a shared fallback pipeline: **Primary (User Pref) -> Secondary -> Local LLM (Ollama) -> MyMemory**.
2. **Flattened Directory Structure**: Removed nested `ai_transcriber_v2/ai_transcriber_v2` folders. All V2 modules are now in the root of `ai_transcriber_v2/`.
3. **Flatpak & Bazzite Compatibility**:
* Added `flatpak-spawn --host` support for `ffmpeg`, `ffprobe`, and `ollama`.
* Updated `mount_truenas.sh` to handle UID/GID mapping for write permissions on remote shares.
4. **Hardware-Accelerated Safety Checks**:
* **Size Validation**: Rejects any remuxed file that drops more than 20% of the original size.
* **Duration Validation**: Rejects any remuxed file where the duration differs by more than 1 second.
* **Zero-Byte Check**: Deletes empty outputs immediately.
5. **Intelligent Skipping**:
* Whisper detects source language; if it matches the target (e.g., English to English), translation is skipped entirely.
6. **Progress Tracking**:
* Integrated `tqdm` progress bars for iterative translation steps.
* Enabled `verbose=True` for Whisper to show live transcription segments.
7. **Stability**:
* Fixed SQLAlchemy `DetachedInstanceError` by expunging objects from sessions in `tracker.py`.
* Added `GracefulKiller` for clean `Ctrl+C` shutdowns (finishes current file, then exits).
## 🛠 Active Setup
- **Environment**: Bazzite (Linux) running VS Code via Flatpak.
- **Local LLM**: Ollama with `llama3` (or `dolphin-llama3`).
- **Media Tools**: Host-level FFmpeg/FFprobe accessible via `flatpak-spawn`.
## 📌 Next Steps / Future Ideas
- Consider batching small SRT segments for Gemini to reduce API calls and improve context.
- Add a GUI or Web dashboard for tracking `job_history.db`.
- Implement auto-retry for specific "failed_validation" files with different model parameters.
+95
View File
@@ -0,0 +1,95 @@
#!/usr/bin/env python3
import sys
import shutil
import subprocess
import torch
import os
def check_flatpak():
return os.path.exists("/.flatpak-info")
def run_cmd(cmd_list):
in_flatpak = check_flatpak()
if in_flatpak:
full_cmd = ["flatpak-spawn", "--host"] + cmd_list
else:
full_cmd = cmd_list
try:
result = subprocess.run(full_cmd, capture_output=True, text=True)
return result.returncode == 0, result.stdout.strip(), result.stderr.strip()
except FileNotFoundError:
return False, "", "Command not found"
except Exception as e:
return False, "", str(e)
def main():
print("========================================")
print(" AI Transcriber Diagnostic Tool")
print("========================================")
in_flatpak = check_flatpak()
print(f"Environment: {'Flatpak Sandbox' if in_flatpak else 'Native Host'}")
print("-" * 40)
# 1. GPU Check (PyTorch)
print("\n[1] GPU Availability (Internal PyTorch)")
if torch.cuda.is_available():
print(f"✅ GPU Detected: {torch.cuda.get_device_name(0)}")
print(f" VRAM: {torch.cuda.get_device_properties(0).total_memory / 1024**3:.2f} GB")
print(f" CUDA Version: {torch.version.cuda}")
else:
print("❌ GPU NOT Detected by PyTorch.")
print(" Whisper will run on CPU (Slow).")
# 2. Nvidia Driver Check (System)
print("\n[2] Nvidia Driver Check (System)")
ok, out, err = run_cmd(["nvidia-smi", "--query-gpu=name,driver_version", "--format=csv,noheader"])
if ok:
print(f"✅ Driver Active: {out}")
else:
print("❌ 'nvidia-smi' failed. Drivers might be missing or inaccessible.")
if err: print(f" Error: {err}")
# 3. FFmpeg Check
print("\n[3] FFmpeg Check")
ok, out, err = run_cmd(["ffmpeg", "-version"])
if ok:
version_line = out.split('\n')[0]
print(f"✅ FFmpeg Ready: {version_line}")
else:
print("❌ FFmpeg NOT found.")
print(" On Bazzite, install it with: 'brew install ffmpeg'")
# 4. FFprobe Check
print("\n[4] FFprobe Check (Required for duration checks)")
ok, out, err = run_cmd(["ffprobe", "-version"])
if ok:
version_line = out.split('\n')[0]
print(f"✅ FFprobe Ready: {version_line}")
else:
print("❌ FFprobe NOT found.")
# 5. Ollama Check
print("\n[5] Local LLM (Ollama)")
ok, out, err = run_cmd(["ollama", "--version"])
if ok:
print(f"✅ Ollama Installed: {out}")
# Check server
import socket
sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
sock.settimeout(1)
result = sock.connect_ex(('127.0.0.1', 11434))
if result == 0:
print("✅ Ollama Server is RUNNING.")
else:
print("⚠️ Ollama Server is NOT running (Scripts will try to auto-start it).")
else:
print("❌ Ollama NOT found.")
print(" Run: ./ai_transcriber_v2/install_local_llm.sh")
print("\n========================================")
if __name__ == "__main__":
main()
@@ -0,0 +1,96 @@
# Note: This feature requires pyannote.audio and a HuggingFace token.
# If these are not present, this module will likely fail or raise errors.
# Due to the complexity and weight of pyannote.audio, this is a placeholder
# for where the logic would sit. Implementing full diarization requires
# downloading models and handling complex segment merging.
import os
import sys
import tracker
import torch
def diarize_audio(audio_path, num_speakers=None, hf_token=None):
"""
Performs speaker diarization using pyannote.audio.
"""
try:
from pyannote.audio import Pipeline
except ImportError:
tracker.logger.error("Error: pyannote.audio not installed. Diarization skipped.")
return None
if not hf_token:
tracker.logger.error("Error: HuggingFace Token (HF_TOKEN) not found. Diarization skipped.")
return None
tracker.logger.info(f"Loading Diarization Pipeline (pyannote/speaker-diarization-3.1)...")
try:
# Fix for PyTorch 2.6+ weights_only issue
torch.serialization.add_safe_globals([torch.torch_version.TorchVersion])
# Note: 'use_auth_token' was deprecated in favor of 'token' in recent versions
pipeline = Pipeline.from_pretrained(
"pyannote/speaker-diarization-3.1",
token=hf_token
)
# Move to GPU if available
if torch.cuda.is_available():
pipeline.to(torch.device("cuda"))
tracker.logger.info(f"Diarizing {audio_path}...")
diarization = pipeline(audio_path, num_speakers=num_speakers)
results = []
for turn, _, speaker in diarization.itertracks(yield_label=True):
results.append({
"start": turn.start,
"end": turn.end,
"speaker": speaker
})
return results
except Exception as e:
tracker.logger.error(f"Error during diarization: {e}")
return None
def merge_diarization_with_transcript(transcript_segments, diarization_segments):
"""
Merges Whisper segments with Diarization speaker labels based on time overlap.
Args:
transcript_segments (list): Whisper segments [{'start': 0.0, 'end': 1.0, 'text': 'Hi'}, ...]
diarization_segments (list): Diarization segments [{'start': 0.1, 'end': 0.9, 'speaker': 'SPEAKER_00'}]
Returns:
list: Enhanced transcript segments with 'speaker' key.
"""
if not diarization_segments:
return transcript_segments
# Simple overlap matching logic
for t_seg in transcript_segments:
# Find diarization segment with max overlap
t_start = t_seg['start']
t_end = t_seg['end']
best_speaker = "Unknown"
max_overlap = 0
for d_seg in diarization_segments:
d_start = d_seg['start']
d_end = d_seg['end']
# Calculate intersection
overlap_start = max(t_start, d_start)
overlap_end = min(t_end, d_end)
overlap_duration = max(0, overlap_end - overlap_start)
if overlap_duration > max_overlap:
max_overlap = overlap_duration
best_speaker = d_seg['speaker']
t_seg['speaker'] = best_speaker
return transcript_segments
@@ -0,0 +1,90 @@
# How to Run Long Jobs in the Background (Detach & Disconnect)
Since transcribing and translating videos can take hours, you likely want to start the job and then disconnect your SSH session or turn off your local computer without stopping the process on the remote server.
Here are the two best ways to do this.
## Method 1: Using `tmux` (Recommended for Wizard)
**Best for:** Using the interactive Wizard (`run_wizard_v2.py`) and seeing the progress bar when you reconnect.
1. **Start a new session:**
```bash
tmux new -s transcriber
```
*(This opens a new terminal "window" managed by the server)*
2. **Run the Wizard:**
```bash
./ai_transcriber_v2/run_wizard_v2.py
```
Answer all the prompts as usual until the job starts.
3. **Detach (Leave it running):**
* Press **`Ctrl` + `B`** together, release them, then press **`d`**.
* You will return to your normal prompt and see `[detached]`.
4. **Disconnect:**
You can now safely close your SSH terminal or shut down your computer.
5. **Reconnect Later:**
Log back into the server and run:
```bash
tmux attach -t transcriber
```
You will see the progress bar exactly where you left it.
---
## Method 2: Using `nohup` (Command Line Only)
**Best for:** Running automated scripts non-interactively. You CANNOT use the wizard with this method because you can't answer the prompts.
1. **Run the command directly:**
Use `main.py` with all arguments provided upfront.
```bash
nohup python3 ai_transcriber_v2/main.py /path/to/videos \
--lang English \
--model medium \
--prefer-local \
--embed \
--cleanup \
> job_output.log 2>&1 &
```
2. **Verify it's running:**
```bash
ps aux | grep main.py
```
3. **Monitor Progress:**
Since you can't see the progress bar, check the log file:
```bash
tail -f job_output.log
```
---
## Method 3: Rescuing a Running Job (The `disown` Method)
**Use this if:** You already started the wizard or script in a normal terminal and now realize you need to disconnect without killing it.
1. **Pause the running job:**
Inside the terminal where the script is active, press **`Ctrl` + `Z`**.
* The script will pause, and you'll see: `[1]+ Stopped ...`
2. **Move it to the background:**
Type the following command and press Enter:
```bash
bg
```
* The script will resume running in the background.
3. **Disown the job:**
Tell the terminal to "forget" about the job so it doesn't kill it when you disconnect:
```bash
disown -h %1
```
*(Note: Use `%1` if your job number was `[1]`, `%2` if it was `[2]`, etc.)*
4. **Disconnect:**
You can now type `exit` or close your SSH window. The job will continue on the server.
**Warning:** Unlike `tmux`, you cannot "reattach" to see the progress bar again. You must monitor the logs in the `logs/` folder to check its status.
@@ -0,0 +1,31 @@
# Technical Overview
## 🏗 Architecture
The project is modularized into specialized Python scripts:
* **`main.py`**: The entry point. Manages the batch processing loop and job tracking.
* **`extractor.py`**: Media handling via FFmpeg. Responsible for audio extraction and subtitle embedding. Contains the "Flatpak Escape" logic.
* **`transcriber.py`**: Integration with `openai-whisper`. Manages GPU health checks and model loading.
* **`translator.py`**: The AI translation engine. Implements the 4-tier fallback logic (Gemini -> Google -> Ollama -> MyMemory).
* **`utils.py`**: Shared utilities for encoding detection (chardet), port checking, and service health monitoring.
* **`tracker.py`**: Persistence layer using SQLite/SQLAlchemy to track job status across runs.
## 🛡 Safety Mechanisms
1. **Duration Match Check**: Uses `ffprobe` to ensure the final subbed video length matches the original source.
2. **File Size Sanity**: Rejects any remux operation that results in a file < 80% of the original size (preventing video stream loss).
3. **SRT Health Check**: Compares the last timestamp of the translated SRT against the original transcript to detect partial/truncated translations.
4. **Encoding Detection**: Uses `chardet` to reliably read foreign subtitle files without manual configuration.
## 🐳 Flatpak / Sandbox Support
Since this project is designed for Bazzite/Atomic distros, all system-level calls (`ffmpeg`, `ffprobe`, `ollama`) are wrapped in a check that detects the presence of `/.flatpak-info`. If found, it automatically prefixes commands with `flatpak-spawn --host` to utilize system-installed binaries.
## 💾 Database Schema
The `job_history.db` tracks:
- `file_path`: Absolute path to source.
- `status`: PENDING, PROCESSING, COMPLETED, FAILED.
- `step_status`: Individual status for Extract, Transcribe, Translate, and Embed steps.
- `error_message`: Captured stack traces for failed jobs.
@@ -0,0 +1,51 @@
# User Guide: AI Video Transcriber & Translator V2
This tool automates the process of extracting audio from videos, transcribing it using OpenAI Whisper, translating the text via AI (Gemini, Llama3, or Google), and embedding the results back into the video as soft subtitles.
## 📋 Prerequisites
1. **FFmpeg**: Must be installed on your host system.
* On Bazzite: `brew install ffmpeg`
2. **Ollama (Optional but Recommended)**: For private, local translation.
* Run `./ai_transcriber_v2/install_local_llm.sh`
* Pull a model: `ollama pull llama3`
3. **Python Packages**:
```bash
pip install -r ai_transcriber_v2/requirements.txt
```
## 🚀 How to Run
### 1. The Easy Way (Wizard)
Perfect for first-time runs or single folders.
```bash
./ai_transcriber_v2/run_wizard_v2.py
```
Follow the interactive prompts to set your languages, model size, and preferences.
### 2. The Power Way (CLI)
For advanced automation.
```bash
python3 ai_transcriber_v2/main.py /path/to/videos --lang English --prefer-local --embed --cleanup
```
### 3. The Library Fixer (Recovery)
If you have a folder with existing transcripts or partial translations that need fixing:
```bash
./ai_transcriber_v2/recover_and_fix_v2.py /path/to/folder
```
This script intelligently scans for missing translations or translations that don't match the video duration.
## 🛠 Features
* **Prefer Local LLM**: Use `--prefer-local` to prioritize your laptop's GPU (via Ollama) for all translations.
* **Safety First**: The script will NEVER replace your original video if the new one is significantly smaller or has a different duration.
* **Diarization**: Use `--diarize` to identify different speakers (requires HuggingFace token).
* **Graceful Exit**: Press `Ctrl+C` once to stop the script. It will finish the current file and save its progress before closing.
## 📂 File Naming Convention
- `video.srt`: Original language transcript.
- `video.English.srt`: Gemini translated subtitles.
- `video.English.deep_translate.srt`: Google Translate subtitles.
- `video.English.local_llm.srt`: Ollama translated subtitles.
- `video.subbed.mp4`: The final result with embedded soft-subs.
@@ -0,0 +1,156 @@
import os
import subprocess
import sys
import json
import tracker
from utils import verify_file_not_empty
def run_ffmpeg(args):
"""
Runs ffmpeg, automatically escaping Flatpak sandbox if necessary.
"""
in_flatpak = os.path.exists("/.flatpak-info")
cmd = ["flatpak-spawn", "--host", "ffmpeg"] + args if in_flatpak else ["ffmpeg"] + args
try:
subprocess.run(cmd, check=True)
return True
except subprocess.CalledProcessError as e:
tracker.logger.error(f"FFmpeg Error: {e}")
return False
except FileNotFoundError:
tracker.logger.error("Error: 'ffmpeg' command not found. Please ensure it is installed on your host system.")
return False
def get_video_duration(file_path):
"""
Gets video duration in seconds using ffprobe.
"""
in_flatpak = os.path.exists("/.flatpak-info")
cmd_base = ["flatpak-spawn", "--host"] if in_flatpak else []
cmd = cmd_base + [
"ffprobe",
"-v", "error",
"-show_entries", "format=duration",
"-of", "default=noprint_wrappers=1:nokey=1",
file_path
]
try:
result = subprocess.run(cmd, capture_output=True, text=True, check=True)
return float(result.stdout.strip())
except Exception:
return 0.0
def extract_audio(video_path, output_path=None):
"""
Extracts audio from a video file using ffmpeg.
"""
if not os.path.exists(video_path):
raise FileNotFoundError(f"Video file not found: {video_path}")
if output_path is None:
base_name = os.path.splitext(video_path)[0]
output_path = f"{base_name}.wav"
if verify_file_not_empty(output_path):
tracker.logger.info(f"Audio file already exists: {output_path}")
return output_path
tracker.logger.info(f"Extracting audio from {video_path}...")
args = [
"-i", video_path,
"-ar", "16000",
"-ac", "1",
"-c:a", "pcm_s16le",
"-y",
"-v", "error",
output_path
]
if run_ffmpeg(args):
if not verify_file_not_empty(output_path):
raise Exception("FFmpeg succeeded but output is empty.")
tracker.logger.info(f"Audio extracted to: {output_path}")
return output_path
else:
sys.exit(1)
def embed_subtitles(video_path, srt_path, output_path=None):
"""
Embeds subtitles into the video file (Soft Subs) and sets them as primary.
Includes strict safety checks (Size & Duration) to prevent replacing videos with corrupted files.
"""
if not os.path.exists(video_path) or not os.path.exists(srt_path):
tracker.logger.error("Error: Video or SRT file not found for embedding.")
return False
# Capture original stats
original_size = os.path.getsize(video_path)
original_duration = get_video_duration(video_path)
if original_size == 0:
tracker.logger.error("Error: Source video is 0 bytes.")
return False
if output_path is None:
base, ext = os.path.splitext(video_path)
output_path = f"{base}.subbed{ext}"
tracker.logger.info(f"Embedding subtitles into: {output_path}...")
sub_codec = "mov_text" if video_path.lower().endswith(".mp4") else "srt"
args = [
"-ignore_editlist", "1",
"-i", video_path,
"-i", srt_path,
"-map", "0:v",
"-map", "0:a",
"-map", "1:0",
"-c", "copy",
"-c:s", sub_codec,
"-disposition:s:0", "default",
"-metadata:s:s:0", "language=eng",
"-metadata:s:s:0", "title=English (AI Translated)",
"-max_interleave_delta", "0",
"-avoid_negative_ts", "make_zero",
"-y",
"-v", "error",
output_path
]
if run_ffmpeg(args):
# --- Safety Checks ---
if not os.path.exists(output_path):
tracker.logger.error("Error: Output file was not created.")
return False
new_size = os.path.getsize(output_path)
new_duration = get_video_duration(output_path)
# 1. Zero Byte Check
if new_size == 0:
tracker.logger.error("❌ CRITICAL: Output file is 0 bytes. Deleting corrupted output.")
os.remove(output_path)
raise Exception("Embedding failed: Output is empty.")
# 2. Significant Size Drop Check
if new_size < (original_size * 0.8):
tracker.logger.error(f"❌ CRITICAL: Output file is significantly smaller than source!")
os.remove(output_path)
raise Exception("Embedding failed: Suspicious file size reduction.")
# 3. Duration Mismatch Check (New)
if abs(original_duration - new_duration) > 1.0:
tracker.logger.error(f"❌ CRITICAL: Duration mismatch detected!")
os.remove(output_path)
raise Exception("Embedding failed: Duration mismatch > 1s.")
tracker.logger.info(f"Subtitles embedded successfully: {output_path}")
return True
else:
tracker.logger.error("Error: Embedding failed.")
return False
+57
View File
@@ -0,0 +1,57 @@
#!/bin/bash
# install_local_llm.sh
# Installs Ollama and a translation-capable model on Linux (Bazzite/Fedora/Debian compatible)
set -e
echo "================================================="
echo " Local LLM Setup for AI Transcriber (Ollama)"
echo "================================================="
# 1. Check if Ollama is already installed
if command -v ollama &> /dev/null; then
echo "✅ Ollama is already installed."
else
echo "⬇️ Installing Ollama..."
# Standard Ollama install script (Works on Bazzite/Silverblue as /usr/local is writable)
curl -fsSL https://ollama.com/install.sh | sh
fi
# 2. Check GPU availability for Ollama
echo "-------------------------------------------------"
if command -v nvidia-smi &> /dev/null; then
echo "✅ Nvidia GPU detected. Ollama should run efficiently."
else
echo "⚠️ Nvidia GPU not found (or drivers missing)."
echo " Ollama will run on CPU, which might be slow for translation."
fi
echo "-------------------------------------------------"
# 3. Start Ollama Server (Background)
# In some dev containers, systemd isn't available, so we try to start it manually if not running.
if ! pgrep -x "ollama" > /dev/null; then
echo "🚀 Starting Ollama server in the background..."
nohup ollama serve > ollama.log 2>&1 &
PID=$!
echo " (PID: $PID) - Waiting 5 seconds for initialization..."
sleep 5
else
echo "✅ Ollama server is already running."
fi
# 4. Pull a Model
# 'llama3' (8B) is a great balance of speed and quality for translation.
# 'gemma:7b' is also good.
MODEL="llama3"
echo "⬇️ Pulling model: $MODEL (This may take a few minutes)..."
ollama pull $MODEL
echo "-------------------------------------------------"
echo "✅ Installation Complete!"
echo ""
echo "You can test it manually with: ollama run $MODEL 'Translate this to Spanish: Hello World'"
echo ""
echo "The AI Transcriber scripts will now detect and use this as a fallback."
echo "================================================="
Binary file not shown.
@@ -0,0 +1,226 @@
2026-01-12 11:10:13,189 - INFO - AFC is enabled with max remote calls: 10.
2026-01-12 11:10:13,323 - INFO - HTTP Request: POST https://generativelanguage.googleapis.com/v1beta/models/gemini-2.0-flash:generateContent "HTTP/1.1 429 Too Many Requests"
2026-01-12 11:11:36,758 - INFO - === Processing: /mnt/truenas_isolation/videos/castingcurvy/Videos/8a0885ce40eab5701744f9af5b6c667d935233640d1add5a2b5fce29683e0dc0.m4v ===
2026-01-12 11:40:21,805 - INFO - AFC is enabled with max remote calls: 10.
2026-01-12 11:40:21,954 - INFO - HTTP Request: POST https://generativelanguage.googleapis.com/v1beta/models/gemini-2.0-flash:generateContent "HTTP/1.1 429 Too Many Requests"
2026-01-12 11:40:31,462 - INFO - === Processing: /mnt/truenas_isolation/videos/castingcurvy/Videos/8a0885ce40eab5701744f9af5b6c667d935233640d1add5a2b5fce29683e0dc0.m4v ===
2026-01-12 11:42:26,576 - INFO - Running Speaker Diarization...
2026-01-12 11:42:28,049 - INFO - Failed to extract font properties from /usr/share/fonts/google-noto-serif-cjk-vf-fonts/NotoSerifCJK-VF.ttc: Can not load face (SFNT font table missing; error code 0x8e)
2026-01-12 11:42:28,055 - INFO - Failed to extract font properties from /usr/share/fonts/google-noto-sans-mono-cjk-vf-fonts/NotoSansMonoCJK-VF.ttc: Can not load face (SFNT font table missing; error code 0x8e)
2026-01-12 11:42:28,088 - INFO - Failed to extract font properties from /usr/share/fonts/abattis-cantarell-vf-fonts/Cantarell-VF.otf: Can not load face (SFNT font table missing; error code 0x8e)
2026-01-12 11:42:28,136 - INFO - Failed to extract font properties from /usr/share/fonts/twemoji/Twemoji.ttf: Can not load face (unknown file format; error code 0x2)
2026-01-12 11:42:28,171 - INFO - generated new fontManager
2026-01-12 11:42:30,653 - INFO - HTTP Request: HEAD https://huggingface.co/pyannote/speaker-diarization-3.1/resolve/main/config.yaml "HTTP/1.1 200 OK"
2026-01-12 11:42:30,686 - INFO - HTTP Request: GET https://huggingface.co/pyannote/speaker-diarization-3.1/resolve/main/config.yaml "HTTP/1.1 200 OK"
2026-01-12 11:42:31,248 - INFO - HTTP Request: HEAD https://huggingface.co/pyannote/segmentation-3.0/resolve/main/pytorch_model.bin "HTTP/1.1 302 Found"
2026-01-12 11:42:31,294 - INFO - HTTP Request: GET https://huggingface.co/api/models/pyannote/segmentation-3.0/xet-read-token/e66f3d3b9eb0873085418a7b813d3b369bf160bb "HTTP/1.1 200 OK"
2026-01-12 11:42:31,893 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,894 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,895 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,896 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,897 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,899 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,900 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,901 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,902 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,903 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,904 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,905 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,906 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,907 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,908 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,909 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,910 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,911 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,912 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,913 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,914 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,915 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,916 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,917 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,918 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,919 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,920 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,921 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,922 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,923 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,924 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,925 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,926 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,927 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,928 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,929 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,930 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,931 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,932 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,933 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,934 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,935 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,936 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,937 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,938 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,939 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,940 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,941 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,942 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,943 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,944 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,945 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,946 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,947 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,948 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,948 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,949 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,950 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,951 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,952 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,953 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,954 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,955 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,956 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,957 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,958 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,959 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,960 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,961 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,962 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,963 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,963 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,964 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,965 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,966 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,967 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,968 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,969 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,970 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,971 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,972 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,973 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,974 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,975 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,976 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,977 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,978 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,979 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,980 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,981 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,982 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,983 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,984 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,985 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,986 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,987 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,988 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,989 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,990 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,991 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,992 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,993 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,994 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,995 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,996 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,997 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,998 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:31,999 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,000 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,001 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,002 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,003 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,004 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,005 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,006 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,007 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,008 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,009 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,010 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,011 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,012 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,013 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,014 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,015 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,016 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,017 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,018 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,019 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,020 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,021 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,022 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,023 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,024 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,025 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,026 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,027 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,028 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,029 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,030 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,031 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,032 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,033 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,034 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,036 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,037 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,038 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,040 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,041 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,042 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,044 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,045 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,046 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,047 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,048 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,049 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,050 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,051 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,052 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,053 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,054 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,055 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,056 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,057 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,058 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,059 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,060 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,061 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,062 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,063 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,063 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,064 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,066 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,067 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,068 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,069 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,070 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,072 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,073 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,074 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,075 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,076 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,077 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,079 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,080 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,081 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,082 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,083 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,084 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,085 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,086 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,087 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,088 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,089 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,090 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,091 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,092 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,093 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,094 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,095 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,096 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,097 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,098 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,100 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,101 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,102 - INFO - HTTP Request: POST http://127.0.0.1:11434/api/chat "HTTP/1.1 404 Not Found"
2026-01-12 11:42:32,137 - INFO - Translation saved to: /mnt/truenas_isolation/videos/castingcurvy/Videos/8a0885ce40eab5701744f9af5b6c667d935233640d1add5a2b5fce29683e0dc0.English.local_llm.srt (Local LLM (Ollama))
2026-01-12 11:42:32,178 - INFO - Validation: Duration match verified (Diff=0.0s)
2026-01-12 11:42:32,185 - INFO - Translation Method: Local LLM (Ollama)
2026-01-12 11:42:32,517 - INFO - Cleanup: Removed /mnt/truenas_isolation/videos/castingcurvy/Videos/8a0885ce40eab5701744f9af5b6c667d935233640d1add5a2b5fce29683e0dc0.wav
2026-01-12 11:42:32,524 - INFO - === Processing: /mnt/truenas_isolation/videos/castingcurvy/Videos/9d013920eb78cd3f3a41cd0d83aa36ed4d47fabcdc6b81fc8fd77ddd0d287ca0.mp4 ===
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,368 @@
import argparse
import os
import sys
import socket
from dotenv import load_dotenv
# Load environment variables from central .env_files directory
script_dir = os.path.dirname(os.path.abspath(__file__))
env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe'))
if os.path.exists(env_path):
load_dotenv(env_path)
else:
# Fallback: check local .env
local_env = os.path.join(script_dir, '.env')
if os.path.exists(local_env):
load_dotenv(local_env)
else:
# Last resort: just try loading generic (cwd)
load_dotenv()
from extractor import extract_audio, embed_subtitles
from transcriber import transcribe_audio, save_as_srt, load_whisper_model
from translator import translate_with_auto_fallback
from utils import validate_and_repair_srt, check_srt_duration_match, GracefulKiller, ensure_ollama_running, check_service_availability, check_path_permissions, LANGUAGE_MAP
from diarizer import diarize_audio, merge_diarization_with_transcript
import tracker
from tracker import JobStatus
from tqdm import tqdm
def save_srt_with_speakers(segments, output_path):
"""Helper to save SRT with speaker labels prepended to text."""
def format_timestamp(seconds: float):
whole_seconds = int(seconds)
milliseconds = int((seconds - whole_seconds) * 1000)
hours = whole_seconds // 3600
minutes = (whole_seconds % 3600) // 60
seconds = whole_seconds % 60
return f"{hours:02d}:{minutes:02d}:{seconds:02d},{milliseconds:03d}"
with open(output_path, "w", encoding="utf-8") as f:
for i, segment in enumerate(segments, start=1):
start = format_timestamp(segment["start"])
end = format_timestamp(segment["end"])
text = segment["text"].strip()
speaker = segment.get("speaker", "")
if speaker and speaker != "Unknown":
text = f"[{speaker}]: {text}"
f.write(f"{i}\n")
f.write(f"{start} --> {end}\n")
f.write(f"{text}\n\n")
tracker.logger.info(f"SRT saved to: {output_path}")
def process_file(file_path, args, source_lang=None, loaded_model=None, service_status=None):
tracker.logger.info(f"=== Processing: {file_path} ===")
# Initialize Job
job = tracker.get_job(file_path)
if job.status == JobStatus.COMPLETED and not args.force:
tracker.logger.info("Job already completed. Skipping.")
return
tracker.update_job_status(file_path, JobStatus.PROCESSING)
try:
# 1. Extract Audio
tracker.update_step(file_path, "step_extract", "processing")
audio_path = extract_audio(file_path)
tracker.update_step(file_path, "step_extract", "done")
# 2. Transcribe (Generate SRT)
tracker.update_step(file_path, "step_transcribe", "processing")
transcript_file = os.path.splitext(file_path)[0] + ".srt"
transcript_exists = os.path.exists(transcript_file) and not args.force
final_srt_path = transcript_file
detected_iso = None
if transcript_exists:
tracker.logger.info(f"Transcript exists: {transcript_file}. Skipping transcription.")
with open(transcript_file, "r", encoding="utf-8") as f:
srt_content = f.read()
else:
# Use loaded_model if available
result = transcribe_audio(audio_path, model_size=args.model, language=source_lang, loaded_model=loaded_model)
segments = result["segments"]
detected_iso = result.get("language")
if args.diarize:
hf_token = args.hf_token or os.getenv("HF_TOKEN")
if hf_token:
tracker.logger.info("Running Speaker Diarization...")
diar_segments = diarize_audio(audio_path, hf_token=hf_token)
if diar_segments:
segments = merge_diarization_with_transcript(segments, diar_segments)
tracker.logger.info("Diarization merged into transcript.")
else:
tracker.logger.warning("Warning: --diarize requested but HF_TOKEN not provided. Skipping.")
if args.diarize:
save_srt_with_speakers(segments, transcript_file)
else:
save_as_srt(result, transcript_file)
validate_and_repair_srt(transcript_file)
with open(transcript_file, "r", encoding="utf-8") as f:
srt_content = f.read()
tracker.update_step(file_path, "step_transcribe", "done")
# 3. Translate
tracker.update_step(file_path, "step_translate", "processing")
base_translated = os.path.splitext(file_path)[0] + f".{args.lang}.srt"
deep_translated = os.path.splitext(file_path)[0] + f".{args.lang}.deep_translate.srt"
local_translated = os.path.splitext(file_path)[0] + f".{args.lang}.local_llm.srt"
# Determine output path logic
target_path_gemini = base_translated
target_path_deep = deep_translated
target_path_local = local_translated
translated_file = None
translation_success = False
method_used = "None"
# Check if source language matches target language
target_iso = LANGUAGE_MAP.get(args.lang)
if detected_iso and target_iso and detected_iso == target_iso:
tracker.logger.info(f"Source language '{detected_iso}' matches target '{target_iso}'. Skipping translation.")
final_srt_path = transcript_file
translation_success = True
method_used = "Source Match"
# Check existing (if not already handled by match)
elif (os.path.exists(base_translated) or os.path.exists(deep_translated) or os.path.exists(local_translated)) and not args.force:
if os.path.exists(local_translated):
translated_file = local_translated
method_used = "Local LLM (Existing)"
elif os.path.exists(deep_translated):
translated_file = deep_translated
method_used = "DeepTranslate (Existing)"
else:
translated_file = base_translated
method_used = "Gemini (Existing)"
tracker.logger.info(f"Translation exists: {translated_file} ({method_used}). Skipping translation.")
final_srt_path = translated_file
translation_success = True
else:
if srt_content:
res_content, method = translate_with_auto_fallback(
srt_content,
target_language=args.lang,
prefer_deep=args.prefer_deep,
prefer_local=args.prefer_local,
available_services=service_status
)
if res_content:
# Save based on method used
if "DeepTranslate" in method:
save_path = target_path_deep
elif "Local LLM" in method:
save_path = target_path_local
else:
save_path = target_path_gemini
with open(save_path, "w", encoding="utf-8") as f:
f.write(res_content)
tracker.logger.info(f"Translation saved to: {save_path} ({method})")
validate_and_repair_srt(save_path)
# Duration Check
is_valid_duration, msg = check_srt_duration_match(transcript_file, save_path)
if is_valid_duration:
tracker.logger.info(f"Validation: {msg}")
final_srt_path = save_path
translation_success = True
method_used = method
else:
tracker.logger.error(f"VALIDATION FAILED: {msg}")
tracker.logger.error("Marking translation as failed due to incomplete coverage.")
hostname = socket.gethostname()
redo_file = os.path.join(os.path.dirname(file_path), f"redo_queue_{hostname}.txt")
with open(redo_file, "a", encoding="utf-8") as rf:
rf.write(f"{file_path} | {msg}\n")
translation_success = False
else:
tracker.logger.error("TRANSLATION FAILED (All methods attempted).")
tracker.update_step(file_path, "step_translate", "failed")
translation_success = False
if translation_success:
tracker.update_step(file_path, "step_translate", "done")
tracker.logger.info(f"Translation Method: {method_used}")
# 4. Embed Subtitles
tracker.update_step(file_path, "step_embed", "processing")
should_embed = args.embed
if args.embed and not translation_success:
tracker.logger.warning("SAFETY HALT: Translation failed. Skipping embedding/deletion.")
should_embed = False
if should_embed:
success_embed = embed_subtitles(file_path, final_srt_path)
if success_embed and args.delete_source:
base, ext = os.path.splitext(file_path)
expected_output = f"{base}.subbed{ext}"
if os.path.exists(expected_output):
try:
os.remove(file_path)
tracker.logger.info(f"SOURCE DELETED: {file_path}")
except OSError as e:
tracker.logger.error(f"Error deleting source: {e}")
else:
tracker.logger.error(f"SAFETY ABORT: Output '{expected_output}' not found.")
tracker.update_step(file_path, "step_embed", "done")
# 5. Cleanup
if args.cleanup:
try:
os.remove(audio_path)
tracker.logger.info(f"Cleanup: Removed {audio_path}")
except OSError as e:
tracker.logger.warning(f"Warning: Could not remove audio: {e}")
# Mark Complete
if translation_success:
tracker.update_job_status(file_path, JobStatus.COMPLETED)
else:
tracker.update_job_status(file_path, JobStatus.FAILED, error="Translation failed")
except Exception as e:
tracker.logger.exception(f"Job Failed for {file_path}")
tracker.update_job_status(file_path, JobStatus.FAILED, error=str(e))
return
def main():
parser = argparse.ArgumentParser(description="AI Video Transcriber & Translator")
parser.add_argument("inputs", nargs='*', help="Path(s) to video file or directory")
parser.add_argument("--model", default="auto", choices=["auto", "tiny", "base", "small", "medium", "large"], help="Whisper model size (default: auto)")
parser.add_argument("--lang", default="English", help="Target language for translation (default: English)")
parser.add_argument("--source-lang", help="Source language of the audio (e.g., 'fr', 'es'). If omitted, you will be prompted.")
parser.add_argument("--force", action="store_true", help="Overwrite existing files")
parser.add_argument("--cleanup", action="store_true", help="Delete temporary .wav file")
parser.add_argument("--embed", action="store_true", help="Embed subtitles (Soft Subs)")
parser.add_argument("--diarize", action="store_true", help="Enable speaker diarization")
parser.add_argument("--hf-token", help="HuggingFace Token")
parser.add_argument("--delete-source", action="store_true", help="Delete original file after embedding")
parser.add_argument("--retry-failed", action="store_true", help="Retry FAILED jobs from DB")
parser.add_argument("--prefer-deep", action="store_true", help="Prefer DeepTranslate (Free) over Gemini")
parser.add_argument("--prefer-local", action="store_true", help="Prefer Local LLM (Ollama) over cloud APIs")
args = parser.parse_args()
if not os.getenv("GEMINI_API_KEY"):
print("Warning: GEMINI_API_KEY environment variable not set. Translation step will fail.")
source_lang = args.source_lang
if args.retry_failed:
print("Retrying failed jobs from database...")
failed_files = tracker.get_failed_jobs()
if not failed_files:
print("No failed jobs found.")
return
if not source_lang:
print("\n--- Audio Configuration ---")
user_input = input("Enter source language (e.g. 'French'). Enter for Auto: ").strip()
source_lang = user_input if user_input else None
# Load model for retries too
loaded_model = load_whisper_model(args.model)
service_status = check_service_availability()
for file_path in failed_files:
if os.path.exists(file_path):
process_file(file_path, args, source_lang, loaded_model=loaded_model, service_status=service_status)
else:
print(f"Skipping missing file: {file_path}")
return
if not args.inputs:
parser.print_help()
sys.exit(1)
if not source_lang:
print("\n--- Audio Configuration ---")
user_input = input("Enter source language (e.g. 'French'). Enter for Auto: ").strip()
source_lang = user_input if user_input else None
print(f"Selected: {source_lang if source_lang else 'Auto-detect'}")
# --- Ensure Ollama is Running ---
ensure_ollama_running()
# --------------------------------
# --- Check Service Health ---
service_status = check_service_availability()
# ----------------------------
# --- Check Path Permissions ---
valid_inputs = []
print("Checking Input Permissions...")
for inp in args.inputs:
ok, msg = check_path_permissions(inp)
print(msg)
if ok:
valid_inputs.append(inp)
if not valid_inputs:
print("\n❌ Error: No valid inputs with read/write permissions found. Exiting.")
return
# ------------------------------
# --- Load Model Once ---
loaded_model = load_whisper_model(args.model)
# -----------------------
# Initialize Graceful Exit Handler
killer = GracefulKiller()
# --- Collect All Files ---
all_files = []
video_extensions = ('.mp4', '.mkv', '.mov', '.avi', '.webm', '.flv', '.wmv', '.m4v')
for input_path in valid_inputs:
if os.path.isfile(input_path):
all_files.append(input_path)
elif os.path.isdir(input_path):
for root, dirs, files in os.walk(input_path):
for file in files:
if file.lower().endswith(video_extensions):
all_files.append(os.path.join(root, file))
if not all_files:
print("No video files found to process.")
return
# --- Batch Process with Progress Bar ---
pbar = tqdm(all_files, desc="Batch Progress", unit="file", dynamic_ncols=True, leave=True)
for file_path in pbar:
if killer.kill_now:
break
# Update progress bar description with current file
filename = os.path.basename(file_path)
pbar.set_description(f"File: {filename[:30]}")
process_file(file_path, args, source_lang, loaded_model=loaded_model, service_status=service_status)
if killer.kill_now:
print("\n🛑 Process stopped by user. Progress saved in database.")
else:
print("\n✅ All jobs finished.")
if __name__ == "__main__":
main()
+69
View File
@@ -0,0 +1,69 @@
#!/bin/bash
# Configuration
MOUNT_POINT="/mnt/truenas_isolation"
SHARE="//truenas.local/isolation"
echo "--- SMB Mount Tool (V2) ---"
# Determine privilege escalation method
PRIV_CMD=""
if [ "$EUID" -eq 0 ]; then
echo "Running as root."
else
if command -v sudo &> /dev/null; then
PRIV_CMD="sudo"
elif command -v flatpak-spawn &> /dev/null; then
echo "Detected Flatpak environment. Attempting to use host permissions via sudo..."
# We need to run sudo ON THE HOST.
# flatpak-spawn --host runs as the current user on the host.
# So we run 'sudo' inside that host shell.
PRIV_CMD="flatpak-spawn --host sudo"
# Note: This requires the flatpak to have permission to talk to the host
else
echo "❌ Error: This script requires root privileges to mount drives."
echo " 'sudo' was not found."
echo " Please run this script as root: su -c ./mount_truenas.sh"
exit 1
fi
fi
# 1. Create mount point if it doesn't exist
if [ ! -d "$MOUNT_POINT" ]; then
echo "Creating directory $MOUNT_POINT..."
# We try to create it. If it fails (e.g. inside read-only flatpak mount namespace), warn user.
$PRIV_CMD mkdir -p "$MOUNT_POINT"
if [ $? -ne 0 ]; then
echo "Error creating directory. If you are in a Flatpak, you might not have access to host /mnt."
exit 1
fi
fi
# 2. Get Credentials
read -p "Enter SMB Username [guest]: " SMB_USER
SMB_USER=${SMB_USER:-guest}
# 3. Mount
echo "Mounting $SHARE to $MOUNT_POINT..."
# IMPORTANT: We force the mount to be owned by the current user (UID 1000 usually)
# This fixes "Permission Denied" errors when writing to the share.
# We also set file_mode/dir_mode to 0777 as a fallback to ensure full access.
MOUNT_OPTS="vers=3.0,uid=$(id -u),gid=$(id -g),file_mode=0777,dir_mode=0777,noperm"
if [ "$SMB_USER" == "guest" ]; then
$PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o "guest,$MOUNT_OPTS"
else
# This will prompt for the SMB password
$PRIV_CMD mount -t cifs "$SHARE" "$MOUNT_POINT" -o "username=$SMB_USER,$MOUNT_OPTS"
fi
# 4. Check result
if [ $? -eq 0 ]; then
echo "✅ Success! Share is now available at $MOUNT_POINT"
echo "Files are now owned by $(id -un):$(id -gn) with full write access."
echo "The mapping will disappear automatically after you reboot."
else
echo "❌ Error: Failed to mount the share."
echo "Ensure 'cifs-utils' is installed and the server is reachable."
fi
File diff suppressed because it is too large Load Diff
+256
View File
@@ -0,0 +1,256 @@
#!/usr/bin/env python3
import os
import sys
import argparse
import subprocess
import socket
from dotenv import load_dotenv
from datetime import datetime
import pysubs2
# Load config
script_dir = os.path.dirname(os.path.abspath(__file__))
env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe'))
if os.path.exists(env_path):
load_dotenv(env_path)
else:
load_dotenv()
from translator import translate_with_auto_fallback
from utils import validate_and_repair_srt, check_srt_duration_match, GracefulKiller, ensure_ollama_running, detect_file_encoding, check_service_availability, check_path_permissions
from extractor import embed_subtitles
def process_recovery(folder_path, target_lang="English", prefer_deep=False, prefer_local=False):
print(f"Scanning {folder_path} for incomplete translations (V2)...")
# Check Permissions
perm_ok, perm_msg = check_path_permissions(folder_path)
print(perm_msg)
if not perm_ok:
print("Aborting due to permission errors.")
return
if prefer_local:
print("Preference: Local LLM (Ollama) > Gemini/Deep")
elif prefer_deep:
print("Preference: DeepTranslate (Google Translate Free) > Gemini")
else:
print("Preference: Gemini (API) > DeepTranslate")
hostname = socket.gethostname()
recovery_log_file = os.path.join(folder_path, f"recovery_status_{hostname}.log")
print(f"Logging actions to: {recovery_log_file}")
# Ensure Ollama is ready
ensure_ollama_running()
# Check Service Health
service_status = check_service_availability()
# Initialize Graceful Exit
killer = GracefulKiller()
count_fixed = 0
video_extensions = ('.mp4', '.mkv', '.mov', '.avi')
for root, dirs, files in os.walk(folder_path):
if killer.kill_now:
break
for file in files:
if killer.kill_now:
break
if file.endswith(".srt") and \
not file.endswith(f".{target_lang}.srt") and \
not file.endswith(f".{target_lang}.deep_translate.srt") and \
not file.endswith(f".{target_lang}.local_llm.srt"):
source_srt_path = os.path.join(root, file)
base_name = os.path.splitext(file)[0]
path_gemini = os.path.join(root, f"{base_name}.{target_lang}.srt")
path_deep = os.path.join(root, f"{base_name}.{target_lang}.deep_translate.srt")
path_local = os.path.join(root, f"{base_name}.{target_lang}.local_llm.srt")
needs_translation = False
existing_translation_path = None
# Check if translation exists
if os.path.exists(path_gemini):
existing_translation_path = path_gemini
elif os.path.exists(path_deep):
existing_translation_path = path_deep
elif os.path.exists(path_local):
existing_translation_path = path_local
if existing_translation_path:
# Validate duration
is_valid, msg = check_srt_duration_match(source_srt_path, existing_translation_path)
if not is_valid:
print(f"\n⚠️ Found partial/broken translation: {existing_translation_path}")
print(f" Reason: {msg}")
print(" -> Queueing for re-translation...")
needs_translation = True
else:
# Missing translation
print(f"\nFound untranslated transcript: {file}")
needs_translation = True
if not needs_translation:
continue
# --- Proceed with Translation ---
content = None
# 1. Try automatic detection
detected_enc = detect_file_encoding(source_srt_path)
try:
with open(source_srt_path, "r", encoding=detected_enc) as f:
content = f.read()
except Exception:
# 2. Fallback to brute force if chardet was wrong
encodings_to_try = ['utf-8', 'shift_jis', 'euc_jp', 'latin-1', 'cp1252', 'utf-16']
for enc in encodings_to_try:
try:
with open(source_srt_path, "r", encoding=enc) as f:
content = f.read()
break # Success
except UnicodeDecodeError:
continue
if content is None:
print(f"❌ Error: Could not decode {file}. Skipping.")
continue
# Use shared translation logic
res_content, method_used = translate_with_auto_fallback(
content,
target_language=target_lang,
prefer_deep=prefer_deep,
prefer_local=prefer_local,
available_services=service_status
)
final_srt_path = None
# --- Helper to save result ---
def save_translation(text, method):
path = None
if "DeepTranslate" in method:
path = path_deep
elif "Local LLM" in method:
path = path_local
else:
path = path_gemini
with open(path, "w", encoding="utf-8") as f:
f.write(text)
return path
if res_content:
final_srt_path = save_translation(res_content, method_used)
if final_srt_path:
# Validate the NEW translation immediately
is_valid_new, msg_new = check_srt_duration_match(source_srt_path, final_srt_path)
if not is_valid_new:
print(f"❌ New translation ({method_used}) failed validation: {msg_new}")
# --- RETRY WITH LOCAL LLM ---
# Only retry if we haven't already used Local LLM and it is available
if "Local LLM" not in method_used and service_status.get("Ollama", False):
print(" -> Retrying with Local LLM (Ollama) as fallback strategy...")
# Force try Ollama
from translator import translate_via_ollama
retry_content = translate_via_ollama(content, target_language=target_lang)
if retry_content:
retry_path = path_local
with open(retry_path, "w", encoding="utf-8") as f:
f.write(retry_content)
# Validate Retry
valid_retry, msg_retry = check_srt_duration_match(source_srt_path, retry_path)
if valid_retry:
print(f" ✅ Local LLM Retry Succeeded! Using: {os.path.basename(retry_path)}")
# Rename/Cleanup the previous failed attempt
invalid_path = final_srt_path + ".invalid"
os.replace(final_srt_path, invalid_path)
final_srt_path = retry_path
method_used = "Local LLM (Retry)"
is_valid_new = True # Mark as valid so we proceed to embedding
else:
print(f" ❌ Local LLM Retry also failed validation: {msg_retry}")
# Cleanup retry attempt
os.replace(retry_path, retry_path + ".invalid")
if not is_valid_new:
# Rename the invalid file so it doesn't sit there as a "fake" good translation
invalid_path = final_srt_path + ".invalid"
if os.path.exists(final_srt_path):
os.replace(final_srt_path, invalid_path)
print(f" -> Moved failed attempt to: {os.path.basename(invalid_path)}")
with open(recovery_log_file, "a", encoding="utf-8") as log:
log.write(f"{datetime.now().isoformat()} | {method_used} | FAILED_VALIDATION | {file}\n")
continue
# Log result
with open(recovery_log_file, "a", encoding="utf-8") as log:
log.write(f"{datetime.now().isoformat()} | {method_used} | FIXED | {file} -> {os.path.basename(final_srt_path)}\n")
validate_and_repair_srt(final_srt_path)
video_candidates = [
os.path.join(root, base_name + ".mp4"),
os.path.join(root, base_name + ".mkv"),
os.path.join(root, base_name + ".subbed.mp4"),
]
found_video = None
for v in video_candidates:
if os.path.exists(v):
found_video = v
break
if found_video:
print(f"Found video to fix: {found_video}")
temp_video_out = found_video + ".temp_fix.mp4"
try:
embed_subtitles(found_video, final_srt_path, output_path=temp_video_out)
os.replace(temp_video_out, found_video)
print(f"✅ Fixed: {found_video}")
count_fixed += 1
except Exception as e:
print(f"Error re-embedding: {e}")
if os.path.exists(temp_video_out):
os.remove(temp_video_out)
else:
print("Warning: Could not find a corresponding video file to embed into.")
else:
print("❌ All translation methods failed. Skipping.")
if killer.kill_now:
print("\n🛑 Recovery process stopped by user.")
else:
print(f"\nRecovery Complete. Fixed {count_fixed} files.")
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Recover and Fix Translations (V2)")
parser.add_argument("folders", nargs='+', help="One or more paths to folders to scan")
parser.add_argument("--lang", default="English", help="Target language (default: English)")
parser.add_argument("--prefer-deep", action="store_true", help="Prefer DeepTranslate (Free) over Gemini API")
parser.add_argument("--prefer-local", action="store_true", help="Prefer Local LLM (Ollama) over cloud APIs")
args = parser.parse_args()
for folder in args.folders:
if os.path.exists(folder):
process_recovery(folder, args.lang, args.prefer_deep, args.prefer_local)
else:
print(f"Error: Folder '{folder}' does not exist. Skipping.")
@@ -0,0 +1,13 @@
openai-whisper
google-genai
ffmpeg-python
torch
numpy
tenacity
pysubs2
pyannote.audio
deep-translator
ollama
chardet
tqdm
python-dotenv
@@ -0,0 +1,35 @@
@echo off
TITLE AI Video Transcriber V2
REM Check if Python is installed
python --version >nul 2>&1
IF %ERRORLEVEL% NEQ 0 (
echo Error: Python is not installed or not in your PATH.
echo Please install Python from https://www.python.org/
PAUSE
EXIT /B
)
REM Check if FFmpeg is installed
ffmpeg -version >nul 2>&1
IF %ERRORLEVEL% NEQ 0 (
echo Error: FFmpeg is not installed or not in your PATH.
echo Please download FFmpeg and add bin folder to your Environment Variables.
PAUSE
EXIT /B
)
REM Check for virtual environment
IF NOT EXIST ".venv" (
echo Creating Virtual Environment...
python -m venv .venv
echo Installing dependencies (this may take a while)...
.venv\Scripts\pip install -r ai_transcriber_v2\requirements.txt
.venv\Scripts\pip install python-dotenv
)
REM Run the V2 launcher
echo Starting Wizard...
.venv\Scripts\python run_v2.py
PAUSE
+184
View File
@@ -0,0 +1,184 @@
#!/usr/bin/env python3
import os
import sys
import subprocess
import shutil
def clear_screen():
os.system('cls' if os.name == 'nt' else 'clear')
def get_input(prompt, default=None):
"""Helper to get input with a default value."""
if default:
user_input = input(f"{prompt} [{default}]: ").strip()
return user_input if user_input else default
else:
return input(f"{prompt}: ").strip()
def get_yes_no(prompt, default="y"):
"""Helper to get boolean input."""
display_default = "Y/n" if default.lower() in ["y", "yes"] else "y/N"
choice = get_input(f"{prompt} ({display_default})", default).lower()
return choice in ["y", "yes", "true", "1"]
def print_header():
print("==========================================")
print(" AI Video Transcriber & Translator V2")
print(" (Powered by Google GenAI SDK)")
print("==========================================")
print("")
def main():
clear_screen()
print_header()
# Try to load the .env file so the wizard knows what's already configured
try:
from dotenv import load_dotenv
script_dir = os.path.dirname(os.path.abspath(__file__))
env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe'))
if os.path.exists(env_path):
load_dotenv(env_path)
except ImportError:
pass
# 1. Input File/Folder (Multiple)
input_paths = []
while True:
prompt_text = "Enter a path to a video file or folder"
if input_paths:
prompt_text += " (or press Enter to finish)"
input_path = get_input(prompt_text)
if not input_path:
if input_paths:
break
else:
print("Error: You must provide at least one path.")
continue
# Clean up input
input_path = input_path.strip('"\'')
input_path = input_path.replace(r'\ ', ' ')
# Expand user (~) and resolve absolute path
input_path = os.path.abspath(os.path.expanduser(input_path))
if os.path.exists(input_path):
input_paths.append(input_path)
print(f"Added: {input_path}")
else:
print(f"Error: Path '{input_path}' does not exist. Please try again.\n")
print("\nSelected Inputs:")
for p in input_paths:
print(f" - {p}")
print("")
# 2. Languages
source_lang = get_input("Source Language (e.g., French, es)", default="auto")
target_lang = get_input("Target Language for translation", default="English")
print("")
# 3. Model Size
print("Model Size Options: tiny, base, small, medium, large, auto")
model_size = get_input("Whisper Model Size", default="auto")
print("")
# 4. Features
do_cleanup = get_yes_no("Cleanup temporary audio files after processing?", default="y")
do_embed = get_yes_no("Embed subtitles into the video file (Soft Subs)?", default="y")
do_diarize = get_yes_no("Enable Speaker Diarization (Identify speakers)?", default="n")
do_delete_source = False
if do_embed:
print("\n⚠️ WARNING: Using this next option will PERMANENTLY DELETE the original video files.")
print(" It will only run if the new subtitled video is successfully created.")
do_delete_source = get_yes_no("Delete original source files after embedding?", default="n")
do_prefer_deep = get_yes_no("Prefer DeepTranslate (Free) over Gemini API?", default="n")
do_prefer_local = get_yes_no("Prefer Local LLM (Ollama) over all cloud options?", default="n")
hf_token = None
if do_diarize:
if not os.getenv("HF_TOKEN"):
print("\nSpeaker Diarization requires a HuggingFace Token.")
hf_token = get_input("Enter your HuggingFace Token (hidden)", default="")
else:
print("Using HF_TOKEN from environment.")
# 5. Build Command
# Point to v2 main script
script_dir = os.path.dirname(os.path.abspath(__file__))
main_script = os.path.join(script_dir, "main.py")
cmd = [sys.executable, main_script]
cmd.extend(input_paths)
cmd.extend(["--lang", target_lang])
cmd.extend(["--model", model_size])
if source_lang != "auto":
cmd.extend(["--source-lang", source_lang])
if do_cleanup:
cmd.append("--cleanup")
if do_embed:
cmd.append("--embed")
if do_delete_source:
cmd.append("--delete-source")
if do_diarize:
cmd.append("--diarize")
if hf_token:
cmd.extend(["--hf-token", hf_token])
if do_prefer_deep:
cmd.append("--prefer-deep")
if do_prefer_local:
cmd.append("--prefer-local")
# 6. Confirmation and Execution
clear_screen()
print_header()
print("Configuration Complete!")
print("-" * 30)
print("Inputs:")
for p in input_paths:
print(f" - {p}")
print(f"Source Lang: {source_lang}")
print(f"Target Lang: {target_lang}")
print(f"Model: {model_size}")
print(f"Cleanup: {do_cleanup}")
print(f"Embed Subs: {do_embed}")
print(f"Delete Src: {do_delete_source}")
print(f"Diarization: {do_diarize}")
print(f"Prefer Deep: {do_prefer_deep}")
print(f"Prefer Local: {do_prefer_local}")
print("-" * 30)
if not get_yes_no("Run this job now?", default="y"):
print("Aborted.")
sys.exit(0)
print("\nStarting Job (V2)...")
try:
# Pass environment variables including HF_TOKEN if set
env = os.environ.copy()
if hf_token:
env["HF_TOKEN"] = hf_token
subprocess.run(cmd, check=True, env=env)
print("\n✅ Job Complete!")
except subprocess.CalledProcessError as e:
print(f"\n❌ Job Failed with error code {e.returncode}")
except KeyboardInterrupt:
print("\nJob interrupted by user.")
if __name__ == "__main__":
main()
+184
View File
@@ -0,0 +1,184 @@
#!/usr/bin/env python3
import os
import sys
import subprocess
import shutil
def clear_screen():
os.system('cls' if os.name == 'nt' else 'clear')
def get_input(prompt, default=None):
"""Helper to get input with a default value."""
if default:
user_input = input(f"{prompt} [{default}]: ").strip()
return user_input if user_input else default
else:
return input(f"{prompt}: ").strip()
def get_yes_no(prompt, default="y"):
"""Helper to get boolean input."""
display_default = "Y/n" if default.lower() in ["y", "yes"] else "y/N"
choice = get_input(f"{prompt} ({display_default})", default).lower()
return choice in ["y", "yes", "true", "1"]
def print_header():
print("==========================================")
print(" AI Video Transcriber & Translator V2")
print(" (Powered by Google GenAI SDK)")
print("==========================================")
print("")
def main():
clear_screen()
print_header()
# Try to load the .env file so the wizard knows what's already configured
try:
from dotenv import load_dotenv
script_dir = os.path.dirname(os.path.abspath(__file__))
env_path = os.path.abspath(os.path.join(script_dir, '../../.env_files/.env.aitranscribe'))
if os.path.exists(env_path):
load_dotenv(env_path)
except ImportError:
pass
# 1. Input File/Folder (Multiple)
input_paths = []
while True:
prompt_text = "Enter a path to a video file or folder"
if input_paths:
prompt_text += " (or press Enter to finish)"
input_path = get_input(prompt_text)
if not input_path:
if input_paths:
break
else:
print("Error: You must provide at least one path.")
continue
# Clean up input
input_path = input_path.strip('"\'')
input_path = input_path.replace(r'\ ', ' ')
# Expand user (~) and resolve absolute path
input_path = os.path.abspath(os.path.expanduser(input_path))
if os.path.exists(input_path):
input_paths.append(input_path)
print(f"Added: {input_path}")
else:
print(f"Error: Path '{input_path}' does not exist. Please try again.\n")
print("\nSelected Inputs:")
for p in input_paths:
print(f" - {p}")
print("")
# 2. Languages
source_lang = get_input("Source Language (e.g., French, es)", default="auto")
target_lang = get_input("Target Language for translation", default="English")
print("")
# 3. Model Size
print("Model Size Options: tiny, base, small, medium, large, auto")
model_size = get_input("Whisper Model Size", default="auto")
print("")
# 4. Features
do_cleanup = get_yes_no("Cleanup temporary audio files after processing?", default="y")
do_embed = get_yes_no("Embed subtitles into the video file (Soft Subs)?", default="y")
do_diarize = get_yes_no("Enable Speaker Diarization (Identify speakers)?", default="n")
do_delete_source = False
if do_embed:
print("\n⚠️ WARNING: Using this next option will PERMANENTLY DELETE the original video files.")
print(" It will only run if the new subtitled video is successfully created.")
do_delete_source = get_yes_no("Delete original source files after embedding?", default="n")
do_prefer_deep = get_yes_no("Prefer DeepTranslate (Free) over Gemini API?", default="n")
do_prefer_local = get_yes_no("Prefer Local LLM (Ollama) over all cloud options?", default="n")
hf_token = None
if do_diarize:
if not os.getenv("HF_TOKEN"):
print("\nSpeaker Diarization requires a HuggingFace Token.")
hf_token = get_input("Enter your HuggingFace Token (hidden)", default="")
else:
print("Using HF_TOKEN from environment.")
# 5. Build Command
# Point to v2 main script
script_dir = os.path.dirname(os.path.abspath(__file__))
main_script = os.path.join(script_dir, "main.py")
cmd = [sys.executable, main_script]
cmd.extend(input_paths)
cmd.extend(["--lang", target_lang])
cmd.extend(["--model", model_size])
if source_lang != "auto":
cmd.extend(["--source-lang", source_lang])
if do_cleanup:
cmd.append("--cleanup")
if do_embed:
cmd.append("--embed")
if do_delete_source:
cmd.append("--delete-source")
if do_diarize:
cmd.append("--diarize")
if hf_token:
cmd.extend(["--hf-token", hf_token])
if do_prefer_deep:
cmd.append("--prefer-deep")
if do_prefer_local:
cmd.append("--prefer-local")
# 6. Confirmation and Execution
clear_screen()
print_header()
print("Configuration Complete!")
print("-" * 30)
print("Inputs:")
for p in input_paths:
print(f" - {p}")
print(f"Source Lang: {source_lang}")
print(f"Target Lang: {target_lang}")
print(f"Model: {model_size}")
print(f"Cleanup: {do_cleanup}")
print(f"Embed Subs: {do_embed}")
print(f"Delete Src: {do_delete_source}")
print(f"Diarization: {do_diarize}")
print(f"Prefer Deep: {do_prefer_deep}")
print(f"Prefer Local: {do_prefer_local}")
print("-" * 30)
if not get_yes_no("Run this job now?", default="y"):
print("Aborted.")
sys.exit(0)
print("\nStarting Job (V2)...")
try:
# Pass environment variables including HF_TOKEN if set
env = os.environ.copy()
if hf_token:
env["HF_TOKEN"] = hf_token
subprocess.run(cmd, check=True, env=env)
print("\n✅ Job Complete!")
except subprocess.CalledProcessError as e:
print(f"\n❌ Job Failed with error code {e.returncode}")
except KeyboardInterrupt:
print("\nJob interrupted by user.")
if __name__ == "__main__":
main()
@@ -0,0 +1 @@
123
@@ -0,0 +1,101 @@
import logging
import os
import socket
from datetime import datetime
from sqlalchemy import create_engine, Column, Integer, String, DateTime, Enum, Text
from sqlalchemy.orm import declarative_base, sessionmaker
import enum
# Get Hostname for namespacing
HOSTNAME = socket.gethostname()
# Setup Logging
log_dir = "logs"
os.makedirs(log_dir, exist_ok=True)
log_file = os.path.join(log_dir, f"transcriber_{HOSTNAME}_{datetime.now().strftime('%Y%m%d')}.log")
# Create formatters
log_formatter = logging.Formatter('%(asctime)s - %(levelname)s - %(message)s')
# File Handler (Full Detail)
file_handler = logging.FileHandler(log_file)
file_handler.setFormatter(log_formatter)
file_handler.setLevel(logging.INFO)
# Stream Handler (Quiet Detail for Terminal)
stream_handler = logging.StreamHandler()
stream_handler.setFormatter(log_formatter)
stream_handler.setLevel(logging.WARNING) # Only warnings/errors to terminal
logging.basicConfig(
level=logging.INFO,
handlers=[file_handler, stream_handler]
)
logger = logging.getLogger(__name__)
# Database Setup
Base = declarative_base()
DB_FILE = f"job_history_{HOSTNAME}.db"
class JobStatus(enum.Enum):
PENDING = "pending"
PROCESSING = "processing"
COMPLETED = "completed"
FAILED = "failed"
class Job(Base):
__tablename__ = 'jobs'
id = Column(Integer, primary_key=True)
file_path = Column(String, unique=True, nullable=False)
status = Column(Enum(JobStatus), default=JobStatus.PENDING)
error_message = Column(Text, nullable=True)
last_updated = Column(DateTime, default=datetime.utcnow, onupdate=datetime.utcnow)
# Track progress of individual steps
step_extract = Column(String, default="pending") # pending, done, failed
step_transcribe = Column(String, default="pending")
step_translate = Column(String, default="pending")
step_embed = Column(String, default="pending")
engine = create_engine(f'sqlite:///{DB_FILE}')
Base.metadata.create_all(engine)
Session = sessionmaker(bind=engine)
def get_job(file_path):
session = Session()
job = session.query(Job).filter_by(file_path=file_path).first()
if not job:
job = Job(file_path=file_path)
session.add(job)
session.commit()
session.refresh(job) # Ensure we have the ID and defaults
# Detach from session so we can use it after session.close()
session.expunge(job)
session.close()
return job
def update_job_status(file_path, status, error=None):
session = Session()
job = session.query(Job).filter_by(file_path=file_path).first()
if job:
job.status = status
if error:
job.error_message = str(error)
session.commit()
session.close()
def update_step(file_path, step_name, status):
session = Session()
job = session.query(Job).filter_by(file_path=file_path).first()
if job:
setattr(job, step_name, status)
session.commit()
session.close()
def get_failed_jobs():
session = Session()
jobs = session.query(Job).filter_by(status=JobStatus.FAILED).all()
session.close()
return [j.file_path for j in jobs]
@@ -0,0 +1,170 @@
import whisper
import os
import sys
import subprocess
import torch
import tracker
def check_gpu_health():
"""
Performs a robust check for GPU availability and prints detailed troubleshooting
info if issues are detected, specific to Bazzite/VS Code environments.
"""
tracker.logger.info("Checking GPU health...")
# 1. Check if the OS/Driver sees the GPU
nvidia_smi_ok = False
try:
in_flatpak = os.path.exists("/.flatpak-info")
cmd = ["flatpak-spawn", "--host", "nvidia-smi"] if in_flatpak else ["nvidia-smi"]
subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True)
nvidia_smi_ok = True
except (subprocess.CalledProcessError, FileNotFoundError):
nvidia_smi_ok = False
# 2. Check if PyTorch sees the GPU
torch_cuda_ok = torch.cuda.is_available()
if torch_cuda_ok:
tracker.logger.info(f"✅ GPU is accessible: {torch.cuda.get_device_name(0)}")
tracker.logger.info(f" CUDA Version: {torch.version.cuda}")
return True
# --- Troubleshooting Block ---
tracker.logger.warning("\n⚠️ WARNING: GPU not detected by PyTorch. Falling back to CPU.")
tracker.logger.warning(" Transcription will be significantly slower.\n")
tracker.logger.info("--- Diagnostic Report ---")
if nvidia_smi_ok:
tracker.logger.info("1. [OK] 'nvidia-smi' command works. The system driver is installed and visible.")
tracker.logger.info("2. [FAIL] PyTorch cannot see the GPU.")
tracker.logger.info(" -> Likely Cause: You might have installed the CPU-only version of PyTorch.")
tracker.logger.info(" -> Solution: Reinstall PyTorch with CUDA support:")
tracker.logger.info(" pip uninstall torch torchvision torchaudio")
tracker.logger.info(" pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cu118")
else:
tracker.logger.info("1. [FAIL] 'nvidia-smi' command failed or not found.")
tracker.logger.info(" -> Likely Cause: Nvidia drivers are missing, or the container/sandbox cannot access the GPU.")
tracker.logger.info("\n --- Bazzite / VS Code / Container Specific Checks ---")
tracker.logger.info(" a. If you are running inside a dev container (DevBox/Distrobox/Toolbox):")
tracker.logger.info(" Ensure the container was created with nvidia support.")
tracker.logger.info(" (Bazzite usually handles this for 'distrobox', but check your config).")
tracker.logger.info(" b. If you are using VS Code Flatpak:")
tracker.logger.info(" Flatpak might be restricting access. Check Flatseal permissions for VS Code.")
tracker.logger.info(" c. Driver Check:")
tracker.logger.info(" Run 'rpm -qa | grep nvidia' in your host terminal to verify drivers are installed.")
tracker.logger.info("-------------------------\n")
return False
def get_vram_gb():
"""Returns the total VRAM in GB of the first CUDA device, or 0 if no CUDA."""
if not torch.cuda.is_available():
return 0
try:
# returns bytes
total_mem = torch.cuda.get_device_properties(0).total_memory
return total_mem / (1024 ** 3)
except Exception:
return 0
def get_optimal_model_size():
"""
Determines the best Whisper model based on available VRAM.
Rough estimates for VRAM usage (fp16):
- large: ~10 GB
- medium: ~5 GB
- small: ~2 GB
- base: ~1 GB
- tiny: ~1 GB
"""
vram = get_vram_gb()
if vram == 0:
return "base"
tracker.logger.info(f"Detected GPU with {vram:.2f} GB VRAM.")
if vram >= 11:
return "large"
elif vram >= 6:
return "medium"
elif vram >= 3:
return "small"
else:
return "base"
def format_timestamp(seconds: float):
"""Converts seconds to SRT timestamp format (HH:MM:SS,mmm)."""
whole_seconds = int(seconds)
milliseconds = int((seconds - whole_seconds) * 1000)
hours = whole_seconds // 3600
minutes = (whole_seconds % 3600) // 60
seconds = whole_seconds % 60
return f"{hours:02d}:{minutes:02d}:{seconds:02d},{milliseconds:03d}"
def save_as_srt(result, output_path):
"""Saves the Whisper transcription result as an SRT file."""
with open(output_path, "w", encoding="utf-8") as f:
for i, segment in enumerate(result["segments"], start=1):
start = format_timestamp(segment["start"])
end = format_timestamp(segment["end"])
text = segment["text"].strip()
f.write(f"{i}\n")
f.write(f"{start} --> {end}\n")
f.write(f"{text}\n\n")
tracker.logger.info(f"SRT saved to: {output_path}")
def load_whisper_model(model_size="auto"):
"""
Loads and returns the Whisper model.
"""
check_gpu_health()
if model_size == "auto":
model_size = get_optimal_model_size()
tracker.logger.info(f"Auto-selected model: '{model_size}'")
tracker.logger.info(f"Loading Whisper model ('{model_size}')...")
device = "cuda" if torch.cuda.is_available() else "cpu"
tracker.logger.info(f"Using device: {device}")
try:
model = whisper.load_model(model_size, device=device)
return model
except Exception as e:
tracker.logger.error(f"Error loading model: {e}")
sys.exit(1)
def transcribe_audio(audio_path, model_size="auto", language=None, loaded_model=None):
"""
Transcribes an audio file using OpenAI's Whisper model.
Args:
audio_path (str): Path to the input audio file.
model_size (str): Size of the Whisper model to use. If "auto", selects based on VRAM.
language (str, optional): Language code (e.g., "en", "fr", "es"). If None, auto-detects.
loaded_model (object, optional): Pre-loaded Whisper model object.
Returns:
dict: The full transcription result containing segments and text.
"""
if not os.path.exists(audio_path):
raise FileNotFoundError(f"Audio file not found: {audio_path}")
model = loaded_model
if model is None:
model = load_whisper_model(model_size)
tracker.logger.info(f"Transcribing {audio_path}...")
try:
# Disable verbose to prevent line-by-line output
result = model.transcribe(audio_path, language=language, verbose=False)
tracker.logger.info("Transcription complete.")
return result
except Exception as e:
tracker.logger.error(f"Error during transcription: {e}")
sys.exit(1)
@@ -0,0 +1,345 @@
import os
import sys
import time
from datetime import datetime
from google import genai
from google.genai import types
from tenacity import retry, stop_after_attempt, wait_exponential, retry_if_exception_type
import pysubs2
from deep_translator import GoogleTranslator, MyMemoryTranslator
import ollama
from tqdm import tqdm
from utils import LANGUAGE_MAP
# Define a retry decorator
# ... (retry_policy remains)
def translate_via_ollama(source_srt_content, target_language="English", model="dolphin-llama3"):
"""
Translates SRT content using a local Ollama model (Line-by-Line for progress).
Includes retries and debug logging.
"""
debug_log_path = "ollama_debug.log"
try:
subs = pysubs2.SSAFile.from_string(source_srt_content)
# Using tqdm for progress bar
# dynamic_ncols=True helps it resize properly.
for line in tqdm(subs, desc=" Ollama Progress", unit="line", dynamic_ncols=True, leave=False):
text = line.text.strip()
# Skip empty, numeric-only, or extremely short non-word text
if not text or text.isdigit() or len(text) < 2:
continue
prompt = (
f"Translate this subtitle text to {target_language}. Output ONLY the translation.\n"
f"Text: {text}"
)
# Retry loop for stability
max_retries = 3
for attempt in range(max_retries):
try:
response = ollama.chat(model=model, messages=[{'role': 'user', 'content': prompt}])
# Check for "model not found" or other soft errors in response if API wraps them
# Usually ollama library raises ResponseError for 404
translated_text = response['message']['content'].strip()
if translated_text:
line.text = translated_text
break # Success, exit retry loop
except Exception as e:
# Log error details
with open(debug_log_path, "a") as log:
log.write(f"[{datetime.now()}] Error on line '{text}': {str(e)}\n")
if attempt < max_retries - 1:
time.sleep(2) # Wait before retry
else:
# If all retries fail, keep original text or empty?
# Keeping original might be safer than silence, or just skip.
# For now, we skip updating 'line.text' so it stays as source language (better than corruption)
pass
return subs.to_string(format_="srt")
except Exception as e:
print(f" [Local LLM] Critical Error: {e}")
return None
def translate_fallback_mymemory(source_srt_content, target_language="en"):
"""
Fallback translation using MyMemory (via deep-translator).
Limit: 1000 words/day roughly for anonymous usage. Good last resort.
"""
try:
subs = pysubs2.SSAFile.from_string(source_srt_content)
# MyMemory uses ISO 639-1 usually
translator = MyMemoryTranslator(source='auto', target=target_language)
for line in tqdm(subs, desc=" MyMemory Progress", unit="line", leave=False):
text = line.text.strip()
# Skip empty, numeric-only, or extremely short non-word text
if not text or text.isdigit() or len(text) < 2:
continue
if text:
if len(text) > 500: # MyMemory has stricter limits often
continue
try:
original_text = text.replace(r"\N", " ")
translated_text = translator.translate(original_text)
if translated_text:
line.text = translated_text
except Exception:
pass
return subs.to_string(format_="srt")
except Exception as e:
print(f" [MyMemory Fallback] Critical Error: {e}")
return None
def translate_fallback_free(source_srt_content, target_language="en"):
"""
Fallback translation using deep-translator (free Google Translate).
Args:
source_srt_content (str): Content of the source SRT file.
target_language (str): Target language code (e.g. 'en', 'fr').
Returns:
str: Translated SRT content, or None if failed.
"""
try:
# Load from string
subs = pysubs2.SSAFile.from_string(source_srt_content)
translator = GoogleTranslator(source='auto', target=target_language)
# Simple line-by-line translation
for line in tqdm(subs, desc=" DeepTranslate Progress", unit="line", leave=False):
text = line.text.strip()
# Skip empty, numeric-only, or extremely short non-word text
if not text or text.isdigit() or len(text) < 2:
continue
if text:
# Sanity check: Skip lines that are too long
if len(text) > 4000:
continue
try:
# pysubs2 text can contain \N for newlines.
original_text = text.replace(r"\N", " ")
translated_text = translator.translate(original_text)
if translated_text:
line.text = translated_text
except Exception:
pass
# Return as string
return subs.to_string(format_="srt")
except Exception as e:
print(f" [Free Fallback] Critical Error: {e}")
return None
# Define a retry decorator
# Waits 2^x * 1 seconds between retries (1s, 2s, 4s...)
# Stop after 15 attempts
# before_sleep logic can print a simple message
def log_retry_attempt(retry_state):
if retry_state.attempt_number > 1:
print(f" [Gemini] Rate limit hit. Retrying in {retry_state.next_action.sleep}s...", end='\r')
retry_policy = retry(
stop=stop_after_attempt(15),
wait=wait_exponential(multiplier=1, min=2, max=60),
retry=retry_if_exception_type(Exception),
reraise=True,
before_sleep=log_retry_attempt
)
@retry_policy
def _generate_with_retry(client, model_name, prompt):
"""Internal function to wrap the API call with retry logic."""
return client.models.generate_content(
model=model_name,
contents=prompt
)
def get_best_available_model(client):
"""
Queries the API to find the best available model for text generation.
Priority: gemini-2.0-flash > gemini-1.5-flash > gemini-1.5-pro
"""
try:
# Priority list (New v2 naming conventions if applicable, but standard models persist)
priorities = [
"gemini-2.0-flash", # Latest
"gemini-1.5-flash",
"gemini-1.5-pro"
]
# In new SDK, client.models.list() returns iterators of Model objects
# We can just try to use the priority one directly, or list them.
# Listing can be slow. Let's just default to a known good priority list.
# If we really want to check:
# available = [m.name for m in client.models.list()]
# For efficiency/speed, we will trust our priority list.
# The API will error if model doesn't exist, which the try/catch block handling generation will catch?
# No, better to pick one that exists.
# Let's return the latest standard one.
return "gemini-2.0-flash" # Assuming 2.0 is available or falling back
except Exception as e:
print(f"Warning: Model selection issue ({e}). Defaulting to 'gemini-1.5-flash'.")
return "gemini-1.5-flash"
def translate_srt(srt_content, target_language="English", api_key=None):
"""
Translates SRT subtitle content using the Google GenAI SDK (v2).
"""
if not srt_content:
return ""
key = api_key or os.getenv("GEMINI_API_KEY")
if not key:
print("Error: GEMINI_API_KEY not found. Please set the environment variable or pass the key.")
sys.exit(1)
# Initialize Client (v2 style)
try:
client = genai.Client(api_key=key)
except Exception as e:
print(f"Error initializing GenAI Client: {e}")
return None
# Automatically select the best model
# Note: v2 SDK might use 'gemini-1.5-flash' directly without 'models/' prefix usually
model_name = "gemini-2.0-flash"
prompt = (
"You are a professional subtitle translator. Your task is to translate the following SRT subtitle file "
f"into {target_language}.\n\n"
"RULES:\n"
"1. PRESERVE the SRT format exactly. Do not modify timestamps (e.g., 00:00:01,000 --> 00:00:04,000) or sequence numbers.\n"
"2. Only translate the dialogue text.\n"
"3. Maintain the original tone and context.\n"
"4. Output ONLY the translated SRT content, no markdown code blocks or explanations.\n\n"
"SRT Content:\n"
f"{srt_content}"
)
try:
# Call the retried internal function
response = _generate_with_retry(client, model_name, prompt)
# Cleanup: sometimes models wrap output in ```srt ... ``` or ``` ... ```
cleaned_text = response.text.strip()
if cleaned_text.startswith("```"):
lines = cleaned_text.split('\n')
if len(lines) >= 2:
cleaned_text = '\n'.join(lines[1:-1])
return cleaned_text
except Exception as e:
print(f"Error during translation after retries: {e}")
# Fallback to older model if 2.0 fails?
if "404" in str(e) and "gemini-2.0" in model_name:
print(" -> gemini-2.0-flash not found, falling back to gemini-1.5-flash")
try:
response = _generate_with_retry(client, "gemini-1.5-flash", prompt)
cleaned_text = response.text.strip()
if cleaned_text.startswith("```"):
lines = cleaned_text.split('\n')
if len(lines) >= 2:
cleaned_text = '\n'.join(lines[1:-1])
return cleaned_text
except Exception as inner_e:
print(f"Fallback failed: {inner_e}")
return None
def translate_with_auto_fallback(srt_content, target_language="English", prefer_deep=False, prefer_local=False, available_services=None):
"""
Attempts to translate SRT content using Gemini, DeepTranslate, and Local LLM with fallback logic.
Args:
srt_content (str): The source SRT content.
target_language (str): Target language name (e.g., "English", "French").
prefer_deep (bool): If True, try DeepTranslate first (among cloud services).
prefer_local (bool): If True, try Local LLM (Ollama) first.
available_services (dict, optional): Result of check_service_availability().
Returns:
tuple: (translated_content, method_name) or (None, None) if all failed.
"""
# Use central language mapping
target_code = LANGUAGE_MAP.get(target_language, "en")
# Determine which services to even try
def is_ok(name):
if available_services is None: return True
return available_services.get(name, True)
def try_gemini():
if not is_ok("Gemini"): return None, None
res = translate_srt(srt_content, target_language=target_language)
if res: return res, "Gemini"
return None, None
def try_deep():
if not is_ok("DeepTranslate"): return None, None
res = translate_fallback_free(srt_content, target_language=target_code)
if res: return res, "DeepTranslate"
return None, None
def try_ollama():
if not is_ok("Ollama"): return None, None
res = translate_via_ollama(srt_content, target_language=target_language)
if res: return res, "Local LLM (Ollama)"
return None, None
def try_mymemory():
res = translate_fallback_mymemory(srt_content, target_language=target_code)
if res: return res, "MyMemory"
return None, None
# Logic flow
attempts = []
if prefer_local:
attempts.append(try_ollama)
if prefer_deep:
attempts.extend([try_deep, try_gemini])
else:
attempts.extend([try_gemini, try_deep])
else:
if prefer_deep:
attempts.extend([try_deep, try_gemini])
else:
attempts.extend([try_gemini, try_deep])
attempts.append(try_ollama)
# Final last resort
attempts.append(try_mymemory)
# Execute attempts
for i, method_func in enumerate(attempts):
# if i > 0:
# print(f" Attempt {i} failed. Trying next fallback...")
content, method = method_func()
if content:
return content, method
return None, None
@@ -0,0 +1,300 @@
import pysubs2
import os
import signal
import sys
import subprocess
import time
import socket
import shutil
import chardet
from tqdm import tqdm
# Global Language Mapping
LANGUAGE_MAP = {
"English": "en", "French": "fr", "Spanish": "es", "German": "de",
"Italian": "it", "Portuguese": "pt", "Russian": "ru",
"Japanese": "ja", "Chinese": "zh-CN", "auto": "auto"
}
class GracefulKiller:
"""
Handles SIGINT (Ctrl+C) and SIGTERM signals.
Allows the application to finish the current task before exiting.
"""
kill_now = False
def __init__(self):
signal.signal(signal.SIGINT, self.exit_gracefully)
signal.signal(signal.SIGTERM, self.exit_gracefully)
def exit_gracefully(self, signum, frame):
if not self.kill_now:
self.kill_now = True
print("\n\n[STOP REQUESTED] The script will exit after the current file finishes processing.")
print("Press Ctrl+C again to force quit immediately (not recommended).\n")
else:
print("\n[FORCE QUIT] Exiting immediately...")
sys.exit(1)
def is_port_open(host, port):
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
s.settimeout(1)
return s.connect_ex((host, port)) == 0
def detect_file_encoding(file_path):
"""
Robustly detects the encoding of a file using chardet.
Returns 'utf-8' if detection fails or confidence is low, as a safe default.
"""
try:
with open(file_path, 'rb') as f:
raw_data = f.read(10000) # Read first 10KB
result = chardet.detect(raw_data)
encoding = result['encoding']
confidence = result['confidence']
if encoding and confidence > 0.7:
# Shift-JIS is often detected as other Japanese variants, which is fine,
# but sometimes we want to be specific. Chardet is usually good.
return encoding
return 'utf-8'
except Exception:
return 'utf-8'
def verify_file_not_empty(file_path):
"""
Checks if a file exists and is larger than 0 bytes.
"""
if os.path.exists(file_path) and os.path.getsize(file_path) > 0:
return True
return False
def ensure_ollama_running(model_name="dolphin-llama3"):
"""
Checks if Ollama is running. If not, attempts to start it.
Supports Flatpak by escaping to host via flatpak-spawn.
"""
in_flatpak = os.path.exists("/.flatpak-info")
def run_cmd(cmd_list, capture=False):
if in_flatpak:
full_cmd = ["flatpak-spawn", "--host"] + cmd_list
else:
full_cmd = cmd_list
try:
if capture:
return subprocess.run(full_cmd, capture_output=True, text=True)
else:
# For serve, we use Popen
return subprocess.Popen(
full_cmd,
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
start_new_session=True
)
except Exception:
return None
# 1. Start Server if port is closed
if not is_port_open("127.0.0.1", 11434):
print("Starting Ollama server (Local LLM)...")
run_cmd(["ollama", "serve"])
print(" Waiting for Ollama to initialize...", end="", flush=True)
for _ in range(10):
if is_port_open("127.0.0.1", 11434):
print(" Done.")
break
time.sleep(1)
print(".", end="", flush=True)
else:
print("\n Warning: Ollama server failed to start or binary not found.")
return False
# 2. Check Model Presence
try:
result = run_cmd(["ollama", "list"], capture=True)
if result and result.returncode == 0:
if model_name not in result.stdout:
print(f" Model '{model_name}' not found. Pulling now (this may take a while)...")
# Pulling can take a long time, so we don't capture but we want to wait
pull_cmd = ["flatpak-spawn", "--host", "ollama", "pull", model_name] if in_flatpak else ["ollama", "pull", model_name]
subprocess.run(pull_cmd, check=True)
print(" Model pulled successfully.")
else:
# If we can't run list, but port is open, we assume it's okay and let the library handle it
pass
except Exception as e:
print(f" Warning: Could not verify/pull Ollama model: {e}")
return True
def check_service_availability(prefer_deep=False):
"""
Checks availability of configured translation services by performing tiny tests.
Returns a dictionary of status.
"""
status = {
"Gemini": False,
"DeepTranslate": False,
"Ollama": False
}
print("Checking Services...")
# 1. Check Gemini (Real Test)
gemini_key = os.getenv("GEMINI_API_KEY")
if gemini_key:
try:
# We import here to avoid global import issues if dependencies are missing
from google import genai
client = genai.Client(api_key=gemini_key)
# Try a very cheap call
client.models.generate_content(
model="gemini-2.0-flash",
contents="Hi"
)
status["Gemini"] = True
except Exception as e:
# Check for rate limit in string representation
if "429" in str(e) or "RESOURCE_EXHAUSTED" in str(e):
# It is technically 'configured' but currently useless
status["Gemini"] = False
else:
status["Gemini"] = False
# 2. Check DeepTranslate (Real Test)
try:
from deep_translator import GoogleTranslator
GoogleTranslator(source='auto', target='en').translate("hola")
status["DeepTranslate"] = True
except Exception:
status["DeepTranslate"] = False
# 3. Check Ollama
if is_port_open("127.0.0.1", 11434):
status["Ollama"] = True
# Print Report
gemini_msg = "[READY]" if status['Gemini'] else "[UNAVAILABLE] (Rate Limited or Key Invalid)"
if not gemini_key: gemini_msg = "[UNAVAILABLE] (Key missing)"
print(f"1. Gemini API: {gemini_msg}")
print(f"2. DeepTranslate: {'[READY]' if status['DeepTranslate'] else '[UNAVAILABLE] (Network/Block)'}")
print(f"3. Local Ollama: {'[READY]' if status['Ollama'] else '[OFFLINE]'}")
return status
def check_path_permissions(directory_path):
"""
Checks if the script has read and write permissions for the given directory.
Returns: (bool, message)
"""
if not os.path.exists(directory_path):
return False, f"Path not found: {directory_path}"
# If it's a file, check parent directory
if os.path.isfile(directory_path):
directory_path = os.path.dirname(directory_path)
test_file = os.path.join(directory_path, ".perm_test_tmp")
try:
# Test Write
with open(test_file, "w") as f:
f.write("test")
# Test Read
with open(test_file, "r") as f:
content = f.read()
# Cleanup
os.remove(test_file)
if content == "test":
return True, f" [Permissions] Read/Write OK: {directory_path}"
else:
return False, f" [Permissions] Read check failed (content mismatch): {directory_path}"
except PermissionError:
return False, f" ❌ [Permissions] DENIED: Cannot write to {directory_path}. Check ownership/mount options."
except Exception as e:
return False, f" ❌ [Permissions] Error checking {directory_path}: {e}"
def validate_and_repair_srt(srt_path):
"""
Validates an SRT file and attempts to repair it using pysubs2.
Args:
srt_path (str): Path to the SRT file.
Returns:
bool: True if valid/repaired, False if critical error.
"""
if not os.path.exists(srt_path):
return False
print(f"Validating SRT: {srt_path}...")
try:
# Load the subtitle file. pysubs2 parser is robust and handles many errors automatically.
subs = pysubs2.load(srt_path)
# Save it back ensures consistent formatting and fixes minor syntax issues
subs.save(srt_path)
print("SRT validation passed (file re-saved with correct formatting).")
return True
except Exception as e:
print(f"Warning: SRT validation failed: {e}")
return False
def check_srt_duration_match(source_srt_path, target_srt_path, tolerance_seconds=30.0, tolerance_percent=0.10):
"""
Compares the duration of two SRT files to ensure they cover roughly the same timeframe.
Useful for detecting partial translations.
Args:
source_srt_path (str): Path to the original language SRT.
target_srt_path (str): Path to the translated SRT.
tolerance_seconds (float): Max allowed difference in seconds.
tolerance_percent (float): Max allowed difference as a percentage of source duration.
Returns:
tuple: (bool, str) -> (passed, message)
"""
if not os.path.exists(source_srt_path) or not os.path.exists(target_srt_path):
return False, "One or both SRT files missing."
try:
source_subs = pysubs2.load(source_srt_path)
target_subs = pysubs2.load(target_srt_path)
except Exception as e:
return False, f"Error parsing SRTs: {e}"
if not source_subs:
return False, "Source SRT is empty."
if not target_subs:
return False, "Target SRT is empty."
# Get the end timestamp of the last event in each file (in milliseconds)
source_end = source_subs[-1].end
target_end = target_subs[-1].end
# Convert to seconds
source_duration = source_end / 1000.0
target_duration = target_end / 1000.0
diff = abs(source_duration - target_duration)
# Check absolute difference
if diff > tolerance_seconds:
# Also check percentage (for very long videos, 30s might be negligible)
if source_duration > 0 and (diff / source_duration) > tolerance_percent:
return False, f"Duration mismatch: Source={source_duration:.1f}s, Target={target_duration:.1f}s (Diff={diff:.1f}s)"
# For short videos, if percentage is high, fail
if source_duration < 300 and (diff / source_duration) > 0.20:
return False, f"Duration mismatch (short video): Source={source_duration:.1f}s, Target={target_duration:.1f}s"
return True, f"Duration match verified (Diff={diff:.1f}s)"