Files
personal_development/video_transcription/ai_transcriber_v2/utils.py
T

301 lines
10 KiB
Python

import pysubs2
import os
import signal
import sys
import subprocess
import time
import socket
import shutil
import chardet
from tqdm import tqdm
# Global Language Mapping
LANGUAGE_MAP = {
"English": "en", "French": "fr", "Spanish": "es", "German": "de",
"Italian": "it", "Portuguese": "pt", "Russian": "ru",
"Japanese": "ja", "Chinese": "zh-CN", "auto": "auto"
}
class GracefulKiller:
"""
Handles SIGINT (Ctrl+C) and SIGTERM signals.
Allows the application to finish the current task before exiting.
"""
kill_now = False
def __init__(self):
signal.signal(signal.SIGINT, self.exit_gracefully)
signal.signal(signal.SIGTERM, self.exit_gracefully)
def exit_gracefully(self, signum, frame):
if not self.kill_now:
self.kill_now = True
print("\n\n[STOP REQUESTED] The script will exit after the current file finishes processing.")
print("Press Ctrl+C again to force quit immediately (not recommended).\n")
else:
print("\n[FORCE QUIT] Exiting immediately...")
sys.exit(1)
def is_port_open(host, port):
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
s.settimeout(1)
return s.connect_ex((host, port)) == 0
def detect_file_encoding(file_path):
"""
Robustly detects the encoding of a file using chardet.
Returns 'utf-8' if detection fails or confidence is low, as a safe default.
"""
try:
with open(file_path, 'rb') as f:
raw_data = f.read(10000) # Read first 10KB
result = chardet.detect(raw_data)
encoding = result['encoding']
confidence = result['confidence']
if encoding and confidence > 0.7:
# Shift-JIS is often detected as other Japanese variants, which is fine,
# but sometimes we want to be specific. Chardet is usually good.
return encoding
return 'utf-8'
except Exception:
return 'utf-8'
def verify_file_not_empty(file_path):
"""
Checks if a file exists and is larger than 0 bytes.
"""
if os.path.exists(file_path) and os.path.getsize(file_path) > 0:
return True
return False
def ensure_ollama_running(model_name="llama3"):
"""
Checks if Ollama is running. If not, attempts to start it.
Supports Flatpak by escaping to host via flatpak-spawn.
"""
in_flatpak = os.path.exists("/.flatpak-info")
def run_cmd(cmd_list, capture=False):
if in_flatpak:
full_cmd = ["flatpak-spawn", "--host"] + cmd_list
else:
full_cmd = cmd_list
try:
if capture:
return subprocess.run(full_cmd, capture_output=True, text=True)
else:
# For serve, we use Popen
return subprocess.Popen(
full_cmd,
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
start_new_session=True
)
except Exception:
return None
# 1. Start Server if port is closed
if not is_port_open("127.0.0.1", 11434):
print("Starting Ollama server (Local LLM)...")
run_cmd(["ollama", "serve"])
print(" Waiting for Ollama to initialize...", end="", flush=True)
for _ in range(10):
if is_port_open("127.0.0.1", 11434):
print(" Done.")
break
time.sleep(1)
print(".", end="", flush=True)
else:
print("\n Warning: Ollama server failed to start or binary not found.")
return False
# 2. Check Model Presence
try:
result = run_cmd(["ollama", "list"], capture=True)
if result and result.returncode == 0:
if model_name not in result.stdout:
print(f" Model '{model_name}' not found. Pulling now (this may take a while)...")
# Pulling can take a long time, so we don't capture but we want to wait
pull_cmd = ["flatpak-spawn", "--host", "ollama", "pull", model_name] if in_flatpak else ["ollama", "pull", model_name]
subprocess.run(pull_cmd, check=True)
print(" Model pulled successfully.")
else:
# If we can't run list, but port is open, we assume it's okay and let the library handle it
pass
except Exception as e:
print(f" Warning: Could not verify/pull Ollama model: {e}")
return True
def check_service_availability(prefer_deep=False):
"""
Checks availability of configured translation services by performing tiny tests.
Returns a dictionary of status.
"""
status = {
"Gemini": False,
"DeepTranslate": False,
"Ollama": False
}
print("Checking Services...")
# 1. Check Gemini (Real Test)
gemini_key = os.getenv("GEMINI_API_KEY")
if gemini_key:
try:
# We import here to avoid global import issues if dependencies are missing
from google import genai
client = genai.Client(api_key=gemini_key)
# Try a very cheap call
client.models.generate_content(
model="gemini-2.0-flash",
contents="Hi"
)
status["Gemini"] = True
except Exception as e:
# Check for rate limit in string representation
if "429" in str(e) or "RESOURCE_EXHAUSTED" in str(e):
# It is technically 'configured' but currently useless
status["Gemini"] = False
else:
status["Gemini"] = False
# 2. Check DeepTranslate (Real Test)
try:
from deep_translator import GoogleTranslator
GoogleTranslator(source='auto', target='en').translate("hola")
status["DeepTranslate"] = True
except Exception:
status["DeepTranslate"] = False
# 3. Check Ollama
if is_port_open("127.0.0.1", 11434):
status["Ollama"] = True
# Print Report
gemini_msg = "[READY]" if status['Gemini'] else "[UNAVAILABLE] (Rate Limited or Key Invalid)"
if not gemini_key: gemini_msg = "[UNAVAILABLE] (Key missing)"
print(f"1. Gemini API: {gemini_msg}")
print(f"2. DeepTranslate: {'[READY]' if status['DeepTranslate'] else '[UNAVAILABLE] (Network/Block)'}")
print(f"3. Local Ollama: {'[READY]' if status['Ollama'] else '[OFFLINE]'}")
return status
def check_path_permissions(directory_path):
"""
Checks if the script has read and write permissions for the given directory.
Returns: (bool, message)
"""
if not os.path.exists(directory_path):
return False, f"Path not found: {directory_path}"
# If it's a file, check parent directory
if os.path.isfile(directory_path):
directory_path = os.path.dirname(directory_path)
test_file = os.path.join(directory_path, ".perm_test_tmp")
try:
# Test Write
with open(test_file, "w") as f:
f.write("test")
# Test Read
with open(test_file, "r") as f:
content = f.read()
# Cleanup
os.remove(test_file)
if content == "test":
return True, f" [Permissions] Read/Write OK: {directory_path}"
else:
return False, f" [Permissions] Read check failed (content mismatch): {directory_path}"
except PermissionError:
return False, f" ❌ [Permissions] DENIED: Cannot write to {directory_path}. Check ownership/mount options."
except Exception as e:
return False, f" ❌ [Permissions] Error checking {directory_path}: {e}"
def validate_and_repair_srt(srt_path):
"""
Validates an SRT file and attempts to repair it using pysubs2.
Args:
srt_path (str): Path to the SRT file.
Returns:
bool: True if valid/repaired, False if critical error.
"""
if not os.path.exists(srt_path):
return False
print(f"Validating SRT: {srt_path}...")
try:
# Load the subtitle file. pysubs2 parser is robust and handles many errors automatically.
subs = pysubs2.load(srt_path)
# Save it back ensures consistent formatting and fixes minor syntax issues
subs.save(srt_path)
print("SRT validation passed (file re-saved with correct formatting).")
return True
except Exception as e:
print(f"Warning: SRT validation failed: {e}")
return False
def check_srt_duration_match(source_srt_path, target_srt_path, tolerance_seconds=30.0, tolerance_percent=0.10):
"""
Compares the duration of two SRT files to ensure they cover roughly the same timeframe.
Useful for detecting partial translations.
Args:
source_srt_path (str): Path to the original language SRT.
target_srt_path (str): Path to the translated SRT.
tolerance_seconds (float): Max allowed difference in seconds.
tolerance_percent (float): Max allowed difference as a percentage of source duration.
Returns:
tuple: (bool, str) -> (passed, message)
"""
if not os.path.exists(source_srt_path) or not os.path.exists(target_srt_path):
return False, "One or both SRT files missing."
try:
source_subs = pysubs2.load(source_srt_path)
target_subs = pysubs2.load(target_srt_path)
except Exception as e:
return False, f"Error parsing SRTs: {e}"
if not source_subs:
return False, "Source SRT is empty."
if not target_subs:
return False, "Target SRT is empty."
# Get the end timestamp of the last event in each file (in milliseconds)
source_end = source_subs[-1].end
target_end = target_subs[-1].end
# Convert to seconds
source_duration = source_end / 1000.0
target_duration = target_end / 1000.0
diff = abs(source_duration - target_duration)
# Check absolute difference
if diff > tolerance_seconds:
# Also check percentage (for very long videos, 30s might be negligible)
if source_duration > 0 and (diff / source_duration) > tolerance_percent:
return False, f"Duration mismatch: Source={source_duration:.1f}s, Target={target_duration:.1f}s (Diff={diff:.1f}s)"
# For short videos, if percentage is high, fail
if source_duration < 300 and (diff / source_duration) > 0.20:
return False, f"Duration mismatch (short video): Source={source_duration:.1f}s, Target={target_duration:.1f}s"
return True, f"Duration match verified (Diff={diff:.1f}s)"