Added revisions to the translation app

This commit is contained in:
2026-01-12 08:52:38 -05:00
parent 64add85920
commit 8181bebaa0
44 changed files with 993 additions and 375 deletions
@@ -115,7 +115,28 @@ def save_as_srt(result, output_path):
f.write(f"{text}\n\n")
print(f"SRT saved to: {output_path}")
def transcribe_audio(audio_path, model_size="auto", language=None):
def load_whisper_model(model_size="auto"):
"""
Loads and returns the Whisper model.
"""
check_gpu_health()
if model_size == "auto":
model_size = get_optimal_model_size()
print(f"Auto-selected model: '{model_size}'")
print(f"Loading Whisper model ('{model_size}')...")
device = "cuda" if torch.cuda.is_available() else "cpu"
print(f"Using device: {device}")
try:
model = whisper.load_model(model_size, device=device)
return model
except Exception as e:
print(f"Error loading model: {e}")
sys.exit(1)
def transcribe_audio(audio_path, model_size="auto", language=None, loaded_model=None):
"""
Transcribes an audio file using OpenAI's Whisper model.
@@ -123,6 +144,7 @@ def transcribe_audio(audio_path, model_size="auto", language=None):
audio_path (str): Path to the input audio file.
model_size (str): Size of the Whisper model to use. If "auto", selects based on VRAM.
language (str, optional): Language code (e.g., "en", "fr", "es"). If None, auto-detects.
loaded_model (object, optional): Pre-loaded Whisper model object.
Returns:
dict: The full transcription result containing segments and text.
@@ -130,37 +152,14 @@ def transcribe_audio(audio_path, model_size="auto", language=None):
if not os.path.exists(audio_path):
raise FileNotFoundError(f"Audio file not found: {audio_path}")
# Run health check once
check_gpu_health()
# Determine model size if auto
if model_size == "auto":
model_size = get_optimal_model_size()
print(f"Auto-selected model: '{model_size}'")
print(f"Loading Whisper model ('{model_size}')...")
# Check for GPU availability
device = "cuda" if torch.cuda.is_available() else "cpu"
print(f"Using device: {device}")
try:
model = whisper.load_model(model_size, device=device)
except RuntimeError as e:
if "out of memory" in str(e).lower():
print("Error: GPU Out of Memory. Try using a smaller model size.")
else:
print(f"Error loading model: {e}")
sys.exit(1)
except Exception as e:
print(f"Error loading model: {e}")
sys.exit(1)
model = loaded_model
if model is None:
model = load_whisper_model(model_size)
print(f"Transcribing {audio_path}...")
try:
# fp16=False is needed for CPU, but we can let whisper handle defaults usually.
# language=None allows auto-detection.
result = model.transcribe(audio_path, language=language)
# Enable verbose=True to show progress in terminal
result = model.transcribe(audio_path, language=language, verbose=True)
print("Transcription complete.")
return result
except Exception as e: