Added revisions to the translation app
This commit is contained in:
@@ -115,7 +115,28 @@ def save_as_srt(result, output_path):
|
||||
f.write(f"{text}\n\n")
|
||||
print(f"SRT saved to: {output_path}")
|
||||
|
||||
def transcribe_audio(audio_path, model_size="auto", language=None):
|
||||
def load_whisper_model(model_size="auto"):
|
||||
"""
|
||||
Loads and returns the Whisper model.
|
||||
"""
|
||||
check_gpu_health()
|
||||
|
||||
if model_size == "auto":
|
||||
model_size = get_optimal_model_size()
|
||||
print(f"Auto-selected model: '{model_size}'")
|
||||
|
||||
print(f"Loading Whisper model ('{model_size}')...")
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
print(f"Using device: {device}")
|
||||
|
||||
try:
|
||||
model = whisper.load_model(model_size, device=device)
|
||||
return model
|
||||
except Exception as e:
|
||||
print(f"Error loading model: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
def transcribe_audio(audio_path, model_size="auto", language=None, loaded_model=None):
|
||||
"""
|
||||
Transcribes an audio file using OpenAI's Whisper model.
|
||||
|
||||
@@ -123,6 +144,7 @@ def transcribe_audio(audio_path, model_size="auto", language=None):
|
||||
audio_path (str): Path to the input audio file.
|
||||
model_size (str): Size of the Whisper model to use. If "auto", selects based on VRAM.
|
||||
language (str, optional): Language code (e.g., "en", "fr", "es"). If None, auto-detects.
|
||||
loaded_model (object, optional): Pre-loaded Whisper model object.
|
||||
|
||||
Returns:
|
||||
dict: The full transcription result containing segments and text.
|
||||
@@ -130,37 +152,14 @@ def transcribe_audio(audio_path, model_size="auto", language=None):
|
||||
if not os.path.exists(audio_path):
|
||||
raise FileNotFoundError(f"Audio file not found: {audio_path}")
|
||||
|
||||
# Run health check once
|
||||
check_gpu_health()
|
||||
|
||||
# Determine model size if auto
|
||||
if model_size == "auto":
|
||||
model_size = get_optimal_model_size()
|
||||
print(f"Auto-selected model: '{model_size}'")
|
||||
|
||||
print(f"Loading Whisper model ('{model_size}')...")
|
||||
|
||||
# Check for GPU availability
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
print(f"Using device: {device}")
|
||||
|
||||
try:
|
||||
model = whisper.load_model(model_size, device=device)
|
||||
except RuntimeError as e:
|
||||
if "out of memory" in str(e).lower():
|
||||
print("Error: GPU Out of Memory. Try using a smaller model size.")
|
||||
else:
|
||||
print(f"Error loading model: {e}")
|
||||
sys.exit(1)
|
||||
except Exception as e:
|
||||
print(f"Error loading model: {e}")
|
||||
sys.exit(1)
|
||||
model = loaded_model
|
||||
if model is None:
|
||||
model = load_whisper_model(model_size)
|
||||
|
||||
print(f"Transcribing {audio_path}...")
|
||||
try:
|
||||
# fp16=False is needed for CPU, but we can let whisper handle defaults usually.
|
||||
# language=None allows auto-detection.
|
||||
result = model.transcribe(audio_path, language=language)
|
||||
# Enable verbose=True to show progress in terminal
|
||||
result = model.transcribe(audio_path, language=language, verbose=True)
|
||||
print("Transcription complete.")
|
||||
return result
|
||||
except Exception as e:
|
||||
|
||||
Reference in New Issue
Block a user