AI Transcriber tool
This commit is contained in:
@@ -0,0 +1,127 @@
|
||||
import os
|
||||
import sys
|
||||
import google.generativeai as genai
|
||||
from tenacity import retry, stop_after_attempt, wait_exponential, retry_if_exception_type
|
||||
|
||||
# Define a retry decorator
|
||||
# Waits 2^x * 1 seconds between retries (1s, 2s, 4s, 8s, 16s, 32s...)
|
||||
# With max=60, it will cap at waiting 60s per try.
|
||||
# Stop after 15 attempts (approx 15 minutes of trying before giving up)
|
||||
retry_policy = retry(
|
||||
stop=stop_after_attempt(15),
|
||||
wait=wait_exponential(multiplier=1, min=2, max=60),
|
||||
retry=retry_if_exception_type(Exception),
|
||||
reraise=True
|
||||
)
|
||||
|
||||
@retry_policy
|
||||
def _generate_with_retry(model, prompt):
|
||||
"""Internal function to wrap the API call with retry logic."""
|
||||
try:
|
||||
return model.generate_content(prompt)
|
||||
except Exception as e:
|
||||
if "429" in str(e) or "Resource has been exhausted" in str(e):
|
||||
print(f" [Rate Limit Hit] Waiting for quota reset... ({e})")
|
||||
raise e
|
||||
|
||||
def get_best_available_model():
|
||||
"""
|
||||
Queries the API to find the best available model for text generation.
|
||||
Priority: gemini-1.5-flash > gemini-1.5-pro > gemini-pro > any 'generateContent' model
|
||||
"""
|
||||
try:
|
||||
available_models = []
|
||||
for m in genai.list_models():
|
||||
if 'generateContent' in m.supported_generation_methods:
|
||||
available_models.append(m.name)
|
||||
|
||||
# Priority list
|
||||
priorities = [
|
||||
"models/gemini-1.5-flash",
|
||||
"models/gemini-1.5-pro",
|
||||
"models/gemini-pro"
|
||||
]
|
||||
|
||||
# Check for priorities first
|
||||
for p in priorities:
|
||||
if p in available_models:
|
||||
return p
|
||||
|
||||
# Fallback: check for aliases without 'models/' prefix just in case
|
||||
for p in priorities:
|
||||
short_name = p.replace("models/", "")
|
||||
# Some libraries might return short names, or custom handling
|
||||
# But genai.list_models() usually returns 'models/name'
|
||||
pass
|
||||
|
||||
# If priority not found, pick the first available gemini model
|
||||
for m in available_models:
|
||||
if "gemini" in m:
|
||||
return m
|
||||
|
||||
if available_models:
|
||||
return available_models[0]
|
||||
|
||||
except Exception as e:
|
||||
print(f"Warning: Could not list models ({e}). Defaulting to 'gemini-pro'.")
|
||||
|
||||
return "gemini-pro"
|
||||
|
||||
def translate_srt(srt_content, target_language="English", api_key=None):
|
||||
"""
|
||||
Translates SRT subtitle content using the Gemini API, preserving timestamps.
|
||||
|
||||
Args:
|
||||
srt_content (str): The raw text content of the SRT file.
|
||||
target_language (str): The target language for translation.
|
||||
api_key (str): Google Gemini API key. If None, checks env var GEMINI_API_KEY.
|
||||
|
||||
Returns:
|
||||
str: The translated SRT content.
|
||||
"""
|
||||
if not srt_content:
|
||||
return ""
|
||||
|
||||
key = api_key or os.getenv("GEMINI_API_KEY")
|
||||
if not key:
|
||||
print("Error: GEMINI_API_KEY not found. Please set the environment variable or pass the key.")
|
||||
sys.exit(1)
|
||||
|
||||
genai.configure(api_key=key)
|
||||
|
||||
# Automatically select the best model
|
||||
model_name = get_best_available_model()
|
||||
print(f"Using Gemini Model: {model_name}")
|
||||
|
||||
model = genai.GenerativeModel(model_name)
|
||||
|
||||
prompt = (
|
||||
"You are a professional subtitle translator. Your task is to translate the following SRT subtitle file "
|
||||
f"into {target_language}.\n\n"
|
||||
"RULES:\n"
|
||||
"1. PRESERVE the SRT format exactly. Do not modify timestamps (e.g., 00:00:01,000 --> 00:00:04,000) or sequence numbers.\n"
|
||||
"2. Only translate the dialogue text.\n"
|
||||
"3. Maintain the original tone and context.\n"
|
||||
"4. Output ONLY the translated SRT content, no markdown code blocks or explanations.\n\n"
|
||||
"SRT Content:\n"
|
||||
f"{srt_content}"
|
||||
)
|
||||
|
||||
print(f"Translating subtitles to {target_language} (with retries)...")
|
||||
try:
|
||||
# Call the retried internal function
|
||||
response = _generate_with_retry(model, prompt)
|
||||
print("Translation complete.")
|
||||
|
||||
# Cleanup: sometimes models wrap output in ```srt ... ``` or ``` ... ```
|
||||
cleaned_text = response.text.strip()
|
||||
if cleaned_text.startswith("```"):
|
||||
# Remove first line (```srt or ```) and last line (```)
|
||||
lines = cleaned_text.split('\n')
|
||||
if len(lines) >= 2:
|
||||
cleaned_text = '\n'.join(lines[1:-1])
|
||||
|
||||
return cleaned_text
|
||||
except Exception as e:
|
||||
print(f"Error during translation after retries: {e}")
|
||||
return None
|
||||
Reference in New Issue
Block a user