added
This commit is contained in:
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -1,6 +1,7 @@
|
|||||||
import argparse
|
import argparse
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
|
import socket
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
|
|
||||||
# Load environment variables from central .env_files directory
|
# Load environment variables from central .env_files directory
|
||||||
@@ -183,7 +184,8 @@ def process_file(file_path, args, source_lang=None, loaded_model=None, service_s
|
|||||||
tracker.logger.error(f"VALIDATION FAILED: {msg}")
|
tracker.logger.error(f"VALIDATION FAILED: {msg}")
|
||||||
tracker.logger.error("Marking translation as failed due to incomplete coverage.")
|
tracker.logger.error("Marking translation as failed due to incomplete coverage.")
|
||||||
|
|
||||||
redo_file = os.path.join(os.path.dirname(file_path), "redo_queue.txt")
|
hostname = socket.gethostname()
|
||||||
|
redo_file = os.path.join(os.path.dirname(file_path), f"redo_queue_{hostname}.txt")
|
||||||
with open(redo_file, "a", encoding="utf-8") as rf:
|
with open(redo_file, "a", encoding="utf-8") as rf:
|
||||||
rf.write(f"{file_path} | {msg}\n")
|
rf.write(f"{file_path} | {msg}\n")
|
||||||
|
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ import os
|
|||||||
import sys
|
import sys
|
||||||
import argparse
|
import argparse
|
||||||
import subprocess
|
import subprocess
|
||||||
|
import socket
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
import pysubs2
|
import pysubs2
|
||||||
@@ -36,7 +37,8 @@ def process_recovery(folder_path, target_lang="English", prefer_deep=False, pref
|
|||||||
else:
|
else:
|
||||||
print("Preference: Gemini (API) > DeepTranslate")
|
print("Preference: Gemini (API) > DeepTranslate")
|
||||||
|
|
||||||
recovery_log_file = os.path.join(folder_path, "recovery_status.log")
|
hostname = socket.gethostname()
|
||||||
|
recovery_log_file = os.path.join(folder_path, f"recovery_status_{hostname}.log")
|
||||||
print(f"Logging actions to: {recovery_log_file}")
|
print(f"Logging actions to: {recovery_log_file}")
|
||||||
|
|
||||||
# Ensure Ollama is ready
|
# Ensure Ollama is ready
|
||||||
|
|||||||
@@ -1,14 +1,18 @@
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
|
import socket
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from sqlalchemy import create_engine, Column, Integer, String, DateTime, Enum, Text
|
from sqlalchemy import create_engine, Column, Integer, String, DateTime, Enum, Text
|
||||||
from sqlalchemy.orm import declarative_base, sessionmaker
|
from sqlalchemy.orm import declarative_base, sessionmaker
|
||||||
import enum
|
import enum
|
||||||
|
|
||||||
|
# Get Hostname for namespacing
|
||||||
|
HOSTNAME = socket.gethostname()
|
||||||
|
|
||||||
# Setup Logging
|
# Setup Logging
|
||||||
log_dir = "logs"
|
log_dir = "logs"
|
||||||
os.makedirs(log_dir, exist_ok=True)
|
os.makedirs(log_dir, exist_ok=True)
|
||||||
log_file = os.path.join(log_dir, f"transcriber_{datetime.now().strftime('%Y%m%d')}.log")
|
log_file = os.path.join(log_dir, f"transcriber_{HOSTNAME}_{datetime.now().strftime('%Y%m%d')}.log")
|
||||||
|
|
||||||
logging.basicConfig(
|
logging.basicConfig(
|
||||||
level=logging.INFO,
|
level=logging.INFO,
|
||||||
@@ -22,7 +26,7 @@ logger = logging.getLogger(__name__)
|
|||||||
|
|
||||||
# Database Setup
|
# Database Setup
|
||||||
Base = declarative_base()
|
Base = declarative_base()
|
||||||
DB_FILE = "job_history.db"
|
DB_FILE = f"job_history_{HOSTNAME}.db"
|
||||||
|
|
||||||
class JobStatus(enum.Enum):
|
class JobStatus(enum.Enum):
|
||||||
PENDING = "pending"
|
PENDING = "pending"
|
||||||
|
|||||||
@@ -22,6 +22,11 @@ def translate_via_ollama(source_srt_content, target_language="English", model="l
|
|||||||
# Using tqdm for progress bar
|
# Using tqdm for progress bar
|
||||||
for line in tqdm(subs, desc=" Ollama Progress", unit="line"):
|
for line in tqdm(subs, desc=" Ollama Progress", unit="line"):
|
||||||
text = line.text.strip()
|
text = line.text.strip()
|
||||||
|
|
||||||
|
# Skip empty, numeric-only, or extremely short non-word text
|
||||||
|
if not text or text.isdigit() or len(text) < 2:
|
||||||
|
continue
|
||||||
|
|
||||||
if text:
|
if text:
|
||||||
prompt = (
|
prompt = (
|
||||||
f"Translate this subtitle text to {target_language}. Output ONLY the translation.\n"
|
f"Translate this subtitle text to {target_language}. Output ONLY the translation.\n"
|
||||||
@@ -54,6 +59,11 @@ def translate_fallback_mymemory(source_srt_content, target_language="en"):
|
|||||||
|
|
||||||
for line in tqdm(subs, desc=" MyMemory Progress", unit="line"):
|
for line in tqdm(subs, desc=" MyMemory Progress", unit="line"):
|
||||||
text = line.text.strip()
|
text = line.text.strip()
|
||||||
|
|
||||||
|
# Skip empty, numeric-only, or extremely short non-word text
|
||||||
|
if not text or text.isdigit() or len(text) < 2:
|
||||||
|
continue
|
||||||
|
|
||||||
if text:
|
if text:
|
||||||
if len(text) > 500: # MyMemory has stricter limits often
|
if len(text) > 500: # MyMemory has stricter limits often
|
||||||
continue
|
continue
|
||||||
@@ -89,6 +99,11 @@ def translate_fallback_free(source_srt_content, target_language="en"):
|
|||||||
# Simple line-by-line translation
|
# Simple line-by-line translation
|
||||||
for line in tqdm(subs, desc=" DeepTranslate Progress", unit="line"):
|
for line in tqdm(subs, desc=" DeepTranslate Progress", unit="line"):
|
||||||
text = line.text.strip()
|
text = line.text.strip()
|
||||||
|
|
||||||
|
# Skip empty, numeric-only, or extremely short non-word text
|
||||||
|
if not text or text.isdigit() or len(text) < 2:
|
||||||
|
continue
|
||||||
|
|
||||||
if text:
|
if text:
|
||||||
# Sanity check: Skip lines that are too long
|
# Sanity check: Skip lines that are too long
|
||||||
if len(text) > 4000:
|
if len(text) > 4000:
|
||||||
|
|||||||
Reference in New Issue
Block a user