776 lines
No EOL
29 KiB
Python
776 lines
No EOL
29 KiB
Python
"""
|
||
Shared translation utilities for DazedMTLTool
|
||
This module provides a centralized translation function that can be used across all modules
|
||
without relying on global variables.
|
||
"""
|
||
|
||
import os
|
||
import re
|
||
import json
|
||
import tiktoken
|
||
import openai
|
||
from pathlib import Path
|
||
from retry import retry
|
||
|
||
|
||
class TranslationConfig:
|
||
"""Configuration class to hold all translation settings"""
|
||
|
||
def __init__(self,
|
||
model=None,
|
||
language=None,
|
||
prompt=None,
|
||
vocab=None,
|
||
langRegex=None,
|
||
batchSize=None,
|
||
maxHistory=10,
|
||
estimateMode=False,
|
||
logFilePath="log/translationHistory.txt",
|
||
mismatchLogPath="log/mismatchHistory.txt"):
|
||
|
||
# Load from environment if not provided
|
||
self.model = model or os.getenv("model")
|
||
self.language = (language or os.getenv("language", "english")).capitalize()
|
||
|
||
# Load prompt and vocab files if not provided
|
||
if prompt is None:
|
||
try:
|
||
self.prompt = Path("prompt.txt").read_text(encoding="utf-8")
|
||
except FileNotFoundError:
|
||
self.prompt = ""
|
||
else:
|
||
self.prompt = prompt
|
||
|
||
if vocab is None:
|
||
try:
|
||
self.vocab = Path("vocab.txt").read_text(encoding="utf-8")
|
||
except FileNotFoundError:
|
||
self.vocab = ""
|
||
else:
|
||
self.vocab = vocab
|
||
|
||
# Set language regex (default is Japanese)
|
||
self.langRegex = langRegex or r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+"
|
||
|
||
# Set batch size based on model if not provided
|
||
if batchSize is None:
|
||
if "gpt-3.5" in self.model:
|
||
self.batchSize = 10
|
||
elif "gpt-4" in self.model:
|
||
self.batchSize = 30
|
||
elif "deepseek" in self.model:
|
||
self.batchSize = 30
|
||
else:
|
||
# Try to get from environment, fallback to 10
|
||
try:
|
||
self.batchSize = int(os.getenv("batchsize", 10))
|
||
except (ValueError, TypeError):
|
||
self.batchSize = 10
|
||
else:
|
||
self.batchSize = batchSize
|
||
|
||
self.maxHistory = maxHistory
|
||
self.estimateMode = estimateMode
|
||
self.logFilePath = logFilePath
|
||
self.mismatchLogPath = mismatchLogPath
|
||
|
||
|
||
def getPricingConfig(model):
|
||
"""
|
||
Get pricing configuration for a given model.
|
||
|
||
Args:
|
||
model: The model name string
|
||
|
||
Returns:
|
||
dict: Dictionary containing inputAPICost, outputAPICost, batchSize, and frequencyPenalty
|
||
"""
|
||
# Pricing - Depends on the model https://openai.com/pricing ($ Price Per 1M)
|
||
# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request
|
||
# If you are getting a MISMATCH LENGTH error, lower the batch size.
|
||
if "gpt-3.5" in model:
|
||
return {
|
||
"inputAPICost": 3.00,
|
||
"outputAPICost": 5.00,
|
||
"batchSize": 10,
|
||
"frequencyPenalty": 0.2
|
||
}
|
||
elif "gpt-4.1-mini" in model:
|
||
return {
|
||
"inputAPICost": 0.40,
|
||
"outputAPICost": 1.60,
|
||
"batchSize": 30,
|
||
"frequencyPenalty": 0.05
|
||
}
|
||
elif "gpt-4.1" in model:
|
||
return {
|
||
"inputAPICost": 2.00,
|
||
"outputAPICost": 8.00,
|
||
"batchSize": 30,
|
||
"frequencyPenalty": 0.05
|
||
}
|
||
elif "gpt-5" in model:
|
||
return {
|
||
"inputAPICost": 1.25,
|
||
"outputAPICost": 10.00,
|
||
"batchSize": 30,
|
||
"frequencyPenalty": 0.05
|
||
}
|
||
elif "deepseek" in model:
|
||
return {
|
||
"inputAPICost": 0.27,
|
||
"outputAPICost": 1.10,
|
||
"batchSize": 30,
|
||
"frequencyPenalty": 0.05
|
||
}
|
||
elif "sonnet" in model:
|
||
return {
|
||
"inputAPICost": 3.00,
|
||
"outputAPICost": 15.00,
|
||
"batchSize": 30,
|
||
"frequencyPenalty": 0.05
|
||
}
|
||
elif "gemini-2.0-flash-lite" in model:
|
||
return {
|
||
"inputAPICost": 0.075,
|
||
"outputAPICost": 0.30,
|
||
"batchSize": 30,
|
||
"frequencyPenalty": 0.0
|
||
}
|
||
elif "gemini-2.0-flash" in model:
|
||
return {
|
||
"inputAPICost": 0.10,
|
||
"outputAPICost": 0.40,
|
||
"batchSize": 30,
|
||
"frequencyPenalty": 0.0
|
||
}
|
||
elif "gemini-2.5-flash-lite" in model:
|
||
return {
|
||
"inputAPICost": 0.10,
|
||
"outputAPICost": 0.40,
|
||
"batchSize": 30,
|
||
"frequencyPenalty": 0.0
|
||
}
|
||
elif "gemini-2.5-flash" in model:
|
||
return {
|
||
"inputAPICost": 0.30,
|
||
"outputAPICost": 2.50,
|
||
"batchSize": 30,
|
||
"frequencyPenalty": 0.0
|
||
}
|
||
elif "gemini-2.5-pro" in model:
|
||
return {
|
||
"inputAPICost": 1.25,
|
||
"outputAPICost": 10.00,
|
||
"batchSize": 30,
|
||
"frequencyPenalty": 0.0
|
||
}
|
||
else:
|
||
# Fallback to environment variables
|
||
return {
|
||
"inputAPICost": float(os.getenv("input_cost", 3.00)),
|
||
"outputAPICost": float(os.getenv("output_cost", 6.00)),
|
||
"batchSize": int(os.getenv("batchsize", 10)),
|
||
"frequencyPenalty": float(os.getenv("frequency_penalty", 0.2))
|
||
}
|
||
|
||
|
||
def batchList(inputList, batchSize):
|
||
"""Split a list into batches of specified size"""
|
||
if not isinstance(batchSize, int) or batchSize <= 0:
|
||
raise ValueError("batchSize must be a positive integer")
|
||
|
||
return [inputList[i : i + batchSize] for i in range(0, len(inputList), batchSize)]
|
||
|
||
|
||
def parseVocabWithCategories(vocabText):
|
||
"""Parse vocabulary text and extract terms with their categories."""
|
||
pairs = []
|
||
seen = set()
|
||
currentCategory = None
|
||
|
||
for line in vocabText.splitlines():
|
||
line = line.strip()
|
||
if not line or line.startswith('```') or line.startswith('Here are some vocabulary'):
|
||
continue
|
||
|
||
# Check if this is a category header
|
||
if line.startswith('#'):
|
||
currentCategory = line
|
||
continue
|
||
|
||
# Parse vocabulary term - extract both Japanese and English parts
|
||
# Format: "Japanese term (English translation)" or "Japanese term – English translation"
|
||
m = re.match(r'^(.+?)(?:\s?[\(–]\s*(.+?)[\)]?\s*$)', line)
|
||
if m and ('(' in line or '–' in line): # Only process lines that actually have parentheses or dashes
|
||
japanese_term = m.group(1).strip()
|
||
english_term = m.group(2).strip().rstrip(')') # Remove trailing parenthesis if exists
|
||
|
||
# Create a tuple with both terms for matching
|
||
term_pair = (japanese_term, english_term)
|
||
if term_pair not in seen:
|
||
pairs.append((term_pair, line, currentCategory))
|
||
seen.add(term_pair)
|
||
elif line and not line.startswith('#'):
|
||
# Fallback for lines without parentheses - treat as single term
|
||
term = line.strip()
|
||
if term and term not in seen:
|
||
pairs.append((term, line, currentCategory))
|
||
seen.add(term)
|
||
|
||
return pairs
|
||
|
||
|
||
def buildMatchedVocabText(vocabPairs, subbedText, history=None):
|
||
"""Build formatted vocabulary text with matched terms organized by category."""
|
||
matchedCategories = {}
|
||
|
||
# Prepare text to search - combine subbedText and history
|
||
textToSearch = str(subbedText)
|
||
if history:
|
||
if isinstance(history, list):
|
||
textToSearch += " " + " ".join(str(h) for h in history)
|
||
else:
|
||
textToSearch += " " + str(history)
|
||
|
||
# Use word boundaries for Japanese if appropriate, or allow substring as before.
|
||
for term, line, category in vocabPairs:
|
||
# Check if term is a tuple (Japanese, English) or a single term
|
||
term_found = False
|
||
if isinstance(term, tuple):
|
||
# Check both Japanese and English terms
|
||
japanese_term, english_term = term
|
||
if japanese_term in textToSearch or english_term in textToSearch:
|
||
term_found = True
|
||
else:
|
||
# Single term check
|
||
if term in textToSearch:
|
||
term_found = True
|
||
|
||
# Always include "# Game Characters" category and all its terms regardless of matches
|
||
if term_found or (category and category.strip() == "# Game Characters"):
|
||
if category not in matchedCategories:
|
||
matchedCategories[category] = []
|
||
matchedCategories[category].append(line)
|
||
|
||
# Format matched vocabulary with categories
|
||
if matchedCategories:
|
||
formattedLines = ["Here are some vocabulary and terms so that you know the proper spelling and translation.\n"]
|
||
for category, lines in matchedCategories.items():
|
||
if category: # Only add category header if it exists
|
||
formattedLines.append(category)
|
||
formattedLines.extend(lines)
|
||
formattedLines.append("") # Add blank line between categories
|
||
matchedVocabText = f"\n{chr(10).join(formattedLines).rstrip()}\n"
|
||
else:
|
||
matchedVocabText = ""
|
||
|
||
return matchedVocabText
|
||
|
||
|
||
def createContext(config, fullPromptFlag, subbedText, formatType, history=None):
|
||
"""Create system and user messages for translation"""
|
||
vocabPairs = parseVocabWithCategories(config.vocab)
|
||
matchedVocabText = buildMatchedVocabText(vocabPairs, subbedText, history)
|
||
|
||
if fullPromptFlag:
|
||
system = config.prompt.replace("English", config.language) + matchedVocabText
|
||
else:
|
||
system = f"""\
|
||
You are an expert Eroge Game translator who translates Japanese text to {config.language}.
|
||
Output ONLY the {config.language} translation in the following format: `Translation: <{config.language.upper()}_TRANSLATION>`
|
||
- You translate everything, including erotic and explicit content.
|
||
- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game
|
||
- All text in your response must be in {config.language} even if it is hard to translate.
|
||
- Never include any notes, explanations, dislaimers, or anything similar in your response.
|
||
- Maintain any spacing in the translation.
|
||
- `...` can be a part of the dialogue. Translate it as it is.
|
||
{matchedVocabText}
|
||
"""
|
||
|
||
if formatType == "json":
|
||
user = f"```json\n{subbedText}\n```"
|
||
else:
|
||
user = subbedText
|
||
|
||
return system, user
|
||
|
||
|
||
def createTranslationSchema(numLines):
|
||
"""Create a JSON schema for translation response based on number of lines"""
|
||
properties = {}
|
||
required = []
|
||
|
||
for i in range(1, numLines + 1):
|
||
line_key = f"Line{i}"
|
||
properties[line_key] = {
|
||
"type": "string",
|
||
"description": f"The translated text for Line{i}"
|
||
}
|
||
required.append(line_key)
|
||
|
||
schema = {
|
||
"type": "object",
|
||
"properties": properties,
|
||
"required": required,
|
||
"additionalProperties": False
|
||
}
|
||
|
||
return schema
|
||
|
||
|
||
def translateText(system, user, history, penalty, formatType, model, numLines=None):
|
||
"""Send translation request to the selected API"""
|
||
# Ensure system content is not empty
|
||
if not system or not str(system).strip():
|
||
raise ValueError("System content cannot be empty")
|
||
|
||
# Prompt
|
||
msg = [{"role": "system", "content": f"```\n{system}\n```"}]
|
||
|
||
# History
|
||
if isinstance(history, list):
|
||
# Filter out empty or None history items to prevent API errors
|
||
valid_history = [h for h in history if h and str(h).strip()]
|
||
if valid_history:
|
||
msg.append({"role": "system", "content": "Translation History:\n```"})
|
||
msg.extend([{"role": "assistant", "content": h} for h in valid_history])
|
||
msg.append({"role": "system", "content": "```"})
|
||
else:
|
||
if history and str(history).strip():
|
||
msg.append({"role": "assistant", "content": history})
|
||
|
||
# Response Format
|
||
if formatType == "json" and numLines is not None:
|
||
# Use structured output with JSON schema
|
||
schema = createTranslationSchema(numLines)
|
||
responseFormat = {
|
||
"type": "json_schema",
|
||
"json_schema": {
|
||
"name": "translation_response",
|
||
"strict": True,
|
||
"schema": schema
|
||
}
|
||
}
|
||
else:
|
||
responseFormat = {"type": "text"}
|
||
|
||
# Content to TL - ensure user content is not empty
|
||
if not user or not str(user).strip():
|
||
raise ValueError("User content cannot be empty")
|
||
msg.append({"role": "user", "content": f"```\n{user}\n```"})
|
||
|
||
# Debug: Check for any empty messages before API call
|
||
for i, message in enumerate(msg):
|
||
if not message.get("content") or not str(message.get("content")).strip():
|
||
raise ValueError(f"Message {i} has empty content: {message}")
|
||
|
||
# --- API Call Logic ---
|
||
api_provider = os.getenv("API_PROVIDER", "openai").lower()
|
||
|
||
# Base parameters for the API call
|
||
params = {
|
||
"model": model,
|
||
"response_format": responseFormat,
|
||
"messages": msg,
|
||
}
|
||
|
||
# Provider-specific parameters
|
||
if api_provider == "gemini":
|
||
params["temperature"] = 0
|
||
|
||
# Handle thinking budget for Gemini
|
||
thinking_budget_str = os.getenv("GEMINI_THINKING_BUDGET")
|
||
if thinking_budget_str:
|
||
try:
|
||
thinking_budget = int(thinking_budget_str)
|
||
params["extra_body"] = {
|
||
'google': {
|
||
'thinking_config': {
|
||
'thinking_budget': thinking_budget
|
||
}
|
||
}
|
||
}
|
||
except (ValueError, TypeError):
|
||
# Ignore if the value is not a valid integer
|
||
pass
|
||
|
||
# frequency_penalty is not supported via the OpenAI compatibility layer for Gemini
|
||
else: # Default to OpenAI behavior
|
||
if "gpt-5" in model:
|
||
params["reasoning_effort"] = "minimal"
|
||
else:
|
||
params["temperature"] = 0
|
||
params["frequency_penalty"] = penalty
|
||
|
||
# Call API
|
||
try:
|
||
response = openai.chat.completions.create(**params)
|
||
except Exception as e:
|
||
# If structured output fails, fallback to json_object
|
||
if formatType == "json" and "json_schema" in str(responseFormat):
|
||
responseFormat = {"type": "json_object"}
|
||
params["response_format"] = responseFormat
|
||
response = openai.chat.completions.create(**params)
|
||
else:
|
||
raise e
|
||
|
||
return response
|
||
|
||
|
||
def cleanTranslatedText(translatedText, language):
|
||
"""Clean and format translated text"""
|
||
placeholders = {
|
||
f"{language} Translation: ": "",
|
||
"Translation: ": "",
|
||
"っ": "",
|
||
"〜": "~",
|
||
"ッ": "",
|
||
"。": ".",
|
||
"「": '\"',
|
||
"」": '\"',
|
||
"- ": "-",
|
||
"—": "―",
|
||
"】": "]",
|
||
"【": "[",
|
||
"é": "e",
|
||
"’": "'",
|
||
"this guy": "this bastard",
|
||
"This guy": "This bastard",
|
||
"Placeholder Text": "",
|
||
"```json": "",
|
||
"```": "",
|
||
}
|
||
|
||
for target, replacement in placeholders.items():
|
||
translatedText = translatedText.replace(target, replacement)
|
||
|
||
# Remove Repeating Characters
|
||
pattern = re.compile(r"(.)\s*\1(?:\s*\1){" + str(20 - 1) + r",}")
|
||
translatedText = pattern.sub(lambda match: match.group(0).replace(" ", "")[:20], translatedText)
|
||
|
||
# Elongate Long Dashes (Since GPT Ignores them...)
|
||
translatedText = elongateCharacters(translatedText)
|
||
return translatedText
|
||
|
||
|
||
def elongateCharacters(text):
|
||
"""Replace ー sequences with elongated characters"""
|
||
# Define a pattern to match one character followed by two or more ー characters
|
||
pattern = r"(?<=(.))ー{2,}"
|
||
|
||
# Define a replacement function that elongates the captured character
|
||
def repl(match):
|
||
char = match.group(1) # The character before the ー sequence
|
||
count = len(match.group(0)) - 1 # Number of ー characters
|
||
return char * count # Replace ー sequence with the character repeated
|
||
|
||
# Use re.sub() to replace the pattern in the text
|
||
return re.sub(pattern, repl, text)
|
||
|
||
|
||
def extractTranslation(translatedTextList, isList, pbar=None):
|
||
"""Extract translation from JSON response.
|
||
|
||
This function is resilient to a few common model mistakes:
|
||
- Wraps output in code fences or outer quotes
|
||
- Uses smart quotes instead of straight quotes
|
||
- Inserts an extra leading quote in values (e.g. :""Word" -> :"Word")
|
||
- Trailing commas before } or ]
|
||
|
||
If strict JSON parsing fails, falls back to a regex-based extractor that
|
||
captures LineN values in numeric order.
|
||
"""
|
||
s = str(translatedTextList or "").strip()
|
||
|
||
# Fast exit
|
||
if not s:
|
||
return None
|
||
|
||
# Remove code fences if present
|
||
if s.startswith("```"):
|
||
s = re.sub(r"^```(?:json)?\s*", "", s, flags=re.IGNORECASE)
|
||
s = re.sub(r"\s*```$", "", s)
|
||
|
||
# Trim wrapping quotes around the whole JSON blob (common in logs)
|
||
if len(s) >= 2 and s[0] == s[-1] and s[0] in {'"', "'"}:
|
||
# Only strip if it still looks like JSON inside
|
||
if s[1:2] == "{" and s[-2:-1] == "}":
|
||
s = s[1:-1]
|
||
|
||
# Normalize quotes
|
||
s = s.replace("“", '"').replace("”", '"').replace("’", "'")
|
||
|
||
# Remove trailing commas before object/array closures
|
||
s = re.sub(r",(\s*[}\]])", r"\1", s)
|
||
|
||
# Repair common doubled leading quote in values: :""Word" -> :"Word"
|
||
# Ensure we don't alter legitimate empty strings (:"")
|
||
s = re.sub(r":\s*\"\"(?=[^\",}\]\s])", r':"', s)
|
||
|
||
# Attempt strict parse first
|
||
try:
|
||
lineDict = json.loads(s)
|
||
|
||
# Build list in numeric order if keys are LineN
|
||
numeric_keys = []
|
||
for k in lineDict.keys():
|
||
m = re.fullmatch(r"Line(\d+)", str(k))
|
||
if m:
|
||
numeric_keys.append(int(m.group(1)))
|
||
|
||
if numeric_keys:
|
||
stringList = [lineDict.get(f"Line{n}", "") for n in sorted(numeric_keys)]
|
||
else:
|
||
# Fallback to values order if no LineN keys found
|
||
stringList = list(lineDict.values())
|
||
|
||
return stringList if isList else (stringList[0] if stringList else None)
|
||
|
||
except Exception as e:
|
||
# Fallback: regex-based extraction tolerant to one or two opening quotes
|
||
# Captures escaped quotes within values too
|
||
try:
|
||
pairs = re.findall(r'"Line(\d+)"\s*:\s*"{1,2}((?:\\.|[^"\\])*)"', s)
|
||
if not pairs:
|
||
raise ValueError("No LineN pairs found")
|
||
|
||
# Sort numerically and unescape JSON string content
|
||
items = []
|
||
for n_str, v in sorted(((int(n), v) for n, v in pairs), key=lambda x: x[0]):
|
||
try:
|
||
# Decode JSON escapes reliably by round-tripping as a JSON string
|
||
decoded = json.loads(f'"{v}"')
|
||
except Exception:
|
||
decoded = v
|
||
items.append(decoded)
|
||
|
||
return items if isList else (items[0] if items else None)
|
||
except Exception as e2:
|
||
if pbar:
|
||
pbar.write(f"extractTranslation Error: {e2} after JSON error {e} on String {translatedTextList}")
|
||
return None
|
||
|
||
|
||
def calculateCost(inputTokens, outputTokens, model):
|
||
"""
|
||
Calculate the cost of translation based on token usage and model pricing.
|
||
|
||
Args:
|
||
inputTokens: Number of input tokens used
|
||
outputTokens: Number of output tokens generated
|
||
model: The model name string
|
||
|
||
Returns:
|
||
float: Total cost in USD
|
||
"""
|
||
pricing = getPricingConfig(model)
|
||
inputCost = (inputTokens / 1000000) * pricing["inputAPICost"]
|
||
outputCost = (outputTokens / 1000000) * pricing["outputAPICost"]
|
||
return inputCost + outputCost
|
||
|
||
|
||
def countTokens(system, user, history):
|
||
"""Count tokens for cost estimation"""
|
||
inputTotalTokens = 0
|
||
outputTotalTokens = 0
|
||
enc = tiktoken.encoding_for_model("gpt-4")
|
||
|
||
# Input
|
||
if isinstance(history, list):
|
||
for line in history:
|
||
inputTotalTokens += len(enc.encode(line))
|
||
else:
|
||
inputTotalTokens += len(enc.encode(history))
|
||
inputTotalTokens += len(enc.encode(system))
|
||
inputTotalTokens += len(enc.encode(user))
|
||
|
||
# Output
|
||
outputTotalTokens += round(len(enc.encode(user)) * 2.5)
|
||
|
||
return [inputTotalTokens, outputTotalTokens]
|
||
|
||
|
||
@retry(exceptions=Exception, tries=5, delay=5)
|
||
def translateAI(text, history, fullPromptFlag, config, filename=None, pbar=None, lock=None, mismatchList=None):
|
||
"""
|
||
Main translation function that can be used across all modules.
|
||
|
||
Args:
|
||
text: Text to translate (string or list)
|
||
history: Translation history for context
|
||
fullPromptFlag: Whether to use full prompt or simplified version
|
||
config: TranslationConfig instance with all settings
|
||
filename: Current filename being processed (for logging)
|
||
pbar: Progress bar instance (optional)
|
||
lock: Threading lock (optional)
|
||
mismatchList: List to track mismatched files (optional)
|
||
|
||
Returns:
|
||
[translatedText, totalTokens]
|
||
"""
|
||
if not text:
|
||
return [text, [0, 0]]
|
||
|
||
with open(config.logFilePath, "a+", encoding="utf-8") as logFile:
|
||
totalTokens = [0, 0]
|
||
|
||
if isinstance(text, list):
|
||
formatType = "json"
|
||
tList = batchList(text, config.batchSize)
|
||
else:
|
||
formatType = "text"
|
||
tList = [text]
|
||
|
||
for index, tItem in enumerate(tList):
|
||
# Check if text contains target language
|
||
if not re.search(config.langRegex, str(tItem)):
|
||
if pbar is not None:
|
||
pbar.update(len(tItem) if isinstance(tItem, list) else 1)
|
||
if isinstance(tItem, list):
|
||
for j in range(len(tItem)):
|
||
tItem[j] = cleanTranslatedText(tItem[j], config.language)
|
||
tList[index] = tItem
|
||
else:
|
||
tList[index] = cleanTranslatedText(tItem, config.language)
|
||
history = tItem[-config.maxHistory:] if isinstance(tItem, list) else tItem
|
||
continue
|
||
|
||
# Format for translation
|
||
if isinstance(tItem, list):
|
||
for j in range(len(tItem)):
|
||
if not tItem[j] or not str(tItem[j]).strip():
|
||
tItem[j] = "Placeholder Text"
|
||
payload = {f"Line{i+1}": string for i, string in enumerate(tItem)}
|
||
payload = json.dumps(payload, indent=4, ensure_ascii=False)
|
||
subbedT = payload
|
||
else:
|
||
# Check for empty/whitespace strings in non-list items
|
||
if not tItem or not str(tItem).strip():
|
||
subbedT = "Placeholder Text"
|
||
else:
|
||
subbedT = tItem
|
||
|
||
# Create context
|
||
system, user = createContext(config, fullPromptFlag, subbedT, formatType, history)
|
||
|
||
# Calculate estimate if in estimate mode
|
||
if config.estimateMode:
|
||
estimate = countTokens(system, user, history)
|
||
totalTokens[0] += estimate[0]
|
||
totalTokens[1] += estimate[1]
|
||
continue
|
||
|
||
# --- Translation and Validation Retry Block ---
|
||
max_retries = 2 # 1 initial attempt + 2 retries
|
||
final_translations = None
|
||
last_raw_translation = ""
|
||
numLines = len(tItem) if isinstance(tItem, list) else None
|
||
|
||
for attempt in range(max_retries + 1):
|
||
is_valid = True
|
||
|
||
# On retries, add a note to the system prompt
|
||
current_system = system
|
||
if attempt > 0:
|
||
current_system += f"\n\nIMPORTANT: Your previous attempt was incorrect or incomplete. Please ensure the entire output is translated to {config.language} and contains no untranslated characters. Translate the following text again, ensuring the JSON structure is correct."
|
||
if pbar:
|
||
pbar.write(f"Retrying translation... (Attempt {attempt + 1}/{max_retries + 1})")
|
||
|
||
# Translate
|
||
response = translateText(current_system, user, history, 0.05, formatType, config.model, numLines)
|
||
translatedText = response.choices[0].message.content
|
||
last_raw_translation = translatedText
|
||
|
||
# Update token count for this attempt
|
||
totalTokens[0] += response.usage.prompt_tokens
|
||
totalTokens[1] += response.usage.completion_tokens
|
||
|
||
# Clean the translation first for consistency
|
||
cleaned_text = cleanTranslatedText(translatedText, config.language)
|
||
|
||
# Process and validate translation result
|
||
if cleaned_text:
|
||
if isinstance(tItem, list):
|
||
extracted = extractTranslation(cleaned_text, True, pbar)
|
||
|
||
# Check 1: Mismatch in length
|
||
if extracted is None or len(tItem) != len(extracted):
|
||
is_valid = False
|
||
|
||
# Check 2: Untranslated content
|
||
else:
|
||
for line in extracted:
|
||
if re.search(config.langRegex, str(line)):
|
||
is_valid = False
|
||
break
|
||
|
||
if is_valid:
|
||
final_translations = extracted
|
||
else:
|
||
# Check for untranslated content in single string
|
||
if re.search(config.langRegex, cleaned_text):
|
||
is_valid = False
|
||
else:
|
||
final_translations = cleaned_text.replace("Placeholder Text", "")
|
||
else:
|
||
is_valid = False
|
||
if pbar: pbar.write(f"AI Refused: {tItem}\n")
|
||
|
||
# If translation is valid, break the retry loop
|
||
if is_valid:
|
||
break
|
||
|
||
# --- End of Retry Block ---
|
||
|
||
# After the loop, handle the final result
|
||
if final_translations is not None: # Success case
|
||
formatted_output = last_raw_translation
|
||
try:
|
||
parsed_json = json.loads(last_raw_translation)
|
||
formatted_output = json.dumps(parsed_json, indent=4, ensure_ascii=False)
|
||
except (json.JSONDecodeError, ValueError):
|
||
pass
|
||
logFile.write(f"Input:\n{subbedT}\n")
|
||
logFile.write(f"Output:\n{formatted_output}\n")
|
||
|
||
if isinstance(tItem, list):
|
||
tList[index] = final_translations
|
||
history = final_translations[-config.maxHistory:]
|
||
else:
|
||
tList[index] = final_translations
|
||
history = final_translations
|
||
|
||
if lock and pbar is not None:
|
||
with lock:
|
||
pbar.update(len(tItem) if isinstance(tItem, list) else 1)
|
||
|
||
else: # Failure case after all retries
|
||
if pbar: pbar.write(f"Translation failed after {max_retries + 1} attempts. Check mismatch log.")
|
||
|
||
formatted_mismatch_output = last_raw_translation
|
||
try:
|
||
parsed_json = json.loads(last_raw_translation)
|
||
formatted_mismatch_output = json.dumps(parsed_json, indent=4, ensure_ascii=False)
|
||
except (json.JSONDecodeError, ValueError):
|
||
pass
|
||
with open(config.mismatchLogPath, "a+", encoding="utf-8") as mismatchFile:
|
||
mismatchFile.write(f"Failed after retries: {filename}\n")
|
||
mismatchFile.write(f"Input:\n{subbedT}\n")
|
||
mismatchFile.write(f"Final Output:\n{formatted_mismatch_output}\n")
|
||
|
||
if filename and mismatchList is not None and filename not in mismatchList:
|
||
mismatchList.append(filename)
|
||
|
||
tList[index] = tItem
|
||
history = text[-config.maxHistory:] if isinstance(text, list) else text
|
||
|
||
# Combine if multilist
|
||
if tList and isinstance(tList[0], list):
|
||
tList = [t for sublist in tList for t in sublist]
|
||
|
||
# Return result
|
||
if formatType == "json":
|
||
return [tList, totalTokens]
|
||
else:
|
||
return [tList[0], totalTokens] |