# Libraries
import os
import re
import textwrap
import threading
import time
import traceback
import tiktoken
import openai
from pathlib import Path
from colorama import Fore
from dotenv import load_dotenv
from retry import retry
from tqdm import tqdm
# Open AI
load_dotenv()
if os.getenv("api").replace(" ", "") != "":
openai.base_url = os.getenv("api")
openai.organization = os.getenv("org")
openai.api_key = os.getenv("key")
# Globals
MODEL = os.getenv("model")
TIMEOUT = int(os.getenv("timeout"))
LANGUAGE = os.getenv("language").capitalize()
PROMPT = Path("prompt.txt").read_text(encoding="utf-8")
VOCAB = Path("vocab.txt").read_text(encoding="utf-8")
THREADS = int(os.getenv("threads"))
LOCK = threading.Lock()
WIDTH = int(os.getenv("width"))
LISTWIDTH = int(os.getenv("listWidth"))
NOTEWIDTH = 70
MAXHISTORY = 10
ESTIMATE = ""
TOKENS = [0, 0]
NAMESLIST = []
NAMES = False # Output a list of all the character names found
BRFLAG = False # If the game uses
instead
FIXTEXTWRAP = True # Overwrites textwrap
IGNORETLTEXT = False # Ignores all translated text.
MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong)
# tqdm Globals
BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}"
POSITION = 0
LEAVE = False
# Pricing - Depends on the model https://openai.com/pricing
# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request
# If you are getting a MISMATCH LENGTH error, lower the batch size.
if "gpt-3.5" in MODEL:
INPUTAPICOST = 0.002
OUTPUTAPICOST = 0.002
BATCHSIZE = 10
elif "gpt-4" in MODEL:
INPUTAPICOST = 0.0025
OUTPUTAPICOST = 0.01
BATCHSIZE = 40
else:
INPUTAPICOST = float(os.getenv("input_cost"))
OUTPUTAPICOST = float(os.getenv("output_cost"))
BATCHSIZE = int(os.getenv("batchsize"))
FREQUENCY_PENALTY = float(os.getenv("frequency_penalty"))
def handleIris(filename, estimate):
global ESTIMATE
ESTIMATE = estimate
if ESTIMATE:
start = time.time()
translatedData = openFiles(filename)
# Print Result
end = time.time()
tqdm.write(getResultString(translatedData, end - start, filename))
with LOCK:
TOKENS[0] += translatedData[1][0]
TOKENS[1] += translatedData[1][1]
# Print Total
totalString = getResultString(["", TOKENS, None], end - start, "TOTAL")
# Print any errors on maps
if len(MISMATCH) > 0:
return totalString + Fore.RED + f"\nMismatch Errors: {MISMATCH}" + Fore.RESET
else:
return totalString
else:
try:
with open("translated/" + filename, "w", encoding="cp932", errors="ignore") as outFile:
start = time.time()
translatedData = openFiles(filename)
# Print Result
end = time.time()
outFile.writelines(translatedData[0])
tqdm.write(getResultString(translatedData, end - start, filename))
with LOCK:
TOKENS[0] += translatedData[1][0]
TOKENS[1] += translatedData[1][1]
except Exception:
traceback.print_exc()
return "Fail"
return getResultString(["", TOKENS, None], end - start, "TOTAL")
def getResultString(translatedData, translationTime, filename):
# File Print String
totalTokenstring = (
Fore.YELLOW + "[Input: " + str(translatedData[1][0]) + "]"
"[Output: "
+ str(translatedData[1][1])
+ "]" "[Cost: ${:,.4f}".format((translatedData[1][0] * 0.001 * INPUTAPICOST) + (translatedData[1][1] * 0.001 * OUTPUTAPICOST))
+ "]"
)
timeString = Fore.BLUE + "[" + str(round(translationTime, 1)) + "s]"
if translatedData[2] == None:
# Success
return filename + ": " + totalTokenstring + timeString + Fore.GREEN + " \u2713 " + Fore.RESET
else:
# Fail
try:
raise translatedData[2]
except Exception as e:
traceback.print_exc()
errorString = str(e) + Fore.RED
return filename + ": " + totalTokenstring + timeString + Fore.RED + " \u2717 " + errorString + Fore.RESET
def openFiles(filename):
with open("files/" + filename, "r", encoding="shift_jis") as readFile:
translatedData = parseIris(readFile, filename)
# Delete lines marked for deletion
finalData = []
for line in translatedData[0]:
if line != "\\d\n":
finalData.append(line)
translatedData[0] = finalData
return translatedData
def parseIris(readFile, filename):
totalTokens = [0, 0]
# Read File into data
data = readFile.readlines()
# Create Progress Bar
with tqdm(bar_format=BAR_FORMAT, position=POSITION, leave=LEAVE) as pbar:
pbar.desc = filename
try:
result = translateIris(data, pbar, filename, [])
totalTokens[0] += result[0]
totalTokens[1] += result[1]
except Exception as e:
traceback.print_exc()
return [data, totalTokens, e]
return [data, totalTokens, None]
def translateIris(data, pbar, filename, translatedList):
stringList = []
currentGroup = []
tokens = [0, 0]
speaker = ""
voice = False
global LOCK, ESTIMATE
i = 0
while i < len(data):
voice = False
speaker = ""
if "#MSGVOICE" in data[i]:
i += 1
voice = True
voiceVar = data[i]
if "#MSG," in data[i] or "#MSG\n" in data[i] or voice == True:
i += 1
# Speaker
if re.search(r'^ ?([^#\/."、。*!!()\(\)\[\] \n]+)\n', data[i]) and len(data[i]) < 30:
match = re.search(r"(.*)", data[i])
if match != None:
speaker = match.group(1)
if speaker[0] == "\u3000":
speaker = speaker[1:]
response = getSpeaker(speaker, pbar, filename)
speaker = response[0]
tokens[0] += response[1][0]
tokens[1] += response[1][1]
if translatedList != []:
speaker = speaker.replace(" ", "\u3000")
data[i] = f"\u3000{speaker}\n"
else:
speaker = ""
i += 1
# Lines
match = re.search(r"(.*)", data[i])
if match != None and match.group(1) != "":
# Pass 1
if translatedList == []:
# Grab Consecutive Strings
jaString = data[i]
if data[i] != "\n":
if data[i][0] == "\u3000":
jaString = data[i][1:]
currentGroup.append(jaString)
i += 1
while data[i] != "\n":
jaString = data[i]
if data[i] != "\n":
jaString = data[i][1:]
currentGroup.append(jaString)
i += 1
# Join up 401 groups for better translation.
if len(currentGroup) > 0:
jaString = "".join(currentGroup)
currentGroup = []
# Remove any textwrap
jaString = jaString.replace("\n", " ")
# Temporarily convert spaces (For Textwrap Later)
jaString = jaString.replace("\u3000", " ")
# Add Speaker (If there is one)
if speaker != "":
jaString = f"{speaker}: {jaString}"
# Add String
stringList.append(jaString.strip())
# Pass 2
else:
# Insert Strings
while data[i] != "\n":
data.pop(i)
# Get Text
if translatedList:
translatedText = translatedList[0]
translatedList.pop(0)
if len(translatedList) <= 0:
translatedList = None
# Remove added speaker
translatedText = re.sub(r"^.+?:\s", "", translatedText)
# Textwrap
translatedText = textwrap.fill(translatedText, width=WIDTH)
translatedText = translatedText.replace("\n", "\n\u3000")
# Replace Whitespace and Commas
translatedText = translatedText.replace(", ", "、")
translatedText = translatedText.replace(",\u3000", "、")
translatedText = translatedText.replace(",", "、")
translatedText = translatedText.replace(" ", "\u3000")
# Set Data
# Game crashes on more than 3 lines. Will need to create a new MSG for long translations
if translatedText.count("\n") > 2:
# Split List
translatedTextList = splitNewlines(translatedText)
# MSG Voice
count = 0
for text in translatedTextList:
if count != 0:
if voice == True:
# MSG for each item in the list
data.insert(i, "#MSGVOICE,\n")
i += 1
data.insert(i, f"{voiceVar}")
i += 1
else:
data.insert(i, "#MSG,\n")
i += 1
if speaker:
data[i] = f"\u3000{speaker}\n"
i += 1
if text[0] == "\u3000":
data.insert(i, f"{text}\n")
else:
data.insert(i, f"\u3000{text}\n")
i += 1
count += 1
if data[i] != "\n":
data.insert(i, "\n")
data[i] = f"\n{data[i]}"
else:
data.insert(i, f"\u3000{translatedText}\n")
i += 1
if data[i] != "\n":
data[i] = f"\n{data[i]}"
elif "#SELECT" in data[i] and translatedList == []:
Iris = r"(.+?) +\d$"
i += 1
match = re.search(Iris, data[i])
if match:
choiceList = []
choiceList.append(match.group(1))
i += 1
match = re.search(Iris, data[i])
while match:
choiceList.append(match.group(1))
i += 1
match = re.search(Iris, data[i])
# Translate
question = stringList[len(stringList) - 1]
response = translateGPT(
choiceList,
f"Previous text for context: {question}\n\nThis will be a dialogue option",
True,
pbar,
filename,
)
tokens[0] += response[1][0]
tokens[1] += response[1][1]
choiceListTL = response[0]
# Set Data
i = i - len(choiceListTL)
for j in range(len(choiceListTL)):
# Replace Whitespace and Commas
choiceListTL[j] = choiceListTL[j].replace(", ", "、")
choiceListTL[j] = choiceListTL[j].replace(",\u3000", "、")
choiceListTL[j] = choiceListTL[j].replace(",", "、")
choiceListTL[j] = choiceListTL[j].replace(" ", "\u3000")
data[i] = data[i].replace(choiceList[j], choiceListTL[j])
i += 1
# Nothing relevant. Skip Line.
else:
i += 1
else:
i += 1
# EOF
if len(stringList) > 0:
# Set Progress
pbar.total = len(stringList)
pbar.refresh()
# Translate
response = translateGPT(stringList, "", True, pbar, filename)
tokens[0] += response[1][0]
tokens[1] += response[1][1]
translatedList = response[0]
# Set Strings
if len(stringList) == len(translatedList):
translateIris(data, pbar, filename, translatedList)
# Mismatch
else:
with LOCK:
if filename not in MISMATCH:
MISMATCH.append(filename)
return tokens
def splitNewlines(text):
parts = []
newline_count = 0 # Counts the number of newline characters encountered
start_index = 0 # Start index of the current string part
for i, char in enumerate(text):
if char == "\n":
newline_count += 1
if newline_count == 3:
# Append the string part from start_index to current index (inclusive)
parts.append(text[start_index : i + 1])
# Reset newline count and update start_index for the next string part
newline_count = 0
start_index = i + 1
# Edge case: if the text does not end with a newline, we still need to append the last part
if start_index < len(text):
parts.append(text[start_index:])
return parts
# Save some money and enter the character before translation
def getSpeaker(speaker, pbar, filename):
match speaker:
case "ファイン":
return ["Fine", [0, 0]]
case "":
return ["", [0, 0]]
case _:
# Store Speaker
if speaker not in str(NAMESLIST):
response = translateGPT(
speaker,
"Reply with only the " + LANGUAGE + " translation of the NPC name.",
False,
pbar,
filename,
)
response[0] = response[0].replace("'S", "'s")
speakerList = [speaker, response[0]]
NAMESLIST.append(speakerList)
return response
# Find Speaker
else:
for i in range(len(NAMESLIST)):
if speaker == NAMESLIST[i][0]:
return [NAMESLIST[i][1], [0, 0]]
return [speaker, [0, 0]]
def subVars(jaString):
jaString = jaString.replace("\u3000", " ")
# Nested
count = 0
nestedList = re.findall(r"[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]", jaString)
nestedList = set(nestedList)
if len(nestedList) != 0:
for icon in nestedList:
jaString = jaString.replace(icon, "[Nested_" + str(count) + "]")
count += 1
# Icons
count = 0
iconList = re.findall(r"[\\]+[iIkKwWaA]+\[[0-9]+\]", jaString)
iconList = set(iconList)
if len(iconList) != 0:
for icon in iconList:
jaString = jaString.replace(icon, "[Ascii_" + str(count) + "]")
count += 1
# Colors
count = 0
colorList = re.findall(r"[\\]+[cC]\[[0-9]+\]", jaString)
colorList = set(colorList)
if len(colorList) != 0:
for color in colorList:
jaString = jaString.replace(color, "[Color_" + str(count) + "]")
count += 1
# Names
count = 0
nameList = re.findall(r"[\\]+[nN]\[.+?\]+", jaString)
nameList = set(nameList)
if len(nameList) != 0:
for name in nameList:
jaString = jaString.replace(name, "[Noun_" + str(count) + "]")
count += 1
# Variables
count = 0
varList = re.findall(r"[\\]+[vV]\[[0-9]+\]", jaString)
varList = set(varList)
if len(varList) != 0:
for var in varList:
jaString = jaString.replace(var, "[Var_" + str(count) + "]")
count += 1
# Formatting
count = 0
formatList = re.findall(r"[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]", jaString)
formatList = set(formatList)
if len(formatList) != 0:
for var in formatList:
jaString = jaString.replace(var, "[FCode_" + str(count) + "]")
count += 1
# Put all lists in list and return
allList = [nestedList, iconList, colorList, nameList, varList, formatList]
return [jaString, allList]
def resubVars(translatedText, allList):
# Fix Spacing and ChatGPT Nonsense
matchList = re.findall(r"\[\s?.+?\s?\]", translatedText)
if len(matchList) > 0:
for match in matchList:
text = match.strip()
translatedText = translatedText.replace(match, text)
# Nested
count = 0
if len(allList[0]) != 0:
for var in allList[0]:
translatedText = translatedText.replace("[Nested_" + str(count) + "]", var)
count += 1
# Icons
count = 0
if len(allList[1]) != 0:
for var in allList[1]:
translatedText = translatedText.replace("[Ascii_" + str(count) + "]", var)
count += 1
# Colors
count = 0
if len(allList[2]) != 0:
for var in allList[2]:
translatedText = translatedText.replace("[Color_" + str(count) + "]", var)
count += 1
# Names
count = 0
if len(allList[3]) != 0:
for var in allList[3]:
translatedText = translatedText.replace("[Noun_" + str(count) + "]", var)
count += 1
# Vars
count = 0
if len(allList[4]) != 0:
for var in allList[4]:
translatedText = translatedText.replace("[Var_" + str(count) + "]", var)
count += 1
# Formatting
count = 0
if len(allList[5]) != 0:
for var in allList[5]:
translatedText = translatedText.replace("[FCode_" + str(count) + "]", var)
count += 1
return translatedText
def batchList(input_list, batch_size):
if not isinstance(batch_size, int) or batch_size <= 0:
raise ValueError("batch_size must be a positive integer")
return [input_list[i : i + batch_size] for i in range(0, len(input_list), batch_size)]
def createContext(fullPromptFlag, subbedT):
characters = "Game Characters:\n\
フィリア (Philia) - Female\n\
アルネット (Annett) - Female\n\
ラピュセナ (Rapusena) - Female\n\
リッカ (Rikka) - Female\n\
アンデリビア (Andelivia) - Female\n\
リリアブルム (Liliabloom) - Female\n\
カルナ (Karna) - Female\n\
ラフィング=スピア (Laughing Spear) - Female\n\
ノーラ (Nora) - Female\n\
"
system = (
PROMPT + VOCAB
if fullPromptFlag
else f"\
You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\
Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\
- You translate everything, including erotic and explicit content.\n\
- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\
- All text in your response must be in {LANGUAGE} even if it is hard to translate.\n\
- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\
- Maintain any spacing in the translation.\n\
- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\
- `...` can be a part of the dialogue. Translate it as it is.\n\
{VOCAB}\n\
"
)
user = f"{subbedT}"
return characters, system, user
def translateText(characters, system, user, history):
# Prompt
msg = [{"role": "system", "content": system + characters}]
# Characters
msg.append({"role": "system", "content": characters})
# History
if isinstance(history, list):
msg.extend([{"role": "system", "content": h} for h in history])
else:
msg.append({"role": "system", "content": history})
# Content to TL
msg.append({"role": "user", "content": f"{user}"})
response = openai.chat.completions.create(
temperature=0.1,
frequency_penalty=0.1,
model=MODEL,
messages=msg,
)
return response
def cleanTranslatedText(translatedText, varResponse):
placeholders = {
f"{LANGUAGE} Translation: ": "",
"Translation: ": "",
"っ": "",
"〜": "~",
"ッ": "",
"。": ".",
"Placeholder Text": "",
# Add more replacements as needed
}
for target, replacement in placeholders.items():
translatedText = translatedText.replace(target, replacement)
# Elongate Long Dashes (Since GPT Ignores them...)
translatedText = elongateCharacters(translatedText)
translatedText = resubVars(translatedText, varResponse[1])
return translatedText
def elongateCharacters(text):
# Define a pattern to match one character followed by one or more `ー` characters
# Using a positive lookbehind assertion to capture the preceding character
pattern = r"(?<=(.))ー+"
# Define a replacement function that elongates the captured character
def repl(match):
char = match.group(1) # The character before the ー sequence
count = len(match.group(0)) - 1 # Number of ー characters
return char * count # Replace ー sequence with the character repeated
# Use re.sub() to replace the pattern in the text
return re.sub(pattern, repl, text)
def extractTranslation(translatedTextList, is_list):
pattern = r"`?<[Ll]ine\d+>([\\]*.*?[\\]*?)<\/?[Ll]ine\d+>`?"
# If it's a batch (i.e., list), extract with tags; otherwise, return the single item.
if is_list:
matchList = re.findall(pattern, translatedTextList)
return matchList
else:
matchList = re.findall(pattern, translatedTextList)
return matchList[0][0] if matchList else translatedTextList
def countTokens(characters, system, user, history):
inputTotalTokens = 0
outputTotalTokens = 0
enc = tiktoken.encoding_for_model("gpt-4")
# Input
if isinstance(history, list):
for line in history:
inputTotalTokens += len(enc.encode(line))
else:
inputTotalTokens += len(enc.encode(history))
inputTotalTokens += len(enc.encode(system))
inputTotalTokens += len(enc.encode(characters))
inputTotalTokens += len(enc.encode(user))
# Output
outputTotalTokens += round(len(enc.encode(user)) * 2)
return [inputTotalTokens, outputTotalTokens]
@retry(exceptions=Exception, tries=5, delay=5)
def translateGPT(text, history, fullPromptFlag, pbar, filename):
mismatch = False
totalTokens = [0, 0]
if isinstance(text, list):
tList = batchList(text, BATCHSIZE)
else:
tList = [text]
for index, tItem in enumerate(tList):
# Before sending to translation, if we have a list of items, add the formatting
if isinstance(tItem, list):
payload = "\n".join([f"`{item}`" for i, item in enumerate(tItem)])
payload = re.sub(r"(<)(\/Line\d+>)", r"\1>Placeholder Text<\3", payload)
varResponse = subVars(payload)
subbedT = varResponse[0]
else:
varResponse = subVars(tItem)
subbedT = varResponse[0]
# Things to Check before starting translation
if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+", subbedT):
continue
# Create Message
characters, system, user = createContext(fullPromptFlag, subbedT)
# Calculate Estimate
if ESTIMATE:
estimate = countTokens(characters, system, user, history)
totalTokens[0] += estimate[0]
totalTokens[1] += estimate[1]
continue
# Translating
response = translateText(characters, system, user, history)
translatedText = response.choices[0].message.content
totalTokens[0] += response.usage.prompt_tokens
totalTokens[1] += response.usage.completion_tokens
# Formatting
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedText, True)
if len(tItem) != len(extractedTranslations):
# Mismatch. Try Again
response = translateText(characters, system, user, history)
translatedText = response.choices[0].message.content
totalTokens[0] += response.usage.prompt_tokens
totalTokens[1] += response.usage.completion_tokens
# Formatting
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedText, True)
if len(tItem) == len(extractedTranslations):
tList[index] = extractedTranslations
else:
MISMATCH.append(filename)
else:
tList[index] = extractedTranslations
# Create History
history = tList[index] # Update history if we have a list
pbar.update(len(tList[index]))
else:
# Ensure we're passing a single string to extractTranslation
extractedTranslations = extractTranslation(translatedText, False)
tList[index] = extractedTranslations
# Combine if multilist
if isinstance(tList[0], list):
tList = [t for sublist in tList for t in sublist]
# Return
if format == "json":
return [tList, totalTokens]
else:
return [tList[0], totalTokens]