DazedTL/modules/irissoft.py

752 lines
27 KiB
Python
Raw Blame History

This file contains invisible Unicode characters

This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

# Libraries
import os
import re
import textwrap
import threading
import time
import traceback
import tiktoken
import openai
from pathlib import Path
from colorama import Fore
from dotenv import load_dotenv
from retry import retry
from tqdm import tqdm
# Open AI
load_dotenv()
if os.getenv("api").replace(" ", "") != "":
openai.base_url = os.getenv("api")
openai.organization = os.getenv("org")
openai.api_key = os.getenv("key")
# Globals
MODEL = os.getenv("model")
TIMEOUT = int(os.getenv("timeout"))
LANGUAGE = os.getenv("language").capitalize()
PROMPT = Path("prompt.txt").read_text(encoding="utf-8")
VOCAB = Path("vocab.txt").read_text(encoding="utf-8")
THREADS = int(os.getenv("threads"))
LOCK = threading.Lock()
WIDTH = int(os.getenv("width"))
LISTWIDTH = int(os.getenv("listWidth"))
NOTEWIDTH = 70
MAXHISTORY = 10
ESTIMATE = ""
TOKENS = [0, 0]
NAMESLIST = []
NAMES = False # Output a list of all the character names found
BRFLAG = False # If the game uses <br> instead
FIXTEXTWRAP = True # Overwrites textwrap
IGNORETLTEXT = False # Ignores all translated text.
MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong)
# tqdm Globals
BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}"
POSITION = 0
LEAVE = False
# Pricing - Depends on the model https://openai.com/pricing
# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request
# If you are getting a MISMATCH LENGTH error, lower the batch size.
if "gpt-3.5" in MODEL:
INPUTAPICOST = 0.002
OUTPUTAPICOST = 0.002
BATCHSIZE = 10
elif "gpt-4" in MODEL:
INPUTAPICOST = 0.0025
OUTPUTAPICOST = 0.01
BATCHSIZE = 40
else:
INPUTAPICOST = float(os.getenv("input_cost"))
OUTPUTAPICOST = float(os.getenv("output_cost"))
BATCHSIZE = int(os.getenv("batchsize"))
FREQUENCY_PENALTY = float(os.getenv("frequency_penalty"))
def handleIris(filename, estimate):
global ESTIMATE
ESTIMATE = estimate
if ESTIMATE:
start = time.time()
translatedData = openFiles(filename)
# Print Result
end = time.time()
tqdm.write(getResultString(translatedData, end - start, filename))
with LOCK:
TOKENS[0] += translatedData[1][0]
TOKENS[1] += translatedData[1][1]
# Print Total
totalString = getResultString(["", TOKENS, None], end - start, "TOTAL")
# Print any errors on maps
if len(MISMATCH) > 0:
return totalString + Fore.RED + f"\nMismatch Errors: {MISMATCH}" + Fore.RESET
else:
return totalString
else:
try:
with open("translated/" + filename, "w", encoding="cp932", errors="ignore") as outFile:
start = time.time()
translatedData = openFiles(filename)
# Print Result
end = time.time()
outFile.writelines(translatedData[0])
tqdm.write(getResultString(translatedData, end - start, filename))
with LOCK:
TOKENS[0] += translatedData[1][0]
TOKENS[1] += translatedData[1][1]
except Exception:
traceback.print_exc()
return "Fail"
return getResultString(["", TOKENS, None], end - start, "TOTAL")
def getResultString(translatedData, translationTime, filename):
# File Print String
totalTokenstring = (
Fore.YELLOW + "[Input: " + str(translatedData[1][0]) + "]"
"[Output: "
+ str(translatedData[1][1])
+ "]" "[Cost: ${:,.4f}".format((translatedData[1][0] * 0.001 * INPUTAPICOST) + (translatedData[1][1] * 0.001 * OUTPUTAPICOST))
+ "]"
)
timeString = Fore.BLUE + "[" + str(round(translationTime, 1)) + "s]"
if translatedData[2] == None:
# Success
return filename + ": " + totalTokenstring + timeString + Fore.GREEN + " \u2713 " + Fore.RESET
else:
# Fail
try:
raise translatedData[2]
except Exception as e:
traceback.print_exc()
errorString = str(e) + Fore.RED
return filename + ": " + totalTokenstring + timeString + Fore.RED + " \u2717 " + errorString + Fore.RESET
def openFiles(filename):
with open("files/" + filename, "r", encoding="shift_jis") as readFile:
translatedData = parseIris(readFile, filename)
# Delete lines marked for deletion
finalData = []
for line in translatedData[0]:
if line != "\\d\n":
finalData.append(line)
translatedData[0] = finalData
return translatedData
def parseIris(readFile, filename):
totalTokens = [0, 0]
# Read File into data
data = readFile.readlines()
# Create Progress Bar
with tqdm(bar_format=BAR_FORMAT, position=POSITION, leave=LEAVE) as pbar:
pbar.desc = filename
try:
result = translateIris(data, pbar, filename, [])
totalTokens[0] += result[0]
totalTokens[1] += result[1]
except Exception as e:
traceback.print_exc()
return [data, totalTokens, e]
return [data, totalTokens, None]
def translateIris(data, pbar, filename, translatedList):
stringList = []
currentGroup = []
tokens = [0, 0]
speaker = ""
voice = False
global LOCK, ESTIMATE
i = 0
while i < len(data):
voice = False
speaker = ""
if "#MSGVOICE" in data[i]:
i += 1
voice = True
voiceVar = data[i]
if "#MSG," in data[i] or "#MSG\n" in data[i] or voice == True:
i += 1
# Speaker
if re.search(r'^ ?([^#\/."、。*!\(\)\[\] \n]+)\n', data[i]) and len(data[i]) < 30:
match = re.search(r"(.*)", data[i])
if match != None:
speaker = match.group(1)
if speaker[0] == "\u3000":
speaker = speaker[1:]
response = getSpeaker(speaker, pbar, filename)
speaker = response[0]
tokens[0] += response[1][0]
tokens[1] += response[1][1]
if translatedList != []:
speaker = speaker.replace(" ", "\u3000")
data[i] = f"\u3000{speaker}\n"
else:
speaker = ""
i += 1
# Lines
match = re.search(r"(.*)", data[i])
if match != None and match.group(1) != "":
# Pass 1
if translatedList == []:
# Grab Consecutive Strings
jaString = data[i]
if data[i] != "\n":
if data[i][0] == "\u3000":
jaString = data[i][1:]
currentGroup.append(jaString)
i += 1
while data[i] != "\n":
jaString = data[i]
if data[i] != "\n":
jaString = data[i][1:]
currentGroup.append(jaString)
i += 1
# Join up 401 groups for better translation.
if len(currentGroup) > 0:
jaString = "".join(currentGroup)
currentGroup = []
# Remove any textwrap
jaString = jaString.replace("\n", " ")
# Temporarily convert spaces (For Textwrap Later)
jaString = jaString.replace("\u3000", " ")
# Add Speaker (If there is one)
if speaker != "":
jaString = f"{speaker}: {jaString}"
# Add String
stringList.append(jaString.strip())
# Pass 2
else:
# Insert Strings
while data[i] != "\n":
data.pop(i)
# Get Text
if translatedList:
translatedText = translatedList[0]
translatedList.pop(0)
if len(translatedList) <= 0:
translatedList = None
# Remove added speaker
translatedText = re.sub(r"^.+?:\s", "", translatedText)
# Textwrap
translatedText = textwrap.fill(translatedText, width=WIDTH)
translatedText = translatedText.replace("\n", "\n\u3000")
# Replace Whitespace and Commas
translatedText = translatedText.replace(", ", "")
translatedText = translatedText.replace(",\u3000", "")
translatedText = translatedText.replace(",", "")
translatedText = translatedText.replace(" ", "\u3000")
# Set Data
# Game crashes on more than 3 lines. Will need to create a new MSG for long translations
if translatedText.count("\n") > 2:
# Split List
translatedTextList = splitNewlines(translatedText)
# MSG Voice
count = 0
for text in translatedTextList:
if count != 0:
if voice == True:
# MSG for each item in the list
data.insert(i, "#MSGVOICE,\n")
i += 1
data.insert(i, f"{voiceVar}")
i += 1
else:
data.insert(i, "#MSG,\n")
i += 1
if speaker:
data[i] = f"\u3000{speaker}\n"
i += 1
if text[0] == "\u3000":
data.insert(i, f"{text}\n")
else:
data.insert(i, f"\u3000{text}\n")
i += 1
count += 1
if data[i] != "\n":
data.insert(i, "\n")
data[i] = f"\n{data[i]}"
else:
data.insert(i, f"\u3000{translatedText}\n")
i += 1
if data[i] != "\n":
data[i] = f"\n{data[i]}"
elif "#SELECT" in data[i] and translatedList == []:
Iris = r"(.+?) +\d$"
i += 1
match = re.search(Iris, data[i])
if match:
choiceList = []
choiceList.append(match.group(1))
i += 1
match = re.search(Iris, data[i])
while match:
choiceList.append(match.group(1))
i += 1
match = re.search(Iris, data[i])
# Translate
question = stringList[len(stringList) - 1]
response = translateGPT(
choiceList,
f"Previous text for context: {question}\n\nThis will be a dialogue option",
True,
pbar,
filename,
)
tokens[0] += response[1][0]
tokens[1] += response[1][1]
choiceListTL = response[0]
# Set Data
i = i - len(choiceListTL)
for j in range(len(choiceListTL)):
# Replace Whitespace and Commas
choiceListTL[j] = choiceListTL[j].replace(", ", "")
choiceListTL[j] = choiceListTL[j].replace(",\u3000", "")
choiceListTL[j] = choiceListTL[j].replace(",", "")
choiceListTL[j] = choiceListTL[j].replace(" ", "\u3000")
data[i] = data[i].replace(choiceList[j], choiceListTL[j])
i += 1
# Nothing relevant. Skip Line.
else:
i += 1
else:
i += 1
# EOF
if len(stringList) > 0:
# Set Progress
pbar.total = len(stringList)
pbar.refresh()
# Translate
response = translateGPT(stringList, "", True, pbar, filename)
tokens[0] += response[1][0]
tokens[1] += response[1][1]
translatedList = response[0]
# Set Strings
if len(stringList) == len(translatedList):
translateIris(data, pbar, filename, translatedList)
# Mismatch
else:
with LOCK:
if filename not in MISMATCH:
MISMATCH.append(filename)
return tokens
def splitNewlines(text):
parts = []
newline_count = 0 # Counts the number of newline characters encountered
start_index = 0 # Start index of the current string part
for i, char in enumerate(text):
if char == "\n":
newline_count += 1
if newline_count == 3:
# Append the string part from start_index to current index (inclusive)
parts.append(text[start_index : i + 1])
# Reset newline count and update start_index for the next string part
newline_count = 0
start_index = i + 1
# Edge case: if the text does not end with a newline, we still need to append the last part
if start_index < len(text):
parts.append(text[start_index:])
return parts
# Save some money and enter the character before translation
def getSpeaker(speaker, pbar, filename):
match speaker:
case "ファイン":
return ["Fine", [0, 0]]
case "":
return ["", [0, 0]]
case _:
# Store Speaker
if speaker not in str(NAMESLIST):
response = translateGPT(
speaker,
"Reply with only the " + LANGUAGE + " translation of the NPC name.",
False,
pbar,
filename,
)
response[0] = response[0].replace("'S", "'s")
speakerList = [speaker, response[0]]
NAMESLIST.append(speakerList)
return response
# Find Speaker
else:
for i in range(len(NAMESLIST)):
if speaker == NAMESLIST[i][0]:
return [NAMESLIST[i][1], [0, 0]]
return [speaker, [0, 0]]
def subVars(jaString):
jaString = jaString.replace("\u3000", " ")
# Nested
count = 0
nestedList = re.findall(r"[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]", jaString)
nestedList = set(nestedList)
if len(nestedList) != 0:
for icon in nestedList:
jaString = jaString.replace(icon, "[Nested_" + str(count) + "]")
count += 1
# Icons
count = 0
iconList = re.findall(r"[\\]+[iIkKwWaA]+\[[0-9]+\]", jaString)
iconList = set(iconList)
if len(iconList) != 0:
for icon in iconList:
jaString = jaString.replace(icon, "[Ascii_" + str(count) + "]")
count += 1
# Colors
count = 0
colorList = re.findall(r"[\\]+[cC]\[[0-9]+\]", jaString)
colorList = set(colorList)
if len(colorList) != 0:
for color in colorList:
jaString = jaString.replace(color, "[Color_" + str(count) + "]")
count += 1
# Names
count = 0
nameList = re.findall(r"[\\]+[nN]\[.+?\]+", jaString)
nameList = set(nameList)
if len(nameList) != 0:
for name in nameList:
jaString = jaString.replace(name, "[Noun_" + str(count) + "]")
count += 1
# Variables
count = 0
varList = re.findall(r"[\\]+[vV]\[[0-9]+\]", jaString)
varList = set(varList)
if len(varList) != 0:
for var in varList:
jaString = jaString.replace(var, "[Var_" + str(count) + "]")
count += 1
# Formatting
count = 0
formatList = re.findall(r"[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]", jaString)
formatList = set(formatList)
if len(formatList) != 0:
for var in formatList:
jaString = jaString.replace(var, "[FCode_" + str(count) + "]")
count += 1
# Put all lists in list and return
allList = [nestedList, iconList, colorList, nameList, varList, formatList]
return [jaString, allList]
def resubVars(translatedText, allList):
# Fix Spacing and ChatGPT Nonsense
matchList = re.findall(r"\[\s?.+?\s?\]", translatedText)
if len(matchList) > 0:
for match in matchList:
text = match.strip()
translatedText = translatedText.replace(match, text)
# Nested
count = 0
if len(allList[0]) != 0:
for var in allList[0]:
translatedText = translatedText.replace("[Nested_" + str(count) + "]", var)
count += 1
# Icons
count = 0
if len(allList[1]) != 0:
for var in allList[1]:
translatedText = translatedText.replace("[Ascii_" + str(count) + "]", var)
count += 1
# Colors
count = 0
if len(allList[2]) != 0:
for var in allList[2]:
translatedText = translatedText.replace("[Color_" + str(count) + "]", var)
count += 1
# Names
count = 0
if len(allList[3]) != 0:
for var in allList[3]:
translatedText = translatedText.replace("[Noun_" + str(count) + "]", var)
count += 1
# Vars
count = 0
if len(allList[4]) != 0:
for var in allList[4]:
translatedText = translatedText.replace("[Var_" + str(count) + "]", var)
count += 1
# Formatting
count = 0
if len(allList[5]) != 0:
for var in allList[5]:
translatedText = translatedText.replace("[FCode_" + str(count) + "]", var)
count += 1
return translatedText
def batchList(input_list, batch_size):
if not isinstance(batch_size, int) or batch_size <= 0:
raise ValueError("batch_size must be a positive integer")
return [input_list[i : i + batch_size] for i in range(0, len(input_list), batch_size)]
def createContext(fullPromptFlag, subbedT):
characters = "Game Characters:\n\
フィリア (Philia) - Female\n\
アルネット (Annett) - Female\n\
ラピュセナ (Rapusena) - Female\n\
リッカ (Rikka) - Female\n\
アンデリビア (Andelivia) - Female\n\
リリアブルム (Liliabloom) - Female\n\
カルナ (Karna) - Female\n\
ラフィング=スピア (Laughing Spear) - Female\n\
ノーラ (Nora) - Female\n\
"
system = (
PROMPT + VOCAB
if fullPromptFlag
else f"\
You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\
Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\
- You translate everything, including erotic and explicit content.\n\
- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\
- All text in your response must be in {LANGUAGE} even if it is hard to translate.\n\
- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\
- Maintain any spacing in the translation.\n\
- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\
- `...` can be a part of the dialogue. Translate it as it is.\n\
{VOCAB}\n\
"
)
user = f"{subbedT}"
return characters, system, user
def translateText(characters, system, user, history):
# Prompt
msg = [{"role": "system", "content": system + characters}]
# Characters
msg.append({"role": "system", "content": characters})
# History
if isinstance(history, list):
msg.extend([{"role": "system", "content": h} for h in history])
else:
msg.append({"role": "system", "content": history})
# Content to TL
msg.append({"role": "user", "content": f"{user}"})
response = openai.chat.completions.create(
temperature=0.1,
frequency_penalty=0.1,
model=MODEL,
messages=msg,
)
return response
def cleanTranslatedText(translatedText, varResponse):
placeholders = {
f"{LANGUAGE} Translation: ": "",
"Translation: ": "",
"": "",
"": "~",
"": "",
"": ".",
"Placeholder Text": "",
# Add more replacements as needed
}
for target, replacement in placeholders.items():
translatedText = translatedText.replace(target, replacement)
# Elongate Long Dashes (Since GPT Ignores them...)
translatedText = elongateCharacters(translatedText)
translatedText = resubVars(translatedText, varResponse[1])
return translatedText
def elongateCharacters(text):
# Define a pattern to match one character followed by one or more `ー` characters
# Using a positive lookbehind assertion to capture the preceding character
pattern = r"(?<=(.))ー+"
# Define a replacement function that elongates the captured character
def repl(match):
char = match.group(1) # The character before the ー sequence
count = len(match.group(0)) - 1 # Number of ー characters
return char * count # Replace ー sequence with the character repeated
# Use re.sub() to replace the pattern in the text
return re.sub(pattern, repl, text)
def extractTranslation(translatedTextList, is_list):
pattern = r"`?<[Ll]ine\d+>([\\]*.*?[\\]*?)<\/?[Ll]ine\d+>`?"
# If it's a batch (i.e., list), extract with tags; otherwise, return the single item.
if is_list:
matchList = re.findall(pattern, translatedTextList)
return matchList
else:
matchList = re.findall(pattern, translatedTextList)
return matchList[0][0] if matchList else translatedTextList
def countTokens(characters, system, user, history):
inputTotalTokens = 0
outputTotalTokens = 0
enc = tiktoken.encoding_for_model("gpt-4")
# Input
if isinstance(history, list):
for line in history:
inputTotalTokens += len(enc.encode(line))
else:
inputTotalTokens += len(enc.encode(history))
inputTotalTokens += len(enc.encode(system))
inputTotalTokens += len(enc.encode(characters))
inputTotalTokens += len(enc.encode(user))
# Output
outputTotalTokens += round(len(enc.encode(user)) * 2)
return [inputTotalTokens, outputTotalTokens]
@retry(exceptions=Exception, tries=5, delay=5)
def translateGPT(text, history, fullPromptFlag, pbar, filename):
mismatch = False
totalTokens = [0, 0]
if isinstance(text, list):
tList = batchList(text, BATCHSIZE)
else:
tList = [text]
for index, tItem in enumerate(tList):
# Before sending to translation, if we have a list of items, add the formatting
if isinstance(tItem, list):
payload = "\n".join([f"`<Line{i}>{item}</Line{i}>`" for i, item in enumerate(tItem)])
payload = re.sub(r"(<Line\d+)(><)(\/Line\d+>)", r"\1>Placeholder Text<\3", payload)
varResponse = subVars(payload)
subbedT = varResponse[0]
else:
varResponse = subVars(tItem)
subbedT = varResponse[0]
# Things to Check before starting translation
if not re.search(r"[一-龠ぁ-ゔァ-ヴーa---]+", subbedT):
continue
# Create Message
characters, system, user = createContext(fullPromptFlag, subbedT)
# Calculate Estimate
if ESTIMATE:
estimate = countTokens(characters, system, user, history)
totalTokens[0] += estimate[0]
totalTokens[1] += estimate[1]
continue
# Translating
response = translateText(characters, system, user, history)
translatedText = response.choices[0].message.content
totalTokens[0] += response.usage.prompt_tokens
totalTokens[1] += response.usage.completion_tokens
# Formatting
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedText, True)
if len(tItem) != len(extractedTranslations):
# Mismatch. Try Again
response = translateText(characters, system, user, history)
translatedText = response.choices[0].message.content
totalTokens[0] += response.usage.prompt_tokens
totalTokens[1] += response.usage.completion_tokens
# Formatting
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedText, True)
if len(tItem) == len(extractedTranslations):
tList[index] = extractedTranslations
else:
MISMATCH.append(filename)
else:
tList[index] = extractedTranslations
# Create History
history = tList[index] # Update history if we have a list
pbar.update(len(tList[index]))
else:
# Ensure we're passing a single string to extractTranslation
extractedTranslations = extractTranslation(translatedText, False)
tList[index] = extractedTranslations
# Combine if multilist
if isinstance(tList[0], list):
tList = [t for sublist in tList for t in sublist]
# Return
if format == "json":
return [tList, totalTokens]
else:
return [tList[0], totalTokens]