# Libraries import os import re import util.dazedwrap as dazedwrap import threading import time import traceback import tiktoken import openai from pathlib import Path from colorama import Fore from dotenv import load_dotenv from retry import retry from tqdm import tqdm # Open AI load_dotenv() if os.getenv("api").replace(" ", "") != "": openai.base_url = os.getenv("api") openai.organization = os.getenv("org") openai.api_key = os.getenv("key") # Globals MODEL = os.getenv("model") TIMEOUT = int(os.getenv("timeout")) LANGUAGE = os.getenv("language").capitalize() PROMPT = Path("prompt.txt").read_text(encoding="utf-8") VOCAB = Path("vocab.txt").read_text(encoding="utf-8") THREADS = int(os.getenv("threads")) LOCK = threading.Lock() WIDTH = int(os.getenv("width")) LISTWIDTH = int(os.getenv("listWidth")) NOTEWIDTH = 70 MAXHISTORY = 10 ESTIMATE = "" TOKENS = [0, 0] NAMESLIST = [] NAMES = False # Output a list of all the character names found BRFLAG = False # If the game uses
instead FIXTEXTWRAP = True # Overwrites textwrap IGNORETLTEXT = False # Ignores all translated text. MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) # tqdm Globals BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" POSITION = 0 LEAVE = False # Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. if "gpt-3.5" in MODEL: INPUTAPICOST = 3.00 OUTPUTAPICOST = 5.00 BATCHSIZE = 10 FREQUENCY_PENALTY = 0.2 elif "gpt-4" in MODEL: INPUTAPICOST = 2.0 OUTPUTAPICOST = 8.00 BATCHSIZE = 30 FREQUENCY_PENALTY = 0.05 elif "deepseek" in MODEL: INPUTAPICOST = 0.27 OUTPUTAPICOST = 1.10 BATCHSIZE = 30 FREQUENCY_PENALTY = 0.05 else: INPUTAPICOST = float(os.getenv("input_cost")) OUTPUTAPICOST = float(os.getenv("output_cost")) BATCHSIZE = int(os.getenv("batchsize")) FREQUENCY_PENALTY = float(os.getenv("frequency_penalty")) def handleIris(filename, estimate): global ESTIMATE ESTIMATE = estimate if ESTIMATE: start = time.time() translatedData = openFiles(filename) # Print Result end = time.time() tqdm.write(getResultString(translatedData, end - start, filename)) with LOCK: TOKENS[0] += translatedData[1][0] TOKENS[1] += translatedData[1][1] # Print Total totalString = getResultString(["", TOKENS, None], end - start, "TOTAL") # Print any errors on maps if len(MISMATCH) > 0: return totalString + Fore.RED + f"\nMismatch Errors: {MISMATCH}" + Fore.RESET else: return totalString else: try: with open("translated/" + filename, "w", encoding="cp932", errors="ignore") as outFile: start = time.time() translatedData = openFiles(filename) # Print Result end = time.time() outFile.writelines(translatedData[0]) tqdm.write(getResultString(translatedData, end - start, filename)) with LOCK: TOKENS[0] += translatedData[1][0] TOKENS[1] += translatedData[1][1] except Exception: traceback.print_exc() return "Fail" return getResultString(["", TOKENS, None], end - start, "TOTAL") def getResultString(translatedData, translationTime, filename): # File Print String totalTokenstring = ( Fore.YELLOW + "[Input: " + str(translatedData[1][0]) + "]" "[Output: " + str(translatedData[1][1]) + "]" "[Cost: ${:,.4f}".format(((translatedData[1][0] / 1000000) * INPUTAPICOST) + ((translatedData[1][1] / 1000000) * OUTPUTAPICOST)) + "]" ) timeString = Fore.BLUE + "[" + str(round(translationTime, 1)) + "s]" if translatedData[2] == None: # Success return filename + ": " + totalTokenstring + timeString + Fore.GREEN + " \u2713 " + Fore.RESET else: # Fail try: raise translatedData[2] except Exception as e: traceback.print_exc() errorString = str(e) + Fore.RED return filename + ": " + totalTokenstring + timeString + Fore.RED + " \u2717 " + errorString + Fore.RESET def openFiles(filename): with open("files/" + filename, "r", encoding="shift_jis") as readFile: translatedData = parseIris(readFile, filename) # Delete lines marked for deletion finalData = [] for line in translatedData[0]: if line != "\\d\n": finalData.append(line) translatedData[0] = finalData return translatedData def parseIris(readFile, filename): totalTokens = [0, 0] # Read File into data data = readFile.readlines() # Create Progress Bar with tqdm(bar_format=BAR_FORMAT, position=POSITION, leave=LEAVE) as pbar: pbar.desc = filename try: result = translateIris(data, pbar, filename, []) totalTokens[0] += result[0] totalTokens[1] += result[1] except Exception as e: traceback.print_exc() return [data, totalTokens, e] return [data, totalTokens, None] def translateIris(data, pbar, filename, translatedList): stringList = [] currentGroup = [] tokens = [0, 0] speaker = "" voice = False global LOCK, ESTIMATE i = 0 while i < len(data): voice = False speaker = "" if "#MSGVOICE" in data[i]: i += 1 voice = True voiceVar = data[i] if "#MSG," in data[i] or "#MSG\n" in data[i] or voice == True: i += 1 # Speaker if re.search(r'^ ?([^#\/."、。*!!()\(\)\[\] \n]+)\n', data[i]) and len(data[i]) < 30: match = re.search(r"(.*)", data[i]) if match != None: speaker = match.group(1) if speaker[0] == "\u3000": speaker = speaker[1:] response = getSpeaker(speaker, pbar, filename) speaker = response[0] tokens[0] += response[1][0] tokens[1] += response[1][1] if translatedList != []: speaker = speaker.replace(" ", "\u3000") data[i] = f"\u3000{speaker}\n" else: speaker = "" i += 1 # Lines match = re.search(r"(.*)", data[i]) if match != None and match.group(1) != "": # Pass 1 if translatedList == []: # Grab Consecutive Strings jaString = data[i] if data[i] != "\n": if data[i][0] == "\u3000": jaString = data[i][1:] currentGroup.append(jaString) i += 1 while data[i] != "\n": jaString = data[i] if data[i] != "\n": jaString = data[i][1:] currentGroup.append(jaString) i += 1 # Join up 401 groups for better translation. if len(currentGroup) > 0: jaString = "".join(currentGroup) currentGroup = [] # Remove any textwrap jaString = jaString.replace("\n", " ") # Temporarily convert spaces (For Textwrap Later) jaString = jaString.replace("\u3000", " ") # Add Speaker (If there is one) if speaker != "": jaString = f"{speaker}: {jaString}" # Add String stringList.append(jaString.strip()) # Pass 2 else: # Insert Strings while data[i] != "\n": data.pop(i) # Get Text if translatedList: translatedText = translatedList[0] translatedList.pop(0) if len(translatedList) <= 0: translatedList = None # Remove added speaker translatedText = re.sub(r"^.+?:\s", "", translatedText) # Textwrap translatedText = dazedwrap.wrapText(translatedText, width=WIDTH) translatedText = translatedText.replace("\n", "\n\u3000") # Replace Whitespace and Commas translatedText = translatedText.replace(", ", "、") translatedText = translatedText.replace(",\u3000", "、") translatedText = translatedText.replace(",", "、") translatedText = translatedText.replace(" ", "\u3000") # Set Data # Game crashes on more than 3 lines. Will need to create a new MSG for long translations if translatedText.count("\n") > 2: # Split List translatedTextList = splitNewlines(translatedText) # MSG Voice count = 0 for text in translatedTextList: if count != 0: if voice == True: # MSG for each item in the list data.insert(i, "#MSGVOICE,\n") i += 1 data.insert(i, f"{voiceVar}") i += 1 else: data.insert(i, "#MSG,\n") i += 1 if speaker: data[i] = f"\u3000{speaker}\n" i += 1 if text[0] == "\u3000": data.insert(i, f"{text}\n") else: data.insert(i, f"\u3000{text}\n") i += 1 count += 1 if data[i] != "\n": data.insert(i, "\n") data[i] = f"\n{data[i]}" else: data.insert(i, f"\u3000{translatedText}\n") i += 1 if data[i] != "\n": data[i] = f"\n{data[i]}" elif "#SELECT" in data[i] and translatedList == []: Iris = r"(.+?) +\d$" i += 1 match = re.search(Iris, data[i]) if match: choiceList = [] choiceList.append(match.group(1)) i += 1 match = re.search(Iris, data[i]) while match: choiceList.append(match.group(1)) i += 1 match = re.search(Iris, data[i]) # Translate question = stringList[len(stringList) - 1] response = translateGPT( choiceList, f"Previous text for context: {question}\n\nThis will be a dialogue option", True, pbar, filename, ) tokens[0] += response[1][0] tokens[1] += response[1][1] choiceListTL = response[0] # Set Data i = i - len(choiceListTL) for j in range(len(choiceListTL)): # Replace Whitespace and Commas choiceListTL[j] = choiceListTL[j].replace(", ", "、") choiceListTL[j] = choiceListTL[j].replace(",\u3000", "、") choiceListTL[j] = choiceListTL[j].replace(",", "、") choiceListTL[j] = choiceListTL[j].replace(" ", "\u3000") data[i] = data[i].replace(choiceList[j], choiceListTL[j]) i += 1 # Nothing relevant. Skip Line. else: i += 1 else: i += 1 # EOF if len(stringList) > 0: # Set Progress pbar.total = len(stringList) pbar.refresh() # Translate response = translateGPT(stringList, "", True, pbar, filename) tokens[0] += response[1][0] tokens[1] += response[1][1] translatedList = response[0] # Set Strings if len(stringList) == len(translatedList): translateIris(data, pbar, filename, translatedList) # Mismatch else: with LOCK: if filename not in MISMATCH: MISMATCH.append(filename) return tokens def splitNewlines(text): parts = [] newline_count = 0 # Counts the number of newline characters encountered start_index = 0 # Start index of the current string part for i, char in enumerate(text): if char == "\n": newline_count += 1 if newline_count == 3: # Append the string part from start_index to current index (inclusive) parts.append(text[start_index : i + 1]) # Reset newline count and update start_index for the next string part newline_count = 0 start_index = i + 1 # Edge case: if the text does not end with a newline, we still need to append the last part if start_index < len(text): parts.append(text[start_index:]) return parts # Save some money and enter the character before translation def getSpeaker(speaker, pbar, filename): match speaker: case "ファイン": return ["Fine", [0, 0]] case "": return ["", [0, 0]] case _: # Store Speaker if speaker not in str(NAMESLIST): response = translateGPT( speaker, "Reply with only the " + LANGUAGE + " translation of the NPC name.", False, pbar, filename, ) response[0] = response[0].replace("'S", "'s") speakerList = [speaker, response[0]] NAMESLIST.append(speakerList) return response # Find Speaker else: for i in range(len(NAMESLIST)): if speaker == NAMESLIST[i][0]: return [NAMESLIST[i][1], [0, 0]] return [speaker, [0, 0]] def subVars(jaString): jaString = jaString.replace("\u3000", " ") # Nested count = 0 nestedList = re.findall(r"[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]", jaString) nestedList = set(nestedList) if len(nestedList) != 0: for icon in nestedList: jaString = jaString.replace(icon, "[Nested_" + str(count) + "]") count += 1 # Icons count = 0 iconList = re.findall(r"[\\]+[iIkKwWaA]+\[[0-9]+\]", jaString) iconList = set(iconList) if len(iconList) != 0: for icon in iconList: jaString = jaString.replace(icon, "[Ascii_" + str(count) + "]") count += 1 # Colors count = 0 colorList = re.findall(r"[\\]+[cC]\[[0-9]+\]", jaString) colorList = set(colorList) if len(colorList) != 0: for color in colorList: jaString = jaString.replace(color, "[Color_" + str(count) + "]") count += 1 # Names count = 0 nameList = re.findall(r"[\\]+[nN]\[.+?\]+", jaString) nameList = set(nameList) if len(nameList) != 0: for name in nameList: jaString = jaString.replace(name, "[Noun_" + str(count) + "]") count += 1 # Variables count = 0 varList = re.findall(r"[\\]+[vV]\[[0-9]+\]", jaString) varList = set(varList) if len(varList) != 0: for var in varList: jaString = jaString.replace(var, "[Var_" + str(count) + "]") count += 1 # Formatting count = 0 formatList = re.findall(r"[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]", jaString) formatList = set(formatList) if len(formatList) != 0: for var in formatList: jaString = jaString.replace(var, "[FCode_" + str(count) + "]") count += 1 # Put all lists in list and return allList = [nestedList, iconList, colorList, nameList, varList, formatList] return [jaString, allList] def resubVars(translatedText, allList): # Fix Spacing and ChatGPT Nonsense matchList = re.findall(r"\[\s?.+?\s?\]", translatedText) if len(matchList) > 0: for match in matchList: text = match.strip() translatedText = translatedText.replace(match, text) # Nested count = 0 if len(allList[0]) != 0: for var in allList[0]: translatedText = translatedText.replace("[Nested_" + str(count) + "]", var) count += 1 # Icons count = 0 if len(allList[1]) != 0: for var in allList[1]: translatedText = translatedText.replace("[Ascii_" + str(count) + "]", var) count += 1 # Colors count = 0 if len(allList[2]) != 0: for var in allList[2]: translatedText = translatedText.replace("[Color_" + str(count) + "]", var) count += 1 # Names count = 0 if len(allList[3]) != 0: for var in allList[3]: translatedText = translatedText.replace("[Noun_" + str(count) + "]", var) count += 1 # Vars count = 0 if len(allList[4]) != 0: for var in allList[4]: translatedText = translatedText.replace("[Var_" + str(count) + "]", var) count += 1 # Formatting count = 0 if len(allList[5]) != 0: for var in allList[5]: translatedText = translatedText.replace("[FCode_" + str(count) + "]", var) count += 1 return translatedText def batchList(input_list, batch_size): if not isinstance(batch_size, int) or batch_size <= 0: raise ValueError("batch_size must be a positive integer") return [input_list[i : i + batch_size] for i in range(0, len(input_list), batch_size)] def createContext(fullPromptFlag, subbedT): characters = "Game Characters:\n\ フィリア (Philia) - Female\n\ アルネット (Annett) - Female\n\ ラピュセナ (Rapusena) - Female\n\ リッカ (Rikka) - Female\n\ アンデリビア (Andelivia) - Female\n\ リリアブルム (Liliabloom) - Female\n\ カルナ (Karna) - Female\n\ ラフィング=スピア (Laughing Spear) - Female\n\ ノーラ (Nora) - Female\n\ " system = ( PROMPT + VOCAB if fullPromptFlag else f"\ You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\ Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\ - You translate everything, including erotic and explicit content.\n\ - 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\ - All text in your response must be in {LANGUAGE} even if it is hard to translate.\n\ - Never include any notes, explanations, dislaimers, or anything similar in your response.\n\ - Maintain any spacing in the translation.\n\ - Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\ - `...` can be a part of the dialogue. Translate it as it is.\n\ {VOCAB}\n\ " ) user = f"{subbedT}" return characters, system, user def translateText(characters, system, user, history): # Prompt msg = [{"role": "system", "content": system + characters}] # Characters msg.append({"role": "system", "content": characters}) # History if isinstance(history, list): msg.extend([{"role": "system", "content": h} for h in history]) else: msg.append({"role": "system", "content": history}) # Content to TL msg.append({"role": "user", "content": f"{user}"}) response = openai.chat.completions.create( temperature=0.1, frequency_penalty=0.1, model=MODEL, messages=msg, ) return response def cleanTranslatedText(translatedText, varResponse): placeholders = { f"{LANGUAGE} Translation: ": "", "Translation: ": "", "っ": "", "〜": "~", "ッ": "", "。": ".", "Placeholder Text": "", # Add more replacements as needed } for target, replacement in placeholders.items(): translatedText = translatedText.replace(target, replacement) # Elongate Long Dashes (Since GPT Ignores them...) translatedText = elongateCharacters(translatedText) translatedText = resubVars(translatedText, varResponse[1]) return translatedText def elongateCharacters(text): # Define a pattern to match one character followed by one or more `ー` characters # Using a positive lookbehind assertion to capture the preceding character pattern = r"(?<=(.))ー+" # Define a replacement function that elongates the captured character def repl(match): char = match.group(1) # The character before the ー sequence count = len(match.group(0)) - 1 # Number of ー characters return char * count # Replace ー sequence with the character repeated # Use re.sub() to replace the pattern in the text return re.sub(pattern, repl, text) def extractTranslation(translatedTextList, is_list): pattern = r"`?<[Ll]ine\d+>([\\]*.*?[\\]*?)<\/?[Ll]ine\d+>`?" # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. if is_list: matchList = re.findall(pattern, translatedTextList) return matchList else: matchList = re.findall(pattern, translatedTextList) return matchList[0][0] if matchList else translatedTextList def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 enc = tiktoken.encoding_for_model("gpt-4") # Input if isinstance(history, list): for line in history: inputTotalTokens += len(enc.encode(line)) else: inputTotalTokens += len(enc.encode(history)) inputTotalTokens += len(enc.encode(system)) inputTotalTokens += len(enc.encode(characters)) inputTotalTokens += len(enc.encode(user)) # Output outputTotalTokens += round(len(enc.encode(user)) * 2) return [inputTotalTokens, outputTotalTokens] @retry(exceptions=Exception, tries=5, delay=5) def translateGPT(text, history, fullPromptFlag, pbar, filename): mismatch = False totalTokens = [0, 0] if isinstance(text, list): tList = batchList(text, BATCHSIZE) else: tList = [text] for index, tItem in enumerate(tList): # Before sending to translation, if we have a list of items, add the formatting if isinstance(tItem, list): payload = "\n".join([f"`{item}`" for i, item in enumerate(tItem)]) payload = re.sub(r"(<)(\/Line\d+>)", r"\1>Placeholder Text<\3", payload) varResponse = subVars(payload) subbedT = varResponse[0] else: varResponse = subVars(tItem) subbedT = varResponse[0] # Things to Check before starting translation if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+", subbedT): continue # Create Message characters, system, user = createContext(fullPromptFlag, subbedT) # Calculate Estimate if ESTIMATE: estimate = countTokens(characters, system, user, history) totalTokens[0] += estimate[0] totalTokens[1] += estimate[1] continue # Translating response = translateText(characters, system, user, history) translatedText = response.choices[0].message.content totalTokens[0] += response.usage.prompt_tokens totalTokens[1] += response.usage.completion_tokens # Formatting translatedText = cleanTranslatedText(translatedText, varResponse) if isinstance(tItem, list): extractedTranslations = extractTranslation(translatedText, True) if len(tItem) != len(extractedTranslations): # Mismatch. Try Again response = translateText(characters, system, user, history) translatedText = response.choices[0].message.content totalTokens[0] += response.usage.prompt_tokens totalTokens[1] += response.usage.completion_tokens # Formatting translatedText = cleanTranslatedText(translatedText, varResponse) if isinstance(tItem, list): extractedTranslations = extractTranslation(translatedText, True) if len(tItem) == len(extractedTranslations): tList[index] = extractedTranslations else: MISMATCH.append(filename) else: tList[index] = extractedTranslations # Create History history = tList[index] # Update history if we have a list pbar.update(len(tList[index])) else: # Ensure we're passing a single string to extractTranslation extractedTranslations = extractTranslation(translatedText, False) tList[index] = extractedTranslations # Combine if multilist if isinstance(tList[0], list): tList = [t for sublist in tList for t in sublist] # Return if format == "json": return [tList, totalTokens] else: return [tList[0], totalTokens]