# Libraries import os import re import textwrap import threading import time import traceback import tiktoken import openai from pathlib import Path from colorama import Fore from dotenv import load_dotenv from retry import retry from tqdm import tqdm # Open AI load_dotenv() if os.getenv("api").replace(" ", "") != "": openai.base_url = os.getenv("api") openai.organization = os.getenv("org") openai.api_key = os.getenv("key") # Globals MODEL = os.getenv("model") TIMEOUT = int(os.getenv("timeout")) LANGUAGE = os.getenv("language").capitalize() PROMPT = Path("prompt.txt").read_text(encoding="utf-8") VOCAB = Path("vocab.txt").read_text(encoding="utf-8") THREADS = int(os.getenv("threads")) LOCK = threading.Lock() WIDTH = int(os.getenv("width")) LISTWIDTH = int(os.getenv("listWidth")) NOTEWIDTH = 70 MAXHISTORY = 10 ESTIMATE = "" TOKENS = [0, 0] NAMESLIST = [] NAMES = False # Output a list of all the character names found BRFLAG = False # If the game uses
instead FIXTEXTWRAP = True # Overwrites textwrap IGNORETLTEXT = False # Ignores all translated text. MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) # tqdm Globals BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" POSITION = 0 LEAVE = False # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. if "gpt-3.5" in MODEL: INPUTAPICOST = 0.002 OUTPUTAPICOST = 0.002 BATCHSIZE = 10 elif "gpt-4" in MODEL: INPUTAPICOST = 0.005 OUTPUTAPICOST = 0.015 BATCHSIZE = 40 def handleIris(filename, estimate): global ESTIMATE ESTIMATE = estimate if ESTIMATE: start = time.time() translatedData = openFiles(filename) # Print Result end = time.time() tqdm.write(getResultString(translatedData, end - start, filename)) with LOCK: TOKENS[0] += translatedData[1][0] TOKENS[1] += translatedData[1][1] # Print Total totalString = getResultString(["", TOKENS, None], end - start, "TOTAL") # Print any errors on maps if len(MISMATCH) > 0: return ( totalString + Fore.RED + f"\nMismatch Errors: {MISMATCH}" + Fore.RESET ) else: return totalString else: try: with open( "translated/" + filename, "w", encoding="cp932", errors="ignore" ) as outFile: start = time.time() translatedData = openFiles(filename) # Print Result end = time.time() outFile.writelines(translatedData[0]) tqdm.write(getResultString(translatedData, end - start, filename)) with LOCK: TOKENS[0] += translatedData[1][0] TOKENS[1] += translatedData[1][1] except Exception: traceback.print_exc() return "Fail" return getResultString(["", TOKENS, None], end - start, "TOTAL") def getResultString(translatedData, translationTime, filename): # File Print String totalTokenstring = ( Fore.YELLOW + "[Input: " + str(translatedData[1][0]) + "]" "[Output: " + str(translatedData[1][1]) + "]" "[Cost: ${:,.4f}".format( (translatedData[1][0] * 0.001 * INPUTAPICOST) + (translatedData[1][1] * 0.001 * OUTPUTAPICOST) ) + "]" ) timeString = Fore.BLUE + "[" + str(round(translationTime, 1)) + "s]" if translatedData[2] == None: # Success return ( filename + ": " + totalTokenstring + timeString + Fore.GREEN + " \u2713 " + Fore.RESET ) else: # Fail try: raise translatedData[2] except Exception as e: traceback.print_exc() errorString = str(e) + Fore.RED return ( filename + ": " + totalTokenstring + timeString + Fore.RED + " \u2717 " + errorString + Fore.RESET ) def openFiles(filename): with open("files/" + filename, "r", encoding="shift_jis") as readFile: translatedData = parseIris(readFile, filename) # Delete lines marked for deletion finalData = [] for line in translatedData[0]: if line != "\\d\n": finalData.append(line) translatedData[0] = finalData return translatedData def parseIris(readFile, filename): totalTokens = [0, 0] # Read File into data data = readFile.readlines() # Create Progress Bar with tqdm(bar_format=BAR_FORMAT, position=POSITION, leave=LEAVE) as pbar: pbar.desc = filename try: result = translateIris(data, pbar, filename, []) totalTokens[0] += result[0] totalTokens[1] += result[1] except Exception as e: traceback.print_exc() return [data, totalTokens, e] return [data, totalTokens, None] def translateIris(data, pbar, filename, translatedList): stringList = [] currentGroup = [] tokens = [0, 0] speaker = "" voice = False global LOCK, ESTIMATE i = 0 while i < len(data): voice = False speaker = "" if "#MSGVOICE" in data[i]: i += 1 voice = True voiceVar = data[i] if "#MSG," in data[i] or "#MSG\n" in data[i] or voice == True: i += 1 # Speaker if ( re.search(r'^ ?([^#\/."、。*!!()\(\)\[\] \n]+)\n', data[i]) and len(data[i]) < 30 ): match = re.search(r"(.*)", data[i]) if match != None: speaker = match.group(1) if speaker[0] == "\u3000": speaker = speaker[1:] response = getSpeaker(speaker, pbar, filename) speaker = response[0] tokens[0] += response[1][0] tokens[1] += response[1][1] if translatedList != []: speaker = speaker.replace(" ", "\u3000") data[i] = f"\u3000{speaker}\n" else: speaker = "" i += 1 # Lines match = re.search(r"(.*)", data[i]) if match != None and match.group(1) != "": # Pass 1 if translatedList == []: # Grab Consecutive Strings jaString = data[i] if data[i] != "\n": if data[i][0] == "\u3000": jaString = data[i][1:] currentGroup.append(jaString) i += 1 while data[i] != "\n": jaString = data[i] if data[i] != "\n": jaString = data[i][1:] currentGroup.append(jaString) i += 1 # Join up 401 groups for better translation. if len(currentGroup) > 0: jaString = "".join(currentGroup) currentGroup = [] # Remove any textwrap jaString = jaString.replace("\n", " ") # Temporarily convert spaces (For Textwrap Later) jaString = jaString.replace("\u3000", " ") # Add Speaker (If there is one) if speaker != "": jaString = f"{speaker}: {jaString}" # Add String stringList.append(jaString.strip()) # Pass 2 else: # Insert Strings while data[i] != "\n": data.pop(i) # Get Text if translatedList: translatedText = translatedList[0] translatedList.pop(0) if len(translatedList) <= 0: translatedList = None # Remove added speaker translatedText = re.sub(r"^.+?:\s", "", translatedText) # Textwrap translatedText = textwrap.fill(translatedText, width=WIDTH) translatedText = translatedText.replace("\n", "\n\u3000") # Replace Whitespace and Commas translatedText = translatedText.replace(", ", "、") translatedText = translatedText.replace(",\u3000", "、") translatedText = translatedText.replace(",", "、") translatedText = translatedText.replace(" ", "\u3000") # Set Data # Game crashes on more than 3 lines. Will need to create a new MSG for long translations if translatedText.count("\n") > 2: # Split List translatedTextList = splitNewlines(translatedText) # MSG Voice count = 0 for text in translatedTextList: if count != 0: if voice == True: # MSG for each item in the list data.insert(i, "#MSGVOICE,\n") i += 1 data.insert(i, f"{voiceVar}") i += 1 else: data.insert(i, "#MSG,\n") i += 1 if speaker: data[i] = f"\u3000{speaker}\n" i += 1 if text[0] == "\u3000": data.insert(i, f"{text}\n") else: data.insert(i, f"\u3000{text}\n") i += 1 count += 1 if data[i] != "\n": data.insert(i, "\n") data[i] = f"\n{data[i]}" else: data.insert(i, f"\u3000{translatedText}\n") i += 1 if data[i] != "\n": data[i] = f"\n{data[i]}" elif "#SELECT" in data[i] and translatedList == []: Iris = r"(.+?) +\d$" i += 1 match = re.search(Iris, data[i]) if match: choiceList = [] choiceList.append(match.group(1)) i += 1 match = re.search(Iris, data[i]) while match: choiceList.append(match.group(1)) i += 1 match = re.search(Iris, data[i]) # Translate question = stringList[len(stringList) - 1] response = translateGPT( choiceList, f"Previous text for context: {question}\n\nThis will be a dialogue option", True, pbar, filename, ) tokens[0] += response[1][0] tokens[1] += response[1][1] choiceListTL = response[0] # Set Data i = i - len(choiceListTL) for j in range(len(choiceListTL)): # Replace Whitespace and Commas choiceListTL[j] = choiceListTL[j].replace(", ", "、") choiceListTL[j] = choiceListTL[j].replace(",\u3000", "、") choiceListTL[j] = choiceListTL[j].replace(",", "、") choiceListTL[j] = choiceListTL[j].replace(" ", "\u3000") data[i] = data[i].replace(choiceList[j], choiceListTL[j]) i += 1 # Nothing relevant. Skip Line. else: i += 1 else: i += 1 # EOF if len(stringList) > 0: # Set Progress pbar.total = len(stringList) pbar.refresh() # Translate response = translateGPT(stringList, "", True, pbar, filename) tokens[0] += response[1][0] tokens[1] += response[1][1] translatedList = response[0] # Set Strings if len(stringList) == len(translatedList): translateIris(data, pbar, filename, translatedList) # Mismatch else: with LOCK: if filename not in MISMATCH: MISMATCH.append(filename) return tokens def splitNewlines(text): parts = [] newline_count = 0 # Counts the number of newline characters encountered start_index = 0 # Start index of the current string part for i, char in enumerate(text): if char == "\n": newline_count += 1 if newline_count == 3: # Append the string part from start_index to current index (inclusive) parts.append(text[start_index : i + 1]) # Reset newline count and update start_index for the next string part newline_count = 0 start_index = i + 1 # Edge case: if the text does not end with a newline, we still need to append the last part if start_index < len(text): parts.append(text[start_index:]) return parts # Save some money and enter the character before translation def getSpeaker(speaker, pbar, filename): match speaker: case "ファイン": return ["Fine", [0, 0]] case "": return ["", [0, 0]] case _: # Store Speaker if speaker not in str(NAMESLIST): response = translateGPT( speaker, "Reply with only the " + LANGUAGE + " translation of the NPC name.", False, pbar, filename, ) response[0] = response[0].replace("'S", "'s") speakerList = [speaker, response[0]] NAMESLIST.append(speakerList) return response # Find Speaker else: for i in range(len(NAMESLIST)): if speaker == NAMESLIST[i][0]: return [NAMESLIST[i][1], [0, 0]] return [speaker, [0, 0]] def subVars(jaString): jaString = jaString.replace("\u3000", " ") # Nested count = 0 nestedList = re.findall(r"[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]", jaString) nestedList = set(nestedList) if len(nestedList) != 0: for icon in nestedList: jaString = jaString.replace(icon, "[Nested_" + str(count) + "]") count += 1 # Icons count = 0 iconList = re.findall(r"[\\]+[iIkKwWaA]+\[[0-9]+\]", jaString) iconList = set(iconList) if len(iconList) != 0: for icon in iconList: jaString = jaString.replace(icon, "[Ascii_" + str(count) + "]") count += 1 # Colors count = 0 colorList = re.findall(r"[\\]+[cC]\[[0-9]+\]", jaString) colorList = set(colorList) if len(colorList) != 0: for color in colorList: jaString = jaString.replace(color, "[Color_" + str(count) + "]") count += 1 # Names count = 0 nameList = re.findall(r"[\\]+[nN]\[.+?\]+", jaString) nameList = set(nameList) if len(nameList) != 0: for name in nameList: jaString = jaString.replace(name, "[Noun_" + str(count) + "]") count += 1 # Variables count = 0 varList = re.findall(r"[\\]+[vV]\[[0-9]+\]", jaString) varList = set(varList) if len(varList) != 0: for var in varList: jaString = jaString.replace(var, "[Var_" + str(count) + "]") count += 1 # Formatting count = 0 formatList = re.findall(r"[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]", jaString) formatList = set(formatList) if len(formatList) != 0: for var in formatList: jaString = jaString.replace(var, "[FCode_" + str(count) + "]") count += 1 # Put all lists in list and return allList = [nestedList, iconList, colorList, nameList, varList, formatList] return [jaString, allList] def resubVars(translatedText, allList): # Fix Spacing and ChatGPT Nonsense matchList = re.findall(r"\[\s?.+?\s?\]", translatedText) if len(matchList) > 0: for match in matchList: text = match.strip() translatedText = translatedText.replace(match, text) # Nested count = 0 if len(allList[0]) != 0: for var in allList[0]: translatedText = translatedText.replace("[Nested_" + str(count) + "]", var) count += 1 # Icons count = 0 if len(allList[1]) != 0: for var in allList[1]: translatedText = translatedText.replace("[Ascii_" + str(count) + "]", var) count += 1 # Colors count = 0 if len(allList[2]) != 0: for var in allList[2]: translatedText = translatedText.replace("[Color_" + str(count) + "]", var) count += 1 # Names count = 0 if len(allList[3]) != 0: for var in allList[3]: translatedText = translatedText.replace("[Noun_" + str(count) + "]", var) count += 1 # Vars count = 0 if len(allList[4]) != 0: for var in allList[4]: translatedText = translatedText.replace("[Var_" + str(count) + "]", var) count += 1 # Formatting count = 0 if len(allList[5]) != 0: for var in allList[5]: translatedText = translatedText.replace("[FCode_" + str(count) + "]", var) count += 1 return translatedText def batchList(input_list, batch_size): if not isinstance(batch_size, int) or batch_size <= 0: raise ValueError("batch_size must be a positive integer") return [ input_list[i : i + batch_size] for i in range(0, len(input_list), batch_size) ] def createContext(fullPromptFlag, subbedT): characters = "Game Characters:\n\ フィリア (Philia) - Female\n\ アルネット (Annett) - Female\n\ ラピュセナ (Rapusena) - Female\n\ リッカ (Rikka) - Female\n\ アンデリビア (Andelivia) - Female\n\ リリアブルム (Liliabloom) - Female\n\ カルナ (Karna) - Female\n\ ラフィング=スピア (Laughing Spear) - Female\n\ ノーラ (Nora) - Female\n\ " system = ( PROMPT + VOCAB if fullPromptFlag else f"\ You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\ Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\ - You translate everything, including erotic and explicit content.\n\ - 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\ - All text in your response must be in {LANGUAGE} even if it is hard to translate.\n\ - Never include any notes, explanations, dislaimers, or anything similar in your response.\n\ - Maintain any spacing in the translation.\n\ - Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\ - `...` can be a part of the dialogue. Translate it as it is.\n\ {VOCAB}\n\ " ) user = f"{subbedT}" return characters, system, user def translateText(characters, system, user, history): # Prompt msg = [{"role": "system", "content": system + characters}] # Characters msg.append({"role": "system", "content": characters}) # History if isinstance(history, list): msg.extend([{"role": "system", "content": h} for h in history]) else: msg.append({"role": "system", "content": history}) # Content to TL msg.append({"role": "user", "content": f"{user}"}) response = openai.chat.completions.create( temperature=0.1, frequency_penalty=0.1, model=MODEL, messages=msg, ) return response def cleanTranslatedText(translatedText, varResponse): placeholders = { f"{LANGUAGE} Translation: ": "", "Translation: ": "", "っ": "", "〜": "~", "ッ": "", "。": ".", "Placeholder Text": "", # Add more replacements as needed } for target, replacement in placeholders.items(): translatedText = translatedText.replace(target, replacement) # Elongate Long Dashes (Since GPT Ignores them...) translatedText = elongateCharacters(translatedText) translatedText = resubVars(translatedText, varResponse[1]) return translatedText def elongateCharacters(text): # Define a pattern to match one character followed by one or more `ー` characters # Using a positive lookbehind assertion to capture the preceding character pattern = r"(?<=(.))ー+" # Define a replacement function that elongates the captured character def repl(match): char = match.group(1) # The character before the ー sequence count = len(match.group(0)) - 1 # Number of ー characters return char * count # Replace ー sequence with the character repeated # Use re.sub() to replace the pattern in the text return re.sub(pattern, repl, text) def extractTranslation(translatedTextList, is_list): pattern = r"`?<[Ll]ine\d+>([\\]*.*?[\\]*?)<\/?[Ll]ine\d+>`?" # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. if is_list: matchList = re.findall(pattern, translatedTextList) return matchList else: matchList = re.findall(pattern, translatedTextList) return matchList[0][0] if matchList else translatedTextList def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 enc = tiktoken.encoding_for_model("gpt-4") # Input if isinstance(history, list): for line in history: inputTotalTokens += len(enc.encode(line)) else: inputTotalTokens += len(enc.encode(history)) inputTotalTokens += len(enc.encode(system)) inputTotalTokens += len(enc.encode(characters)) inputTotalTokens += len(enc.encode(user)) # Output outputTotalTokens += round(len(enc.encode(user)) * 2) return [inputTotalTokens, outputTotalTokens] def combineList(tlist, text): if isinstance(text, list): return [t for sublist in tlist for t in sublist] return tlist[0] @retry(exceptions=Exception, tries=5, delay=5) def translateGPT(text, history, fullPromptFlag, pbar, filename): mismatch = False totalTokens = [0, 0] if isinstance(text, list): tList = batchList(text, BATCHSIZE) else: tList = [text] for index, tItem in enumerate(tList): # Before sending to translation, if we have a list of items, add the formatting if isinstance(tItem, list): payload = "\n".join( [f"`{item}`" for i, item in enumerate(tItem)] ) payload = re.sub( r"(<)(\/Line\d+>)", r"\1>Placeholder Text<\3", payload ) varResponse = subVars(payload) subbedT = varResponse[0] else: varResponse = subVars(tItem) subbedT = varResponse[0] # Things to Check before starting translation if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+", subbedT): continue # Create Message characters, system, user = createContext(fullPromptFlag, subbedT) # Calculate Estimate if ESTIMATE: estimate = countTokens(characters, system, user, history) totalTokens[0] += estimate[0] totalTokens[1] += estimate[1] continue # Translating response = translateText(characters, system, user, history) translatedText = response.choices[0].message.content totalTokens[0] += response.usage.prompt_tokens totalTokens[1] += response.usage.completion_tokens # Formatting translatedText = cleanTranslatedText(translatedText, varResponse) if isinstance(tItem, list): extractedTranslations = extractTranslation(translatedText, True) if len(tItem) != len(extractedTranslations): # Mismatch. Try Again response = translateText(characters, system, user, history) translatedText = response.choices[0].message.content totalTokens[0] += response.usage.prompt_tokens totalTokens[1] += response.usage.completion_tokens # Formatting translatedText = cleanTranslatedText(translatedText, varResponse) if isinstance(tItem, list): extractedTranslations = extractTranslation(translatedText, True) if len(tItem) == len(extractedTranslations): tList[index] = extractedTranslations else: MISMATCH.append(filename) else: tList[index] = extractedTranslations # Create History history = tList[index] # Update history if we have a list pbar.update(len(tList[index])) else: # Ensure we're passing a single string to extractTranslation extractedTranslations = extractTranslation(translatedText, False) tList[index] = extractedTranslations finalList = combineList(tList, text) return [finalList, totalTokens]