# Libraries import os import re import textwrap import threading import time import traceback import tiktoken import openai from pathlib import Path from colorama import Fore from dotenv import load_dotenv from retry import retry from tqdm import tqdm # Open AI load_dotenv() if os.getenv("api").replace(" ", "") != "": openai.base_url = os.getenv("api") openai.organization = os.getenv("org") openai.api_key = os.getenv("key") # Globals MODEL = os.getenv("model") TIMEOUT = int(os.getenv("timeout")) LANGUAGE = os.getenv("language").capitalize() PROMPT = Path("prompt.txt").read_text(encoding="utf-8") VOCAB = Path("vocab.txt").read_text(encoding="utf-8") THREADS = int(os.getenv("threads")) LOCK = threading.Lock() WIDTH = int(os.getenv("width")) LISTWIDTH = int(os.getenv("listWidth")) NOTEWIDTH = 70 MAXHISTORY = 10 ESTIMATE = "" TOKENS = [0, 0] NAMESLIST = [] NAMES = False # Output a list of all the character names found BRFLAG = False # If the game uses
instead FIXTEXTWRAP = True # Overwrites textwrap IGNORETLTEXT = False # Ignores all translated text. MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) # tqdm Globals BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" POSITION = 0 LEAVE = False # Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. if "gpt-3.5" in MODEL: INPUTAPICOST = 3.00 OUTPUTAPICOST = 5.00 BATCHSIZE = 10 FREQUENCY_PENALTY = 0.2 elif "gpt-4" in MODEL: INPUTAPICOST = 2.50 OUTPUTAPICOST = 10.00 BATCHSIZE = 30 FREQUENCY_PENALTY = 0.05 elif "deepseek" in MODEL: INPUTAPICOST = 0.14 OUTPUTAPICOST = 0.28 BATCHSIZE = 30 FREQUENCY_PENALTY = 0.05 else: INPUTAPICOST = float(os.getenv("input_cost")) OUTPUTAPICOST = float(os.getenv("output_cost")) BATCHSIZE = int(os.getenv("batchsize")) FREQUENCY_PENALTY = float(os.getenv("frequency_penalty")) def handleAlice(filename, estimate): global ESTIMATE totalTokens = [0, 0] ESTIMATE = estimate if estimate: start = time.time() translatedData = openFiles(filename) # Print Result end = time.time() tqdm.write(getResultString(translatedData, end - start, filename)) with LOCK: totalTokens[0] += translatedData[1][0] totalTokens[1] += translatedData[1][1] # Print Total totalString = getResultString(["", totalTokens, None], end - start, "TOTAL") # Print any errors on maps if len(MISMATCH) > 0: return totalString + Fore.RED + f"\nMismatch Errors: {MISMATCH}" + Fore.RESET else: return totalString else: try: with open("translated/" + filename, "w", encoding="UTF-8") as outFile: start = time.time() translatedData = openFiles(filename) # Print Result end = time.time() outFile.writelines(translatedData[0]) tqdm.write(getResultString(translatedData, end - start, filename)) with LOCK: totalTokens[0] += translatedData[1][0] totalTokens[1] += translatedData[1][1] except Exception: traceback.print_exc() return "Fail" return getResultString(["", totalTokens, None], end - start, "TOTAL") def openFiles(filename): with open("files/" + filename, "r", encoding="UTF-8") as f: translatedData = parseText(f, filename) return translatedData def getResultString(translatedData, translationTime, filename): # File Print String totalTokenstring = ( Fore.YELLOW + "[Input: " + str(translatedData[1][0]) + "]" "[Output: " + str(translatedData[1][1]) + "]" "[Cost: ${:,.4f}".format(((translatedData[1][0] / 1000000) * INPUTAPICOST) + ((translatedData[1][1] / 1000000) * OUTPUTAPICOST)) + "]" ) timeString = Fore.BLUE + "[" + str(round(translationTime, 1)) + "s]" if translatedData[2] == None: # Success return filename + ": " + totalTokenstring + timeString + Fore.GREEN + " \u2713 " + Fore.RESET else: # Fail try: raise translatedData[2] except Exception as e: traceback.print_exc() errorString = str(e) + Fore.RED return filename + ": " + totalTokenstring + timeString + Fore.RED + " \u2717 " + errorString + Fore.RESET def parseText(data, filename): # Get total for progress bar linesList = data.readlines() totalTokens = [0, 0] totalLines = len(linesList) global LOCK with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar: pbar.desc = filename pbar.total = totalLines try: result = translateLines(linesList, pbar) totalTokens[0] += result[1][0] totalTokens[1] += result[1][1] except Exception as e: traceback.print_exc() return [linesList, totalTokens, e] return [linesList, totalTokens, None] # Grab scenario data from text file def translateLines(linesList, pbar): currentGroup = [] batch = [] textHistory = [] tokens = [0, 0] batchStartIndex = 0 insertBool = False multiLine = False i = 0 try: while i < len(linesList): # Check if Proper Message match = re.findall(r"s\[[0-9]+\] = \"(.*)\"", linesList[i]) if len(match) > 0: jaString = match[0] # Skip Files if "/" in jaString: i += 1 continue ### Translate # Remove any textwrap jaString = re.sub(r"\\n", " ", jaString) # Grab Speaker speakerMatch = re.findall(r"s\[[0-9]+\] = \"([^/]+)\"", linesList[i - 1]) if len(speakerMatch) > 0: # If there isn't any Japanese in the text just skip if re.search(r"[一-龠]+|[ぁ-ゔ]+|[ァ-ヴー]+", jaString) and "_" not in speakerMatch[0]: speaker = speakerMatch[0] else: speaker = "" else: speaker = "" # Grab rest of the messages currentGroup.append(jaString) # Check if next line should be merged if insertBool is True: linesList[i] = re.sub(r"(s\[[0-9]+\]) = \"(.+)\"", r'\1 = ""', linesList[i]) linesList[i] = linesList[i].replace(";", "") start = i while len(linesList) > i + 1 and re.search(r"s\[[0-9]+\] = \"\s+(.*)\"", linesList[i + 1]) != None: multiLine = True i += 1 match = re.findall(r"s\[[0-9]+\] = \"\s+(.*)\"", linesList[i]) currentGroup.append(match[0]) if insertBool is True: linesList[i] = re.sub(r"(s\[[0-9]+\]) = \"\s+(.+)\"", r'\1 = ""', linesList[i]) linesList[i] = linesList[i].replace(";", "") i += 1 # Combine Groups and Add Speaker finalJAString = " ".join(currentGroup) if speaker != "": finalJAString = f"{speaker}: {finalJAString}" else: finalJAString = f"{finalJAString}" # [Passthrough 1] Pulling From File if insertBool is False: # Append to List and Clear Values batch.append(finalJAString) # Translate Batch if Full if len(batch) == BATCHSIZE or i >= len(linesList) - 1: # Translate response = translateGPT(batch, textHistory, True) tokens[0] += response[1][0] tokens[1] += response[1][1] translatedBatch = response[0] textHistory = translatedBatch[-10:] # Set Values if len(batch) == len(translatedBatch): i = batchStartIndex insertBool = True # Mismatch else: pbar.write(f"Mismatch: {batchStartIndex} - {i}") MISMATCH.append(batch) batchStartIndex = i batch.clear() multiLine = False currentGroup = [] # [Passthrough 2] Setting Data else: # Get Text translatedText = translatedBatch[0] # Remove added speaker and quotes translatedText = re.sub(r"^.+?:\s", "", translatedText) # Textwrap translatedText = translatedText.replace('"', '\\"') translatedText = textwrap.fill(translatedText, width=WIDTH) # Set Data if multiLine: textList = translatedText.split("\n") for t in textList: translatedText = translatedText.replace(";", "") translatedText = re.sub( r"(s\[[0-9]+\]) = \"(.*)\"", rf'\1 = "{t}"', linesList[start], ) translatedText = translatedText.replace(";", "") linesList[start] = translatedText pbar.update(1) start += 1 multiLine = False translatedText = translatedText.replace(";", "") translatedBatch.pop(0) else: # Remove any textwrap translatedText = translatedText.replace("\n", " ") translatedText = re.sub( r"(s\[[0-9]+\]) = \"(.*)\"", rf'\1 = "{translatedText}"', linesList[start], ) translatedText = translatedText.replace(";", "") linesList[start] = translatedText pbar.update(1) translatedBatch.pop(0) # If Batch is empty. Move on. if len(translatedBatch) == 0: insertBool = False batchStartIndex = i pbar.update(1) batch.clear() currentGroup = [] else: if insertBool is True: pbar.update(1) i += 1 return [linesList, tokens] except Exception: traceback.print_exc() return [linesList, tokens] def subVars(jaString): jaString = jaString.replace("\u3000", " ") # Nested count = 0 nestedList = re.findall(r"[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]", jaString) nestedList = set(nestedList) if len(nestedList) != 0: for icon in nestedList: jaString = jaString.replace(icon, "{Nested_" + str(count) + "}") count += 1 # Icons count = 0 iconList = re.findall(r"[\\]+[iIkKwWaA]+\[[0-9]+\]", jaString) iconList = set(iconList) if len(iconList) != 0: for icon in iconList: jaString = jaString.replace(icon, "{Ascii_" + str(count) + "}") count += 1 # Colors count = 0 colorList = re.findall(r"[\\]+[cC]\[[0-9]+\]", jaString) colorList = set(colorList) if len(colorList) != 0: for color in colorList: jaString = jaString.replace(color, "{Color_" + str(count) + "}") count += 1 # Names count = 0 nameList = re.findall(r"[\\]+[nN]\[.+?\]+", jaString) nameList = set(nameList) if len(nameList) != 0: for name in nameList: jaString = jaString.replace(name, "{Noun_" + str(count) + "}") count += 1 # Variables count = 0 varList = re.findall(r"[\\]+[vV]\[[0-9]+\]", jaString) varList = set(varList) if len(varList) != 0: for var in varList: jaString = jaString.replace(var, "{Var_" + str(count) + "}") count += 1 # Formatting count = 0 formatList = re.findall(r"[\\]+[\w]+\[.+?\]", jaString) formatList = set(formatList) if len(formatList) != 0: for var in formatList: jaString = jaString.replace(var, "{FCode_" + str(count) + "}") count += 1 # Put all lists in list and return allList = [nestedList, iconList, colorList, nameList, varList, formatList] return [jaString, allList] def resubVars(translatedText, allList): # Fix Spacing and ChatGPT Nonsense matchList = re.findall(r"\[\s?.+?\s?\]", translatedText) if len(matchList) > 0: for match in matchList: text = match.strip() translatedText = translatedText.replace(match, text) # Nested count = 0 if len(allList[0]) != 0: for var in allList[0]: translatedText = translatedText.replace("{Nested_" + str(count) + "}", var) count += 1 # Icons count = 0 if len(allList[1]) != 0: for var in allList[1]: translatedText = translatedText.replace("{Ascii_" + str(count) + "}", var) count += 1 # Colors count = 0 if len(allList[2]) != 0: for var in allList[2]: translatedText = translatedText.replace("{Color_" + str(count) + "}", var) count += 1 # Names count = 0 if len(allList[3]) != 0: for var in allList[3]: translatedText = translatedText.replace("{Noun_" + str(count) + "}", var) count += 1 # Vars count = 0 if len(allList[4]) != 0: for var in allList[4]: translatedText = translatedText.replace("{Var_" + str(count) + "}", var) count += 1 # Formatting count = 0 if len(allList[5]) != 0: for var in allList[5]: translatedText = translatedText.replace("{FCode_" + str(count) + "}", var) count += 1 return translatedText def batchList(input_list, batch_size): if not isinstance(batch_size, int) or batch_size <= 0: raise ValueError("batch_size must be a positive integer") return [input_list[i : i + batch_size] for i in range(0, len(input_list), batch_size)] def createContext(fullPromptFlag, subbedT): characters = "Game Characters:\n\ 林つかさ (Tsukasa Hayashi) - Female\n\ 山田美兎 (Miyato Yamada) - Female\n\ 鈴木赤音 (Akane Suzuki) - Female\n\ 佐藤莉伊南 (Riina Satou) - Female\n\ 佐々木万梨美 (Marimi Sasaki) - Female\n\ 渡辺登樹子 (Tokiko Watanabe) - Female\n\ 桃乃夢 (Yume Momono) - Female\n\ 吉浦美雪 (Miyuki Yoshiura) - Female\n\ 三ツ門まあな (Maana Mitsukado) - Female\n\ モリー・ボイド (Molly Boyd) - Female\n\ オルガ・ブヤチッチ (Olga Buyachich) - Female\n\ アッチャラー ギッティ (Atchara Gitti) - Female\n\ " system = ( PROMPT if fullPromptFlag else f"Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`" ) user = f"{subbedT}" return characters, system, user def translateText(characters, system, user, history): # Prompt msg = [{"role": "system", "content": system + characters}] # Characters msg.append({"role": "system", "content": characters}) # History if isinstance(history, list): msg.extend([{"role": "assistant", "content": h} for h in history]) else: msg.append({"role": "assistant", "content": history}) # Content to TL msg.append({"role": "user", "content": f"{user}"}) response = openai.chat.completions.create( temperature=0.1, frequency_penalty=0.1, model=MODEL, messages=msg, ) return response def cleanTranslatedText(translatedText, varResponse): placeholders = { f"{LANGUAGE} Translation: ": "", "Translation: ": "", "っ": "", "〜": "~", "ッ": "", "。": ".", "Placeholder Text": "", # Add more replacements as needed } for target, replacement in placeholders.items(): translatedText = translatedText.replace(target, replacement) translatedText = resubVars(translatedText, varResponse[1]) if "\n" in translatedText: return [line for line in translatedText.split("\n") if line] else: return [line for line in translatedText.split("\\n") if line] def extractTranslation(translatedTextList, is_list): pattern = r"[\\]*`?(.*?)[\\]*?`?" # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. if is_list: return [re.findall(pattern, line)[0][1] for line in translatedTextList if re.search(pattern, line)] else: matchList = re.findall(pattern, translatedTextList) return matchList[0][1] if matchList else translatedTextList def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 enc = tiktoken.encoding_for_model("gpt-4") # Input if isinstance(history, list): for line in history: inputTotalTokens += len(enc.encode(line)) else: inputTotalTokens += len(enc.encode(history)) inputTotalTokens += len(enc.encode(system)) inputTotalTokens += len(enc.encode(characters)) inputTotalTokens += len(enc.encode(user)) # Output outputTotalTokens += round(len(enc.encode(user)) * 3) return [inputTotalTokens, outputTotalTokens] @retry(exceptions=Exception, tries=5, delay=5) def translateGPT(text, history, fullPromptFlag): totalTokens = [0, 0] if isinstance(text, list): tList = batchList(text, BATCHSIZE) else: tList = [text] for index, tItem in enumerate(tList): # Before sending to translation, if we have a list of items, add the formatting if isinstance(tItem, list): payload = "\n".join([f"`{item}`" for i, item in enumerate(tItem)]) payload = payload.replace("``", "`Placeholder Text`") varResponse = subVars(payload) subbedT = varResponse[0] else: varResponse = subVars(tItem) subbedT = varResponse[0] # Things to Check before starting translation if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+", subbedT): continue # Create Message characters, system, user = createContext(fullPromptFlag, subbedT) # Calculate Estimate if ESTIMATE: estimate = countTokens(characters, system, user, history) totalTokens[0] += estimate[0] totalTokens[1] += estimate[1] continue # Translating response = translateText(characters, system, user, history) translatedText = response.choices[0].message.content totalTokens[0] += response.usage.prompt_tokens totalTokens[1] += response.usage.completion_tokens # Formatting translatedTextList = cleanTranslatedText(translatedText, varResponse) if isinstance(tItem, list): extractedTranslations = extractTranslation(translatedTextList, True) tList[index] = extractedTranslations if len(tItem) != len(translatedTextList): mismatch = True # Just here so breakpoint can be set history = extractedTranslations[-10:] # Update history if we have a list else: # Ensure we're passing a single string to extractTranslation extractedTranslations = extractTranslation("\n".join(translatedTextList), False) tList[index] = extractedTranslations # Combine if multilist if isinstance(tList[0], list): tList = [t for sublist in tList for t in sublist] # Return if format == "json": return [tList, totalTokens] else: return [tList[0], totalTokens]