From 90f0ee5d25eca61e88094e42d4295a00d74ce1c4 Mon Sep 17 00:00:00 2001 From: DazedAnon Date: Tue, 22 Oct 2024 12:48:05 -0500 Subject: [PATCH] Create kirikiri module --- modules/kirikiri.py | 599 ++++++++++++++++++++++++++++++++++++++++++++ modules/main.py | 2 + 2 files changed, 601 insertions(+) create mode 100644 modules/kirikiri.py diff --git a/modules/kirikiri.py b/modules/kirikiri.py new file mode 100644 index 0000000..5cd4afb --- /dev/null +++ b/modules/kirikiri.py @@ -0,0 +1,599 @@ +# Libraries +import json +import os +import re +import textwrap +import threading +import time +import traceback +import tiktoken +import openai +from pathlib import Path +from colorama import Fore +from dotenv import load_dotenv +from retry import retry +from tqdm import tqdm + +# Open AI +load_dotenv() +if os.getenv("api").replace(" ", "") != "": + openai.base_url = os.getenv("api") +openai.organization = os.getenv("org") +openai.api_key = os.getenv("key") + +# Globals +MODEL = os.getenv("model") +TIMEOUT = int(os.getenv("timeout")) +LANGUAGE = os.getenv("language").capitalize() +PROMPT = Path("prompt.txt").read_text(encoding="utf-8") +VOCAB = Path("vocab.txt").read_text(encoding="utf-8") +THREADS = int(os.getenv("threads")) +LOCK = threading.Lock() +WIDTH = int(os.getenv("width")) +LISTWIDTH = int(os.getenv("listWidth")) +NOTEWIDTH = 70 +MAXHISTORY = 10 +ESTIMATE = "" +TOKENS = [0, 0] +NAMESLIST = [] +NAMES = False # Output a list of all the character names found +BRFLAG = False # If the game uses
instead +FIXTEXTWRAP = True # Overwrites textwrap +IGNORETLTEXT = False # Ignores all translated text. +MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) +PBAR = None +FILENAME = None + +# tqdm Globals +BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" +POSITION = 0 +LEAVE = False + +# Pricing - Depends on the model https://openai.com/pricing +# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request +# If you are getting a MISMATCH LENGTH error, lower the batch size. +if "gpt-3.5" in MODEL: + INPUTAPICOST = 0.002 + OUTPUTAPICOST = 0.002 + BATCHSIZE = 10 +elif "gpt-4" in MODEL: + INPUTAPICOST = 0.0025 + OUTPUTAPICOST = 0.01 + BATCHSIZE = 40 + + +def handleKirikiri(filename, estimate): + global ESTIMATE, FILENAME + ESTIMATE = estimate + FILENAME = filename + + if ESTIMATE: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + + # Print Total + totalString = getResultString(["", TOKENS, None], end - start, "TOTAL") + + # Print any errors on maps + if len(MISMATCH) > 0: + return ( + totalString + Fore.RED + f"\nMismatch Errors: {MISMATCH}" + Fore.RESET + ) + else: + return totalString + + else: + try: + with open( + "translated/" + filename, "w", encoding="utf16", errors="ignore" + ) as outFile: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + outFile.writelines(translatedData[0]) + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + except Exception: + traceback.print_exc() + return "Fail" + + return getResultString(["", TOKENS, None], end - start, "TOTAL") + + +def getResultString(translatedData, translationTime, filename): + # File Print String + totalTokenstring = ( + Fore.YELLOW + "[Input: " + str(translatedData[1][0]) + "]" + "[Output: " + + str(translatedData[1][1]) + + "]" "[Cost: ${:,.4f}".format( + (translatedData[1][0] * 0.001 * INPUTAPICOST) + + (translatedData[1][1] * 0.001 * OUTPUTAPICOST) + ) + + "]" + ) + timeString = Fore.BLUE + "[" + str(round(translationTime, 1)) + "s]" + + if translatedData[2] == None: + # Success + return ( + filename + + ": " + + totalTokenstring + + timeString + + Fore.GREEN + + " \u2713 " + + Fore.RESET + ) + + else: + # Fail + try: + raise translatedData[2] + except Exception as e: + traceback.print_exc() + errorString = str(e) + Fore.RED + return ( + filename + + ": " + + totalTokenstring + + timeString + + Fore.RED + + " \u2717 " + + errorString + + Fore.RESET + ) + + +def openFiles(filename): + with open("files/" + filename, "r", encoding="utf16") as readFile: + translatedData = parseKiriKiri(readFile, filename) + + # Delete lines marked for deletion + finalData = [] + for line in translatedData[0]: + if line != "\\d\n": + finalData.append(line) + translatedData[0] = finalData + + return translatedData + + +def parseKiriKiri(readFile, filename): + global PBAR + totalTokens = [0, 0] + + # Read File into data + data = readFile.readlines() + + # Create Progress Bar + with tqdm(bar_format=BAR_FORMAT, position=POSITION, leave=LEAVE) as PBAR: + PBAR.desc = filename + + try: + result = translateKiriKiri(data, PBAR, filename, []) + totalTokens[0] += result[0] + totalTokens[1] += result[1] + except Exception as e: + traceback.print_exc() + return [data, totalTokens, e] + return [data, totalTokens, None] + + +def translateKiriKiri(data, pbar, filename, translatedList): + stringList = [] + tokens = [0, 0] + speaker = "" + global LOCK, ESTIMATE + i = 0 + + # Regex + speakerRegex = r'【(.*)】\[CR\]' + dialogueRegex = r'^\[text\](.*).*\[KeyWait\]|\[v\](.*)\[\/v\].*\[KeyWait\]' + furiganaRegex = r'(\[eruby\sstr="(.*?)"\stext.*?\])' + + while i < len(data): + speaker = "" + # Speaker + match = re.search(speakerRegex, data[i]) + if match: + speakerJA = match.group(1) + response = getSpeaker(speakerJA) + speaker = response[0] + tokens[0] += response[1][0] + tokens[1] += response[1][1] + data[i] = data[i].replace(speakerJA, speaker) + i += 1 + + # Dialogue + match = re.search(dialogueRegex, data[i]) + if match: + jaString = match.group(1) + if not jaString: + jaString = match.group(2) + # Pass 1 + if translatedList == []: + # Remove any textwrap + jaString = jaString.replace("[r]", " ") + + # Remove Furigana + matchList = re.findall(furiganaRegex, jaString) + if matchList: + for match in matchList: + jaString = jaString.replace(match[0], match[1]) + + # Add String + if speaker: + stringList.append(f"[{speaker}]: {jaString.strip()}") + else: + stringList.append(f"{jaString.strip()}") + + # Pass 2 + else: + # Grab and Pop + translatedText = translatedList[0] + translatedList.pop(0) + + # Set to None if empty list + if len(translatedList) <= 0: + translatedList = None + + # Remove Speaker + translatedText = translatedText.replace(f"[{speaker}]: ", "") + + # Textwrap + translatedText = textwrap.fill(translatedText, width=WIDTH) + translatedText = translatedText.replace("\n", "[r]") + + # Set Data + data[i] = data[i].replace(jaString, translatedText) + + # Next Line + i += 1 + + # EOF + if len(stringList) > 0: + # Set Progress + pbar.total = len(stringList) + pbar.refresh() + + # Translate + response = translateGPT( + stringList, + "", + True, + ) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + translatedList = response[0] + + # Set Strings + if len(stringList) == len(translatedList): + translateKiriKiri(data, pbar, filename, translatedList) + + # Mismatch + else: + with LOCK: + if filename not in MISMATCH: + MISMATCH.append(filename) + return tokens + + +# Save some money and enter the character before translation +def getSpeaker(speaker): + match speaker: + case "ファイン": + return ["Fine", [0, 0]] + case "": + return ["", [0, 0]] + case _: + # Find Speaker + for i in range(len(NAMESLIST)): + if speaker == NAMESLIST[i][0]: + return [NAMESLIST[i][1], [0, 0]] + + # Translate and Store Speaker + response = translateGPT( + f"{speaker}", + "Reply with the " + LANGUAGE + " translation of the NPC name.", + True, + ) + response[0] = response[0].title() + response[0] = response[0].replace("'S", "'s") + response[0] = response[0].replace("Speaker: ", "") + + # Retry if name doesn't translate for some reason + if re.search(r"([a-zA-Z??])", response[0]) == None: + response = translateGPT( + f"{speaker}", + "Reply with the " + LANGUAGE + " translation of the NPC name.", + False, + ) + response[0] = response[0].title() + response[0] = response[0].replace("'S", "'s") + + speakerList = [speaker, response[0]] + NAMESLIST.append(speakerList) + return response + return [speaker, [0, 0]] + + +def subVars(jaString): + jaString = jaString.replace("\u3000", " ") + + # Formatting + codeList = re.findall(r"([\\]*(\w+)\[(\d+)\])|([\\]*(\w+)\[[\\]*\\w+\[(\d+)\]\])", jaString) + codeList = set(codeList) + if len(codeList) != 0: + for var in codeList: + if var[2]: + jaString = jaString.replace(var[0], f"[{var[1]}Code_" + f"{var[2]}]") + else: + jaString = jaString.replace(var[3], f"[{var[4]}Code_" + f"{var[5]}]") + + # Put all lists in list and return + return [jaString, codeList] + +def resubVars(translatedText, codeList): + # Formatting + for var in codeList: + if var[2]: + translatedText = translatedText.replace(f"[{var[1]}Code_" + f"{var[2]}]", var[0]) + else: + translatedText = translatedText.replace(f"[{var[4]}Code_" + f"{var[5]}]", var[3]) + + return translatedText + + +def batchList(input_list, batch_size): + if not isinstance(batch_size, int) or batch_size <= 0: + raise ValueError("batch_size must be a positive integer") + + return [ + input_list[i : i + batch_size] for i in range(0, len(input_list), batch_size) + ] + + +def createContext(fullPromptFlag, subbedT, format): + system = ( + PROMPT + VOCAB + if fullPromptFlag + else f"\ +You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\ +Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\ +- You translate everything, including erotic and explicit content.\n\ +- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\ +- All text in your response must be in {LANGUAGE} even if it is hard to translate.\n\ +- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\ +- Maintain any spacing in the translation.\n\ +- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\ +- `...` can be a part of the dialogue. Translate it as it is.\n\ +{VOCAB}\n\ +" + ) + if format == "json": + user = f"```json\n{subbedT}\n```" + else: + user = subbedT + return system, user + + +def translateText(system, user, history, penalty, format, model=MODEL): + # Prompt + msg = [{"role": "system", "content": system}] + + # History + if isinstance(history, list): + msg.extend([{"role": "system", "content": h} for h in history]) + else: + msg.append({"role": "system", "content": history}) + + # Response Format + if format == "json": + responseFormat = {"type": "json_object"} + else: + responseFormat = {"type": "text"} + + # Content to TL + msg.append({"role": "user", "content": f"{user}"}) + response = openai.chat.completions.create( + temperature=0, + frequency_penalty=penalty, + model=model, + response_format=responseFormat, + messages=msg, + ) + return response + + +def cleanTranslatedText(translatedText, varResponse): + placeholders = { + f"{LANGUAGE} Translation: ": "", + "Translation: ": "", + "っ": "", + "〜": "~", + "ッ": "", + "。": ".", + "「": '\\"', + "」": '\\"', + "- ": "-", + "】": "]", + "【": "[", + "Placeholder Text": "", + # Add more replacements as needed + } + for target, replacement in placeholders.items(): + translatedText = translatedText.replace(target, replacement) + + # Elongate Long Dashes (Since GPT Ignores them...) + translatedText = elongateCharacters(translatedText) + translatedText = resubVars(translatedText, varResponse[1]) + return translatedText + + +def elongateCharacters(text): + # Define a pattern to match one character followed by one or more `ー` characters + # Using a positive lookbehind assertion to capture the preceding character + pattern = r"(?<=(.))ー+" + + # Define a replacement function that elongates the captured character + def repl(match): + char = match.group(1) # The character before the ー sequence + count = len(match.group(0)) - 1 # Number of ー characters + return char * count # Replace ー sequence with the character repeated + + # Use re.sub() to replace the pattern in the text + return re.sub(pattern, repl, text) + + +def extractTranslation(translatedTextList, is_list): + try: + translatedTextList = re.sub(r'\\"+\"([^,\n}])', r'\\"\1', translatedTextList) + translatedTextList = re.sub(r"(?