diff --git a/modules/linebyline.py b/modules/linebyline.py new file mode 100644 index 0000000..449881f --- /dev/null +++ b/modules/linebyline.py @@ -0,0 +1,566 @@ +# Libraries +import json +import os +import re +import textwrap +import threading +import time +import traceback +import tiktoken +import openai +from pathlib import Path +from colorama import Fore +from dotenv import load_dotenv +from retry import retry +from tqdm import tqdm + +# Open AI +load_dotenv() +if os.getenv("api").replace(" ", "") != "": + openai.base_url = os.getenv("api") +openai.organization = os.getenv("org") +openai.api_key = os.getenv("key") + +# Globals +MODEL = os.getenv("model") +TIMEOUT = int(os.getenv("timeout")) +LANGUAGE = os.getenv("language").capitalize() +PROMPT = Path("prompt.txt").read_text(encoding="utf-8") +VOCAB = Path("vocab.txt").read_text(encoding="utf-8") +THREADS = int(os.getenv("threads")) +LOCK = threading.Lock() +WIDTH = int(os.getenv("width")) +LISTWIDTH = int(os.getenv("listWidth")) +NOTEWIDTH = 70 +MAXHISTORY = 10 +ESTIMATE = "" +TOKENS = [0, 0] +NAMESLIST = [] +NAMES = False # Output a list of all the character names found +BRFLAG = False # If the game uses
instead +FIXTEXTWRAP = True # Overwrites textwrap +IGNORETLTEXT = False # Ignores all translated text. +MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) +FILENAME = None + +# tqdm Globals +BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" +POSITION = 0 +LEAVE = False +PBAR = None + +# Pricing - Depends on the model https://openai.com/pricing +# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request +# If you are getting a MISMATCH LENGTH error, lower the batch size. +if "gpt-3.5" in MODEL: + INPUTAPICOST = 0.002 + OUTPUTAPICOST = 0.002 + BATCHSIZE = 10 +elif "gpt-4" in MODEL: + INPUTAPICOST = 0.0025 + OUTPUTAPICOST = 0.01 + BATCHSIZE = 20 + +def handleText(filename, estimate): + global ESTIMATE, TOKENS, FILENAME + ESTIMATE = estimate + FILENAME = filename + + # Translate + start = time.time() + translatedData = openFiles(filename) + + # Translate + if not estimate: + try: + with open("translated/" + filename, "w", encoding="utf-8") as outFile: + outFile.writelines(translatedData[0]) + except Exception: + traceback.print_exc() + return "Fail" + + # Print File + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + + # Print Total + totalString = getResultString(["", TOKENS, None], end - start, "TOTAL") + + # Print any errors on maps + if len(MISMATCH) > 0: + return totalString + Fore.RED + f"\nMismatch Errors: {MISMATCH}" + Fore.RESET + else: + return totalString + + +def getResultString(translatedData, translationTime, filename): + # File Print String + totalTokenstring = ( + Fore.YELLOW + "[Input: " + str(translatedData[1][0]) + "]" + "[Output: " + + str(translatedData[1][1]) + + "]" "[Cost: ${:,.4f}".format( + (translatedData[1][0] * 0.001 * INPUTAPICOST) + + (translatedData[1][1] * 0.001 * OUTPUTAPICOST) + ) + + "]" + ) + timeString = Fore.BLUE + "[" + str(round(translationTime, 1)) + "s]" + + if translatedData[2] == None: + # Success + return ( + filename + + ": " + + totalTokenstring + + timeString + + Fore.GREEN + + " \u2713 " + + Fore.RESET + ) + + else: + # Fail + try: + raise translatedData[2] + except Exception as e: + traceback.print_exc() + errorString = str(e) + Fore.RED + return ( + filename + + ": " + + totalTokenstring + + timeString + + Fore.RED + + " \u2717 " + + errorString + + Fore.RESET + ) + + +def openFiles(filename): + with open("files/" + filename, "r", encoding="utf8") as readFile: + translatedData = parseText(readFile, filename) + + # Delete lines marked for deletion + finalData = [] + for line in translatedData[0]: + if line != "\\d\n": + finalData.append(line) + translatedData[0] = finalData + + return translatedData + + +def parseText(readFile, filename): + global PBAR + totalTokens = [0, 0] + + # Read File into data + data = readFile.readlines() + + # Create Progress Bar + with tqdm(bar_format=BAR_FORMAT, position=POSITION, leave=LEAVE) as pbar: + pbar.desc = filename + PBAR = pbar + + try: + result = translateTxt(data, []) + totalTokens[0] += result[0] + totalTokens[1] += result[1] + except Exception as e: + traceback.print_exc() + return [data, totalTokens, e] + return [data, totalTokens, None] + + +def translateTxt(data, translatedList): + if translatedList: + stringList = translatedList[0] + choiceList = translatedList[1] + else: + stringList = [] + choiceList = [] + tokens = [0, 0] + speaker = "" + global LOCK, ESTIMATE, FILENAME, PBAR, MISMATCH + i = 0 + + while i < len(data): + lineTextRegex = r"(.*)" + + # Dialogue + match = re.search(lineTextRegex, data[i]) + jaString = None + if match: + # Set String + jaString = match.group(1) + originalString = jaString + + # Pass 1 + if not translatedList: + # Strip Spaces + jaString = jaString.strip() + + if jaString: + if speaker: + stringList.append(f"[{speaker}]: {jaString}") + else: + stringList.append(jaString) + + # Pass 2 + else: + # Get Text + if stringList: + # Grab and Pop + translatedText = stringList[0] + stringList.pop(0) + + # Set to None if empty list + if len(stringList) <= 0: + stringList = None + + # Remove speaker + translatedText = re.sub( + r"^\[?(.+?)\]?\s?[|:]\s?", "", translatedText + ) + + # Escape Quotes + translatedText = re.sub(r'(?`\n\ +- You translate everything, including erotic and explicit content.\n\ +- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\ +- All text in your response must be in {LANGUAGE} even if it is hard to translate.\n\ +- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\ +- Maintain any spacing in the translation.\n\ +- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\ +- `...` can be a part of the dialogue. Translate it as it is.\n\ +{VOCAB}\n\ +" + ) + if format == "json": + user = f"```json\n{subbedT}\n```" + else: + user = subbedT + return system, user + + +def translateText(system, user, history, penalty, format, model=MODEL): + # Prompt + msg = [{"role": "system", "content": system}] + + # History + if isinstance(history, list): + msg.extend([{"role": "system", "content": h} for h in history]) + else: + msg.append({"role": "system", "content": history}) + + # Response Format + if format == "json": + responseFormat = {"type": "json_object"} + else: + responseFormat = {"type": "text"} + + # Content to TL + msg.append({"role": "user", "content": f"{user}"}) + response = openai.chat.completions.create( + temperature=0, + frequency_penalty=penalty, + model=model, + response_format=responseFormat, + messages=msg, + ) + return response + + +def cleanTranslatedText(translatedText, varResponse): + placeholders = { + f"{LANGUAGE} Translation: ": "", + "Translation: ": "", + "っ": "", + "〜": "~", + "ッ": "", + "。": ".", + "「": '\\"', + "」": '\\"', + "- ": "-", + "—": "-", + "】": "]", + "【": "[", + "Placeholder Text": "", + # Add more replacements as needed + } + for target, replacement in placeholders.items(): + translatedText = translatedText.replace(target, replacement) + + # Elongate Long Dashes (Since GPT Ignores them...) + translatedText = elongateCharacters(translatedText) + return translatedText + + +def elongateCharacters(text): + # Define a pattern to match one character followed by one or more `ー` characters + # Using a positive lookbehind assertion to capture the preceding character + pattern = r"(?<=(.))ー+" + + # Define a replacement function that elongates the captured character + def repl(match): + char = match.group(1) # The character before the ー sequence + count = len(match.group(0)) - 1 # Number of ー characters + return char * count # Replace ー sequence with the character repeated + + # Use re.sub() to replace the pattern in the text + return re.sub(pattern, repl, text) + + +def extractTranslation(translatedTextList, is_list): + try: + translatedTextList = re.sub(r'\\"+\"([^,\n}])', r'\\"\1', translatedTextList) + translatedTextList = re.sub(r"(?