From ddb3d9196afea149429c1172c1f38e37a04de789 Mon Sep 17 00:00:00 2001 From: Dazed Date: Sun, 26 Nov 2023 08:13:44 -0600 Subject: [PATCH] Changes to VN scripts --- modules/anim.py | 401 +++++++++++++++++++++++++++++++++++++++++++++ modules/atelier.py | 15 +- modules/main.py | 28 +++- modules/tyrano.py | 33 ++-- 4 files changed, 455 insertions(+), 22 deletions(-) create mode 100644 modules/anim.py diff --git a/modules/anim.py b/modules/anim.py new file mode 100644 index 0000000..44409b3 --- /dev/null +++ b/modules/anim.py @@ -0,0 +1,401 @@ +import json +import os +from pathlib import Path +import re +import sys +import textwrap +import threading +import time +import traceback +import tiktoken + +from colorama import Fore +from dotenv import load_dotenv +import openai +from retry import retry +from tqdm import tqdm + +#Globals +load_dotenv() +if os.getenv('api').replace(' ', '') != '': + openai.api_base = os.getenv('api') + +openai.organization = os.getenv('org') +openai.api_key = os.getenv('key') +MODEL = os.getenv('model') +TIMEOUT = int(os.getenv('timeout')) +LANGUAGE=os.getenv('language').capitalize() + +INPUTAPICOST = .002 # Depends on the model https://openai.com/pricing +OUTPUTAPICOST = .002 +PROMPT = Path('prompt.txt').read_text(encoding='utf-8') +THREADS = int(os.getenv('threads')) # Controls how many threads are working on a single file (May have to drop this) +LOCK = threading.Lock() +WIDTH = int(os.getenv('width')) +LISTWIDTH = int(os.getenv('listWidth')) +NOTEWIDTH = 50 +MAXHISTORY = 10 +ESTIMATE = '' +totalTokens = [0, 0] +NAMESLIST = [] + +#tqdm Globals +BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' +POSITION=0 +LEAVE=False +BRFLAG = False # If the game uses
instead +FIXTEXTWRAP = True +IGNORETLTEXT = True + +def handleAnim(filename, estimate): + global ESTIMATE, totalTokens + ESTIMATE = estimate + + if estimate: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + totalTokens[0] += translatedData[1][0] + totalTokens[1] += translatedData[1][1] + + return getResultString(['', totalTokens, None], end - start, 'TOTAL') + + else: + try: + with open('translated/' + filename, 'w', encoding='UTF-8') as outFile: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + json.dump(translatedData[0], outFile, ensure_ascii=False) + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + totalTokens[0] += translatedData[1][0] + totalTokens[1] += translatedData[1][1] + except Exception as e: + return 'Fail' + + return getResultString(['', totalTokens, None], end - start, 'TOTAL') + +def openFiles(filename): + with open('files/' + filename, 'r', encoding='UTF-8-sig') as f: + data = json.load(f) + + # Map Files + if '.json' in filename: + translatedData = parseJSON(data, filename) + + else: + raise NameError(filename + ' Not Supported') + + return translatedData + +def getResultString(translatedData, translationTime, filename): + # File Print String + totalTokenstring =\ + Fore.YELLOW +\ + '[Input: ' + str(translatedData[1][0]) + ']'\ + '[Output: ' + str(translatedData[1][1]) + ']'\ + '[Cost: ${:,.4f}'.format((translatedData[1][0] * .001 * INPUTAPICOST) +\ + (translatedData[1][1] * .001 * OUTPUTAPICOST)) + ']' + timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' + + if translatedData[2] == None: + # Success + return filename + ': ' + totalTokenstring + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET + + else: + # Fail + try: + raise translatedData[2] + except Exception as e: + traceback.print_exc() + errorString = str(e) + Fore.RED + return filename + ': ' + totalTokenstring + timeString + Fore.RED + u' \u2717 ' +\ + errorString + Fore.RESET + +def parseJSON(data, filename): + totalTokens = [0, 0] + totalLines = 0 + totalLines = len(data) + global LOCK + + with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar: + pbar.desc=filename + pbar.total=totalLines + try: + result = translateJSON(data, pbar) + totalTokens[0] += result[0] + totalTokens[1] += result[1] + except Exception as e: + return [data, totalTokens, e] + return [data, totalTokens, None] + +def translateJSON(data, pbar): + textHistory = [] + maxHistory = MAXHISTORY + tokens = [0, 0] + + for key, value in data.items(): + # Text + if value == "": + jaString = key + else: + jaString = value + + # Check if TLed + # If there isn't any Japanese in the text just skip + if IGNORETLTEXT is True: + if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴー]+', jaString): + pbar.update(1) + continue + + # Remove any textwrap + if FIXTEXTWRAP == True: + jaString = jaString.replace('\n', ' ') + + # Translate + if jaString != '': + response = translateGPT(f'{jaString}', textHistory, True) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + translatedText = response[0] + textHistory.append('\"' + translatedText + '\"') + else: + translatedText = jaString + textHistory.append('\"' + translatedText + '\"') + + # Remove added speaker + translatedText = re.sub(r'^.+?\s\|\s?', '', translatedText) + + # Textwrap + if '\n' not in translatedText: + translatedText = textwrap.fill(translatedText, width=WIDTH) + + # Set Data + data[key] = translatedText + + # Keep textHistory list at length maxHistory + if len(textHistory) > maxHistory: + textHistory.pop(0) + currentGroup = [] + pbar.update(1) + + return tokens + +def subVars(jaString): + jaString = jaString.replace('\u3000', ' ') + + # Nested + count = 0 + nestedList = re.findall(r'[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]', jaString) + nestedList = set(nestedList) + if len(nestedList) != 0: + for icon in nestedList: + jaString = jaString.replace(icon, '{Nested_' + str(count) + '}') + count += 1 + + # Icons + count = 0 + iconList = re.findall(r'[\\]+[iIkKwWaA]+\[[0-9]+\]', jaString) + iconList = set(iconList) + if len(iconList) != 0: + for icon in iconList: + jaString = jaString.replace(icon, '{Ascii_' + str(count) + '}') + count += 1 + + # Colors + count = 0 + colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString) + colorList = set(colorList) + if len(colorList) != 0: + for color in colorList: + jaString = jaString.replace(color, '{Color_' + str(count) + '}') + count += 1 + + # Names + count = 0 + nameList = re.findall(r'[\\]+[nN]\[.+?\]+', jaString) + nameList = set(nameList) + if len(nameList) != 0: + for name in nameList: + jaString = jaString.replace(name, '{N_' + str(count) + '}') + count += 1 + + # Variables + count = 0 + varList = re.findall(r'[\\]+[vV]\[[0-9]+\]', jaString) + varList = set(varList) + if len(varList) != 0: + for var in varList: + jaString = jaString.replace(var, '{Var_' + str(count) + '}') + count += 1 + + # Formatting + count = 0 + if '笑えるよね.' in jaString: + print('t') + formatList = re.findall(r'[\\]+[\w]+\[.+?\]', jaString) + formatList = set(formatList) + if len(formatList) != 0: + for var in formatList: + jaString = jaString.replace(var, '{FCode_' + str(count) + '}') + count += 1 + + # Put all lists in list and return + allList = [nestedList, iconList, colorList, nameList, varList, formatList] + return [jaString, allList] + +def resubVars(translatedText, allList): + # Fix Spacing and ChatGPT Nonsense + matchList = re.findall(r'\[\s?.+?\s?\]', translatedText) + if len(matchList) > 0: + for match in matchList: + text = match.strip() + translatedText = translatedText.replace(match, text) + + # Nested + count = 0 + if len(allList[0]) != 0: + for var in allList[0]: + translatedText = translatedText.replace('{Nested_' + str(count) + '}', var) + count += 1 + + # Icons + count = 0 + if len(allList[1]) != 0: + for var in allList[1]: + translatedText = translatedText.replace('{Ascii_' + str(count) + '}', var) + count += 1 + + # Colors + count = 0 + if len(allList[2]) != 0: + for var in allList[2]: + translatedText = translatedText.replace('{Color_' + str(count) + '}', var) + count += 1 + + # Names + count = 0 + if len(allList[3]) != 0: + for var in allList[3]: + translatedText = translatedText.replace('{N_' + str(count) + '}', var) + count += 1 + + # Vars + count = 0 + if len(allList[4]) != 0: + for var in allList[4]: + translatedText = translatedText.replace('{Var_' + str(count) + '}', var) + count += 1 + + # Formatting + count = 0 + if len(allList[5]) != 0: + for var in allList[5]: + translatedText = translatedText.replace('{FCode_' + str(count) + '}', var) + count += 1 + + # Remove Color Variables Spaces + # if '\\c' in translatedText: + # translatedText = re.sub(r'\s*(\\+c\[[1-9]+\])\s*', r' \1', translatedText) + # translatedText = re.sub(r'\s*(\\+c\[0+\])', r'\1', translatedText) + return translatedText + +@retry(exceptions=Exception, tries=5, delay=5) +def translateGPT(t, history, fullPromptFlag): + # Sub Vars + varResponse = subVars(t) + subbedT = varResponse[0] + + # If there isn't any Japanese in the text just skip + if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴ]+|[\uFF00-\uFFEF]', subbedT): + return(t, [0,0]) + + # If ESTIMATE is True just count this as an execution and return. + if ESTIMATE: + enc = tiktoken.encoding_for_model(MODEL) + historyRaw = '' + if isinstance(history, list): + for line in history: + historyRaw += line + else: + historyRaw = history + + inputTotalTokens = len(enc.encode(historyRaw)) + len(enc.encode(PROMPT)) + outputTotalTokens = len(enc.encode(t)) * 2 # Estimating 2x the size of the original text + totalTokens = [inputTotalTokens, outputTotalTokens] + return (t, totalTokens) + + # Characters + context = 'Game Characters:\ + Character: リッカ == Ricca - Gender: Female\ + Character: シーナ == Sina - Gender: Female\ + Character: ヘレナ == Helena - Gender: Female\ + Character: Miko == Miko - Gender: Female' + + # Prompt + if fullPromptFlag: + system = PROMPT + user = 'Line to Translate = ' + subbedT + else: + system = 'Output ONLY the '+ LANGUAGE +' translation in the following format: `Translation: <'+ LANGUAGE.upper() +'_TRANSLATION>`' + user = 'Line to Translate = ' + subbedT + + # Create Message List + msg = [] + msg.append({"role": "system", "content": system}) + msg.append({"role": "user", "content": context}) + if isinstance(history, list): + for line in history: + msg.append({"role": "user", "content": line}) + else: + msg.append({"role": "user", "content": history}) + msg.append({"role": "user", "content": user}) + + response = openai.ChatCompletion.create( + temperature=0, + frequency_penalty=0.2, + presence_penalty=0.2, + model=MODEL, + messages=msg, + request_timeout=TIMEOUT, + ) + + # Save Translated Text + translatedText = response.choices[0].message.content + totalTokens = [response.usage.prompt_tokens, response.usage.completion_tokens] + + # Resub Vars + translatedText = resubVars(translatedText, varResponse[1]) + + # Remove Placeholder Text + translatedText = translatedText.replace(LANGUAGE +' Translation: ', '') + translatedText = translatedText.replace('Translation: ', '') + translatedText = translatedText.replace('Line to Translate = ', '') + translatedText = translatedText.replace('Translation = ', '') + translatedText = translatedText.replace('Translate = ', '') + translatedText = translatedText.replace(LANGUAGE +' Translation:', '') + translatedText = translatedText.replace('Translation:', '') + translatedText = translatedText.replace('Line to Translate =', '') + translatedText = translatedText.replace('Translation =', '') + translatedText = translatedText.replace('Translate =', '') + translatedText = translatedText.replace('っ', '') + translatedText = translatedText.replace('ッ', '') + translatedText = translatedText.replace('ぁ', '') + translatedText = translatedText.replace('。', '.') + translatedText = translatedText.replace('、', ',') + translatedText = translatedText.replace('?', '?') + translatedText = translatedText.replace('!', '!') + translatedText = translatedText.replace('\n', '') + + # Return Translation + if len(translatedText) > 15 * len(t) or "I'm sorry, but I'm unable to assist with that translation" in translatedText: + raise Exception + else: + return [translatedText, totalTokens] diff --git a/modules/atelier.py b/modules/atelier.py index d214749..79c8227 100644 --- a/modules/atelier.py +++ b/modules/atelier.py @@ -64,9 +64,10 @@ def handleAtelier(filename, estimate): else: try: - with open('translated/' + filename, 'w', encoding='utf-8'): + with open('translated/' + filename, 'w', encoding='utf-8') as outFile: start = time.time() translatedData = openFiles(filename) + outFile.writelines(translatedData[0]) # Print Result end = time.time() @@ -158,8 +159,8 @@ def translateText(data, pbar): textHistory.pop(0) # Textwrap - translatedText = translatedText.replace('\"', '\\"') translatedText = textwrap.fill(translatedText, width=WIDTH) + translatedText = translatedText.replace('\n', '\\n') # Write data[i] = data[i].replace(match[0], translatedText) @@ -314,10 +315,12 @@ def translateGPT(t, history, fullPromptFlag): # Characters context = 'Game Characters:\ - Character: リッカ == Ricca - Gender: Female\ - Character: シーナ == Sina - Gender: Female\ - Character: ヘレナ == Helena - Gender: Female\ - Character: Miko == Miko - Gender: Female' + Character: Surname:久高 Name:有史 == Surname:Kudaka Name:Yuushi - Gender: Male\ + Character: Surname:葛城 Name:碧璃 == Surname:Katsuragi Name:Midori - Gender: Female\ + Character: Surname:葛城 Name:依理子 == Surname:Katsuragi Name:Yoriko - Gender: Female\ + Character: Surname:桐乃木 Name:奏 == Surname:Kirinogi Name:Kanade - Gender: Female\ + Character: Surname:葛城 Name:光男 == Surname:Katsuragi Name:Mitsuo - Gender: Male\ + Character: Surname:尾木 Name:優真 == Surname:Ogi Name:Yuuma - Gender: Male' # Prompt if fullPromptFlag: diff --git a/modules/main.py b/modules/main.py index 8520a6d..2581309 100644 --- a/modules/main.py +++ b/modules/main.py @@ -14,6 +14,7 @@ from modules.json import handleJSON from modules.kansen import handleKansen from modules.lune2 import handleLuneTxt from modules.atelier import handleAtelier +from modules.anim import handleAnim # For GPT4 rate limit will be hit if you have more than 1 thread. # 1 Thread for each file. Controls how many files are worked on at once. @@ -40,7 +41,18 @@ def main(): totalCost = 0 version = '' while version == '': - version = input('Select the RPGMaker Version:\n\n1. MV/MZ\n2. ACE\n3. CSV (From Translator++)\n4. Text (Custom)\n5. Tyrano\n6. JSON\n7. Kansen\n8. Lune\n9. Atelier\n') + version = input('Select the RPGMaker Version:\n\n\ +1. MV/MZ\n\ +2. ACE\n\ +3. CSV (From Translator++)\n\ +4. Text (Custom)\n\ +5. Tyrano\n\ +6. JSON\n\ +7. Kansen\n\ +8. Lune\n\ +9. Atelier\n\ +10. Anim\n' + ) match version: case '1': # Open File (Threads) @@ -166,6 +178,20 @@ def main(): tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno) tqdm.write(Fore.RED + str(e) + '|' + tracebackLineNo + Fore.RESET) + case '10': + # Open File (Threads) + with ThreadPoolExecutor(max_workers=THREADS) as executor: + futures = [executor.submit(handleAnim, filename, estimate) \ + for filename in os.listdir("files") if filename.endswith('json')] + + for future in as_completed(futures): + try: + totalCost = future.result() + + except Exception as e: + tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno) + tqdm.write(Fore.RED + str(e) + '|' + tracebackLineNo + Fore.RESET) + case _: version = '' diff --git a/modules/tyrano.py b/modules/tyrano.py index 8b8e607..0f12dd0 100644 --- a/modules/tyrano.py +++ b/modules/tyrano.py @@ -73,7 +73,7 @@ def handleTyrano(filename, estimate): else: try: - with open("translated/" + filename, "w", encoding="utf-16") as outFile: + with open("translated/" + filename, "w", encoding="utf-8") as outFile: start = time.time() translatedData = openFiles(filename) outFile.writelines(translatedData[0]) @@ -135,7 +135,7 @@ def getResultString(translatedData, translationTime, filename): def openFiles(filename): - with open("files/" + filename, "r", encoding="utf-16") as readFile: + with open("files/" + filename, "r", encoding="utf-8") as readFile: translatedData = parseTyrano(readFile, filename) # Delete lines marked for deletion @@ -203,9 +203,7 @@ def translateTyrano(data, pbar): continue # Speaker - matchList = re.findall(r"^\[(.+)\sstorage=.+\]", data[i]) - if len(matchList) == 0: - matchList = re.findall(r"^\[([^/].+)\]$", data[i]) + matchList = re.findall(r"^\[([^=\".,!?>]+?)\]$", data[i]) if len(matchList) > 0: if "主人公" in matchList[0]: speaker = "Protagonist" @@ -268,16 +266,17 @@ def translateTyrano(data, pbar): data[i] = translatedText # Grab Lines - matchList = re.findall(r"^([^\n;@*\{\[].+[^;'{}\[]$)", data[i]) - if len(matchList) > 0 and (re.search(r'^\[(.+)\sstorage=.+\],', data[i-1]) or re.search(r'^\[(.+)\]$', data[i-1]) or re.search(r'^《(.+)》', data[i-1])): + matchList = re.findall(r"(.+)\[[rpcm]+\]$", data[i]) + if len(matchList) > 0: currentGroup.append(matchList[0]) + # Grab All Lines in a Row if len(data) > i + 1: - matchList = re.findall(r"^([^\n;@*\{\[].+[^;'{}\[]$)", data[i + 1]) + matchList = re.findall(r"(.+)\[[rpcm]+\]$", data[i + 1]) while len(matchList) > 0: delFlag = True data[i] = "\d\n" # \d Marks line for deletion i += 1 - matchList = re.findall(r"^([^\n;@*\{\[].+[^;'{}\[]$)", data[i]) + matchList = re.findall(r"(.+)\[[rpcm]+\]$", data[i]) if len(matchList) > 0: currentGroup.append(matchList[0]) @@ -287,7 +286,7 @@ def translateTyrano(data, pbar): # Remove any textwrap if FIXTEXTWRAP is True: - finalJAString = finalJAString.replace("_", " ") + finalJAString = finalJAString.replace("[r]", " ") # Check Speaker if speaker == "": @@ -317,9 +316,13 @@ def translateTyrano(data, pbar): translatedText = translatedText.replace("]", "") # Wordwrap Text - if "_" not in translatedText: - translatedText = textwrap.fill(translatedText, width=WIDTH) - translatedText = translatedText.replace("\n", "_") + if "[r]" not in translatedText: + translatedTextList = textwrap.wrap(translatedText, width=WIDTH) + for j in range(len(translatedTextList)): + translatedTextList[j] = translatedTextList[j] + '[r]\n' + j += 1 + translatedTextList[j - 1] = translatedTextList[j - 1].replace('[r]\n', '[pcm]\n') + translatedText = ''.join(translatedTextList) # Set if delFlag is True: @@ -347,12 +350,12 @@ def translateTyrano(data, pbar): originalText = matchList[0][1] currentGroup.append(matchList[0][1]) if len(data) > i + 1: - matchList = re.findall(r"^([^\n;@*\{\[].+[^;'{}\[]$)", data[i + 1]) + matchList = re.findall(r"(.+)\[[rpcm]+\]$", data[i + 1]) while len(matchList) > 0: delFlag = True data[i] = "\d\n" # \d Marks line for deletion i += 1 - matchList = re.findall(r"^([^\n;@*\{\[].+[^;'{}\[]$)", data[i]) + matchList = re.findall(r"(.+)\[[rpcm]+\]$", data[i]) if len(matchList) > 0: currentGroup.append(matchList[0])