diff --git a/modules/main.py b/modules/main.py index 0206107..9c0d2ee 100644 --- a/modules/main.py +++ b/modules/main.py @@ -8,6 +8,7 @@ from modules.rpgmakermvmz import handleMVMZ from modules.rpgmakerace import handleACE from modules.csvtl import handleCSV from modules.textfile import handleTextfile +from modules.tyrano import handleTyrano THREADS = 10 # For GPT4 rate limit will be hit if you have more than 1 thread. @@ -29,7 +30,7 @@ def main(): totalCost = 0 version = '' while version == '': - version = input('Select the RPGMaker Version:\n\n1. MV/MZ\n2. ACE\n3. CSV (From Translator++)\n4. Text (Custom)\n') + version = input('Select the RPGMaker Version:\n\n1. MV/MZ\n2. ACE\n3. CSV (From Translator++)\n4. Text (Custom)\n5. Tyrano\n') match version: case '1': # Open File (Threads) @@ -85,6 +86,20 @@ def main(): tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno) print(Fore.RED + str(e) + '|' + tracebackLineNo + Fore.RESET) + case '5': + # Open File (Threads) + with ThreadPoolExecutor(max_workers=THREADS) as executor: + futures = [executor.submit(handleTyrano, filename, estimate) \ + for filename in os.listdir("files") if filename.endswith('ks')] + + for future in as_completed(futures): + try: + totalCost = future.result() + + except Exception as e: + tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno) + print(Fore.RED + str(e) + '|' + tracebackLineNo + Fore.RESET) + case _: version = '' diff --git a/modules/tyrano.py b/modules/tyrano.py new file mode 100644 index 0000000..191dd5f --- /dev/null +++ b/modules/tyrano.py @@ -0,0 +1,294 @@ +from concurrent.futures import ThreadPoolExecutor, as_completed +import json +import os +from pathlib import Path +import re +import sys +import textwrap +import threading +import time +import traceback +import tiktoken + +from colorama import Fore +from dotenv import load_dotenv +import openai +from retry import retry +from tqdm import tqdm + +#Globals +load_dotenv() +openai.organization = os.getenv('org') +openai.api_key = os.getenv('key') + +APICOST = .002 # Depends on the model https://openai.com/pricing +PROMPT = Path('prompt.txt').read_text(encoding='utf-8') +THREADS = 10 # For GPT4 rate limit will be hit if you have more than 1 thread. +LOCK = threading.Lock() +WIDTH = 60 +LISTWIDTH = 60 +MAXHISTORY = 10 +ESTIMATE = '' +TOTALCOST = 0 +TOKENS = 0 +TOTALTOKENS = 0 +NAMESLIST = [] + +#tqdm Globals +BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' +POSITION=0 +LEAVE=False + +# Flags +NAMES = True # Output a list of all the character names found +BRFLAG = False # If the game uses
instead +FIXTEXTWRAP = False + +def handleTyrano(filename, estimate): + global ESTIMATE, TOKENS, TOTALTOKENS, TOTALCOST + ESTIMATE = estimate + + if estimate: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + tqdm.write(getResultString(['', TOKENS, None], end - start, filename)) + if NAMES == True: + tqdm.write(str(NAMESLIST)) + with LOCK: + TOTALCOST += TOKENS * .001 * APICOST + TOTALTOKENS += TOKENS + TOKENS = 0 + + return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL') + + else: + try: + with open('translated/' + filename, 'w', encoding='UTF-8') as outFile: + start = time.time() + translatedData = openFiles(filename, outFile) + + # Print Result + outFile.writeLines(translatedData[0]) + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOTALCOST += translatedData[1] * .001 * APICOST + TOTALTOKENS += translatedData[1] + except Exception as e: + traceback.print_exc() + return 'Fail' + + return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL') + +def openFiles(filename, outFile): + with open('files/' + filename, 'r', encoding='utf-8') as readFile: + translatedData = parseTyrano(readFile, outFile, filename) + + return translatedData + +def parseTyrano(readFile, outFile, filename): + totalTokens = 0 + totalLines = 0 + global LOCK + + # Get total for progress bar + data = readFile.readlines() + totalLines = len(data) + + with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar: + pbar.desc=filename + pbar.total=totalLines + + try: + final = translateTyrano(data, outFile, pbar) + except Exception as e: + traceback.print_exc() + return [data, totalTokens, e] + return [readFile, totalTokens, None] + +def translateTyrano(data, outFile, pbar): + textHistory = [] + maxHistory = MAXHISTORY + tokens = 0 + currentGroup = [] + syncIndex = 0 + global LOCK, ESTIMATE + + for i in range(len(data)): + if syncIndex > i: + i = syncIndex + + # Speaker + if '#' in data[i]: + matchList = re.findall(r'#(.+)', data[i]) + if len(matchList) != 0: + response = translateGPT(matchList[0], 'Reply with only the english translation of the NPC name', True) + speaker = response[0] + tokens += response[1] + data[i] = '#' + speaker + else: + speaker = '' + + # Lines + elif '[p]' in data[i]: + matchList = re.findall(r'(.+?)\[p\]', data[i]) + if len(matchList) > 0: + matchList[0] = matchList[0].replace('「', '') + matchList[0] = matchList[0].replace('」', '') + currentGroup.append(matchList[0]) + if len(data) > i+1: + while '[p]' in data[i+1]: + i += 1 + matchList = re.findall(r'(.+?)\[p\]', data[i]) + if len(matchList) > 0: + matchList[0] = matchList[0].replace('「', '') + matchList[0] = matchList[0].replace('」', '') + currentGroup.append(matchList[0]) + # Join up 401 groups for better translation. + if len(currentGroup) > 0: + finalJAString = ''.join(currentGroup) + oldjaString = finalJAString + + #Check Speaker + if speaker == '': + response = translateGPT(finalJAString, 'Previous Dialogue: ' + '\n\n'.join(textHistory), True) + tokens += response[1] + translatedText = response[0] + textHistory.append('\"' + translatedText + '\"') + else: + response = translateGPT(speaker + ': ' + finalJAString, 'Previous Dialogue: ' + '\n\n'.join(textHistory), True) + tokens += response[1] + translatedText = response[0] + textHistory.append('\"' + translatedText + '\"') + + # Remove added speaker + translatedText = translatedText.replace(speaker + ': ', '') + + # Set Data + translatedText = translatedText.replace('\"', '') + data[i] = translatedText + '[p]' + + currentGroup = [] + pbar.update(1) + syncIndex = i + 1 + + return [tokens, data] + +def getResultString(translatedData, translationTime, filename): + # File Print String + tokenString = Fore.YELLOW + '[' + str(translatedData[1]) + \ + ' Tokens/${:,.4f}'.format(translatedData[1] * .001 * APICOST) + ']' + timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' + + if translatedData[2] == None: + # Success + return filename + ': ' + tokenString + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET + + else: + # Fail + try: + raise translatedData[2] + except Exception as e: + traceback.print_exc() + errorString = str(e) + Fore.RED + return filename + ': ' + tokenString + timeString + Fore.RED + u' \u2717 ' +\ + errorString + Fore.RESET + +def subVars(jaString): + varRegex = r'\\+[a-zA-Z]+\[[0-9a-zA-Z\\\[\]]+\]+|[\\]+[#a-zA-Z]+|[\\.<>]+' + count = 0 + + varList = re.findall(varRegex, jaString) + if len(varList) != 0: + for var in varList: + jaString = jaString.replace(var, '[v' + str(count) + ']') + count += 1 + + return [jaString, varList] + +def resubVars(translatedText, varList): + count = 0 + + # Fix Spacing and ChatGPT Nonsense + matchList = re.findall(r'@\s?[0-9]+?', translatedText) + if len(matchList) > 0: + for match in matchList: + text = match.replace(' ', '') + translatedText = translatedText.replace(match, text) + + if len(varList) != 0: + for var in varList: + translatedText = translatedText.replace('@' + str(count) + '', var) + count += 1 + + # Remove Color Variables Spaces + # if '\\c' in translatedText: + # translatedText = re.sub(r'\s*(\\+c\[[1-9]+\])\s*', r' \1', translatedText) + # translatedText = re.sub(r'\s*(\\+c\[0+\])', r'\1', translatedText) + return translatedText + +@retry(exceptions=Exception, tries=5, delay=5) +def translateGPT(t, history, fullPromptFlag): + with LOCK: + # If ESTIMATE is True just count this as an execution and return. + if ESTIMATE: + global TOKENS + enc = tiktoken.encoding_for_model("gpt-3.5-turbo") + TOKENS += len(enc.encode(t)) * 2 + len(enc.encode(history)) + len(enc.encode(PROMPT)) + return (t, 0) + + # Sub Vars + varResponse = subVars(t) + subbedT = varResponse[0] + + # If there isn't any Japanese in the text just skip + if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴー]+', subbedT): + return(t, 0) + + """Translate text using GPT""" + context = 'Eroge Names Context: 桐嶋 香織 == Kaori Kirishima | Female, 肉山 猛 == Takeshi Nikuyama' + if fullPromptFlag: + system = PROMPT + user = 'Line to Translate: ' + subbedT + else: + system = 'You are an expert translator who translates everything to English. Reply with only the English Translation of the text.' + user = 'Line to Translate: ' + subbedT + response = openai.ChatCompletion.create( + temperature=0, + frequency_penalty=0.2, + presence_penalty=0.2, + model="gpt-3.5-turbo", + messages=[ + {"role": "system", "content": system}, + {"role": "user", "content": context}, + {"role": "user", "content": history}, + {"role": "user", "content": user} + ], + request_timeout=30, + ) + + # Save Translated Text + translatedText = response.choices[0].message.content + tokens = response.usage.total_tokens + + # Resub Vars + translatedText = resubVars(translatedText, varResponse[1]) + + # Remove Placeholder Text + translatedText = translatedText.replace('English Translation: ', '') + translatedText = translatedText.replace('Translation: ', '') + translatedText = translatedText.replace('Line to Translate: ', '') + translatedText = translatedText.replace('English Translation:', '') + translatedText = translatedText.replace('Translation:', '') + translatedText = translatedText.replace('Line to Translate:', '') + translatedText = re.sub(r'\n\nPast Translated Text:.*', '', translatedText, 0, re.DOTALL) + translatedText = re.sub(r'Note:.*', '', translatedText) + + # Return Translation + if len(translatedText) > 15 * len(t) or "I'm sorry, but I'm unable to assist with that translation" in translatedText: + return [t, response.usage.total_tokens] + else: + return [translatedText, tokens] \ No newline at end of file