From b56f387b0ec51ca7989afd5a34f44ba932305653 Mon Sep 17 00:00:00 2001 From: Dazed Date: Mon, 2 Oct 2023 10:17:37 -0500 Subject: [PATCH] Changing some filenames around and some tweaks to a few scripts --- README.md | 2 +- modules/{csvtl.py => csv.py} | 0 modules/json.py | 255 ++++++++++++++++++++++++++++++++ modules/main.py | 25 +++- modules/rpgmakerace.py | 19 +-- modules/{textfile.py => txt.py} | 2 +- 6 files changed, 287 insertions(+), 16 deletions(-) rename modules/{csvtl.py => csv.py} (100%) create mode 100644 modules/json.py rename modules/{textfile.py => txt.py} (99%) diff --git a/README.md b/README.md index 488413f..5a17513 100644 --- a/README.md +++ b/README.md @@ -73,7 +73,7 @@ A breakdown of what all the different files are, this is important. * rpgmakermvmz.py - Translation Script for the RPGMaker MV/MZ Engine. * rpgmakerace.py - Translation Script for the RPGMaker ACE Engine. (Requires rvpacker to unpack rvdata files) * csvtl.py - Translation Script for CSV Files. Requires at least 2 columns to work. - * textfile.py - Translation Script for Other game engines. (More of a custom script I change depending on the game) + * TXT.py - Translation Script for Other game engines. (More of a custom script I change depending on the game) * .env.example - An example env file. This gets renamed to .env and holds your PRIVATE API and Organization key. Do not EVER upload this information. * RPGMakerEventCodes.info - Information on the various types of event codes in RPGMaker. More on this later. * prompt.example - Holds an example prompt ChatGPT uses to determine what to do with text you give it. Change this as you please. diff --git a/modules/csvtl.py b/modules/csv.py similarity index 100% rename from modules/csvtl.py rename to modules/csv.py diff --git a/modules/json.py b/modules/json.py new file mode 100644 index 0000000..37b1d06 --- /dev/null +++ b/modules/json.py @@ -0,0 +1,255 @@ +from concurrent.futures import ThreadPoolExecutor, as_completed +import json +import os +from pathlib import Path +import re +import sys +import textwrap +import threading +import time +import traceback +import tiktoken + +from colorama import Fore +from dotenv import load_dotenv +import openai +from retry import retry +from tqdm import tqdm + +#Globals +load_dotenv() +openai.organization = os.getenv('org') +openai.api_key = os.getenv('key') + +APICOST = .002 # Depends on the model https://openai.com/pricing +PROMPT = Path('prompt.txt').read_text(encoding='utf-8') +THREADS = 10 # For GPT4 rate limit will be hit if you have more than 1 thread. +LOCK = threading.Lock() +WIDTH = 90 +LISTWIDTH = 60 +MAXHISTORY = 10 +ESTIMATE = '' +TOTALCOST = 0 +TOTALTOKENS = 0 +NAMESLIST = [] + +#tqdm Globals +BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' +POSITION=0 +LEAVE=False +BRFLAG = False # If the game uses
instead +FIXTEXTWRAP = True + +def handleJSON(filename, estimate): + global ESTIMATE, TOKENS, TOTALTOKENS, TOTALCOST + ESTIMATE = estimate + + if estimate: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOTALCOST += translatedData[1] * .001 * APICOST + TOTALTOKENS += translatedData[1] + + return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL') + + else: + try: + with open('translated/' + filename, 'w', encoding='UTF-8') as outFile: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + json.dump(translatedData[0], outFile, ensure_ascii=False) + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOTALCOST += translatedData[1] * .001 * APICOST + TOTALTOKENS += translatedData[1] + except Exception as e: + traceback.print_exc() + return 'Fail' + + return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL') + +def openFiles(filename): + with open('files/' + filename, 'r', encoding='UTF-8-sig') as f: + data = json.load(f) + + # Map Files + if 'script' in filename: + translatedData = parseJSON(data, filename) + + else: + raise NameError(filename + ' Not Supported') + + return translatedData + +def getResultString(translatedData, translationTime, filename): + # File Print String + tokenString = Fore.YELLOW + '[' + str(translatedData[1]) + \ + ' Tokens/${:,.4f}'.format(translatedData[1] * .001 * APICOST) + ']' + timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' + + if translatedData[2] == None: + # Success + return filename + ': ' + tokenString + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET + + else: + # Fail + try: + raise translatedData[2] + except Exception as e: + errorString = str(e) + Fore.RED + return filename + ': ' + tokenString + timeString + Fore.RED + u' \u2717 ' +\ + errorString + Fore.RESET + +def parseJSON(data, filename): + totalTokens = 0 + totalLines = 0 + totalLines = len(data) + global LOCK + + with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar: + pbar.desc=filename + pbar.total=totalLines + try: + totalTokens += translateJSON(data, pbar) + except Exception as e: + return [data, totalTokens, e] + return [data, totalTokens, None] + +def translateJSON(data, pbar): + textHistory = [] + maxHistory = MAXHISTORY + tokens = 0 + + for key, value in data.items(): + # Remove any textwrap + if FIXTEXTWRAP == True: + value = re.sub(r'@b', ' ', value) + + # Translate + if value == '': + response = translateGPT(key, 'Past Translated Text: ' + '|\n\n'.join(textHistory), True) + tokens += response[1] + translatedText = response[0] + textHistory.append('\"' + translatedText + '\"') + else: + translatedText = value + textHistory.append('\"' + translatedText + '\"') + + # Textwrap + translatedText = textwrap.fill(translatedText, width=WIDTH) + translatedText = translatedText.replace('\n', '@b') + + # Set Data + data[key] = translatedText + + # Keep textHistory list at length maxHistory + if len(textHistory) > maxHistory: + textHistory.pop(0) + currentGroup = [] + pbar.update(1) + + return tokens + +def subVars(jaString): + jaString = jaString.replace('\u3000', ' ') + varRegex = r'[\\]+[\w.\\\s]+?\[.+?\]]?|[\\]+[\w.\\]+?\<.+?\>\>?|[\\]+[#{}<>.]' + count = 0 + + varList = re.findall(varRegex, jaString) + varList = set(varList) + if len(varList) != 0: + for var in varList: + jaString = jaString.replace(var, '@' + str(count) + '') + count += 1 + + return [jaString, varList] + +def resubVars(translatedText, varList): + count = 0 + + # Fix Spacing and ChatGPT Nonsense + matchList = re.findall(r'@\s?[0-9]+?', translatedText) + if len(matchList) > 0: + for match in matchList: + text = match.replace(' ', '') + translatedText = translatedText.replace(match, text) + + if len(varList) != 0: + for var in varList: + translatedText = translatedText.replace('@' + str(count) + '', var) + count += 1 + + # Remove Color Variables Spaces + # if '\\c' in translatedText: + # translatedText = re.sub(r'\s*(\\+c\[[1-9]+\])\s*', r' \1', translatedText) + # translatedText = re.sub(r'\s*(\\+c\[0+\])', r'\1', translatedText) + return translatedText + +@retry(exceptions=Exception, tries=5, delay=5) +def translateGPT(t, history, fullPromptFlag): + # If ESTIMATE is True just count this as an execution and return. + if ESTIMATE: + enc = tiktoken.encoding_for_model("gpt-3.5-turbo") + tokens = len(enc.encode(t)) * 2 + len(enc.encode(history)) + len(enc.encode(PROMPT)) + return (t, tokens) + + # Sub Vars + varResponse = subVars(t) + subbedT = varResponse[0] + + # If there isn't any Japanese in the text just skip + if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴー]+', subbedT): + return(t, 0) + + """Translate text using GPT""" + context = 'Eroge Names Context: Name: 稲盛 楓 == Inamori Kaede\nNicknames: かえちゃん == Kae-chan or かえねぇ == Kae-nee\nGender: Female,\nName: 稲盛 真守 == Inamori Mamoru\nNicknames: まーくん == Maa-kun\nGender: Male,\nName: 蓮見 雄次郎 == Hasumi Yujiro\nGender: Male,\nName: 桐谷 拓馬 == Kiriya Takuma\nNicknames: たっくん == Tak-kun\nGender: Male,\nName: 稲盛 美玖 == Inamori Yuki\nGender: Female,\nName: 奥さん == Missus\nGender: Female' + if fullPromptFlag: + system = PROMPT + user = 'Line to Translate: ' + subbedT + else: + system = 'You are an expert translator who translates everything to English. Reply with only the English Translation of the text.' + user = 'Line to Translate: ' + subbedT + response = openai.ChatCompletion.create( + temperature=0, + frequency_penalty=0.2, + presence_penalty=0.2, + model="gpt-3.5-turbo", + messages=[ + {"role": "system", "content": system}, + {"role": "user", "content": context}, + {"role": "user", "content": history}, + {"role": "user", "content": user} + ], + request_timeout=30, + ) + + # Save Translated Text + translatedText = response.choices[0].message.content + tokens = response.usage.total_tokens + + # Resub Vars + translatedText = resubVars(translatedText, varResponse[1]) + + # Remove Placeholder Text + translatedText = translatedText.replace('English Translation: ', '') + translatedText = translatedText.replace('Translation: ', '') + translatedText = translatedText.replace('Line to Translate: ', '') + translatedText = translatedText.replace('English Translation:', '') + translatedText = translatedText.replace('Translation:', '') + translatedText = translatedText.replace('Line to Translate:', '') + translatedText = re.sub(r'\n\nPast Translated Text:.*', '', translatedText, 0, re.DOTALL) + translatedText = re.sub(r'Note:.*', '', translatedText) + + # Return Translation + if len(translatedText) > 15 * len(t) or "I'm sorry, but I'm unable to assist with that translation" in translatedText: + return [t, response.usage.total_tokens] + else: + return [translatedText, tokens] \ No newline at end of file diff --git a/modules/main.py b/modules/main.py index 18659bb..02bcbe4 100644 --- a/modules/main.py +++ b/modules/main.py @@ -6,9 +6,10 @@ import os from modules.rpgmakermvmz import handleMVMZ from modules.rpgmakerace import handleACE -from modules.csvtl import handleCSV -from modules.textfile import handleTextfile +from modules.csv import handleCSV +from modules.txt import handleTXT from modules.tyrano import handleTyrano +from modules.json import handleJSON THREADS = 10 # For GPT4 rate limit will be hit if you have more than 1 thread. @@ -30,7 +31,7 @@ def main(): totalCost = 0 version = '' while version == '': - version = input('Select the RPGMaker Version:\n\n1. MV/MZ\n2. ACE\n3. CSV (From Translator++)\n4. Text (Custom)\n5. Tyrano\n') + version = input('Select the RPGMaker Version:\n\n1. MV/MZ\n2. ACE\n3. CSV (From Translator++)\n4. Text (Custom)\n5. Tyrano\n6. JSON\n') match version: case '1': # Open File (Threads) @@ -75,7 +76,7 @@ def main(): case '4': # Open File (Threads) with ThreadPoolExecutor(max_workers=THREADS) as executor: - futures = [executor.submit(handleTextfile, filename, estimate) \ + futures = [executor.submit(handleTXT, filename, estimate) \ for filename in os.listdir("files") if filename.endswith('txt')] for future in as_completed(futures): @@ -100,6 +101,20 @@ def main(): tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno) print(Fore.RED + str(e) + '|' + tracebackLineNo + Fore.RESET) + case '6': + # Open File (Threads) + with ThreadPoolExecutor(max_workers=THREADS) as executor: + futures = [executor.submit(handleJSON, filename, estimate) \ + for filename in os.listdir("files") if filename.endswith('json')] + + for future in as_completed(futures): + try: + totalCost = future.result() + + except Exception as e: + tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno) + print(Fore.RED + str(e) + '|' + tracebackLineNo + Fore.RESET) + case _: version = '' @@ -110,7 +125,7 @@ def main(): # Prevent immediately closing of CLI print(totalCost) - input('Done! Press Enter to close.') + # input('Done! Press Enter to close.') def deleteFolderFiles(folderPath): for filename in os.listdir(folderPath): diff --git a/modules/rpgmakerace.py b/modules/rpgmakerace.py index 5284dd6..caad0aa 100644 --- a/modules/rpgmakerace.py +++ b/modules/rpgmakerace.py @@ -54,9 +54,10 @@ CODE324 = False CODE111 = False CODE408 = False CODE108 = False -NAMES = False # Output a list of all the character names found +NAMES = False # Output a list of all the character names found BRFLAG = False # If the game uses
instead -FIXTEXTWRAP = True +FIXTEXTWRAP = True # Adjust wordwrap of text (IGNORETLTEXT must be False) +IGNORETLTEXT = True # Leave this False if you need to adjust the wordwrap def handleACE(filename, estimate): global ESTIMATE, TOTALTOKENS, TOTALCOST @@ -193,12 +194,6 @@ def parseMap(data, filename): events = data['events'] global LOCK - # Translate displayName for Map files - if 'Map' in filename: - response = translateGPT(data['displayName'], 'Reply with only the english translation of the RPG location name', False) - totalTokens += response[1] - data['displayName'] = response[0].replace('\"', '') - # Get total for progress bar for key in events: if key is not None: @@ -523,6 +518,12 @@ def searchCodes(page, pbar): jaString = codeList[i]['p'][0] firstJAString = jaString + # If there isn't any Japanese in the text just skip + if IGNORETLTEXT == True: + if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴー]+', jaString): + textHistory.append('\"' + jaString + '\"') + continue + # Using this to keep track of 401's in a row. Throws IndexError at EndOfList (Expected Behavior) currentGroup.append(jaString) @@ -1362,7 +1363,7 @@ def translateGPT(t, history, fullPromptFlag): return(t, 0) """Translate text using GPT""" - context = 'Eroge Names Context: アサギ == Asagi | Female, ウィップ == Whip | Female, ウラ == Ura | Female, ブレイド == Blade | Female' + context = 'Eroge Character Names Context: アサギ == Asagi | Female, ウィップ == Whip | Female, ウラ == Ura | Female, ブレイド == Blade | Female' if fullPromptFlag: system = PROMPT user = 'Line to Translate: ' + subbedT diff --git a/modules/textfile.py b/modules/txt.py similarity index 99% rename from modules/textfile.py rename to modules/txt.py index a542d45..fe6f165 100644 --- a/modules/textfile.py +++ b/modules/txt.py @@ -49,7 +49,7 @@ CODE356 = False CODE320 = False CODE111 = False -def handleTextfile(filename, estimate): +def handleTXT(filename, estimate): global ESTIMATE, TOKENS, TOTALTOKENS, TOTALCOST ESTIMATE = estimate