From de591011c65c2fb47d8fc45765b41ddf128f3e7b Mon Sep 17 00:00:00 2001 From: Dazed Date: Mon, 23 Oct 2023 15:27:00 -0500 Subject: [PATCH] Add Kikiriki support --- modules/kikiriki.py | 404 ++++++++++++++++++++++++++++++++++++++++ modules/main.py | 17 +- modules/rpgmakermvmz.py | 8 +- modules/tyrano.py | 70 +++---- 4 files changed, 453 insertions(+), 46 deletions(-) create mode 100644 modules/kikiriki.py diff --git a/modules/kikiriki.py b/modules/kikiriki.py new file mode 100644 index 0000000..f63114f --- /dev/null +++ b/modules/kikiriki.py @@ -0,0 +1,404 @@ +from concurrent.futures import ThreadPoolExecutor, as_completed +import os +from pathlib import Path +import re +import threading +import time +import traceback +import tiktoken + +from colorama import Fore +from dotenv import load_dotenv +import openai +from retry import retry +from tqdm import tqdm + +#Globals +load_dotenv() +openai.organization = os.getenv('org') +openai.api_key = os.getenv('key') + +APICOST = .002 # Depends on the model https://openai.com/pricing +PROMPT = Path('prompt.txt').read_text(encoding='utf-8') +THREADS = 10 # For GPT4 rate limit will be hit if you have more than 1 thread. +LOCK = threading.Lock() +WIDTH = 60 +LISTWIDTH = 60 +MAXHISTORY = 10 +ESTIMATE = '' +TOTALCOST = 0 +TOKENS = 0 +TOTALTOKENS = 0 +NAMESLIST = [] + +#tqdm Globals +BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' +POSITION=0 +LEAVE=False + +# Flags +NAMES = False # Output a list of all the character names found +BRFLAG = False # If the game uses
instead +FIXTEXTWRAP = False + +def handleKikiriki(filename, estimate): + global ESTIMATE, TOKENS, TOTALTOKENS, TOTALCOST + ESTIMATE = estimate + + if estimate: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOTALCOST += translatedData[1] * .001 * APICOST + TOTALTOKENS += translatedData[1] + + return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL') + + else: + try: + with open('translated/' + filename, 'w', encoding='shift-jis') as outFile: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + outFile.writelines(translatedData[0]) + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOTALCOST += translatedData[1] * .001 * APICOST + TOTALTOKENS += translatedData[1] + except Exception as e: + traceback.print_exc() + return 'Fail' + + return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL') + +def openFiles(filename): + with open('files/' + filename, 'r', encoding='shift-jis') as readFile: + translatedData = parseTyrano(readFile, filename) + + # Delete lines marked for deletion + finalData = [] + for line in translatedData[0]: + if line != '\\d\n': + finalData.append(line) + translatedData[0] = finalData + + return translatedData + +def parseTyrano(readFile, filename): + totalTokens = 0 + totalLines = 0 + + # Get total for progress bar + data = readFile.readlines() + totalLines = len(data) + + with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar: + pbar.desc=filename + pbar.total=totalLines + + try: + totalTokens += translateTyrano(data, pbar) + except Exception as e: + traceback.print_exc() + return [data, totalTokens, e] + return [data, totalTokens, None] + +def translateTyrano(data, pbar): + textHistory = [] + maxHistory = MAXHISTORY + tokens = 0 + currentGroup = [] + syncIndex = 0 + speaker = '' + global LOCK, ESTIMATE + + for i in range(len(data)): + if syncIndex > i: + i = syncIndex + + # Speaker + if '[ns]' in data[i]: + matchList = re.findall(r'#(.+)', data[i]) + if len(matchList) != 0: + response = translateGPT(matchList[0], 'Reply with only the english translation of the NPC name', True) + speaker = response[0] + tokens += response[1] + data[i] = '#' + speaker + '\n' + else: + speaker = '' + + # Choices + elif 'glink' in data[i]: + matchList = re.findall(r'text=\"(.+?)\"', data[i]) + if len(matchList) != 0: + if len(textHistory) > 0: + response = translateGPT(matchList[0], 'Past Translated Text: ' + textHistory[len(textHistory)-1] + '\n\nReply in the style of a dialogue option.', True) + else: + response = translateGPT(matchList[0], '', False) + translatedText = response[0] + tokens += response[1] + + # Remove characters that may break scripts + charList = ['.', '\"', '\\n'] + for char in charList: + translatedText = translatedText.replace(char, '') + + # Set Data + translatedText = 'text=\"' + translatedText.replace(' ', ' ') + '\"' + data[i] = re.sub(r'text=\"(.+?)\"', translatedText, data[i]) + + # Lines + elif '[r]' in data[i]: + matchList = re.findall(r'(.+?)\[r\]', data[i]) + if len(matchList) > 0: + matchList[0] = matchList[0].replace('「', '') + matchList[0] = matchList[0].replace('」', '') + currentGroup.append(matchList[0]) + if len(data) > i+1: + while '[r]' in data[i+1]: + data[i] = '\d\n' + i += 1 + matchList = re.findall(r'(.+?)\[r\]', data[i]) + if len(matchList) > 0: + matchList[0] = matchList[0].replace('「', '') + matchList[0] = matchList[0].replace('」', '') + currentGroup.append(matchList[0]) + # Join up 401 groups for better translation. + if len(currentGroup) > 0: + finalJAString = ''.join(currentGroup) + oldjaString = finalJAString + + #Check Speaker + if speaker == '': + response = translateGPT(finalJAString, 'Previous Dialogue: ' + '\n\n'.join(textHistory), True) + tokens += response[1] + translatedText = response[0] + textHistory.append('\"' + translatedText + '\"') + else: + response = translateGPT(speaker + ': ' + finalJAString, 'Previous Dialogue: ' + '\n\n'.join(textHistory), True) + tokens += response[1] + translatedText = response[0] + textHistory.append('\"' + translatedText + '\"') + + # Remove added speaker + translatedText = re.sub(r'^.+:\s?', '', translatedText) + + # Set Data + translatedText = translatedText.replace('ッ', '') + translatedText = translatedText.replace('っ', '') + translatedText = translatedText.replace('ー', '') + translatedText = translatedText.replace('\"', '') + + # Format Text + matchList = re.findall(r'(.+?[)\.\?\!)。・]+)', translatedText) + translatedText = re.sub(r'(.+?[)\.\?\!)。・]+)', '', translatedText) + + # Combine Lists + for k in range(len(matchList)): + matchList[k] = matchList[k].strip() + j=0 + while(len(matchList) > j+1): + while len(matchList[j]) < 100 and len(matchList) > j: + matchList[j:j+2] = [' '.join(matchList[j:j+2])] + if len(matchList) == j+1: + matchList[j] = matchList[j] + ' ' + translatedText + translatedText = '' + break + j+=1 + + if len(matchList) > 0: + data[i] = '\d\n' + for line in matchList: + data.insert(i, line.strip() + '[r]\n') + i+=1 + # else: + # print ('No Matches') + if translatedText != '': + data[i] = translatedText.strip() + '[r]\n' + + # Keep textHistory list at length maxHistory + if len(textHistory) > maxHistory: + textHistory.pop(0) + currentGroup = [] + + currentGroup = [] + pbar.update(1) + if len(data) > i+1: + syncIndex = i+1 + else: + break + + return tokens + +def getResultString(translatedData, translationTime, filename): + # File Print String + tokenString = Fore.YELLOW + '[' + str(translatedData[1]) + \ + ' Tokens/${:,.4f}'.format(translatedData[1] * .001 * APICOST) + ']' + timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' + + if translatedData[2] == None: + # Success + return filename + ': ' + tokenString + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET + + else: + # Fail + try: + raise translatedData[2] + except Exception as e: + traceback.print_exc() + errorString = str(e) + Fore.RED + return filename + ': ' + tokenString + timeString + Fore.RED + u' \u2717 ' +\ + errorString + Fore.RESET + +def subVars(jaString): + jaString = jaString.replace('\u3000', ' ') + + # Icons + count = 0 + iconList = re.findall(r'[\\]+[iIkKwW]+\[[0-9]+\]', jaString) + iconList = set(iconList) + if len(iconList) != 0: + for icon in iconList: + jaString = jaString.replace(icon, '[Icon' + str(count) + ']') + count += 1 + + # Colors + count = 0 + colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString) + colorList = set(colorList) + if len(colorList) != 0: + for color in colorList: + jaString = jaString.replace(color, '[Color' + str(count) + ']') + count += 1 + + # Names + count = 0 + nameList = re.findall(r'[\\]+[nN]\[[0-9]+\]', jaString) + nameList = set(nameList) + if len(nameList) != 0: + for name in nameList: + jaString = jaString.replace(name, '[Name' + str(count) + ']') + count += 1 + + # Variables + count = 0 + varList = re.findall(r'[\\]+[vV]\[[0-9]+\]', jaString) + varList = set(varList) + if len(varList) != 0: + for var in varList: + jaString = jaString.replace(var, '[Var' + str(count) + ']') + count += 1 + + # Put all lists in list and return + allList = [iconList, colorList, nameList, varList] + return [jaString, allList] + +def resubVars(translatedText, allList): + # Fix Spacing and ChatGPT Nonsense + matchList = re.findall(r'\[\s?.+?\s?\]', translatedText) + if len(matchList) > 0: + for match in matchList: + text = match.replace(' ', '') + translatedText = translatedText.replace(match, text) + + # Icons + count = 0 + if len(allList[0]) != 0: + for var in allList[0]: + translatedText = translatedText.replace('[Icon' + str(count) + ']', var) + count += 1 + + # Colors + count = 0 + if len(allList[1]) != 0: + for var in allList[1]: + translatedText = translatedText.replace('[Color' + str(count) + ']', var) + count += 1 + + # Names + count = 0 + if len(allList[2]) != 0: + for var in allList[2]: + translatedText = translatedText.replace('[Name' + str(count) + ']', var) + count += 1 + + # Vars + count = 0 + if len(allList[3]) != 0: + for var in allList[3]: + translatedText = translatedText.replace('[Var' + str(count) + ']', var) + count += 1 + + # Remove Color Variables Spaces + # if '\\c' in translatedText: + # translatedText = re.sub(r'\s*(\\+c\[[1-9]+\])\s*', r' \1', translatedText) + # translatedText = re.sub(r'\s*(\\+c\[0+\])', r'\1', translatedText) + return translatedText + +@retry(exceptions=Exception, tries=5, delay=5) +def translateGPT(t, history, fullPromptFlag): + # If ESTIMATE is True just count this as an execution and return. + if ESTIMATE: + enc = tiktoken.encoding_for_model("gpt-3.5-turbo") + tokens = len(enc.encode(t)) * 2 + len(enc.encode(history)) + len(enc.encode(PROMPT)) + return (t, tokens) + + # Sub Vars + varResponse = subVars(t) + subbedT = varResponse[0] + + # If there isn't any Japanese in the text just skip + if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴ]+|[\uFF00-\uFFEF]', subbedT): + return(t, 0) + + """Translate text using GPT""" + context = 'Eroge Names Context: (Name: サクラ == Sakura\nGender: Female,\n\nName: ルリカ == Rurika\nGender: Female,\n\nName: ケンイチ == Kenichi\nGender: Male)' + if fullPromptFlag: + system = PROMPT + user = 'Line to Translate: ' + subbedT + else: + system = 'You are an expert translator who translates everything to English. Reply with only the English Translation of the text.' + user = 'Line to Translate: ' + subbedT + response = openai.ChatCompletion.create( + temperature=0, + frequency_penalty=0.2, + presence_penalty=0.2, + model="gpt-3.5-turbo", + messages=[ + {"role": "system", "content": system}, + {"role": "user", "content": context}, + {"role": "user", "content": history}, + {"role": "user", "content": user} + ], + request_timeout=30, + ) + + # Save Translated Text + translatedText = response.choices[0].message.content + tokens = response.usage.total_tokens + + # Resub Vars + translatedText = resubVars(translatedText, varResponse[1]) + + # Remove Placeholder Text + translatedText = translatedText.replace('English Translation: ', '') + translatedText = translatedText.replace('Translation: ', '') + translatedText = translatedText.replace('Line to Translate: ', '') + translatedText = translatedText.replace('English Translation:', '') + translatedText = translatedText.replace('Translation:', '') + translatedText = translatedText.replace('Line to Translate:', '') + translatedText = re.sub(r'\n\nPast Translated Text:.*', '', translatedText, 0, re.DOTALL) + translatedText = re.sub(r'Note:.*', '', translatedText) + translatedText = translatedText.replace('っ', '') + + # Return Translation + if len(translatedText) > 15 * len(t) or "I'm sorry, but I'm unable to assist with that translation" in translatedText: + raise Exception + else: + return [translatedText, tokens] \ No newline at end of file diff --git a/modules/main.py b/modules/main.py index 3a106d4..0a1f6cc 100644 --- a/modules/main.py +++ b/modules/main.py @@ -10,6 +10,7 @@ from modules.csv import handleCSV from modules.txt import handleTXT from modules.tyrano import handleTyrano from modules.json import handleJSON +from modules.kikiriki import handleKikiriki THREADS = 9 # For GPT4 rate limit will be hit if you have more than 1 thread. @@ -31,7 +32,7 @@ def main(): totalCost = 0 version = '' while version == '': - version = input('Select the RPGMaker Version:\n\n1. MV/MZ\n2. ACE\n3. CSV (From Translator++)\n4. Text (Custom)\n5. Tyrano\n6. JSON\n') + version = input('Select the RPGMaker Version:\n\n1. MV/MZ\n2. ACE\n3. CSV (From Translator++)\n4. Text (Custom)\n5. Tyrano\n6. JSON\n7. Kikiriki\n') match version: case '1': # Open File (Threads) @@ -115,6 +116,20 @@ def main(): tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno) print(Fore.RED + str(e) + '|' + tracebackLineNo + Fore.RESET) + case '7': + # Open File (Threads) + with ThreadPoolExecutor(max_workers=THREADS) as executor: + futures = [executor.submit(handleKikiriki, filename, estimate) \ + for filename in os.listdir("files") if filename.endswith('ks')] + + for future in as_completed(futures): + try: + totalCost = future.result() + + except Exception as e: + tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno) + print(Fore.RED + str(e) + '|' + tracebackLineNo + Fore.RESET) + case _: version = '' diff --git a/modules/rpgmakermvmz.py b/modules/rpgmakermvmz.py index 6e29ffd..6af2003 100644 --- a/modules/rpgmakermvmz.py +++ b/modules/rpgmakermvmz.py @@ -710,7 +710,7 @@ def searchCodes(page, pbar): # Translate if speaker == '' and finalJAString != '': - response = translateGPT(finalJAString, 'Past Translated Text: (' + ' || '.join(textHistory) + ')', True) + response = translateGPT(finalJAString, 'Past Translated Text: (' + ', '.join(textHistory) + ')', True) tokens += response[1] translatedText = response[0] textHistory.append('\"' + translatedText + '\"') @@ -1420,7 +1420,7 @@ def searchSS(state, pbar): def searchSystem(data, pbar): tokens = 0 - context = 'Reply with only the english translation of the UI textbox.\nTranslate "逃げる" as "Escape".\nTranslate"大事なもの" as "Key Items.\nTranslate "最強装備" as "Optimize".\nTranslate "攻撃力" as "Attack".\nTranslate "回避率" as "Evasion".\nTranslate "最大HP" as "Max HP".\nTranslate "経験値" as "EXP".\n Translate "購入する" as "Buy"' + context = 'Reply with only the english translation of the UI textbox.\nTranslate "逃げる" as "Escape".\nTranslate"大事なもの" as "Key Items.\nTranslate "最強装備" as "Optimize".\nTranslate "攻撃力" as "Attack".\nTranslate "最大HP" as "Max HP".\nTranslate "経験値" as "EXP".\n Translate "購入する" as "Buy"' # Title response = translateGPT(data['gameTitle'], ' Reply with the English translation of the game title name', False) @@ -1470,7 +1470,7 @@ def searchSystem(data, pbar): # Messages messages = (data['terms']['messages']) for key, value in messages.items(): - response = translateGPT(value, 'Reply with only the english translation of the battle text.', False) + response = translateGPT(value, 'Reply with only the english translation of the battle text.\nTranslate "常時ダッシュ" as "Always Dash".', False) translatedText = response[0] # Remove characters that may break scripts @@ -1582,7 +1582,7 @@ def translateGPT(t, history, fullPromptFlag): subbedT = varResponse[0] # If there isn't any Japanese in the text just skip - if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴ]+', subbedT): + if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴ]+|[\uFF00-\uFFEF]', subbedT): return(t, 0) """Translate text using GPT""" diff --git a/modules/tyrano.py b/modules/tyrano.py index 94019b4..b5a4308 100644 --- a/modules/tyrano.py +++ b/modules/tyrano.py @@ -264,56 +264,47 @@ def subVars(jaString): # Icons count = 0 - iconList = re.findall(r'[\\]+[iI]\[[0-9]+\]', jaString) + iconList = re.findall(r'[\\]+[iIkKwW]+\[[0-9]+\]', jaString) iconList = set(iconList) if len(iconList) != 0: for icon in iconList: - jaString = jaString.replace(icon, '') + jaString = jaString.replace(icon, '[Icon' + str(count) + ']') count += 1 # Colors count = 0 colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString) colorList = set(colorList) - if len(iconList) != 0: + if len(colorList) != 0: for color in colorList: - jaString = jaString.replace(color, '') + jaString = jaString.replace(color, '[Color' + str(count) + ']') count += 1 # Names count = 0 nameList = re.findall(r'[\\]+[nN]\[[0-9]+\]', jaString) nameList = set(nameList) - if len(iconList) != 0: + if len(nameList) != 0: for name in nameList: - jaString = jaString.replace(name, '') + jaString = jaString.replace(name, '[Name' + str(count) + ']') count += 1 # Variables count = 0 varList = re.findall(r'[\\]+[vV]\[[0-9]+\]', jaString) varList = set(varList) - if len(iconList) != 0: + if len(varList) != 0: for var in varList: - jaString = jaString.replace(var, '') - count += 1 - - # Formatting - count = 0 - formatList = re.findall(r'[\\]+[!.]', jaString) - formatList = set(formatList) - if len(formatList) != 0: - for format in formatList: - jaString = jaString.replace(format, '') + jaString = jaString.replace(var, '[Var' + str(count) + ']') count += 1 # Put all lists in list and return - allList = [iconList, colorList, nameList, varList, formatList] + allList = [iconList, colorList, nameList, varList] return [jaString, allList] def resubVars(translatedText, allList): # Fix Spacing and ChatGPT Nonsense - matchList = re.findall(r'<\s?.+?\s?>', translatedText) + matchList = re.findall(r'\[\s?.+?\s?\]', translatedText) if len(matchList) > 0: for match in matchList: text = match.replace(' ', '') @@ -323,57 +314,54 @@ def resubVars(translatedText, allList): count = 0 if len(allList[0]) != 0: for var in allList[0]: - translatedText = translatedText.replace('', var) + translatedText = translatedText.replace('[Icon' + str(count) + ']', var) count += 1 # Colors count = 0 if len(allList[1]) != 0: for var in allList[1]: - translatedText = translatedText.replace('', var) + translatedText = translatedText.replace('[Color' + str(count) + ']', var) count += 1 # Names count = 0 if len(allList[2]) != 0: for var in allList[2]: - translatedText = translatedText.replace('', var) + translatedText = translatedText.replace('[Name' + str(count) + ']', var) count += 1 # Vars count = 0 if len(allList[3]) != 0: for var in allList[3]: - translatedText = translatedText.replace('', var) - count += 1 - - # Formatting - count = 0 - if len(allList[4]) != 0: - for var in allList[4]: - translatedText = translatedText.replace('', var) + translatedText = translatedText.replace('[Var' + str(count) + ']', var) count += 1 + # Remove Color Variables Spaces + # if '\\c' in translatedText: + # translatedText = re.sub(r'\s*(\\+c\[[1-9]+\])\s*', r' \1', translatedText) + # translatedText = re.sub(r'\s*(\\+c\[0+\])', r'\1', translatedText) + return translatedText + @retry(exceptions=Exception, tries=5, delay=5) def translateGPT(t, history, fullPromptFlag): - global LOCK, TOKENS - with LOCK: - # If ESTIMATE is True just count this as an execution and return. - if ESTIMATE: - enc = tiktoken.encoding_for_model("gpt-3.5-turbo") - tokens = len(enc.encode(t)) * 2 + len(enc.encode(history)) + len(enc.encode(PROMPT)) - return (t, tokens) + # If ESTIMATE is True just count this as an execution and return. + if ESTIMATE: + enc = tiktoken.encoding_for_model("gpt-3.5-turbo") + tokens = len(enc.encode(t)) * 2 + len(enc.encode(history)) + len(enc.encode(PROMPT)) + return (t, tokens) # Sub Vars varResponse = subVars(t) subbedT = varResponse[0] # If there isn't any Japanese in the text just skip - if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴー]+', subbedT): + if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴ]+|[\uFF00-\uFFEF]', subbedT): return(t, 0) """Translate text using GPT""" - context = 'Eroge Names Context: 桐嶋 香織 == Kaori Kirishima | Female, 肉山 猛 == Takeshi Nikuyama' + context = 'Eroge Names Context: (Name: サクラ == Sakura\nGender: Female,\n\nName: ルリカ == Rurika\nGender: Female,\n\nName: ケンイチ == Kenichi\nGender: Male)' if fullPromptFlag: system = PROMPT user = 'Line to Translate: ' + subbedT @@ -410,10 +398,10 @@ def translateGPT(t, history, fullPromptFlag): translatedText = translatedText.replace('Line to Translate:', '') translatedText = re.sub(r'\n\nPast Translated Text:.*', '', translatedText, 0, re.DOTALL) translatedText = re.sub(r'Note:.*', '', translatedText) + translatedText = translatedText.replace('っ', '') # Return Translation if len(translatedText) > 15 * len(t) or "I'm sorry, but I'm unable to assist with that translation" in translatedText: - return [t, response.usage.total_tokens] + raise Exception else: - return [translatedText, tokens] \ No newline at end of file