diff --git a/modules/kansen.py b/modules/kansen.py index 77c81b4..f7c31a7 100644 --- a/modules/kansen.py +++ b/modules/kansen.py @@ -354,36 +354,36 @@ def translateTyrano(data, pbar, totalLines): # Save some money and enter the character before translation def getSpeaker(speaker): match speaker: - case '大介': - return ['Daisuke', [0,0]] - case '眞琴': + case '誠': return ['Makoto', [0,0]] - case '翔': - return ['Shou', [0,0]] - case '冴子': - return ['Saeko', [0,0]] - case '絢': - return ['Aya', [0,0]] - case '梢': - return ['Kozue', [0,0]] - case '壬': - return ['Jin', [0,0]] - case '緒織': - return ['Inori', [0,0]] - case '浩助': - return ['Kousuke', [0,0]] - case '太宰': - return ['Dazai', [0,0]] - case '大嶋': - return ['Oshimi', [0,0]] - case 'セスカ': - return ['Sesuka', [0,0]] - case '重吉': - return ['Shigeyoshi', [0,0]] - case '忠彦': - return ['Tadahiko', [0,0]] - case '和歌': - return ['Waka', [0,0]] + case '夏都': + return ['Natsu', [0,0]] + case '宗一郎': + return ['Souichirou', [0,0]] + case '彩月': + return ['Satsuki', [0,0]] + case '茜梨': + return ['Akari', [0,0]] + case 'ポホヨネン': + return ['Pohjonen', [0,0]] + case '荒井': + return ['Arai', [0,0]] + case '愛梨': + return ['Airi', [0,0]] + case '朋美': + return ['Tomomi', [0,0]] + case '玄治郎': + return ['Genjirou', [0,0]] + case '稼津央': + return ['Kazuo', [0,0]] + case '美沙緒': + return ['Misao', [0,0]] + case '怜': + return ['Sato', [0,0]] + case 'オス': + return ['Oz', [0,0]] + case '穂村': + return ['Homura', [0,0]] case '吉野': return ['Yoshino', [0,0]] case '忠彦': @@ -512,12 +512,17 @@ def batchList(input_list, batch_size): def createContext(fullPromptFlag, subbedT): characters = 'Game Characters:\n\ -仙道 大介 (Sendou Daisuke) - Male\n\ -鐙 眞琴 (Abumi Makoto) - Female\n\ -石郷岡 翔 (Ishigooka Shou) - Male\n\ -桐越 冴子 (Kirikoshi Saeko) - Female\n\ -真坂 絢 (Masaka Aya) - Female\n\ -能登屋 梢 (Notoya Kozue) - Female\n\ +中澤 誠 (Nakazawa Makoto) - Male\n\ +日向 夏都 (Hyuuga Natsu) - Female\n\ +出渕 宗一郎 (Izubuchi Souichirou) - Male\n\ +南 彩月 (Minami Satsuki) - Female\n\ +越智 茜梨 (Ochi Akari) - Female\n\ +ターヤ ポホヨネン (Tarja Pohjonen) - Female\n\ +マルガリータ バスクェス 穂村 (Margarita Vasquez Homura) - Female\n\ +花沢 愛梨 (Hanazawa Airi) - Female\n\ +五十嵐 朋美 (Igarashi Tomomi) - Female\n\ +前田 美沙緒 (Maeda Misao) - Female\n\ +村上 怜 (Murakami Sato) - Female\n\ ' system = PROMPT if fullPromptFlag else \ diff --git a/modules/tyrano.py b/modules/tyrano.py index 3dfd708..195f42b 100644 --- a/modules/tyrano.py +++ b/modules/tyrano.py @@ -1,13 +1,6 @@ -import os -import re -import textwrap -import threading -import time -import traceback +# Libraries +import json, os, re, textwrap, threading, time, traceback, tiktoken, openai from pathlib import Path - -import openai -import tiktoken from colorama import Fore from dotenv import load_dotenv from retry import retry @@ -15,428 +8,453 @@ from tqdm import tqdm # Open AI load_dotenv() -if os.getenv("api").replace(" ", "") != "": - openai.api_base = os.getenv("api") -openai.organization = os.getenv("org") -openai.api_key = os.getenv("key") +if os.getenv('api').replace(' ', '') != '': + openai.api_base = os.getenv('api') +openai.organization = os.getenv('org') +openai.api_key = os.getenv('key') -# Globals -MODEL = os.getenv("model") -TIMEOUT = int(os.getenv("timeout")) -LANGUAGE = os.getenv("language").capitalize() -INPUTAPICOST = 0.002 # Depends on the model https://openai.com/pricing -OUTPUTAPICOST = 0.002 -PROMPT = Path("prompt.txt").read_text(encoding="utf-8") -THREADS = int( - os.getenv("threads") -) # Controls how many threads are working on a single file (May have to drop this) +#Globals +MODEL = os.getenv('model') +TIMEOUT = int(os.getenv('timeout')) +LANGUAGE = os.getenv('language').capitalize() +PROMPT = Path('prompt.txt').read_text(encoding='utf-8') +THREADS = int(os.getenv('threads')) LOCK = threading.Lock() -WIDTH = int(os.getenv("width")) -LISTWIDTH = int(os.getenv("listWidth")) -NOTEWIDTH = 40 +WIDTH = int(os.getenv('width')) +LISTWIDTH = int(os.getenv('listWidth')) +NOTEWIDTH = 70 MAXHISTORY = 10 -ESTIMATE = "" -totalTokens = [0, 0] +ESTIMATE = '' +TOKENS = [0, 0] NAMESLIST = [] +NAMES = False # Output a list of all the character names found +BRFLAG = False # If the game uses
instead +FIXTEXTWRAP = False # Overwrites textwrap +IGNORETLTEXT = False # Ignores all translated text. +MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) -# tqdm Globals -BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" +#tqdm Globals +BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' POSITION = 0 LEAVE = False -# Flags -NAMES = False # Output a list of all the character names found -BRFLAG = False # If the game uses
instead -FIXTEXTWRAP = True -IGNORETLTEXT = False - +# Pricing - Depends on the model https://openai.com/pricing +# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request +# If you are getting a MISMATCH LENGTH error, lower the batch size. +if 'gpt-3.5' in MODEL: + INPUTAPICOST = .002 + OUTPUTAPICOST = .002 + BATCHSIZE = 10 +elif 'gpt-4' in MODEL: + INPUTAPICOST = .01 + OUTPUTAPICOST = .03 + BATCHSIZE = 40 def handleTyrano(filename, estimate): global ESTIMATE - totalTokens = [0, 0] ESTIMATE = estimate - if estimate: + if ESTIMATE: start = time.time() translatedData = openFiles(filename) # Print Result end = time.time() tqdm.write(getResultString(translatedData, end - start, filename)) - if NAMES is True: - tqdm.write(str(NAMESLIST)) with LOCK: - totalTokens[0] += translatedData[1][0] - totalTokens[1] += translatedData[1][1] + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] - return getResultString(["", totalTokens, None], end - start, "TOTAL") + # Print Total + totalString = getResultString(['', TOKENS, None], end - start, 'TOTAL') + # Print any errors on maps + if len(MISMATCH) > 0: + return totalString + Fore.RED + f'\nMismatch Errors: {MISMATCH}' + Fore.RESET + else: + return totalString + else: try: - with open("translated/" + filename, "w", encoding="utf-8") as outFile: + with open('translated/' + filename, 'w', encoding='shift_jis', errors='ignore') as outFile: start = time.time() translatedData = openFiles(filename) - outFile.writelines(translatedData[0]) # Print Result end = time.time() + outFile.writelines(translatedData[0]) tqdm.write(getResultString(translatedData, end - start, filename)) with LOCK: - totalTokens[0] += translatedData[1][0] - totalTokens[1] += translatedData[1][1] - except Exception: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + except Exception as e: traceback.print_exc() - return "Fail" - - return getResultString(["", totalTokens, None], end - start, "TOTAL") + return 'Fail' + return getResultString(['', TOKENS, None], end - start, 'TOTAL') def getResultString(translatedData, translationTime, filename): # File Print String - totalTokenstring = ( - Fore.YELLOW + "[Input: " + str(translatedData[1][0]) + "]" - "[Output: " + str(translatedData[1][1]) + "]" - "[Cost: ${:,.4f}".format( - (translatedData[1][0] * 0.001 * INPUTAPICOST) - + (translatedData[1][1] * 0.001 * OUTPUTAPICOST) - ) - + "]" - ) - timeString = Fore.BLUE + "[" + str(round(translationTime, 1)) + "s]" + totalTokenstring =\ + Fore.YELLOW +\ + '[Input: ' + str(translatedData[1][0]) + ']'\ + '[Output: ' + str(translatedData[1][1]) + ']'\ + '[Cost: ${:,.4f}'.format((translatedData[1][0] * .001 * INPUTAPICOST) +\ + (translatedData[1][1] * .001 * OUTPUTAPICOST)) + ']' + timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' - if translatedData[2] is None: + if translatedData[2] == None: # Success - return ( - filename - + ": " - + totalTokenstring - + timeString - + Fore.GREEN - + " \u2713 " - + Fore.RESET - ) + return filename + ': ' + totalTokenstring + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET else: # Fail try: raise translatedData[2] except Exception as e: + traceback.print_exc() errorString = str(e) + Fore.RED - return ( - filename - + ": " - + totalTokenstring - + timeString - + Fore.RED - + " \u2717 " - + errorString - + Fore.RESET - ) - + return filename + ': ' + totalTokenstring + timeString + Fore.RED + u' \u2717 ' +\ + errorString + Fore.RESET def openFiles(filename): - with open("files/" + filename, "r", encoding="utf-8") as readFile: + with open('files/' + filename, 'r', encoding='cp932') as readFile: translatedData = parseTyrano(readFile, filename) # Delete lines marked for deletion finalData = [] for line in translatedData[0]: - if line != "\\d\n": + if line != '\\d\n': finalData.append(line) translatedData[0] = finalData - + return translatedData - def parseTyrano(readFile, filename): - totalTokens = [0, 0] + totalTokens = [0,0] totalLines = 0 # Get total for progress bar data = readFile.readlines() totalLines = len(data) - with tqdm( - bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE - ) as pbar: - pbar.desc = filename - pbar.total = totalLines + with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar: + pbar.desc=filename + pbar.total=totalLines try: - response = translateTyrano(data, pbar) - totalTokens[0] = response[0] - totalTokens[1] = response[1] + result = translateTyrano(data, pbar, totalLines) + totalTokens[0] += result[0] + totalTokens[1] += result[1] except Exception as e: traceback.print_exc() return [data, totalTokens, e] return [data, totalTokens, None] - -def translateTyrano(data, pbar): +def translateTyrano(data, pbar, totalLines): textHistory = [] - maxHistory = MAXHISTORY - tokens = [0, 0] + batch = [] currentGroup = [] - syncIndex = 0 - speaker = "" - delFlag = False + maxHistory = MAXHISTORY + tokens = [0,0] + speaker = '' + insertBool = False global LOCK, ESTIMATE + i = 0 + batchStartIndex = 0 - for i in range(len(data)): - currentGroup = [] - matchList = [] - - if syncIndex > i: - i = syncIndex - - if '[▼]' in data[i]: - data[i] = data[i].replace('[▼]'.strip(), '[page]\n') - - # If there isn't any Japanese in the text just skip - if IGNORETLTEXT is True: - if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴー]+', data[i]): - # Keep textHistory list at length maxHistory - textHistory.append('\"' + data[i] + '\"') - if len(textHistory) > maxHistory: - textHistory.pop(0) - currentGroup = [] - continue - + while i < len(data): # Speaker - matchList = re.findall(r"^\[([^=\".,!?>]+?)\]$", data[i]) - if len(matchList) > 0: - if "主人公" in matchList[0]: - speaker = "Protagonist" - elif "思考" in matchList[0]: - speaker = "Protagonist Inner Thoughts" - elif "地の文" in matchList[0]: - speaker = "Narrator" - elif "マコ" in matchList[0]: - speaker = "Mako" - elif '少年' in matchList[0]: - speaker = "Boy" - elif '友達' in matchList[0]: - speaker = "Friend" - elif '少女' in matchList[0]: - speaker = "Girl" - else: - response = translateGPT( - matchList[0], - "Reply with only the " - + LANGUAGE - + " translation of the NPC name", - True, - ) - # speaker = response[0] + if '[ns]' in data[i]: + matchList = re.findall(r'\[ns\](.+?)\[', data[i]) + if len(matchList) != 0: + response = getSpeaker(matchList[0]) + speaker = response[0] tokens[0] += response[1][0] tokens[1] += response[1][1] - # data[i] = '#' + speaker + '\n' + data[i] = '[ns]' + speaker + '[nse]\n' + else: + speaker = '' # Choices - elif "glink" in data[i]: - matchList = re.findall(r"\[glink.+text=\"(.+?)\".+", data[i]) + elif '[eval exp="f.seltext' in data[i]: + matchList = re.findall(r'\[eval exp=.+?\'(.+)\'', data[i]) if len(matchList) != 0: + originalText = matchList[0] if len(textHistory) > 0: - response = translateGPT( - matchList[0], - "Past Translated Text: " - + textHistory[len(textHistory) - 1] - + "\n\nReply in the style of a dialogue option.", - True, - ) + response = translateGPT(matchList[0], 'Keep your translation as brief as possible. Previous text for context: ' + textHistory[len(textHistory)-1] + '\n\nReply in the style of a dialogue option.', False) else: - response = translateGPT(matchList[0], "", False) + response = translateGPT(matchList[0], '\n\nReply in the style of a dialogue option.', False) translatedText = response[0] tokens[0] += response[1][0] tokens[1] += response[1][1] # Remove characters that may break scripts - charList = [".", '"', "\\n"] + charList = ['.', '\"', '\\n'] for char in charList: - translatedText = translatedText.replace(char, "") + translatedText = translatedText.replace(char, '') # Escape all ' - translatedText = translatedText.replace("\\", "") - translatedText = translatedText.replace("'", "\\'") + translatedText = translatedText.replace('\\', '') + translatedText = translatedText.replace("'", "\\\'") # Set Data - translatedText = data[i].replace( - matchList[0], translatedText.replace(" ", "\u00A0") - ) - data[i] = translatedText + translatedText = data[i].replace(originalText, translatedText) + data[i] = translatedText - # Grab Lines - matchList = re.findall(r"(.+)\[[rpcm]+\]$", data[i]) + # Lines + matchList = re.findall(r'(.+?)\[[rpcms]+\]$', data[i]) if len(matchList) > 0: currentGroup.append(matchList[0]) - data[i] = "\d\n" - - # Grab All Lines in a Row - while len(matchList) > 0 and i + 1 < len(data): - i += 1 - # Skip Blank Lines - if data[i] == '\n': - data[i] = "\d\n" - continue - - # Append line to list if match - matchList = re.findall(r"(.+)\[[rpcm]+\]$", data[i]) - if len(matchList) > 0: - currentGroup.append(matchList[0]) - data[i] = "\d\n" - + if len(data) > i+1: + while '[r]' in data[i+1]: + if insertBool is True: + data[i] = '\d\n' + pbar.update(1) + i += 1 + matchList = re.findall(r'(.+?)\[r\]', data[i]) + if len(matchList) > 0: + currentGroup.append(matchList[0]) + while '[pcms]' in data[i+1]: + if insertBool is True: + data[i] = '\d\n' + pbar.update(1) + i += 1 + matchList = re.findall(r'(.+?)\[pcms\]', data[i]) + if len(matchList) > 0: + currentGroup.append(matchList[0]) # Join up 401 groups for better translation. if len(currentGroup) > 0: - finalJAString = " ".join(currentGroup) + finalJAString = ' '.join(currentGroup) + oldjaString = finalJAString # Remove any textwrap - if FIXTEXTWRAP is True: - finalJAString = finalJAString.replace("[r]", " ") + if FIXTEXTWRAP == True: + finalJAString = finalJAString.replace('[r]', ' ') - # Check Speaker - if speaker == "": - response = translateGPT(finalJAString, textHistory, True) - tokens[0] += response[1][0] - tokens[1] += response[1][1] - translatedText = response[0] - textHistory.append('"' + translatedText + '"') - else: - response = translateGPT( - speaker + ": " + finalJAString, textHistory, True - ) - tokens[0] += response[1][0] - tokens[1] += response[1][1] - translatedText = response[0] - textHistory.append('"' + translatedText + '"') + # Remove Extra Stuff bad for translation. + finalJAString = finalJAString.replace('゙', '') + finalJAString = finalJAString.replace('・', '.') + finalJAString = finalJAString.replace('‶', '') + finalJAString = finalJAString.replace('”', '') + finalJAString = finalJAString.replace('―', '-') + finalJAString = finalJAString.replace('ー', '-') + finalJAString = finalJAString.replace('…', '...') + finalJAString = re.sub(r'(\.{3}\.+)', '...', finalJAString) + finalJAString = finalJAString.replace(' ', ' ') - # Set Data - translatedText = translatedText.replace("ッ", "") - translatedText = translatedText.replace("っ", "") - translatedText = translatedText.replace("ー", "") - translatedText = translatedText.replace('"', "\"") - translatedText = translatedText.replace("[", "") - translatedText = translatedText.replace("]", "") - - # Split final string into full sentences. - matchList = re.findall(r'(.+?[\".?!)。・\n]+)[\s]?', translatedText) - for l in range(len(matchList)): - if any(t in matchList[l] for t in ['Mr.', 'Ms.', 'Mrs.', '...']): - if len(matchList) > l+1: - matchList[l] = matchList[l] + ' ' + matchList[l+1] - matchList[l+1] = '\d' - - # Delete lines marked for deletion - finalData = [] - for line in matchList: - if line != "\\d": - finalData.append(line) - matchList = finalData - - # Get rid of whitespace for each item and add wordwrap - for k in range(len(matchList)): - matchList[k] = matchList[k].strip() - - # Combine Sentences with a max limit (Wordwrap for sentences basically) - j = 0 - while(len(matchList) > j+1): - if len(matchList[j]) + len(matchList[j+1]) < WIDTH * 2 and len(matchList) > j: - matchList[j:j+2] = [' '.join(matchList[j:j+2])] - else: - j += 1 - - - # Set Data + # Furigana Removal + matchList = re.findall(r'(\[ruby\stext=.+text=\"(.+)\"\])', finalJAString) if len(matchList) > 0: - for line in matchList: + finalJAString = finalJAString.replace(matchList[0][0], matchList[0][1]) + + # Add Speaker (If there is one) + if speaker != '': + finalJAString = f'{speaker}: {finalJAString}' + + # [Passthrough 1] Pulling From File + if insertBool is False: + # Append to List and Clear Values + batch.append(finalJAString) + speaker = '' + + # Translate Batch if Full + if len(batch) == BATCHSIZE: + # Translate + response = translateGPT(batch, textHistory, True) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + translatedBatch = response[0] + textHistory = translatedBatch[-10:] + + # Set Values + if len(batch) == len(translatedBatch): + i = batchStartIndex + insertBool = True + + # Mismatch + else: + pbar.write(f'Mismatch: {batchStartIndex} - {i}') + MISMATCH.append(batch) + batchStartIndex = i + batch.clear() + + i += 1 + if insertBool is True: + pbar.update(1) + currentGroup = [] + + # [Passthrough 2] Setting Data + else: + # Get Text + translatedText = translatedBatch[0] + translatedText = translatedText.replace('\\"', '\"') + translatedText = translatedText.replace('[', '(') + translatedText = translatedText.replace(']', ')') + + # Remove added speaker + translatedText = re.sub(r'^.+?:\s', '', translatedText) + + # Textwrap + translatedText = textwrap.fill(translatedText, width=WIDTH) + textList = translatedText.split('\n') + + # Set Text + data[i] = '\d\n' + for line in textList: # Wordwrap Text if '[r]' not in line: line = textwrap.fill(line, width=WIDTH) line = line.replace('\n', '[r]') - # Insert Line - data.insert(i, line.strip() + '[p][cm]\n') + # Set + data.insert(i, line.strip() + '[r]\n') i+=1 + data[i-1] = data[i-1].replace('[r]', '[pcms]') + translatedBatch.pop(0) + speaker = '' + currentGroup = [] - # Keep textHistory list at length maxHistory - if len(textHistory) > maxHistory: - textHistory.pop(0) - currentGroup = [] - speaker = "" + # If Batch is empty. Move on. + if len(translatedBatch) == 0: + insertBool = False + batchStartIndex = i + batch.clear() - pbar.update(1) - if len(data) > i + 1: - syncIndex = i + 1 + # Nothing relevant. Skip Line. else: - break + i += 1 + if insertBool is True: + pbar.update(1) + + # Translate Batch if not empty and EOF + if len(batch) != 0 and i >= len(data): + # Translate + response = translateGPT(batch, textHistory, True) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + translatedBatch = response[0] + textHistory = translatedBatch[-10:] + + # Set Values + if len(batch) == len(translatedBatch): + i = batchStartIndex + insertBool = True + + # Mismatch + else: + pbar.write(f'Mismatch: {batchStartIndex} - {i}') + MISMATCH.append(batch) + batchStartIndex = i + batch.clear() + + currentGroup = [] return tokens +# Save some money and enter the character before translation +def getSpeaker(speaker): + match speaker: + case '誠': + return ['Makoto', [0,0]] + case '夏都': + return ['Natsu', [0,0]] + case '宗一郎': + return ['Souichirou', [0,0]] + case '彩月': + return ['Satsuki', [0,0]] + case '茜梨': + return ['Akari', [0,0]] + case 'ポホヨネン': + return ['Pohjonen', [0,0]] + case '荒井': + return ['Arai', [0,0]] + case '愛梨': + return ['Airi', [0,0]] + case '朋美': + return ['Tomomi', [0,0]] + case '玄治郎': + return ['Genjirou', [0,0]] + case '稼津央': + return ['Kazuo', [0,0]] + case '美沙緒': + return ['Misao', [0,0]] + case '怜': + return ['Sato', [0,0]] + case 'オス': + return ['Oz', [0,0]] + case '穂村': + return ['Homura', [0,0]] + case '吉野': + return ['Yoshino', [0,0]] + case '忠彦': + return ['Tadahiko', [0,0]] + case _: + return translateGPT(speaker, 'Reply with only the '+ LANGUAGE +' translation of the NPC name.', False) + def subVars(jaString): - jaString = jaString.replace("\u3000", " ") + jaString = jaString.replace('\u3000', ' ') # Nested count = 0 - nestedList = re.findall(r"[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]", jaString) + nestedList = re.findall(r'[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]', jaString) nestedList = set(nestedList) if len(nestedList) != 0: for icon in nestedList: - jaString = jaString.replace(icon, "{Nested_" + str(count) + "}") + jaString = jaString.replace(icon, '{Nested_' + str(count) + '}') count += 1 # Icons count = 0 - iconList = re.findall(r"[\\]+[iIkKwWaA]+\[[0-9]+\]", jaString) + iconList = re.findall(r'[\\]+[iIkKwWaA]+\[[0-9]+\]', jaString) iconList = set(iconList) if len(iconList) != 0: for icon in iconList: - jaString = jaString.replace(icon, "{Ascii_" + str(count) + "}") + jaString = jaString.replace(icon, '{Ascii_' + str(count) + '}') count += 1 # Colors count = 0 - colorList = re.findall(r"[\\]+[cC]\[[0-9]+\]", jaString) + colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString) colorList = set(colorList) if len(colorList) != 0: for color in colorList: - jaString = jaString.replace(color, "{Color_" + str(count) + "}") + jaString = jaString.replace(color, '{Color_' + str(count) + '}') count += 1 # Names count = 0 - nameList = re.findall(r"[\\]+[nN]\[.+?\]+", jaString) + nameList = re.findall(r'[\\]+[nN]\[.+?\]+', jaString) nameList = set(nameList) if len(nameList) != 0: for name in nameList: - jaString = jaString.replace(name, "{N_" + str(count) + "}") + jaString = jaString.replace(name, '{Noun_' + str(count) + '}') count += 1 # Variables count = 0 - varList = re.findall(r"[\\]+[vV]\[[0-9]+\]", jaString) + varList = re.findall(r'[\\]+[vV]\[[0-9]+\]', jaString) varList = set(varList) if len(varList) != 0: for var in varList: - jaString = jaString.replace(var, "{Var_" + str(count) + "}") + jaString = jaString.replace(var, '{Var_' + str(count) + '}') count += 1 # Formatting count = 0 - if "笑えるよね." in jaString: - print("t") - formatList = re.findall(r"[\\]+[\w]+\[.+?\]", jaString) + formatList = re.findall(r'[\\]+[\w]+\[.+?\]', jaString) formatList = set(formatList) if len(formatList) != 0: for var in formatList: - jaString = jaString.replace(var, "{FCode_" + str(count) + "}") + jaString = jaString.replace(var, '{FCode_' + str(count) + '}') count += 1 # Put all lists in list and return allList = [nestedList, iconList, colorList, nameList, varList, formatList] return [jaString, allList] - def resubVars(translatedText, allList): # Fix Spacing and ChatGPT Nonsense - matchList = re.findall(r"\[\s?.+?\s?\]", translatedText) + matchList = re.findall(r'\[\s?.+?\s?\]', translatedText) if len(matchList) > 0: for match in matchList: text = match.strip() @@ -446,148 +464,201 @@ def resubVars(translatedText, allList): count = 0 if len(allList[0]) != 0: for var in allList[0]: - translatedText = translatedText.replace("{Nested_" + str(count) + "}", var) + translatedText = translatedText.replace('{Nested_' + str(count) + '}', var) count += 1 # Icons count = 0 if len(allList[1]) != 0: for var in allList[1]: - translatedText = translatedText.replace("{Ascii_" + str(count) + "}", var) + translatedText = translatedText.replace('{Ascii_' + str(count) + '}', var) count += 1 # Colors count = 0 if len(allList[2]) != 0: for var in allList[2]: - translatedText = translatedText.replace("{Color_" + str(count) + "}", var) + translatedText = translatedText.replace('{Color_' + str(count) + '}', var) count += 1 # Names count = 0 if len(allList[3]) != 0: for var in allList[3]: - translatedText = translatedText.replace("{N_" + str(count) + "}", var) + translatedText = translatedText.replace('{Noun_' + str(count) + '}', var) count += 1 # Vars count = 0 if len(allList[4]) != 0: for var in allList[4]: - translatedText = translatedText.replace("{Var_" + str(count) + "}", var) + translatedText = translatedText.replace('{Var_' + str(count) + '}', var) count += 1 - + # Formatting count = 0 if len(allList[5]) != 0: for var in allList[5]: - translatedText = translatedText.replace("{FCode_" + str(count) + "}", var) + translatedText = translatedText.replace('{FCode_' + str(count) + '}', var) count += 1 - # Remove Color Variables Spaces - # if '\\c' in translatedText: - # translatedText = re.sub(r'\s*(\\+c\[[1-9]+\])\s*', r' \1', translatedText) - # translatedText = re.sub(r'\s*(\\+c\[0+\])', r'\1', translatedText) return translatedText +def batchList(input_list, batch_size): + if not isinstance(batch_size, int) or batch_size <= 0: + raise ValueError("batch_size must be a positive integer") + + return [input_list[i:i + batch_size] for i in range(0, len(input_list), batch_size)] -@retry(exceptions=Exception, tries=5, delay=5) -def translateGPT(t, history, fullPromptFlag): - # Sub Vars - varResponse = subVars(t) - subbedT = varResponse[0] +def createContext(fullPromptFlag, subbedT): + characters = 'Game Characters:\n\ +中澤 誠 (Nakazawa Makoto) - Male\n\ +日向 夏都 (Hyuuga Natsu) - Female\n\ +出渕 宗一郎 (Izubuchi Souichirou) - Male\n\ +南 彩月 (Minami Satsuki) - Female\n\ +越智 茜梨 (Ochi Akari) - Female\n\ +ターヤ ポホヨネン (Tarja Pohjonen) - Female\n\ +マルガリータ バスクェス 穂村 (Margarita Vasquez Homura) - Female\n\ +花沢 愛梨 (Hanazawa Airi) - Female\n\ +五十嵐 朋美 (Igarashi Tomomi) - Female\n\ +前田 美沙緒 (Maeda Misao) - Female\n\ +村上 怜 (Murakami Sato) - Female\n\ +' + + system = PROMPT if fullPromptFlag else \ + f'Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`' + user = f'{subbedT}' + return characters, system, user - # If there isn't any Japanese in the text just skip - if not re.search(r"[一-龠]+|[ぁ-ゔ]+|[ァ-ヴ]+|[\uFF00-\uFFEF]", subbedT): - return (t, [0, 0]) - - # If ESTIMATE is True just count this as an execution and return. - if ESTIMATE: - enc = tiktoken.encoding_for_model(MODEL) - historyRaw = "" - if isinstance(history, list): - for line in history: - historyRaw += line - else: - historyRaw = history - - inputTotalTokens = len(enc.encode(historyRaw)) + len(enc.encode(PROMPT)) - outputTotalTokens = ( - len(enc.encode(t)) * 2 - ) # Estimating 2x the size of the original text - totalTokens = [inputTotalTokens, outputTotalTokens] - return (t, totalTokens) +def translateText(characters, system, user, history): + # Prompt + msg = [{"role": "system", "content": system + characters}] # Characters - context = "Game Characters:\ - Character: マコ == Mako - Gender: Female\ - Character: 主人公 == Protagonist - Gender: Male" + msg.append({"role": "system", "content": characters}) - # Prompt - if fullPromptFlag: - system = PROMPT - user = "Line to Translate = " + subbedT - else: - system = ( - "Output ONLY the " - + LANGUAGE - + " translation in the following format: `Translation: <" - + LANGUAGE.upper() - + "_TRANSLATION>`" - ) - user = "Line to Translate = " + subbedT - - # Create Message List - msg = [] - msg.append({"role": "system", "content": system}) - msg.append({"role": "user", "content": context}) + # History if isinstance(history, list): - for line in history: - msg.append({"role": "user", "content": line}) + msg.extend([{"role": "assistant", "content": h} for h in history]) else: - msg.append({"role": "user", "content": history}) - msg.append({"role": "user", "content": user}) - - response = openai.ChatCompletion.create( - temperature=0, - frequency_penalty=0.2, - presence_penalty=0.2, + msg.append({"role": "assistant", "content": history}) + + # Content to TL + msg.append({"role": "user", "content": f'{user}'}) + response = openai.chat.completions.create( + temperature=0.1, + frequency_penalty=0.1, model=MODEL, messages=msg, - request_timeout=TIMEOUT, ) + return response - # Save Translated Text - translatedText = response.choices[0].message.content - totalTokens = [response.usage.prompt_tokens, response.usage.completion_tokens] +def cleanTranslatedText(translatedText, varResponse): + placeholders = { + f'{LANGUAGE} Translation: ': '', + 'Translation: ': '', + 'っ': '', + '〜': '~', + 'ー': '-', + 'ッ': '', + '。': '.', + 'Placeholder Text': '' + # Add more replacements as needed + } + for target, replacement in placeholders.items(): + translatedText = translatedText.replace(target, replacement) - # Resub Vars translatedText = resubVars(translatedText, varResponse[1]) - - # Remove Placeholder Text - translatedText = translatedText.replace(LANGUAGE + " Translation: ", "") - translatedText = translatedText.replace("Translation: ", "") - translatedText = translatedText.replace("Line to Translate = ", "") - translatedText = translatedText.replace("Translation = ", "") - translatedText = translatedText.replace("Translate = ", "") - translatedText = translatedText.replace(LANGUAGE + " Translation:", "") - translatedText = translatedText.replace("Translation:", "") - translatedText = translatedText.replace("Line to Translate =", "") - translatedText = translatedText.replace("Translation =", "") - translatedText = translatedText.replace("Translate =", "") - translatedText = translatedText.replace("っ", "") - translatedText = translatedText.replace("ッ", "") - translatedText = translatedText.replace("ぁ", "") - translatedText = translatedText.replace("。", ".") - translatedText = translatedText.replace("、", ",") - translatedText = translatedText.replace("?", "?") - translatedText = translatedText.replace("!", "!") - - # Return Translation - if ( - len(translatedText) > 15 * len(t) - or "I'm sorry, but I'm unable to assist with that translation" in translatedText - ): - raise Exception + if '\n' in translatedText: + return [line for line in translatedText.split('\n') if line] else: - return [translatedText, totalTokens] + return [line for line in translatedText.split('\\n') if line] + +def extractTranslation(translatedTextList, is_list): + pattern = r'[\\]*`?(.*?)[\\]*?`?' + # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. + if is_list: + return [re.findall(pattern, line)[0][1] for line in translatedTextList if re.search(pattern, line)] + else: + matchList = re.findall(pattern, translatedTextList) + return matchList[0][1] if matchList else translatedTextList + +def countTokens(characters, system, user, history): + inputTotalTokens = 0 + outputTotalTokens = 0 + enc = tiktoken.encoding_for_model(MODEL) + + # Input + if isinstance(history, list): + for line in history: + inputTotalTokens += len(enc.encode(line)) + else: + inputTotalTokens += len(enc.encode(history)) + inputTotalTokens += len(enc.encode(system)) + inputTotalTokens += len(enc.encode(characters)) + inputTotalTokens += len(enc.encode(user)) + + # Output + outputTotalTokens += round(len(enc.encode(user))/1.5) + + return [inputTotalTokens, outputTotalTokens] + +def combineList(tlist, text): + if isinstance(text, list): + return [t for sublist in tlist for t in sublist] + return tlist[0] + +@retry(exceptions=Exception, tries=5, delay=5) +def translateGPT(text, history, fullPromptFlag): + totalTokens = [0, 0] + if isinstance(text, list): + tList = batchList(text, BATCHSIZE) + else: + tList = [text] + + for index, tItem in enumerate(tList): + # Before sending to translation, if we have a list of items, add the formatting + if isinstance(tItem, list): + payload = '\n'.join([f'`{item}`' for i, item in enumerate(tItem)]) + payload = payload.replace('``', '`Placeholder Text`') + varResponse = subVars(payload) + subbedT = varResponse[0] + else: + varResponse = subVars(tItem) + subbedT = varResponse[0] + + # Things to Check before starting translation + if not re.search(r'[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+', subbedT): + continue + + # Create Message + characters, system, user = createContext(fullPromptFlag, subbedT) + + # Calculate Estimate + if ESTIMATE: + estimate = countTokens(characters, system, user, history) + totalTokens[0] += estimate[0] + totalTokens[1] += estimate[1] + continue + + # Translating + response = translateText(characters, system, user, history) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Formatting + translatedTextList = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedTextList, True) + tList[index] = extractedTranslations + if len(tItem) != len(translatedTextList): + mismatch = True # Just here so breakpoint can be set + history = extractedTranslations[-10:] # Update history if we have a list + else: + # Ensure we're passing a single string to extractTranslation + extractedTranslations = extractTranslation('\n'.join(translatedTextList), False) + tList[index] = extractedTranslations + + finalList = combineList(tList, text) + return [finalList, totalTokens]