From 2210cc1436d211653c2b28c74563933456e48002 Mon Sep 17 00:00:00 2001 From: Dazed Date: Fri, 2 Feb 2024 12:33:28 -0600 Subject: [PATCH] Create nscript parser --- modules/main.py | 2 + modules/nscript.py | 647 ++++++++++++++++++++++++++++++++++++++++ modules/rpgmakermvmz.py | 23 +- modules/txt.py | 390 ------------------------ vocab.txt | 2 + 5 files changed, 663 insertions(+), 401 deletions(-) create mode 100644 modules/nscript.py delete mode 100644 modules/txt.py diff --git a/modules/main.py b/modules/main.py index ae3672d..f18659a 100644 --- a/modules/main.py +++ b/modules/main.py @@ -24,6 +24,7 @@ from modules.kansen import handleKansen from modules.lune import handleLune from modules.atelier import handleAtelier from modules.anim import handleAnim +from modules.nscript import handleNScript # For GPT4 rate limit will be hit if you have more than 1 thread. # 1 Thread for each file. Controls how many files are worked on at once. @@ -41,6 +42,7 @@ MODULES = [ ["Lune", "json", handleLune], ["Atelier", "txt", handleAtelier], ["Anim", "json", handleAnim], + ["NScript", "txt", handleNScript], ] # Info Message diff --git a/modules/nscript.py b/modules/nscript.py new file mode 100644 index 0000000..dbdde5f --- /dev/null +++ b/modules/nscript.py @@ -0,0 +1,647 @@ +# Libraries +import os, re, textwrap, threading, time, traceback, tiktoken, openai +from pathlib import Path +from colorama import Fore +from dotenv import load_dotenv +from retry import retry +from tqdm import tqdm + +# Open AI +load_dotenv() +if os.getenv('api').replace(' ', '') != '': + openai.api_base = os.getenv('api') +openai.organization = os.getenv('org') +openai.api_key = os.getenv('key') + +#Globals +MODEL = os.getenv('model') +TIMEOUT = int(os.getenv('timeout')) +LANGUAGE = os.getenv('language').capitalize() +PROMPT = Path('prompt.txt').read_text(encoding='utf-8') +VOCAB = Path('vocab.txt').read_text(encoding='utf-8') +THREADS = int(os.getenv('threads')) +LOCK = threading.Lock() +WIDTH = int(os.getenv('width')) +LISTWIDTH = int(os.getenv('listWidth')) +NOTEWIDTH = 70 +MAXHISTORY = 10 +ESTIMATE = '' +TOKENS = [0, 0] +NAMESLIST = [] +NAMES = False # Output a list of all the character names found +BRFLAG = False # If the game uses
instead +FIXTEXTWRAP = True # Overwrites textwrap +IGNORETLTEXT = False # Ignores all translated text. +MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) + +#tqdm Globals +BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' +POSITION = 0 +LEAVE = False + +# Pricing - Depends on the model https://openai.com/pricing +# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request +# If you are getting a MISMATCH LENGTH error, lower the batch size. +if 'gpt-3.5' in MODEL: + INPUTAPICOST = .002 + OUTPUTAPICOST = .002 + BATCHSIZE = 10 +elif 'gpt-4' in MODEL: + INPUTAPICOST = .01 + OUTPUTAPICOST = .03 + BATCHSIZE = 10 + +def handleNScript(filename, estimate): + global ESTIMATE + ESTIMATE = estimate + + if ESTIMATE: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + + # Print Total + totalString = getResultString(['', TOKENS, None], end - start, 'TOTAL') + + # Print any errors on maps + if len(MISMATCH) > 0: + return totalString + Fore.RED + f'\nMismatch Errors: {MISMATCH}' + Fore.RESET + else: + return totalString + + else: + try: + with open('translated/' + filename, 'w', encoding='shift_jis', errors='ignore') as outFile: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + outFile.writelines(translatedData[0]) + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + except Exception as e: + traceback.print_exc() + return 'Fail' + + return getResultString(['', TOKENS, None], end - start, 'TOTAL') + +def getResultString(translatedData, translationTime, filename): + # File Print String + totalTokenstring =\ + Fore.YELLOW +\ + '[Input: ' + str(translatedData[1][0]) + ']'\ + '[Output: ' + str(translatedData[1][1]) + ']'\ + '[Cost: ${:,.4f}'.format((translatedData[1][0] * .001 * INPUTAPICOST) +\ + (translatedData[1][1] * .001 * OUTPUTAPICOST)) + ']' + timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' + + if translatedData[2] == None: + # Success + return filename + ': ' + totalTokenstring + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET + + else: + # Fail + try: + raise translatedData[2] + except Exception as e: + traceback.print_exc() + errorString = str(e) + Fore.RED + return filename + ': ' + totalTokenstring + timeString + Fore.RED + u' \u2717 ' +\ + errorString + Fore.RESET + +def openFiles(filename): + with open('files/' + filename, 'r', encoding='cp932') as readFile: + translatedData = parseNScript(readFile, filename) + + # Delete lines marked for deletion + finalData = [] + for line in translatedData[0]: + if line != '\\d\n': + finalData.append(line) + translatedData[0] = finalData + + return translatedData + +def parseNScript(readFile, filename): + totalTokens = [0,0] + totalLines = 0 + + # Get total for progress bar + data = readFile.readlines() + totalLines = len(data) + + with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar: + pbar.desc=filename + pbar.total=totalLines + + try: + result = translateNScript(data, pbar, totalLines) + totalTokens[0] += result[0] + totalTokens[1] += result[1] + except Exception as e: + traceback.print_exc() + return [data, totalTokens, e] + return [data, totalTokens, None] + +def translateNScript(data, pbar, totalLines): + textHistory = [] + batch = [] + currentGroup = [] + maxHistory = MAXHISTORY + tokens = [0,0] + speaker = '' + insertBool = False + global LOCK, ESTIMATE + i = 0 + batchStartIndex = 0 + + while i < len(data): + # Speaker + matchList = re.findall(r'^【\s+(.*?)\s+】$', data[i]) + if len(matchList) != 0: + response = getSpeaker(matchList[0]) + speaker = response[0] + tokens[0] += response[1][0] + tokens[1] += response[1][1] + data[i] = '【 ' + speaker + ' 】\n' + i += 1 + else: + speaker = '' + + # Choices + matchList = re.findall(r'\[sel.+text="(.+?)".+', data[i]) + if len(matchList) != 0: + originalText = matchList[0] + if len(textHistory) > 0: + response = translateGPT(matchList[0], 'Keep your translation as brief as possible. Previous text for context: ' + textHistory[len(textHistory)-1] + '\n\nReply in the style of a dialogue option.', False) + else: + response = translateGPT(matchList[0], '\n\nReply in the style of a dialogue option.', False) + translatedText = response[0] + tokens[0] += response[1][0] + tokens[1] += response[1][1] + + # Remove characters that may break scripts + charList = ['.', '\"', '\\n'] + for char in charList: + translatedText = translatedText.replace(char, '') + + # Escape all ' + translatedText = translatedText.replace('\\', '') + # translatedText = translatedText.replace("'", "\\\'") + + # Set Data + translatedText = data[i].replace(originalText, translatedText) + data[i] = translatedText + + # Lines + matchList = re.findall(r'^[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9「」 >].*', data[i]) + if len(matchList) > 0: + currentGroup.append(matchList[0]) + if len(data) > i+1: + if speaker == '': + while '\n' != data[i+1] and '【' not in data[i+1]: + if insertBool is True: + data[i] = '\d\n' + pbar.update(1) + i += 1 + matchList = re.findall(r'^[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9「」 >].*', data[i]) + if len(matchList) > 0: + currentGroup.append(matchList[0]) + else: + while ' ' in data[i+1][0] or '"' in data[i+1]: + if insertBool is True: + data[i] = '\d\n' + pbar.update(1) + i += 1 + matchList = re.findall(r'^[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9「」 >].*', data[i]) + if len(matchList) > 0: + currentGroup.append(matchList[0]) + + # Join up 401 groups for better translation. + if len(currentGroup) > 0: + finalJAString = ' '.join(currentGroup) + oldjaString = finalJAString + + # Remove any textwrap + if FIXTEXTWRAP == True: + finalJAString = finalJAString.replace('>', '') + + # Remove Extra Stuff bad for translation. + finalJAString = finalJAString.replace('゙', '') + finalJAString = finalJAString.replace('・', '.') + finalJAString = finalJAString.replace('‶', '') + finalJAString = finalJAString.replace('”', '') + finalJAString = finalJAString.replace('―', '-') + finalJAString = finalJAString.replace('ー', '-') + finalJAString = finalJAString.replace('…', '...') + finalJAString = re.sub(r'(\.{3}\.+)', '...', finalJAString) + finalJAString = finalJAString.replace(' ', ' ') + + # Furigana Removal + matchList = re.findall(r'『\((.+)/.*?』', finalJAString) + if len(matchList) > 0: + finalJAString = finalJAString.replace(matchList[0][0], matchList[0][1]) + + # Add Speaker (If there is one) + if speaker != '': + finalJAString = f'{speaker}: {finalJAString}' + + # [Passthrough 1] Pulling From File + if insertBool is False: + # Append to List and Clear Values + batch.append(finalJAString) + speaker = '' + + # Translate Batch if Full + if len(batch) == BATCHSIZE: + # Translate + response = translateGPT(batch, textHistory, True) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + translatedBatch = response[0] + textHistory = translatedBatch[-10:] + + # Set Values + if len(batch) == len(translatedBatch): + i = batchStartIndex + insertBool = True + + # Mismatch + else: + pbar.write(f'Mismatch: {batchStartIndex} - {i}') + MISMATCH.append(batch) + batchStartIndex = i + batch.clear() + + i += 1 + if insertBool is True: + pbar.update(1) + currentGroup = [] + + # [Passthrough 2] Setting Data + else: + # Get Text + translatedText = translatedBatch[0] + translatedText = translatedText.replace('\\"', '\"') + translatedText = translatedText.replace('[', '(') + translatedText = translatedText.replace(']', ')') + + # Remove added speaker + translatedText = re.sub(r'^.+?:\s', '', translatedText) + + # Textwrap + translatedText = textwrap.fill(translatedText, width=WIDTH) + textList = translatedText.split('\n') + + # Set Text + data[i] = '\d\n' + for line in textList: + # Wordwrap Text + line = textwrap.fill(line, width=WIDTH) + + # Set + data.insert(i, '>' + line.strip() + '\n') + i+=1 + translatedBatch.pop(0) + speaker = '' + currentGroup = [] + + # If Batch is empty. Move on. + if len(translatedBatch) == 0: + insertBool = False + batchStartIndex = i + batch.clear() + + # Nothing relevant. Skip Line. + else: + i += 1 + if insertBool is True: + pbar.update(1) + + # Translate Batch if not empty and EOF + if len(batch) != 0 and i >= len(data): + # Translate + response = translateGPT(batch, textHistory, True) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + translatedBatch = response[0] + textHistory = translatedBatch[-10:] + + # Set Values + if len(batch) == len(translatedBatch): + i = batchStartIndex + insertBool = True + + # Mismatch + else: + pbar.write(f'Mismatch: {batchStartIndex} - {i}') + MISMATCH.append(batch) + batchStartIndex = i + batch.clear() + + currentGroup = [] + return tokens + +# Save some money and enter the character before translation +def getSpeaker(speaker): + match speaker: + case '央': + return ['Akira', [0,0]] + case '累': + return ['Rui', [0,0]] + case '梨里': + return ['Riri', [0,0]] + case '純': + return ['Jun', [0,0]] + case '美鈴': + return ['Misuzu', [0,0]] + case '須田': + return ['Suda', [0,0]] + case '高橋': + return ['Takahashi', [0,0]] + case '勇二': + return ['Yuuji', [0,0]] + case _: + return translateGPT(speaker, 'Reply with only the '+ LANGUAGE +' translation of the NPC name.', False) + +def subVars(jaString): + jaString = jaString.replace('\u3000', ' ') + + # Nested + count = 0 + nestedList = re.findall(r'[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]', jaString) + nestedList = set(nestedList) + if len(nestedList) != 0: + for icon in nestedList: + jaString = jaString.replace(icon, '{Nested_' + str(count) + '}') + count += 1 + + # Icons + count = 0 + iconList = re.findall(r'[\\]+[iIkKwWaA]+\[[0-9]+\]', jaString) + iconList = set(iconList) + if len(iconList) != 0: + for icon in iconList: + jaString = jaString.replace(icon, '{Ascii_' + str(count) + '}') + count += 1 + + # Colors + count = 0 + colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString) + colorList = set(colorList) + if len(colorList) != 0: + for color in colorList: + jaString = jaString.replace(color, '{Color_' + str(count) + '}') + count += 1 + + # Names + count = 0 + nameList = re.findall(r'[\\]+[nN]\[.+?\]+', jaString) + nameList = set(nameList) + if len(nameList) != 0: + for name in nameList: + jaString = jaString.replace(name, '{Noun_' + str(count) + '}') + count += 1 + + # Variables + count = 0 + varList = re.findall(r'[\\]+[vV]\[[0-9]+\]', jaString) + varList = set(varList) + if len(varList) != 0: + for var in varList: + jaString = jaString.replace(var, '{Var_' + str(count) + '}') + count += 1 + + # Formatting + count = 0 + formatList = re.findall(r'[\\]+[\w]+\[.+?\]', jaString) + formatList = set(formatList) + if len(formatList) != 0: + for var in formatList: + jaString = jaString.replace(var, '{FCode_' + str(count) + '}') + count += 1 + + # Put all lists in list and return + allList = [nestedList, iconList, colorList, nameList, varList, formatList] + return [jaString, allList] + +def resubVars(translatedText, allList): + # Fix Spacing and ChatGPT Nonsense + matchList = re.findall(r'\[\s?.+?\s?\]', translatedText) + if len(matchList) > 0: + for match in matchList: + text = match.strip() + translatedText = translatedText.replace(match, text) + + # Nested + count = 0 + if len(allList[0]) != 0: + for var in allList[0]: + translatedText = translatedText.replace('{Nested_' + str(count) + '}', var) + count += 1 + + # Icons + count = 0 + if len(allList[1]) != 0: + for var in allList[1]: + translatedText = translatedText.replace('{Ascii_' + str(count) + '}', var) + count += 1 + + # Colors + count = 0 + if len(allList[2]) != 0: + for var in allList[2]: + translatedText = translatedText.replace('{Color_' + str(count) + '}', var) + count += 1 + + # Names + count = 0 + if len(allList[3]) != 0: + for var in allList[3]: + translatedText = translatedText.replace('{Noun_' + str(count) + '}', var) + count += 1 + + # Vars + count = 0 + if len(allList[4]) != 0: + for var in allList[4]: + translatedText = translatedText.replace('{Var_' + str(count) + '}', var) + count += 1 + + # Formatting + count = 0 + if len(allList[5]) != 0: + for var in allList[5]: + translatedText = translatedText.replace('{FCode_' + str(count) + '}', var) + count += 1 + + return translatedText + +def batchList(input_list, batch_size): + if not isinstance(batch_size, int) or batch_size <= 0: + raise ValueError("batch_size must be a positive integer") + + return [input_list[i:i + batch_size] for i in range(0, len(input_list), batch_size)] + +def createContext(fullPromptFlag, subbedT): + characters = 'Game Characters:\n\ +渋江 央 (Shibue Akira) - Male\n\ +蘆名 累 (Ashina Rui) - Female\n\ +清原 梨里 (Kiyohara Riri) - Female\n\ +五十嵐 純 (Igarashi Jun) - Female\n\ +子野日 美鈴 (Nenohi Misuzu) - Female\n\ +須田 (Suda) - Male\n\ +高橋 (Takahashi) - Female\n\ +勇二 (Yuuji) - Male\n\ +' + + system = PROMPT + VOCAB if fullPromptFlag else \ + f"\ +You are an expert Eroge Game translator who translates Japanese text to English.\n\ +You are going to be translating text from a videogame.\n\ +I will give you lines of text, and you must translate each line to the best of your ability.\n\ +{VOCAB}\n\ +Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\ +" + user = f'{subbedT}' + return characters, system, user + +def translateText(characters, system, user, history): + # Prompt + msg = [{"role": "system", "content": system + characters}] + + # Characters + msg.append({"role": "system", "content": characters}) + + # History + if isinstance(history, list): + msg.extend([{"role": "system", "content": h} for h in history]) + else: + msg.append({"role": "system", "content": history}) + + # Content to TL + msg.append({"role": "user", "content": f'{user}'}) + response = openai.chat.completions.create( + temperature=0.1, + frequency_penalty=0.1, + presence_penalty=0.1, + model=MODEL, + messages=msg, + ) + return response + +def cleanTranslatedText(translatedText, varResponse): + placeholders = { + f'{LANGUAGE} Translation: ': '', + 'Translation: ': '', + 'っ': '', + '〜': '~', + 'ー': '-', + 'ッ': '', + '。': '.', + 'Placeholder Text': '' + # Add more replacements as needed + } + for target, replacement in placeholders.items(): + translatedText = translatedText.replace(target, replacement) + + translatedText = resubVars(translatedText, varResponse[1]) + return [line for line in translatedText.replace('\\n', '\n').split('\n') if line] + +def extractTranslation(translatedTextList, is_list): + pattern = r'`?([\\]*.*?[\\]*?)<\/?Line\d+>`?' + # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. + if is_list: + return [re.findall(pattern, line)[0][1] for line in translatedTextList if re.search(pattern, line)] + else: + matchList = re.findall(pattern, translatedTextList) + return matchList[0][1] if matchList else translatedTextList + +def countTokens(characters, system, user, history): + inputTotalTokens = 0 + outputTotalTokens = 0 + enc = tiktoken.encoding_for_model(MODEL) + + # Input + if isinstance(history, list): + for line in history: + inputTotalTokens += len(enc.encode(line)) + else: + inputTotalTokens += len(enc.encode(history)) + inputTotalTokens += len(enc.encode(system)) + inputTotalTokens += len(enc.encode(characters)) + inputTotalTokens += len(enc.encode(user)) + + # Output + outputTotalTokens += round(len(enc.encode(user))/1.5) + + return [inputTotalTokens, outputTotalTokens] + +def combineList(tlist, text): + if isinstance(text, list): + return [t for sublist in tlist for t in sublist] + return tlist[0] + +@retry(exceptions=Exception, tries=5, delay=5) +def translateGPT(text, history, fullPromptFlag): + totalTokens = [0, 0] + if isinstance(text, list): + tList = batchList(text, BATCHSIZE) + else: + tList = [text] + + for index, tItem in enumerate(tList): + # Before sending to translation, if we have a list of items, add the formatting + if isinstance(tItem, list): + payload = '\n'.join([f'`{item}`' for i, item in enumerate(tItem)]) + payload = payload.replace('``', '`Placeholder Text`') + varResponse = subVars(payload) + subbedT = varResponse[0] + else: + varResponse = subVars(tItem) + subbedT = varResponse[0] + + # Things to Check before starting translation + if not re.search(r'[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+', subbedT): + continue + + # Create Message + characters, system, user = createContext(fullPromptFlag, subbedT) + + # Calculate Estimate + if ESTIMATE: + estimate = countTokens(characters, system, user, history) + totalTokens[0] += estimate[0] + totalTokens[1] += estimate[1] + continue + + # Translating + response = translateText(characters, system, user, history) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Formatting + translatedTextList = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedTextList, True) + tList[index] = extractedTranslations + if len(tItem) != len(translatedTextList): + mismatch = True # Just here so breakpoint can be set + history = extractedTranslations[-10:] # Update history if we have a list + else: + # Ensure we're passing a single string to extractTranslation + extractedTranslations = extractTranslation('\n'.join(translatedTextList), False) + tList[index] = extractedTranslations + + finalList = combineList(tList, text) + return [finalList, totalTokens] diff --git a/modules/rpgmakermvmz.py b/modules/rpgmakermvmz.py index 0ed510d..c55215a 100644 --- a/modules/rpgmakermvmz.py +++ b/modules/rpgmakermvmz.py @@ -47,7 +47,7 @@ if 'gpt-3.5' in MODEL: elif 'gpt-4' in MODEL: INPUTAPICOST = .01 OUTPUTAPICOST = .03 - BATCHSIZE = 20 + BATCHSIZE = 40 FREQUENCY_PENALTY = 0.1 #tqdm Globals @@ -56,11 +56,11 @@ POSITION = 0 LEAVE = False # Dialogue / Scroll -CODE401 = False -CODE405 = False +CODE401 = True +CODE405 = True # Choices -CODE102 = False +CODE102 = True # Variables CODE122 = False @@ -69,7 +69,7 @@ CODE122 = False CODE101 = False # Other -CODE355655 = True +CODE355655 = False CODE357 = False CODE657 = False CODE356 = False @@ -1708,9 +1708,6 @@ def searchCodes(page, pbar, fillList, filename): for choice in range(len(codeList[i]['parameters'][0])): jaString = codeList[i]['parameters'][0][choice] jaString = jaString.replace(' 。', '.') - # Things to Check before starting translation - if re.search(r'[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+', jaString): - print('Test') # Avoid Empty Strings if jaString == '': @@ -2161,9 +2158,13 @@ def batchList(input_list, batch_size): def createContext(fullPromptFlag, subbedT): characters = 'Game Characters:\n\ -神田 瑠唯 (Kanda Rui) - Female\n\ -ルイ (Rui) - Female\n\ -チュベロス (Tuberose) - Male\n\ +ルース (Ruth) - Male\n\ +リーザ (Reeza) - Female\n\ +エリル (Eril) - Female\n\ +シアン (Cyan) - Female\n\ +トリス (Tris) - Female\n\ +エリル (Eril) - Female\n\ +エリル (Eril) - Female\n\ ' system = PROMPT + VOCAB if fullPromptFlag else \ diff --git a/modules/txt.py b/modules/txt.py deleted file mode 100644 index dfe5e3c..0000000 --- a/modules/txt.py +++ /dev/null @@ -1,390 +0,0 @@ -from concurrent.futures import ThreadPoolExecutor, as_completed -import json -import os -from pathlib import Path -import re -import sys -import textwrap -import threading -import time -import traceback -import tiktoken - -from colorama import Fore -from dotenv import load_dotenv -import openai -from retry import retry -from tqdm import tqdm - -#Globals -load_dotenv() -if os.getenv('api').replace(' ', '') != '': - openai.api_base = os.getenv('api') - -openai.organization = os.getenv('org') -openai.api_key = os.getenv('key') -MODEL = os.getenv('model') -TIMEOUT = int(os.getenv('timeout')) -LANGUAGE=os.getenv('language').capitalize() - -APICOST = .002 # Depends on the model https://openai.com/pricing -PROMPT = Path('prompt.txt').read_text(encoding='utf-8') -VOCAB = Path('vocab.txt').read_text(encoding='utf-8') -THREADS = int(os.getenv('threads')) -LOCK = threading.Lock() -WIDTH = int(os.getenv('width')) -LISTWIDTH = int(os.getenv('listWidth')) -MAXHISTORY = 10 -ESTIMATE = '' -TOTALCOST = 0 -TOKENS = 0 -TOTALTOKENS = 0 - -#tqdm Globals -BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' -POSITION=0 -LEAVE=False - -# Flags -CODE401 = True -CODE102 = True -CODE122 = False -CODE101 = False -CODE355655 = False -CODE357 = False -CODE356 = False -CODE320 = False -CODE111 = False - -def handleTXT(filename, estimate): - global ESTIMATE, TOKENS, TOTALTOKENS, TOTALCOST - ESTIMATE = estimate - - if estimate: - start = time.time() - translatedData = openFiles(filename) - - # Print Result - end = time.time() - tqdm.write(getResultString(['', TOKENS, None], end - start, filename)) - with LOCK: - TOTALCOST += TOKENS * .001 * APICOST - TOTALTOKENS += TOKENS - TOKENS = 0 - - return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL') - - else: - with open('translated/' + filename, 'w', encoding='UTF-8') as outFile: - start = time.time() - translatedData = openFiles(filename) - - # Print Result - end = time.time() - outFile.writelines(translatedData[0]) - tqdm.write(getResultString(translatedData, end - start, filename)) - with LOCK: - TOTALCOST += translatedData[1] * .001 * APICOST - TOTALTOKENS += translatedData[1] - - return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL') - -def openFiles(filename): - with open('files/' + filename, 'r', encoding='UTF-8') as f: - translatedData = parseText(f, filename) - - return translatedData - -def getResultString(translatedData, translationTime, filename): - # File Print String - tokenString = Fore.YELLOW + '[' + str(translatedData[1]) + \ - ' Tokens/${:,.4f}'.format(translatedData[1] * .001 * APICOST) + ']' - timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' - - if translatedData[2] == None: - # Success - return filename + ': ' + tokenString + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET - - else: - # Fail - try: - raise translatedData[2] - except Exception as e: - errorString = str(e) + Fore.RED - return filename + ': ' + tokenString + timeString + Fore.RED + u' \u2717 ' +\ - errorString + Fore.RESET - -def parseText(data, filename): - totalTokens = 0 - totalLines = 0 - global LOCK - - # Get total for progress bar - linesList = data.readlines() - totalLines = len(linesList) - - with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar: - pbar.desc=filename - pbar.total=totalLines - try: - response = translateText(linesList, pbar) - except Exception as e: - traceback.print_exc() - return [linesList, 0, e] - return [response[0], response[1], None] - -def translateText(data, pbar): - textHistory = [] - maxHistory = MAXHISTORY - tokens = 0 - speaker = '' - speakerFlag = False - currentGroup = [] - syncIndex = 0 - - for i in range(len(data)): - if i != syncIndex: - continue - - match = re.findall(r'm\[[0-9]+\] = \"(.*)\"', data[i]) - if len(match) > 0: - jaString = match[0] - - ### Translate - # Remove any textwrap - jaString = re.sub(r'\\n', ' ', jaString) - - # Grab Speaker - speakerMatch = re.findall(r's\[[0-9]+\] = \"(.+?)[/\"]', data[i-1]) - if len(speakerMatch) > 0: - # If there isn't any Japanese in the text just skip - if re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴー]+', jaString) and '_' not in speakerMatch[0]: - speaker = '' - else: - speaker = '' - else: - speaker = '' - - # Grab rest of the messages - currentGroup.append(jaString) - start = i - data[i] = re.sub(r'(m\[[0-9]+\]) = \"(.+)\"', rf'\1 = ""', data[i]) - while (len(data) > i+1 and re.search(r'm\[[0-9]+\] = \"(.*)\"', data[i+1]) != None): - i+=1 - match = re.findall(r'm\[[0-9]+\] = \"(.*)\"', data[i]) - currentGroup.append(match[0]) - data[i] = re.sub(r'(m\[[0-9]+\]) = \"(.+)\"', rf'\1 = ""', data[i]) - finalJAString = ' '.join(currentGroup) - - # Translate - if speaker != '': - response = translateGPT(f'{speaker}: {finalJAString}', 'Previous Text for Context: ' + ' '.join(textHistory), True) - else: - response = translateGPT(finalJAString, 'Previous Text for Context: ' + ' '.join(textHistory), True) - tokens += response[1] - translatedText = response[0] - - # Remove added speaker and quotes - translatedText = re.sub(r'^.+?:\s', '', translatedText) - - # TextHistory is what we use to give GPT Context, so thats appended here. - # rawTranslatedText = re.sub(r'[\\<>]+[a-zA-Z]+\[[a-zA-Z0-9]+\]', '', translatedText) - if speaker != '': - textHistory.append(speaker + ': ' + translatedText) - elif speakerFlag == False: - textHistory.append('\"' + translatedText + '\"') - - # Keep textHistory list at length maxHistory - if len(textHistory) > maxHistory: - textHistory.pop(0) - currentGroup = [] - - # Textwrap - translatedText = translatedText.replace('\"', '\\"') - translatedText = textwrap.fill(translatedText, width=WIDTH) - - # Write - textList = translatedText.split("\n") - for t in textList: - data[start] = re.sub(r'(m\[[0-9]+\]) = \"(.*)\"', rf'\1 = "{t}"', data[start]) - start+=1 - - syncIndex = i + 1 - pbar.update() - return [data, tokens] - -def subVars(jaString): - jaString = jaString.replace('\u3000', ' ') - - # Icons - count = 0 - iconList = re.findall(r'[\\]+[iI]\[[0-9]+\]', jaString) - iconList = set(iconList) - if len(iconList) != 0: - for icon in iconList: - jaString = jaString.replace(icon, '') - count += 1 - - # Colors - count = 0 - colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString) - colorList = set(colorList) - if len(colorList) != 0: - for color in colorList: - jaString = jaString.replace(color, '') - count += 1 - - # Names - count = 0 - nameList = re.findall(r'[\\]+[nN]\[[0-9]+\]', jaString) - nameList = set(nameList) - if len(nameList) != 0: - for name in nameList: - jaString = jaString.replace(name, '') - count += 1 - - # Variables - count = 0 - varList = re.findall(r'[\\]+[vV]\[[0-9]+\]', jaString) - varList = set(varList) - if len(varList) != 0: - for var in varList: - jaString = jaString.replace(var, '') - count += 1 - - # Formatting - count = 0 - formatList = re.findall(r'[\\]+[!.]', jaString) - formatList = set(formatList) - if len(formatList) != 0: - for format in formatList: - jaString = jaString.replace(format, '') - count += 1 - - # Put all lists in list and return - allList = [iconList, colorList, nameList, varList, formatList] - return [jaString, allList] - -def resubVars(translatedText, allList): - # Fix Spacing and ChatGPT Nonsense - matchList = re.findall(r'<\s?.+?\s?>', translatedText) - if len(matchList) > 0: - for match in matchList: - text = match.strip() - translatedText = translatedText.replace(match, text) - - # Icons - count = 0 - if len(allList[0]) != 0: - for var in allList[0]: - translatedText = translatedText.replace('', var) - count += 1 - - # Colors - count = 0 - if len(allList[1]) != 0: - for var in allList[1]: - translatedText = translatedText.replace('', var) - count += 1 - - # Names - count = 0 - if len(allList[2]) != 0: - for var in allList[2]: - translatedText = translatedText.replace('', var) - count += 1 - - # Vars - count = 0 - if len(allList[3]) != 0: - for var in allList[3]: - translatedText = translatedText.replace('', var) - count += 1 - - # Formatting - count = 0 - if len(allList[4]) != 0: - for var in allList[4]: - translatedText = translatedText.replace('', var) - count += 1 - -@retry(exceptions=Exception, tries=5, delay=5) -def translateGPT(t, history, fullPromptFlag): - # If ESTIMATE is True just count this as an execution and return. - if ESTIMATE: - enc = tiktoken.encoding_for_model(MODEL) - tokens = len(enc.encode(t)) * 2 + len(enc.encode(str(history))) + len(enc.encode(PROMPT)) - return (t, tokens) - - # Sub Vars - varResponse = subVars(t) - subbedT = varResponse[0] - - # If there isn't any Japanese in the text just skip - if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴ]+|[\uFF00-\uFFEF]', subbedT): - return(t, 0) - - # Characters - context = '```\ - Game Characters:\ - Character: 池ノ上 拓海 == Ikenoue Takumi - Gender: Male\ - Character: 福永 こはる == Fukunaga Koharu - Gender: Female\ - Character: 神泉 理央 == Kamiizumi Rio - Gender: Female\ - Character: 吉祥寺 アリサ == Kisshouji Arisa - Gender: Female\ - Character: 久我 友里子 == Kuga Yuriko - Gender: Female\ - ```' - - # Prompt - if fullPromptFlag: - system = PROMPT - user = 'Line to Translate = ' + subbedT - else: - system = 'Output ONLY the '+ LANGUAGE +' translation in the following format: `Translation: <'+ LANGUAGE.upper() +'_TRANSLATION>`' - user = 'Line to Translate = ' + subbedT - - # Create Message List - msg = [] - msg.append({"role": "system", "content": system}) - msg.append({"role": "user", "content": context}) - if isinstance(history, list): - for line in history: - msg.append({"role": "user", "content": line}) - else: - msg.append({"role": "user", "content": history}) - msg.append({"role": "user", "content": user}) - - response = openai.ChatCompletion.create( - temperature=0.1, - frequency_penalty=0.2, - presence_penalty=0.2, - model=MODEL, - messages=msg, - request_timeout=TIMEOUT, - ) - - # Save Translated Text - translatedText = response.choices[0].message.content - tokens = response.usage.total_tokens - - # Resub Vars - translatedText = resubVars(translatedText, varResponse[1]) - - # Remove Placeholder Text - translatedText = translatedText.replace(LANGUAGE +' Translation: ', '') - translatedText = translatedText.replace('Translation: ', '') - translatedText = translatedText.replace('Line to Translate = ', '') - translatedText = translatedText.replace('Translation = ', '') - translatedText = translatedText.replace('Translate = ', '') - translatedText = translatedText.replace(LANGUAGE +' Translation:', '') - translatedText = translatedText.replace('Translation:', '') - translatedText = translatedText.replace('Line to Translate =', '') - translatedText = translatedText.replace('Translation =', '') - translatedText = translatedText.replace('Translate =', '') - translatedText = re.sub(r'Note:.*', '', translatedText) - translatedText = translatedText.replace('っ', '') - - # Return Translation - if len(translatedText) > 15 * len(t) or "I'm sorry, but I'm unable to assist with that translation" in translatedText: - raise Exception - else: - return [translatedText, tokens] diff --git a/vocab.txt b/vocab.txt index c0063ec..957bbd6 100644 --- a/vocab.txt +++ b/vocab.txt @@ -34,6 +34,8 @@ Here are some vocabulary and terms so that you know the proper spelling and tran 魔法力 (M. Power) 命中率 (Accuracy) %1 の%2を獲得! (Gained %1 %2!) +持っている数 (Owned) +ME 音量 (ME Volume) # Other ソル (Sol)