From 36b72346997b1d81d5f0c2d150227c699dc0bd0e Mon Sep 17 00:00:00 2001 From: DazedAnon Date: Thu, 6 Jun 2024 12:15:35 -0500 Subject: [PATCH] Add module for RPGM plugins --- modules/main.py | 2 + modules/rpgmakermvmz.py | 205 +++++++------- modules/rpgmakerplugin.py | 568 ++++++++++++++++++++++++++++++++++++++ 3 files changed, 667 insertions(+), 108 deletions(-) create mode 100644 modules/rpgmakerplugin.py diff --git a/modules/main.py b/modules/main.py index cee7592..2f11848 100644 --- a/modules/main.py +++ b/modules/main.py @@ -33,6 +33,7 @@ from modules.wolf2 import handleWOLF2 from modules.javascript import handleJavascript from modules.irissoft import handleIris from modules.regex import handleRegex +from modules.rpgmakerplugin import handlePlugin # For GPT4 rate limit will be hit if you have more than 1 thread. # 1 Thread for each file. Controls how many files are worked on at once. @@ -41,6 +42,7 @@ THREADS = int(os.getenv('fileThreads')) # [Display name, file extension, handle function] MODULES = [ ["RPGMaker MV/MZ", "json", handleMVMZ], + ["RPGMaker Plugins", "js", handlePlugin], ["RPGMaker ACE", "yaml", handleACE], ["CSV (From Translator++)", "csv", handleCSV], ["Eushully", "txt", handleEushully], diff --git a/modules/rpgmakermvmz.py b/modules/rpgmakermvmz.py index 2e1ce28..33daf4f 100644 --- a/modules/rpgmakermvmz.py +++ b/modules/rpgmakermvmz.py @@ -29,10 +29,11 @@ MAXHISTORY = 10 ESTIMATE = '' TOKENS = [0, 0] NAMESLIST = [] +FIRSTLINESPEAKERS = False # If 1st line of dialogue is a speaker, set to True NAMES = False # Output a list of all the character names found BRFLAG = False # If the game uses
instead FIXTEXTWRAP = True # Overwrites textwrap -IGNORETLTEXT = True # Ignores all translated text. +IGNORETLTEXT = False # Ignores all translated text. MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) BRACKETNAMES = False PBAR = None @@ -48,7 +49,7 @@ if 'gpt-3.5' in MODEL: elif 'gpt-4o' in MODEL: INPUTAPICOST = .005 OUTPUTAPICOST = .015 - BATCHSIZE = 20 + BATCHSIZE = 50 FREQUENCY_PENALTY = 0.1 #tqdm Globals @@ -57,12 +58,12 @@ POSITION = 0 LEAVE = False # Dialogue / Scroll -CODE401 = True -CODE405 = True +CODE401 = False +CODE405 = False CODE408 = False # Choices -CODE102 = True +CODE102 = False # Variables CODE122 = False @@ -78,7 +79,7 @@ CODE356 = False CODE320 = False CODE324 = False CODE111 = False -CODE108 = False +CODE108 = True def handleMVMZ(filename, estimate): global ESTIMATE, TOKENS @@ -479,7 +480,8 @@ def searchNames(data, pbar, context): if context in ['Armors', 'Weapons', 'Items']: if len(nameList) < BATCHSIZE: nameList.append(data[i]['name']) - descriptionList.append(data[i]['description'].replace('\n', ' ')) + if 'description' in data[i]: + descriptionList.append(data[i]['description'].replace('\n', ' ')) if '') totalTokens[0] += tokensResponse[0] @@ -551,7 +553,9 @@ def searchNames(data, pbar, context): if context in ['Enemies', 'Classes', 'MapInfos']: if len(nameList) < BATCHSIZE: nameList.append(data[i]['name']) - + # tokensResponse = translateNote(data[i], r'.+') + # totalTokens[0] += tokensResponse[0] + # totalTokens[1] += tokensResponse[1] i += 1 else: batchFull = True @@ -627,7 +631,8 @@ def searchNames(data, pbar, context): else: # Get Text data[j]['name'] = translatedNameBatch[0] - data[j]['description'] = textwrap.fill(translatedDescriptionBatch[0], LISTWIDTH) + if 'description' in data[j]: + data[j]['description'] = textwrap.fill(translatedDescriptionBatch[0], LISTWIDTH) translatedNameBatch.pop(0) translatedDescriptionBatch.pop(0) @@ -683,12 +688,16 @@ def searchNames(data, pbar, context): def searchCodes(page, pbar, jobList, filename): if len(jobList) > 0: - docList = jobList[0] - scriptList = jobList[1] + list401 = jobList[0] + list122 = jobList[1] + list355655 = jobList[2] + list108 = jobList[3] setData = True else: - docList = [] - scriptList = [] + list401 = [] + list122 = [] + list355655 = [] + list108 = [] setData = False currentGroup = [] textHistory = [] @@ -753,7 +762,7 @@ def searchCodes(page, pbar, jobList, filename): speakerList = re.findall(r'^【(.*?)】$', jaString) # None - if len(speakerList) == 0: + if len(speakerList) == 0 and FIRSTLINESPEAKERS is True: if len(jaString) < 40 \ and 'code' in codeList[i+1] \ and codeList[i+1]['code'] in [401, 405, -1] \ @@ -848,10 +857,6 @@ def searchCodes(page, pbar, jobList, filename): codeList[i]['parameters'] = [finalJAString + nametag] elif nCase == 1: codeList[i]['parameters'] = [nametag + finalJAString] - - ### Brackets - matchList = re.findall\ - (r'^([\\]+[cC]\[[0-9]+\]【?(.+?)】?[\\]+[cC]\[[0-9]+\])|^(【(.+)】)', finalJAString) # Handle both cases of the regex if len(matchList) != 0 and BRACKETNAMES is True: @@ -949,11 +954,11 @@ def searchCodes(page, pbar, jobList, filename): # 1st Passthrough (Grabbing Data) if setData == False: if speaker == '' and finalJAString != '': - docList.append(finalJAString) + list401.append(finalJAString) elif finalJAString != '': - docList.append(f'[{speaker}]: {finalJAString}') + list401.append(f'[{speaker}]: {finalJAString}') else: - docList.append(speaker) + list401.append(speaker) speaker = '' match = [] currentGroup = [] @@ -962,8 +967,8 @@ def searchCodes(page, pbar, jobList, filename): # 2nd Passthrough (Setting Data) else: # Grab Translated String - if len(docList) > 0: - translatedText = docList[0] + if len(list401) > 0: + translatedText = list401[0] # Remove speaker if speaker != '': @@ -1011,12 +1016,12 @@ def searchCodes(page, pbar, jobList, filename): match = [] currentGroup = [] syncIndex = i + 1 - docList.pop(0) + list401.pop(0) ## Event Code: 122 [Set Variables] if 'code' in codeList[i] and codeList[i]['code'] == 122 and CODE122 is True: # This is going to be the var being set. (IMPORTANT) - if codeList[i]['parameters'][0] not in list(range(0, 20)): + if codeList[i]['parameters'][0] not in list(range(0, 100)): i += 1 continue @@ -1039,13 +1044,13 @@ def searchCodes(page, pbar, jobList, filename): # Pass 1 if setData == False: - scriptList.append(finalJAString) + list122.append(finalJAString) # Pass 2 else: - if len(scriptList) > 0: + if len(list122) > 0: # Grab and Replace - translatedText = scriptList[0] + translatedText = list122[0] translatedText = jaString.replace(jaString, translatedText) # Remove characters that may break scripts @@ -1060,7 +1065,7 @@ def searchCodes(page, pbar, jobList, filename): # Set codeList[i]['parameters'][4] = translatedText - scriptList.pop(0) + list122.pop(0) ## Event Code: 357 [Picture Text] [Optional] if 'code' in codeList[i] and codeList[i]['code'] == 357 and CODE357 is True: @@ -1227,70 +1232,22 @@ def searchCodes(page, pbar, jobList, filename): ## Event Code: 355 or 655 Scripts [Optional] if 'code' in codeList[i] and (codeList[i]['code'] == 355 or codeList[i]['code'] == 655) and CODE355655 is True: - matchList = [] jaString = codeList[i]['parameters'][0] + + # Var Text + if 'text =' in jaString or '$gameVariables.setValue(' in jaString: + # Pass 1 + if setData is False: + list355655.append(jaString) - # Skip Console Logs - if 'console.log' in jaString: - i += 1 - continue - # Skip if - if 'if(' in jaString: - i += 1 - continue - # Skip if - if 'list' in jaString: - i += 1 - continue + # Pass 2 + else: + # Grab and Replace + translatedText = list355655[0] - stringList = [] - matchList = re.findall(r"this.BLogAdd\(.+?\"(.+?)\"", jaString) - if len(matchList) > 0: - for match in matchList: - # Remove Textwrap - match = match.replace('\\n', ' ') - stringList.append(match) - - # If there isn't any Japanese in the text just skip - # if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴー]+', jaString): - # continue - - # Skip These - # if 'this.' in jaString: - # continue - - # Need to remove outside code and put it back later - # matchList = re.findall(r'.+"(.*?)".*[;,]$', jaString) - - # Want to translate this script - if 'this.BLogAdd' not in jaString: - i += 1 - continue - - # Translate - if len(matchList) > 0: - - response = translateGPT(stringList, 'Reply with the '+ LANGUAGE +' translation of the text.', True) - totalTokens[0] += response[1][0] - totalTokens[1] += response[1][1] - translatedTextList = response[0] - translatedText = jaString - - # Replace Each Instance - for j in range(len(translatedTextList)): - # Remove characters that may break scripts - translatedTextList[j] = translatedTextList[j].replace('"', r'\"') - translatedTextList[j] = translatedTextList[j].replace("'", r"\'") - translatedTextList[j] = translatedTextList[j].replace(".", r"\.") - - # Wordwrap - translatedTextList[j] = textwrap.fill(translatedTextList[j], width=WIDTH).replace('\n', '\\n') - - # Replace Instance - translatedText = translatedText.replace(matchList[j], translatedTextList[j]) - - # Set Data - codeList[i]['parameters'][0] = translatedText + # Set + codeList[i]['parameters'][0] = translatedText + list355655.pop(0) ## Event Code: 408 (Script) if 'code' in codeList[i] and (codeList[i]['code'] == 408) and CODE408 is True: @@ -1356,12 +1313,15 @@ def searchCodes(page, pbar, jobList, filename): # Need to remove outside code and put it back later matchList = re.findall(regex, jaString) - # Translate - if len(matchList) > 0: - response = translateGPT(matchList[0], 'Reply with the '+ LANGUAGE +' translation of the text.', False) - totalTokens[0] += response[1][0] - totalTokens[1] += response[1][1] - translatedText = response[0] + # Pass 1 + if setData is False: + list108.append(matchList[0]) + + # Pass 2 + else: + # Grab and Replace + translatedText = list108[0] + list108.pop(0) # Remove characters that may break scripts charList = ['.', '\"'] @@ -1698,18 +1658,20 @@ def searchCodes(page, pbar, jobList, filename): i += 1 # End of the line - docListTL = [] - scriptListTL = [] + list401TL = [] + list122TL = [] + list355655TL = [] + list108TL = [] setData = False PBAR = pbar # 401 - if len(docList) > 0: - response = translateGPT(docList, textHistory, True) - docListTL = response[0] + if len(list401) > 0: + response = translateGPT(list401, textHistory, True) + list401TL = response[0] totalTokens[0] += response[1][0] totalTokens[1] += response[1][1] - if len(docListTL) != len(docList): + if len(list401TL) != len(list401): with LOCK: if filename not in MISMATCH: MISMATCH.append(filename) @@ -1717,12 +1679,38 @@ def searchCodes(page, pbar, jobList, filename): setData = True # 122 - if len(scriptList) > 0: - response = translateGPT(scriptList, textHistory, True) - scriptListTL = response[0] + if len(list122) > 0: + response = translateGPT(list122, textHistory, True) + list122TL = response[0] totalTokens[0] += response[1][0] totalTokens[1] += response[1][1] - if len(scriptListTL) != len(scriptList): + if len(list122TL) != len(list122): + with LOCK: + if filename not in MISMATCH: + MISMATCH.append(filename) + else: + setData = True + + # 355/655 + if len(list355655) > 0: + response = translateGPT(list355655, textHistory, True) + list355655TL = response[0] + totalTokens[0] += response[1][0] + totalTokens[1] += response[1][1] + if len(list355655TL) != len(list355655): + with LOCK: + if filename not in MISMATCH: + MISMATCH.append(filename) + else: + setData = True + + # 108 + if len(list108) > 0: + response = translateGPT(list108, textHistory, True) + list108TL = response[0] + totalTokens[0] += response[1][0] + totalTokens[1] += response[1][1] + if len(list108TL) != len(list108): with LOCK: if filename not in MISMATCH: MISMATCH.append(filename) @@ -1731,7 +1719,7 @@ def searchCodes(page, pbar, jobList, filename): # Start Pass 2 if setData: - searchCodes(page, pbar, [docListTL, scriptListTL], filename) + searchCodes(page, pbar, [list401TL, list122TL, list355655TL, list108TL], filename) # Delete all -1 codes codeListFinal = [] @@ -2063,7 +2051,8 @@ def batchList(input_list, batch_size): def createContext(fullPromptFlag, subbedT): characters = 'Game Characters:\n\ -グレイス (Grace) - Female\n\ +ティアナ (Tiana) - Female\n\ +キャサリン (Catherine) - Female\n\ ' system = PROMPT + VOCAB if fullPromptFlag else \ diff --git a/modules/rpgmakerplugin.py b/modules/rpgmakerplugin.py new file mode 100644 index 0000000..f950de5 --- /dev/null +++ b/modules/rpgmakerplugin.py @@ -0,0 +1,568 @@ +# Libraries +import os, re, textwrap, threading, time, traceback, tiktoken, openai +from pathlib import Path +from colorama import Fore +from dotenv import load_dotenv +from retry import retry +from tqdm import tqdm + +# Open AI +load_dotenv() +if os.getenv('api').replace(' ', '') != '': + openai.base_url = os.getenv('api') +openai.organization = os.getenv('org') +openai.api_key = os.getenv('key') + +#Globals +MODEL = os.getenv('model') +TIMEOUT = int(os.getenv('timeout')) +LANGUAGE = os.getenv('language').capitalize() +PROMPT = Path('prompt.txt').read_text(encoding='utf-8') +VOCAB = Path('vocab.txt').read_text(encoding='utf-8') +THREADS = int(os.getenv('threads')) +LOCK = threading.Lock() +WIDTH = int(os.getenv('width')) +LISTWIDTH = int(os.getenv('listWidth')) +NOTEWIDTH = 70 +MAXHISTORY = 10 +ESTIMATE = '' +TOKENS = [0, 0] +NAMESLIST = [] +NAMES = False # Output a list of all the character names found +BRFLAG = False # If the game uses
instead +FIXTEXTWRAP = True # Overwrites textwrap +IGNORETLTEXT = False # Ignores all translated text. +MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) + +#tqdm Globals +BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' +POSITION = 0 +LEAVE = False + +# Pricing - Depends on the model https://openai.com/pricing +# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request +# If you are getting a MISMATCH LENGTH error, lower the batch size. +if 'gpt-3.5' in MODEL: + INPUTAPICOST = .002 + OUTPUTAPICOST = .002 + BATCHSIZE = 10 +elif 'gpt-4' in MODEL: + INPUTAPICOST = .005 + OUTPUTAPICOST = .015 + BATCHSIZE = 40 + +def handlePlugin(filename, estimate): + global ESTIMATE + ESTIMATE = estimate + + if ESTIMATE: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + + # Print Total + totalString = getResultString(['', TOKENS, None], end - start, 'TOTAL') + + # Print any errors on maps + if len(MISMATCH) > 0: + return totalString + Fore.RED + f'\nMismatch Errors: {MISMATCH}' + Fore.RESET + else: + return totalString + + else: + try: + with open('translated/' + filename, 'w', encoding='utf_8', errors='ignore') as outFile: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + outFile.writelines(translatedData[0]) + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + except Exception as e: + traceback.print_exc() + return 'Fail' + + return getResultString(['', TOKENS, None], end - start, 'TOTAL') + +def getResultString(translatedData, translationTime, filename): + # File Print String + totalTokenstring =\ + Fore.YELLOW +\ + '[Input: ' + str(translatedData[1][0]) + ']'\ + '[Output: ' + str(translatedData[1][1]) + ']'\ + '[Cost: ${:,.4f}'.format((translatedData[1][0] * .001 * INPUTAPICOST) +\ + (translatedData[1][1] * .001 * OUTPUTAPICOST)) + ']' + timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' + + if translatedData[2] == None: + # Success + return filename + ': ' + totalTokenstring + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET + + else: + # Fail + try: + raise translatedData[2] + except Exception as e: + traceback.print_exc() + errorString = str(e) + Fore.RED + return filename + ': ' + totalTokenstring + timeString + Fore.RED + u' \u2717 ' +\ + errorString + Fore.RESET + +def openFiles(filename): + with open('files/' + filename, 'r', encoding='utf_8') as readFile: + translatedData = parsePlugin(readFile, filename) + + # Delete lines marked for deletion + finalData = [] + for line in translatedData[0]: + if line != '\\d\n': + finalData.append(line) + translatedData[0] = finalData + + return translatedData + +def parsePlugin(readFile, filename): + totalTokens = [0,0] + + # Read File into data + data = readFile.readlines() + + # Create Progress Bar + with tqdm(bar_format=BAR_FORMAT, position=POSITION, leave=LEAVE) as pbar: + pbar.desc=filename + + try: + result = translatePlugin(data, pbar, filename, []) + totalTokens[0] += result[0] + totalTokens[1] += result[1] + except Exception as e: + traceback.print_exc() + return [data, totalTokens, e] + return [data, totalTokens, None] + +def translatePlugin(data, pbar, filename, translatedList): + stringList = [] + currentGroup = [] + tokens = [0,0] + speaker = '' + voice = False + global LOCK, ESTIMATE + i = 0 + + while i < len(data): + voice = False + speaker = '' + + """ + Plugin List + Quest Name: [\\]+"QuestName[\\]+":[\\]+"(.*?)[\\]+" + Quest Client: [\\]+"QuestClientName[\\]+":[\\]+"[\\]+"[\\]+"(.*?)[\\]+" + Quest Location: [\\]+"QuestionLocation[\\]+":[\\]+"[\\]+"[\\]+"(.*?)[\\]+" + Quest Targe Location: [\\]+"PlaceInformation[\\]+":[\\]+"(.*?)[\\]+" + Quest Summary: [\\]+"QuestContent[\\]+":[\\]+"(.*?)[\\]+" + Quest Goal: [\\]+"ObjectiveContent[\\]+":[\\]+"[\\]+"[\\]+"(.*?)[\\]+" + Quest Goal 2: [\\]+"ObjectiveContent[\\]+":[\\]+"[\\]+"[\\]+".*?[\\]+"(.*?)[\\]+" + + TODO TL all of the above in one call instead of multiple + """ + # Lines + matchList = re.findall(r'[\\]+"PlaceInformation[\\]+":[\\]+"(.*?)[\\]+"', data[i]) + if len(matchList) > 0: + for match in matchList: + # Save Original String + originalString = match + + # Remove any textwrap + match = match.replace(r'\\\\\\\\n', ' ') + + # Pass 1 + if translatedList == []: + # Add String + stringList.append(match.strip()) + + # Pass 2 + else: + # Get Text + if translatedList: + # Grab and Pop + translatedText = translatedList[0] + translatedList.pop(0) + + # Set to None if empty list + if len(translatedList) <= 0: + translatedList = None + + # Textwrap + translatedText = textwrap.fill(translatedText, width=WIDTH) + translatedText = translatedText.replace('\n', r'\\\\\\\\n') + + # Replace Single Quotes + translatedText = translatedText.replace("'", "\\'") + + # Set Data + data[i] = data[i].replace(originalString, translatedText) + # Next Line + i += 1 + + # EOF + if len(stringList) > 0: + # Set Progress + pbar.total = len(stringList) + pbar.refresh() + + # Translate + response = translateGPT(stringList, 'The following lines are quest locations', True, pbar, filename) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + translatedList = response[0] + + # Set Strings + if len(stringList) == len(translatedList): + translatePlugin(data, pbar, filename, translatedList) + + # Mismatch + else: + with LOCK: + if filename not in MISMATCH: + MISMATCH.append(filename) + return tokens + +# Save some money and enter the character before translation +def getSpeaker(speaker, pbar, filename): + match speaker: + case 'ファイン': + return ['Fine', [0,0]] + case '': + return ['', [0,0]] + case _: + # Store Speaker + if speaker not in str(NAMESLIST): + response = translateGPT(speaker, 'Reply with only the '+ LANGUAGE +' translation of the NPC name.', False, pbar, filename) + response[0] = response[0].replace("'S", "'s") + speakerList = [speaker, response[0]] + NAMESLIST.append(speakerList) + return response + + # Find Speaker + else: + for i in range(len(NAMESLIST)): + if speaker == NAMESLIST[i][0]: + return [NAMESLIST[i][1],[0,0]] + + return [speaker,[0,0]] + +def subVars(jaString): + jaString = jaString.replace('\u3000', ' ') + + # Nested + count = 0 + nestedList = re.findall(r'[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]', jaString) + nestedList = set(nestedList) + if len(nestedList) != 0: + for icon in nestedList: + jaString = jaString.replace(icon, '[Nested_' + str(count) + ']') + count += 1 + + # Icons + count = 0 + iconList = re.findall(r'[\\]+[iIkKwWaA]+\[[0-9]+\]', jaString) + iconList = set(iconList) + if len(iconList) != 0: + for icon in iconList: + jaString = jaString.replace(icon, '[Ascii_' + str(count) + ']') + count += 1 + + # Colors + count = 0 + colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString) + colorList = set(colorList) + if len(colorList) != 0: + for color in colorList: + jaString = jaString.replace(color, '[Color_' + str(count) + ']') + count += 1 + + # Names + count = 0 + nameList = re.findall(r'[\\]+[nN]\[.+?\]+', jaString) + nameList = set(nameList) + if len(nameList) != 0: + for name in nameList: + jaString = jaString.replace(name, '[Noun_' + str(count) + ']') + count += 1 + + # Variables + count = 0 + varList = re.findall(r'[\\]+[vV]\[[0-9]+\]', jaString) + varList = set(varList) + if len(varList) != 0: + for var in varList: + jaString = jaString.replace(var, '[Var_' + str(count) + ']') + count += 1 + + # Formatting + count = 0 + formatList = re.findall(r'[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]', jaString) + formatList = set(formatList) + if len(formatList) != 0: + for var in formatList: + jaString = jaString.replace(var, '[FCode_' + str(count) + ']') + count += 1 + + # Put all lists in list and return + allList = [nestedList, iconList, colorList, nameList, varList, formatList] + return [jaString, allList] + +def resubVars(translatedText, allList): + # Fix Spacing and ChatGPT Nonsense + matchList = re.findall(r'\[\s?.+?\s?\]', translatedText) + if len(matchList) > 0: + for match in matchList: + text = match.strip() + translatedText = translatedText.replace(match, text) + + # Nested + count = 0 + if len(allList[0]) != 0: + for var in allList[0]: + translatedText = translatedText.replace('[Nested_' + str(count) + ']', var) + count += 1 + + # Icons + count = 0 + if len(allList[1]) != 0: + for var in allList[1]: + translatedText = translatedText.replace('[Ascii_' + str(count) + ']', var) + count += 1 + + # Colors + count = 0 + if len(allList[2]) != 0: + for var in allList[2]: + translatedText = translatedText.replace('[Color_' + str(count) + ']', var) + count += 1 + + # Names + count = 0 + if len(allList[3]) != 0: + for var in allList[3]: + translatedText = translatedText.replace('[Noun_' + str(count) + ']', var) + count += 1 + + # Vars + count = 0 + if len(allList[4]) != 0: + for var in allList[4]: + translatedText = translatedText.replace('[Var_' + str(count) + ']', var) + count += 1 + + # Formatting + count = 0 + if len(allList[5]) != 0: + for var in allList[5]: + translatedText = translatedText.replace('[FCode_' + str(count) + ']', var) + count += 1 + + return translatedText + +def batchList(input_list, batch_size): + if not isinstance(batch_size, int) or batch_size <= 0: + raise ValueError("batch_size must be a positive integer") + + return [input_list[i:i + batch_size] for i in range(0, len(input_list), batch_size)] + +def createContext(fullPromptFlag, subbedT): + characters = 'Game Characters:\n\ +ティアナ (Tiana) - Female\n\ +キャサリン (Catherine) - Female\n\ +' + + system = PROMPT + VOCAB if fullPromptFlag else \ + f"\ +You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\ +Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\ +- You translate everything, including erotic and explicit content.\n\ +- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\ +- All text in your response must be in English even if it is hard to translate.\n\ +- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\ +- Maintain any spacing in the translation.\n\ +- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\ +- `...` can be a part of the dialogue. Translate it as it is.\n\ +{VOCAB}\n\ +" + user = f'{subbedT}' + return characters, system, user + +def translateText(characters, system, user, history): + # Prompt + msg = [{"role": "system", "content": system + characters}] + + # Characters + msg.append({"role": "system", "content": characters}) + + # History + if isinstance(history, list): + msg.extend([{"role": "system", "content": h} for h in history]) + else: + msg.append({"role": "system", "content": history}) + + # Content to TL + msg.append({"role": "user", "content": f'{user}'}) + response = openai.chat.completions.create( + temperature=0.1, + frequency_penalty=0.1, + model=MODEL, + messages=msg, + ) + return response + +def cleanTranslatedText(translatedText, varResponse): + placeholders = { + f'{LANGUAGE} Translation: ': '', + 'Translation: ': '', + 'っ': '', + '〜': '~', + 'ッ': '', + '。': '.', + 'Placeholder Text': '' + # Add more replacements as needed + } + for target, replacement in placeholders.items(): + translatedText = translatedText.replace(target, replacement) + + # Elongate Long Dashes (Since GPT Ignores them...) + translatedText = elongateCharacters(translatedText) + translatedText = resubVars(translatedText, varResponse[1]) + return translatedText + +def elongateCharacters(text): + # Define a pattern to match one character followed by one or more `ー` characters + # Using a positive lookbehind assertion to capture the preceding character + pattern = r'(?<=(.))ー+' + + # Define a replacement function that elongates the captured character + def repl(match): + char = match.group(1) # The character before the ー sequence + count = len(match.group(0)) - 1 # Number of ー characters + return char * count # Replace ー sequence with the character repeated + + # Use re.sub() to replace the pattern in the text + return re.sub(pattern, repl, text) + +def extractTranslation(translatedTextList, is_list): + pattern = r'`?<[Ll]ine\d+>([\\]*.*?[\\]*?)<\/?[Ll]ine\d+>`?' + # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. + if is_list: + matchList = re.findall(pattern, translatedTextList) + return matchList + else: + matchList = re.findall(pattern, translatedTextList) + return matchList[0][0] if matchList else translatedTextList + +def countTokens(characters, system, user, history): + inputTotalTokens = 0 + outputTotalTokens = 0 + enc = tiktoken.encoding_for_model('gpt-4') + + # Input + if isinstance(history, list): + for line in history: + inputTotalTokens += len(enc.encode(line)) + else: + inputTotalTokens += len(enc.encode(history)) + inputTotalTokens += len(enc.encode(system)) + inputTotalTokens += len(enc.encode(characters)) + inputTotalTokens += len(enc.encode(user)) + + # Output + outputTotalTokens += round(len(enc.encode(user))*2) + + return [inputTotalTokens, outputTotalTokens] + +def combineList(tlist, text): + if isinstance(text, list): + return [t for sublist in tlist for t in sublist] + return tlist[0] + +@retry(exceptions=Exception, tries=5, delay=5) +def translateGPT(text, history, fullPromptFlag, pbar, filename): + mismatch = False + totalTokens = [0, 0] + if isinstance(text, list): + tList = batchList(text, BATCHSIZE) + else: + tList = [text] + + for index, tItem in enumerate(tList): + # Before sending to translation, if we have a list of items, add the formatting + if isinstance(tItem, list): + payload = '\n'.join([f'`{item}`' for i, item in enumerate(tItem)]) + payload = re.sub(r'(<)(\/Line\d+>)', r'\1>Placeholder Text<\3', payload) + varResponse = subVars(payload) + subbedT = varResponse[0] + else: + varResponse = subVars(tItem) + subbedT = varResponse[0] + + # Things to Check before starting translation + if not re.search(r'[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+', subbedT): + continue + + # Create Message + characters, system, user = createContext(fullPromptFlag, subbedT) + + # Calculate Estimate + if ESTIMATE: + estimate = countTokens(characters, system, user, history) + totalTokens[0] += estimate[0] + totalTokens[1] += estimate[1] + continue + + # Translating + response = translateText(characters, system, user, history) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Formatting + translatedText = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedText, True) + if len(tItem) != len(extractedTranslations): + # Mismatch. Try Again + response = translateText(characters, system, user, history) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Formatting + translatedText = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedText, True) + if len(tItem) == len(extractedTranslations): + tList[index] = extractedTranslations + else: + MISMATCH.append(filename) + else: + tList[index] = extractedTranslations + + # Create History + history = tList[index] # Update history if we have a list + pbar.update(len(tList[index])) + + else: + # Ensure we're passing a single string to extractTranslation + extractedTranslations = extractTranslation(translatedText, False) + tList[index] = extractedTranslations + + finalList = combineList(tList, text) + return [finalList, totalTokens]