From bf9ed0bd542d2c22c429d5b0327a82796658c36b Mon Sep 17 00:00:00 2001 From: DazedAnon Date: Fri, 24 May 2024 10:39:40 -0500 Subject: [PATCH] Create Ushully script and update onscripter --- modules/eushully.py | 726 ++++++++++++++++++++++++++++++++++++++++++++ modules/main.py | 6 +- modules/nscript.py | 399 +++++++++--------------- vocab.txt | 18 +- 4 files changed, 898 insertions(+), 251 deletions(-) create mode 100644 modules/eushully.py diff --git a/modules/eushully.py b/modules/eushully.py new file mode 100644 index 0000000..61d7fd8 --- /dev/null +++ b/modules/eushully.py @@ -0,0 +1,726 @@ +# Libraries +import json, os, re, textwrap, threading, time, traceback, tiktoken, openai, csv +from concurrent.futures import ThreadPoolExecutor, as_completed +from pathlib import Path +from colorama import Fore +from dotenv import load_dotenv +from retry import retry +from tqdm import tqdm + +# Open AI +load_dotenv() +if os.getenv('api').replace(' ', '') != '': + openai.base_url = os.getenv('api') +openai.organization = os.getenv('org') +openai.api_key = os.getenv('key') + +#Globals +MODEL = os.getenv('model') +TIMEOUT = int(os.getenv('timeout')) +LANGUAGE = os.getenv('language').capitalize() +PROMPT = Path('prompt.txt').read_text(encoding='utf-8') +VOCAB = Path('vocab.txt').read_text(encoding='utf-8') +THREADS = int(os.getenv('threads')) +LOCK = threading.Lock() +WIDTH = int(os.getenv('width')) +LISTWIDTH = int(os.getenv('listWidth')) +NOTEWIDTH = int(os.getenv('noteWidth')) +MAXHISTORY = 10 +ESTIMATE = '' +TOKENS = [0, 0] +NAMESLIST = [] +NAMES = False # Output a list of all the character names found +BRFLAG = False # If the game uses
instead +FIXTEXTWRAP = True # Overwrites textwrap +IGNORETLTEXT = True # Ignores all translated text. +MISMATCH = [] # Lists files that thdata a mismatch error (Length of GPT list response is wrong) +BRACKETNAMES = False +TOTALLINES = 0 +PBAR = None + +# Pricing - Depends on the model https://openai.com/pricing +# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request +# If you are getting a MISMATCH LENGTH error, lower the batch size. +if 'gpt-3.5' in MODEL: + INPUTAPICOST = .002 + OUTPUTAPICOST = .002 + BATCHSIZE = 10 + FREQUENCY_PENALTY = 0.2 +elif 'gpt-4' in MODEL: + INPUTAPICOST = .005 + OUTPUTAPICOST = .015 + BATCHSIZE = 20 + FREQUENCY_PENALTY = 0.1 + +#tqdm Globals +BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' +POSITION = 0 +LEAVE = False + +def handleEushully(filename, estimate): + global ESTIMATE, TOKENS + ESTIMATE = estimate + + if not ESTIMATE: + with open('translated/' + filename, 'w+t', newline='', encoding='utf-8') as writeFile: + # Translate + start = time.time() + translatedData = openFiles(filename, writeFile) + + # Print Result + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + else: + # Translate + start = time.time() + translatedData = openFilesEstimate(filename) + + # Print Result + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + + + # Print Total + totalString = getResultString(['', TOKENS, None], end - start, 'TOTAL') + + # Print any errors on maps + if len(MISMATCH) > 0: + return totalString + Fore.RED + f'\nMismatch Errors: {MISMATCH}' + Fore.RESET + else: + return totalString + +def openFiles(filename, writeFile): + with open('files/' + filename, 'r', encoding='utf-8') as readFile, writeFile: + translatedData = parseCSV(readFile, writeFile, filename) + + return translatedData + +def openFilesEstimate(filename): + with open('files/' + filename, 'r', encoding='utf-8') as readFile: + translatedData = parseCSV(readFile, '', filename) + + return translatedData + +def getResultString(translatedData, translationTime, filename): + # File Print String + totalTokenstring =\ + Fore.YELLOW +\ + '[Input: ' + str(translatedData[1][0]) + ']'\ + '[Output: ' + str(translatedData[1][1]) + ']'\ + '[Lines: ' + str(TOTALLINES) + ']'\ + '[Cost: ${:,.4f}'.format((translatedData[1][0] * .001 * INPUTAPICOST) +\ + (translatedData[1][1] * .001 * OUTPUTAPICOST)) + ']' + timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' + + if translatedData[2] is None: + # Success + return filename + ': ' + totalTokenstring + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET + else: + # Fail + try: + raise translatedData[2] + except Exception as e: + traceback.print_exc() + errorString = str(e) + Fore.RED + return filename + ': ' + totalTokenstring + timeString + Fore.RED + u' \u2717 ' +\ + errorString + Fore.RESET + +def parseCSV(readFile, writeFile, filename): + totalTokens = [0,0] + totalLines = 0 + textHistory = [] + global LOCK + + # Get total for progress bar + totalLines = len(readFile.readlines()) + readFile.seek(0) + data = [] + + reader = csv.reader(readFile, delimiter=',',) + if not ESTIMATE: + writer = csv.writer(writeFile, delimiter=',', quoting=csv.QUOTE_ALL) + else: + writer = '' + + # Write All Rows to Data + for row in reader: + data.append(row) + + with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar: + pbar.desc=filename + pbar.total=totalLines + try: + if 'SC' == filename[0:2] or 'SP' == filename[0:2]: + response = translateDialogue(data, pbar, writer, format, filename, []) + totalTokens[0] = response[0] + totalTokens[1] = response[1] + elif 'UI' == filename[0:2]: + response = translateUI(data, pbar, writer, format, filename, []) + totalTokens[0] = response[0] + totalTokens[1] = response[1] + except Exception as e: + traceback.print_exc() + return [reader, totalTokens, None] + +def translateDialogue(data, pbar, writer, format, filename, translatedList): + global LOCK, ESTIMATE + tokens = [0,0] + stringList = [None] * 2 + i = 0 + + try: + # Set Variables + speakerColumn = 0 + textSourceColumn = 3 + textTargetColumn = 3 + + # Lists + dialogueList = [] + setStringList = [] + + # Parse Data + while i in range(len(data)): + # Dialogue + if len(data[i][speakerColumn]) > 0 and data[i][speakerColumn][0].isupper() \ + or 'show-text' in data[i][speakerColumn] \ + or 'concat' in data[i][speakerColumn]: + # Speaker + speaker = '' + if data[i][speakerColumn][0].isupper(): + if speakerColumn != None: + response = getSpeaker(data[i][speakerColumn]) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + speaker = response[0] + + # Dialogue + jaString = data[i][textSourceColumn] + + # Remove Textwrap + jaString = jaString.replace('\n', ' ') + + # Replace Unicode + jaString = jaString.replace('\ue000', '...') + + # Pass 1 + if translatedList == []: + # Add to list + if speaker: + dialogueList.append(f'[{speaker}]: {jaString}') + else: + dialogueList.append(f'[InnerVoice]: {jaString}') + stringList[0] = dialogueList + + # Pass 2 + else: + if translatedList[0]: + # Grab and Pop + translatedText = translatedList[0][0] + translatedList[0].pop(0) + + # Set to None if empty list + if len(translatedList[0]) <= 0: + translatedList[0] = None + + # Remove speaker + translatedText = re.sub(r'^\[(.+?)\]\s?[|:]\s?', '', translatedText) + + # Set Data + data[i][textTargetColumn] = f'{translatedText}' + + # Set String Command + if 'set-string' in data[i][speakerColumn]: + jaString = data[i][textSourceColumn] + + # Pass 1 + if translatedList == []: + setStringList.append(jaString) + stringList[1] = setStringList + + # Pass 2 + else: + if len(translatedList) > 1 and translatedList[1]: + # Grab and Pop + translatedText = translatedList[1][0] + translatedList[1].pop(0) + + # Set to None if empty list + if len(translatedList[1]) <= 0: + translatedList[1] = None + + # Textwrap + translatedText = textwrap.fill(translatedText, WIDTH) + + # Set Data + data[i][textTargetColumn] = f'{translatedText}' + + # Iterate + i += 1 + + # EOF + stringList = [x for x in stringList if x is not None] + if len(stringList) > 0: + # Translate + pbar.total = 0 + for i in range(len(stringList)): + # Set Progress + pbar.total += len(stringList[i]) + pbar.refresh() + PBAR = pbar + response = translateGPT(stringList[i], '', True) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + translatedList.append(response[0]) + + # Set Strings + if len(stringList) == len(translatedList): + translateDialogue(data, pbar, writer, format, filename, translatedList) + + # Write all Data + with LOCK: + if not ESTIMATE: + for row in data: + writer.writerow(row) + + except Exception as e: + traceback.print_exc() + + return tokens + +def translateUI(data, pbar, writer, format, filename, translatedList): + global LOCK, ESTIMATE + tokens = [0,0] + stringList = [None] * 1 + i = 0 + + try: + # Lists + textList = [] + + # Parse Data + while i in range(len(data)): + # Text + for j in range(len(data[i])): + jaString = data[i][j] + + # If Japanese Text, Translate it. + if not re.search(r'[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+', jaString): + continue + + # Replace Unicode + jaString = jaString.replace('', '...') + + # Pass 1 + if translatedList == []: + # Add to list + textList.append(f'{jaString}') + stringList[0] = textList + + # Pass 2 + else: + if translatedList[0]: + # Grab and Pop + translatedText = translatedList[0][0] + translatedList[0].pop(0) + + # Set to None if empty list + if len(translatedList[0]) <= 0: + translatedList[0] = None + + # Set Data + if len(data[i]) > j + 1: + data[i][j+1] = f'{translatedText}' + + # Iterate + i += 1 + + # EOF + stringList = [x for x in stringList if x is not None] + if len(stringList) > 0: + # Translate + pbar.total = 0 + for i in range(len(stringList)): + # Set Progress + pbar.total += len(stringList[i]) + pbar.refresh() + response = translateGPT(stringList[i], '', True) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + translatedList.append(response[0]) + + # Set Strings + if len(stringList) == len(translatedList): + translateUI(data, pbar, writer, format, filename, translatedList) + + # Write all Data + with LOCK: + if not ESTIMATE: + for row in data: + writer.writerow(row) + + except Exception as e: + traceback.print_exc() + + return tokens + + +# Save some money and enter the character before translation +def getSpeaker(speaker): + match speaker: + case 'ファイン': + return ['Fine', [0,0]] + case '': + return ['', [0,0]] + case _: + # Store Speaker + if speaker not in str(NAMESLIST): + response = translateGPT(speaker, 'Reply with only the '+ LANGUAGE +' translation of the NPC name.', False) + response[0] = response[0].replace("'S", "'s") + speakerList = [speaker, response[0]] + NAMESLIST.append(speakerList) + return response + + # Find Speaker + else: + for i in range(len(NAMESLIST)): + if speaker == NAMESLIST[i][0]: + return [NAMESLIST[i][1],[0,0]] + + return [speaker,[0,0]] + +def subVars(jaString): + jaString = jaString.replace('\u3000', ' ') + + # Nested + count = 0 + nestedList = re.findall(r'[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]', jaString) + nestedList = set(nestedList) + if len(nestedList) != 0: + for icon in nestedList: + jaString = jaString.replace(icon, '[Nested_' + str(count) + ']') + count += 1 + + # Icons + count = 0 + iconList = re.findall(r'[\\]+[iIkKwWaA]+\[[0-9]+\]', jaString) + iconList = set(iconList) + if len(iconList) != 0: + for icon in iconList: + jaString = jaString.replace(icon, '[Ascii_' + str(count) + ']') + count += 1 + + # Colors + count = 0 + colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString) + colorList = set(colorList) + if len(colorList) != 0: + for color in colorList: + jaString = jaString.replace(color, '[Color_' + str(count) + ']') + count += 1 + + # Names + count = 0 + nameList = re.findall(r'[\\]+[nN]\[.+?\]+', jaString) + nameList = set(nameList) + if len(nameList) != 0: + for name in nameList: + jaString = jaString.replace(name, '[Noun_' + str(count) + ']') + count += 1 + + # Variables + count = 0 + varList = re.findall(r'[\\]+[vV]\[[0-9]+\]', jaString) + varList = set(varList) + if len(varList) != 0: + for var in varList: + jaString = jaString.replace(var, '[Var_' + str(count) + ']') + count += 1 + + # Formatting + count = 0 + formatList = re.findall(r'[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]', jaString) + formatList = set(formatList) + if len(formatList) != 0: + for var in formatList: + jaString = jaString.replace(var, '[FCode_' + str(count) + ']') + count += 1 + + # Put all lists in list and return + allList = [nestedList, iconList, colorList, nameList, varList, formatList] + return [jaString, allList] + +def resubVars(translatedText, allList): + # Fix Spacing and ChatGPT Nonsense + matchList = re.findall(r'\[\s?.+?\s?\]', translatedText) + if len(matchList) > 0: + for match in matchList: + text = match.strip() + translatedText = translatedText.replace(match, text) + + # Nested + count = 0 + if len(allList[0]) != 0: + for var in allList[0]: + translatedText = translatedText.replace('[Nested_' + str(count) + ']', var) + count += 1 + + # Icons + count = 0 + if len(allList[1]) != 0: + for var in allList[1]: + translatedText = translatedText.replace('[Ascii_' + str(count) + ']', var) + count += 1 + + # Colors + count = 0 + if len(allList[2]) != 0: + for var in allList[2]: + translatedText = translatedText.replace('[Color_' + str(count) + ']', var) + count += 1 + + # Names + count = 0 + if len(allList[3]) != 0: + for var in allList[3]: + translatedText = translatedText.replace('[Noun_' + str(count) + ']', var) + count += 1 + + # Vars + count = 0 + if len(allList[4]) != 0: + for var in allList[4]: + translatedText = translatedText.replace('[Var_' + str(count) + ']', var) + count += 1 + + # Formatting + count = 0 + if len(allList[5]) != 0: + for var in allList[5]: + translatedText = translatedText.replace('[FCode_' + str(count) + ']', var) + count += 1 + + return translatedText + +def batchList(input_list, batch_size): + if not isinstance(batch_size, int) or batch_size <= 0: + raise ValueError("batch_size must be a positive integer") + + return [input_list[i:i + batch_size] for i in range(0, len(input_list), batch_size)] + +def createContext(fullPromptFlag, subbedT): + characters = 'Game Characters:\n\ +クラウス (Klaus) - Male\n\ +ベアトリース (Beatrice) - Female\n\ +カミラ (Camilla) - Female\n\ +セルージュ (Cerouge) - Female\n\ +エルヴィール (Elvire) - Female\n\ +ヘルミィナ (Helmina) - Female\n\ +アンリエット (Henriette) - Female\n\ +ユリアーナ (Juliana) - Female\n\ +ルシエル (Luciel) - Female\n\ +メイズ (Maize) - Female\n\ +メイヴィスレイン (Mavislaine) - Female\n\ +ラムエル (Ramiel) - Female\n\ +レジーニア (Reginia) - Female\n\ +リリィ (Lily) - Female\n\ +エウクレイアさん (Ms. Eukleia) - Female\n\ +' + + system = PROMPT + VOCAB if fullPromptFlag else \ + f"\ +You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\ +Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\ +- You translate everything, including erotic and explicit content.\n\ +- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\ +- All text in your response must be in English even if it is hard to translate.\n\ +- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\ +- Maintain any spacing in the translation.\n\ +- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\ +- `...` can be a part of the dialogue. Translate it as it is.\n\ +{VOCAB}\n\ +" + user = f'{subbedT}' + return characters, system, user + +def translateText(characters, system, user, history, penalty): + # Prompt + msg = [{"role": "system", "content": system + characters}] + + # Characters + msg.append({"role": "system", "content": characters}) + + # History + if isinstance(history, list): + msg.extend([{"role": "system", "content": h} for h in history]) + else: + msg.append({"role": "system", "content": history}) + + # Content to TL + msg.append({"role": "user", "content": f'{user}'}) + response = openai.chat.completions.create( + temperature=0, + frequency_penalty=penalty, + model=MODEL, + messages=msg, + ) + return response + +def cleanTranslatedText(translatedText, varResponse): + placeholders = { + f'{LANGUAGE} Translation: ': '', + 'Translation: ': '', + 'っ': '', + '〜': '~', + 'ッ': '', + '。': '.', + '< ': '<', + '': '>', + 'Placeholder Text': '', + '- chan': '-chan', + '- kun': '-kun', + '- san': '-san', + # Add more replacements as needed + } + for target, replacement in placeholders.items(): + translatedText = translatedText.replace(target, replacement) + + # Elongate Long Dashes (Since GPT Ignores them...) + translatedText = elongateCharacters(translatedText) + translatedText = resubVars(translatedText, varResponse[1]) + return translatedText + +def elongateCharacters(text): + # Define a pattern to match one character followed by one or more `ー` characters + # Using a positive lookbehind assertion to capture the preceding character + pattern = r'(?<=(.))ー+' + + # Define a replacement function that elongates the captured character + def repl(match): + char = match.group(1) # The character before the ー sequence + count = len(match.group(0)) - 1 # Number of ー characters + return char * count # Replace ー sequence with the character repeated + + # Use re.sub() to replace the pattern in the text + return re.sub(pattern, repl, text) + +def extractTranslation(translatedTextList, is_list): + pattern = r'`?([\\]*.*?[\\]*?)<\/?Line\d+>`?' + # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. + if is_list: + matchList = re.findall(pattern, translatedTextList) + return matchList + else: + matchList = re.findall(pattern, translatedTextList) + return matchList[0][0] if matchList else translatedTextList + +def countTokens(characters, system, user, history): + inputTotalTokens = 0 + outputTotalTokens = 0 + enc = tiktoken.encoding_for_model('gpt-4') + + # Input + if isinstance(history, list): + for line in history: + inputTotalTokens += len(enc.encode(line)) + else: + inputTotalTokens += len(enc.encode(history)) + inputTotalTokens += len(enc.encode(system)) + inputTotalTokens += len(enc.encode(characters)) + inputTotalTokens += len(enc.encode(user)) + + # Output + outputTotalTokens += round(len(enc.encode(user))*3) + + return [inputTotalTokens, outputTotalTokens] + +def combineList(tlist, text): + if isinstance(text, list): + return [t for sublist in tlist for t in sublist] + return tlist[0] + +@retry(exceptions=Exception, tries=5, delay=5) +def translateGPT(text, history, fullPromptFlag): + global PBAR + + mismatch = False + totalTokens = [0, 0] + if isinstance(text, list): + tList = batchList(text, BATCHSIZE) + else: + tList = [text] + + for index, tItem in enumerate(tList): + # Before sending to translation, if we have a list of items, add the formatting + if isinstance(tItem, list): + payload = '\n'.join([f'`{item}`' for i, item in enumerate(tItem)]) + payload = re.sub(r'(<)(\/Line\d+>)', r'\1>Placeholder Text<\3', payload) + varResponse = subVars(payload) + subbedT = varResponse[0] + else: + varResponse = subVars(tItem) + subbedT = varResponse[0] + + # Things to Check before starting translation + if not re.search(r'[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+', subbedT): + if PBAR is not None: + PBAR.update(len(tItem)) + continue + + # Create Message + characters, system, user = createContext(fullPromptFlag, subbedT) + + # Calculate Estimate + if ESTIMATE: + estimate = countTokens(characters, system, user, history) + totalTokens[0] += estimate[0] + totalTokens[1] += estimate[1] + continue + + # Translating + response = translateText(characters, system, user, history, 0.02) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Formatting + translatedText = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedText, True) + tList[index] = extractedTranslations + if len(tItem) != len(extractedTranslations): + # Mismatch. Try Again + response = translateText(characters, system, user, history, 0.2) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Formatting + translatedText = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedText, True) + tList[index] = extractedTranslations + if len(tItem) != len(extractedTranslations): + mismatch = True # Just here for breakpoint + + # Create History + with LOCK: + if PBAR is not None: + PBAR.update(len(tItem)) + if not mismatch: + history = extractedTranslations[-10:] # Update history if we have a list + else: + history = text[-10:] + else: + # Ensure we're passing a single string to extractTranslation + extractedTranslations = extractTranslation(translatedText, False) + tList[index] = extractedTranslations + + finalList = combineList(tList, text) + return [finalList, totalTokens] diff --git a/modules/main.py b/modules/main.py index 980cca3..abc2cb5 100644 --- a/modules/main.py +++ b/modules/main.py @@ -19,6 +19,7 @@ these values using an .env file, for an example see .env.example') from modules.rpgmakermvmz import handleMVMZ from modules.rpgmakerace import handleACE from modules.csv import handleCSV +from modules.eushully import handleEushully from modules.alice import handleAlice from modules.tyrano import handleTyrano from modules.json import handleJSON @@ -26,7 +27,7 @@ from modules.kansen import handleKansen from modules.lune import handleLune from modules.atelier import handleAtelier from modules.anim import handleAnim -from modules.nscript import handleNScript +from modules.nscript import handleOnscripter from modules.wolf import handleWOLF from modules.wolf2 import handleWOLF2 from modules.javascript import handleJavascript @@ -42,6 +43,7 @@ MODULES = [ ["RPGMaker MV/MZ", "json", handleMVMZ], ["RPGMaker ACE", "yaml", handleACE], ["CSV (From Translator++)", "csv", handleCSV], + ["Eushully", "csv", handleEushully], ["Alice", "txt", handleAlice], ["Tyrano", "ks", handleTyrano], ["JSON", "json", handleJSON], @@ -49,7 +51,7 @@ MODULES = [ ["Lune", "json", handleLune], ["Atelier", "txt", handleAtelier], ["Anim", "json", handleAnim], - ["NScript", "txt", handleNScript], + ["NScript", "txt", handleOnscripter], ["Wolf", "json", handleWOLF], ["Wolf", "txt", handleWOLF2], ["Javascript", "js", handleJavascript], diff --git a/modules/nscript.py b/modules/nscript.py index 46a8469..4bd4742 100644 --- a/modules/nscript.py +++ b/modules/nscript.py @@ -51,7 +51,7 @@ elif 'gpt-4' in MODEL: OUTPUTAPICOST = .015 BATCHSIZE = 40 -def handleNScript(filename, estimate): +def handleOnscripter(filename, estimate): global ESTIMATE ESTIMATE = estimate @@ -77,7 +77,7 @@ def handleNScript(filename, estimate): else: try: - with open('translated/' + filename, 'w', encoding='utf8', errors='ignore') as outFile: + with open('translated/' + filename, 'w', encoding='cp932', errors='ignore') as outFile: start = time.time() translatedData = openFiles(filename) @@ -120,7 +120,7 @@ def getResultString(translatedData, translationTime, filename): def openFiles(filename): with open('files/' + filename, 'r', encoding='cp932') as readFile: - translatedData = parseNScript(readFile, filename) + translatedData = parseOnscripter(readFile, filename) # Delete lines marked for deletion finalData = [] @@ -131,20 +131,18 @@ def openFiles(filename): return translatedData -def parseNScript(readFile, filename): +def parseOnscripter(readFile, filename): totalTokens = [0,0] - totalLines = 0 - # Get total for progress bar + # Read File into data data = readFile.readlines() - totalLines = len(data) - with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar: + # Create Progress Bar + with tqdm(bar_format=BAR_FORMAT, position=POSITION, leave=LEAVE) as pbar: pbar.desc=filename - pbar.total=totalLines try: - result = translateNScript(data, pbar, totalLines) + result = translateOnscripter(data, pbar, filename, []) totalTokens[0] += result[0] totalTokens[1] += result[1] except Exception as e: @@ -152,240 +150,108 @@ def parseNScript(readFile, filename): return [data, totalTokens, e] return [data, totalTokens, None] -def translateNScript(data, pbar, totalLines): - textHistory = [] - batch = [] +def translateOnscripter(data, pbar, filename, translatedList): + stringList = [] currentGroup = [] - maxHistory = MAXHISTORY tokens = [0,0] speaker = '' - insertBool = False + voice = False global LOCK, ESTIMATE i = 0 - batchStartIndex = 0 while i < len(data): - # Speaker - matchList = re.findall(r'^【\s+(.*?)\s+】$', data[i]) - if len(matchList) != 0: - response = getSpeaker(matchList[0]) - speaker = response[0] - tokens[0] += response[1][0] - tokens[1] += response[1][1] - data[i] = '>[' + speaker + ']\n' - i += 1 - else: - speaker = '' - - # Choices - if 'select' in data[i]: - matchList = re.findall(r'\"(.*?)\"', data[i]) - if len(matchList) != 0: - originalTextList = matchList - if len(textHistory) > 0: - response = translateGPT(matchList, 'Keep your translation as brief as possible. Previous text for context: ' + textHistory[len(textHistory)-1] + '\n\nReply in the style of a dialogue option.', True) - else: - response = translateGPT(matchList, '\n\nReply in the style of a dialogue option.', True) - translatedTextList = response[0] - tokens[0] += response[1][0] - tokens[1] += response[1][1] - - for choice in range(len(translatedTextList)): - translatedText = translatedTextList[choice] - - # Remove characters that may break scripts - charList = ['.', '\"', '\\n'] - for char in charList: - translatedText = translatedText.replace(char, '') - - # Escape all ' - translatedText = translatedText.replace('\\', '') - translatedText = translatedText.replace(' ', ' ') - - # Set Data - translatedText = data[i].replace(originalTextList[choice], translatedText) - data[i] = translatedText - pbar.update(1) - i += 1 - else: - pbar.update(1) - i += 1 - - # Lines - matchList = re.findall(r'^[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9「」『』 >()].*', data[i]) - if len(matchList) > 0: - currentGroup.append(matchList[0]) - if len(data) > i+1: - if speaker == '': - while '\n' != data[i+1] and '【' not in data[i+1]: - if insertBool is True: - data[i] = r'\d\n' - pbar.update(1) - i += 1 - matchList = re.findall(r'^[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9「」『』 >()].*', data[i]) - if len(matchList) > 0: - currentGroup.append(matchList[0]) - else: - while ' ' in data[i+1][0] or '"' in data[i+1] or ')' in data[i+1] or ')' in data[i+1]: - if insertBool is True: - data[i] = r'\d\n' - pbar.update(1) - i += 1 - matchList = re.findall(r'^[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9「」『』 >()].*', data[i]) - if len(matchList) > 0: - currentGroup.append(matchList[0]) + voice = False + speaker = '' + if re.search(r'^caption|^event|^operation_caption|^manual|^mes\s\d+', data[i]): + # Lines + match = re.search(r'="(.+?)"', data[i]) + if match == None: + match = re.search(r'mes\s\d+,"(.+?)"', data[i]) + if match != None and match.group(1) != '': + originalString = match.group(1) + # Pass 1 + if translatedList == []: + # Grab Consecutive Strings + jaString = match.group(1) - # Join up 401 groups for better translation. - if len(currentGroup) > 0: - finalJAString = ' '.join(currentGroup) - oldjaString = finalJAString + # Remove any textwrap + jaString = jaString.replace('\\n', ' ') - # Remove any textwrap - if FIXTEXTWRAP == True: - finalJAString = finalJAString.replace('>', '') - finalJAString = finalJAString.replace('\\', ' ') + # Remove Furigana + furiMatch = re.findall(r'({(.+?)\/(.+?)})', jaString) + if furiMatch: + for match in furiMatch: + jaString = jaString.replace(match[0], match[2]) - # Remove Extra Stuff bad for translation. - finalJAString = finalJAString.replace('゙', '') - finalJAString = finalJAString.replace('・', '.') - finalJAString = finalJAString.replace('‶', '') - finalJAString = finalJAString.replace('”', '') - finalJAString = finalJAString.replace('―', '-') - finalJAString = finalJAString.replace('…', '...') - finalJAString = re.sub(r'(\.{3}\.+)', '...', finalJAString) - finalJAString = finalJAString.replace(' ', ' ') + # Add String + stringList.append(jaString.strip()) + + # Pass 2 + else: + # Get Text + if translatedList: + # Grab and Pop + translatedText = translatedList[0] + translatedList.pop(0) - # Furigana Removal - matchList = re.findall(r'『\((.+)/.*?』', finalJAString) - if len(matchList) > 0: - finalJAString = finalJAString.replace(matchList[0][0], matchList[0][1]) + # Set to None if empty list + if len(translatedList) <= 0: + translatedList = None - # Add Speaker (If there is one) - if speaker != '': - finalJAString = f'{speaker}: {finalJAString}' - - # [Passthrough 1] Pulling From File - if insertBool is False: - # Append to List and Clear Values - batch.append(finalJAString) - speaker = '' - - # Translate Batch if Full - if len(batch) == BATCHSIZE: - # Translate - response = translateGPT(batch, textHistory, True) - tokens[0] += response[1][0] - tokens[1] += response[1][1] - translatedBatch = response[0] - textHistory = translatedBatch[-10:] - - # Set Values - if len(batch) == len(translatedBatch): - i = batchStartIndex - insertBool = True - - # Mismatch - else: - pbar.write(f'Mismatch: {batchStartIndex} - {i}') - MISMATCH.append(batch) - batchStartIndex = i - batch.clear() + # Textwrap + translatedText = textwrap.fill(translatedText, width=WIDTH) + translatedText = translatedText.replace('\n', '\\n') + translatedText = translatedText.replace('\"', '\'') + # Set Data + data[i] = data[i].replace(originalString, translatedText) i += 1 - if insertBool is True: - pbar.update(1) - currentGroup = [] - # [Passthrough 2] Setting Data + # Nothing relevant. Skip Line. else: - # Get Text - translatedText = translatedBatch[0] - translatedText = translatedText.replace('\\"', '\"') - translatedText = translatedText.replace('[', '(') - translatedText = translatedText.replace(']', ')') - - # Remove added speaker - translatedText = re.sub(r'^.+?:\s', '', translatedText) - - # Textwrap - translatedText = textwrap.fill(translatedText, width=WIDTH) - textList = translatedText.split('\n') - - # Set Text - data[i] = r'\d\n' - counter = 0 - for line in textList: - # Wordwrap Text - line = textwrap.fill(line, width=WIDTH) - - # Set - data.insert(i, '>' + line.strip() + '\n') - counter += 1 - i+=1 - - # Go to new window if too long - if counter >= 4: - data[i-1] = data[i-1].replace('\n', '\\\n') - counter = 0 - if '\\' not in data[i-1]: - data[i-1] = data[i-1].replace('\n', '\\\n') - translatedBatch.pop(0) - speaker = '' - currentGroup = [] - - # If Batch is empty. Move on. - if len(translatedBatch) == 0: - insertBool = False - batchStartIndex = i - batch.clear() - - # Nothing relevant. Skip Line. + i += 1 else: i += 1 - if insertBool is False: - pbar.update(1) - # Translate Batch if not empty and EOF - if len(batch) != 0 and i >= len(data): - # Translate - response = translateGPT(batch, textHistory, True) - tokens[0] += response[1][0] - tokens[1] += response[1][1] - translatedBatch = response[0] - textHistory = translatedBatch[-10:] + # EOF + if len(stringList) > 0: + # Set Progress + pbar.total = len(stringList) + pbar.refresh() + + # Translate + response = translateGPT(stringList, '', True, pbar, filename) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + translatedList = response[0] - # Set Values - if len(batch) == len(translatedBatch): - i = batchStartIndex - insertBool = True + # Set Strings + if len(stringList) == len(translatedList): + translateOnscripter(data, pbar, filename, translatedList) - # Mismatch - else: - pbar.write(f'Mismatch: {batchStartIndex} - {i}') - MISMATCH.append(batch) - batchStartIndex = i - batch.clear() - - currentGroup = [] + # Mismatch + else: + with LOCK: + if filename not in MISMATCH: + MISMATCH.append(filename) return tokens # Save some money and enter the character before translation -def getSpeaker(speaker): +def getSpeaker(speaker, pbar, filename): match speaker: - case 'ルイ': - return ['Rui', [0,0]] - case 'チュベロス': - return ['Tuberose', [0,0]] + case 'ファイン': + return ['Fine', [0,0]] case '': return ['', [0,0]] case _: # Store Speaker if speaker not in str(NAMESLIST): - response = translateGPT(speaker, 'Reply with only the '+ LANGUAGE +' translation of the NPC name.', False) - response[0] = response[0].title() + response = translateGPT(speaker, 'Reply with only the '+ LANGUAGE +' translation of the NPC name.', False, pbar, filename) + response[0] = response[0].replace("'S", "'s") speakerList = [speaker, response[0]] NAMESLIST.append(speakerList) return response + # Find Speaker else: for i in range(len(NAMESLIST)): @@ -393,7 +259,7 @@ def getSpeaker(speaker): return [NAMESLIST[i][1],[0,0]] return [speaker,[0,0]] - + def subVars(jaString): jaString = jaString.replace('\u3000', ' ') @@ -403,7 +269,7 @@ def subVars(jaString): nestedList = set(nestedList) if len(nestedList) != 0: for icon in nestedList: - jaString = jaString.replace(icon, '{Nested_' + str(count) + '}') + jaString = jaString.replace(icon, '[Nested_' + str(count) + ']') count += 1 # Icons @@ -412,7 +278,7 @@ def subVars(jaString): iconList = set(iconList) if len(iconList) != 0: for icon in iconList: - jaString = jaString.replace(icon, '{Ascii_' + str(count) + '}') + jaString = jaString.replace(icon, '[Ascii_' + str(count) + ']') count += 1 # Colors @@ -421,7 +287,7 @@ def subVars(jaString): colorList = set(colorList) if len(colorList) != 0: for color in colorList: - jaString = jaString.replace(color, '{Color_' + str(count) + '}') + jaString = jaString.replace(color, '[Color_' + str(count) + ']') count += 1 # Names @@ -430,7 +296,7 @@ def subVars(jaString): nameList = set(nameList) if len(nameList) != 0: for name in nameList: - jaString = jaString.replace(name, '{Noun_' + str(count) + '}') + jaString = jaString.replace(name, '[Noun_' + str(count) + ']') count += 1 # Variables @@ -439,16 +305,16 @@ def subVars(jaString): varList = set(varList) if len(varList) != 0: for var in varList: - jaString = jaString.replace(var, '{Var_' + str(count) + '}') + jaString = jaString.replace(var, '[Var_' + str(count) + ']') count += 1 # Formatting count = 0 - formatList = re.findall(r'[\\]+[\w]+\[.+?\]', jaString) + formatList = re.findall(r'[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]', jaString) formatList = set(formatList) if len(formatList) != 0: for var in formatList: - jaString = jaString.replace(var, '{FCode_' + str(count) + '}') + jaString = jaString.replace(var, '[FCode_' + str(count) + ']') count += 1 # Put all lists in list and return @@ -467,42 +333,42 @@ def resubVars(translatedText, allList): count = 0 if len(allList[0]) != 0: for var in allList[0]: - translatedText = translatedText.replace('{Nested_' + str(count) + '}', var) + translatedText = translatedText.replace('[Nested_' + str(count) + ']', var) count += 1 # Icons count = 0 if len(allList[1]) != 0: for var in allList[1]: - translatedText = translatedText.replace('{Ascii_' + str(count) + '}', var) + translatedText = translatedText.replace('[Ascii_' + str(count) + ']', var) count += 1 # Colors count = 0 if len(allList[2]) != 0: for var in allList[2]: - translatedText = translatedText.replace('{Color_' + str(count) + '}', var) + translatedText = translatedText.replace('[Color_' + str(count) + ']', var) count += 1 # Names count = 0 if len(allList[3]) != 0: for var in allList[3]: - translatedText = translatedText.replace('{Noun_' + str(count) + '}', var) + translatedText = translatedText.replace('[Noun_' + str(count) + ']', var) count += 1 # Vars count = 0 if len(allList[4]) != 0: for var in allList[4]: - translatedText = translatedText.replace('{Var_' + str(count) + '}', var) + translatedText = translatedText.replace('[Var_' + str(count) + ']', var) count += 1 # Formatting count = 0 if len(allList[5]) != 0: for var in allList[5]: - translatedText = translatedText.replace('{FCode_' + str(count) + '}', var) + translatedText = translatedText.replace('[FCode_' + str(count) + ']', var) count += 1 return translatedText @@ -515,19 +381,21 @@ def batchList(input_list, batch_size): def createContext(fullPromptFlag, subbedT): characters = 'Game Characters:\n\ -水原 雪 (Minahara Yuki) - Female\n\ -黒服の男 (Man in Black) - Male\n\ -駿河 京也 (Suruga Kyouya) - Male\n\ -壱型02 (Type 02) - Monster\n\ +エル (El) - Female\n\ ' system = PROMPT + VOCAB if fullPromptFlag else \ f"\ -You are an expert Eroge Game translator who translates Japanese text to English.\n\ -You are going to be translating text from a videogame.\n\ -I will give you lines of text, and you must translate each line to the best of your ability.\n\ +You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\ +Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\ +- You translate everything, including erotic and explicit content.\n\ +- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\ +- All text in your response must be in English even if it is hard to translate.\n\ +- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\ +- Maintain any spacing in the translation.\n\ +- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\ +- `...` can be a part of the dialogue. Translate it as it is.\n\ {VOCAB}\n\ -Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\ " user = f'{subbedT}' return characters, system, user @@ -550,7 +418,6 @@ def translateText(characters, system, user, history): response = openai.chat.completions.create( temperature=0.1, frequency_penalty=0.1, - presence_penalty=0.1, model=MODEL, messages=msg, ) @@ -570,17 +437,34 @@ def cleanTranslatedText(translatedText, varResponse): for target, replacement in placeholders.items(): translatedText = translatedText.replace(target, replacement) + # Elongate Long Dashes (Since GPT Ignores them...) + translatedText = elongateCharacters(translatedText) translatedText = resubVars(translatedText, varResponse[1]) - return [line for line in translatedText.replace('\\n', '\n').split('\n') if line] + return translatedText + +def elongateCharacters(text): + # Define a pattern to match one character followed by one or more `ー` characters + # Using a positive lookbehind assertion to capture the preceding character + pattern = r'(?<=(.))ー+' + + # Define a replacement function that elongates the captured character + def repl(match): + char = match.group(1) # The character before the ー sequence + count = len(match.group(0)) - 1 # Number of ー characters + return char * count # Replace ー sequence with the character repeated + + # Use re.sub() to replace the pattern in the text + return re.sub(pattern, repl, text) def extractTranslation(translatedTextList, is_list): - pattern = r'`?([\\]*.*?[\\]*?)<\/?Line\d+>`?' + pattern = r'`?([\\]*.*?[\\]*?)<\/?Line\d+>`?' # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. if is_list: - return [re.findall(pattern, line)[0][1] for line in translatedTextList if re.search(pattern, line)] + matchList = re.findall(pattern, translatedTextList) + return matchList else: matchList = re.findall(pattern, translatedTextList) - return matchList[0][1] if matchList else translatedTextList + return matchList[0][0] if matchList else translatedTextList def countTokens(characters, system, user, history): inputTotalTokens = 0 @@ -598,7 +482,7 @@ def countTokens(characters, system, user, history): inputTotalTokens += len(enc.encode(user)) # Output - outputTotalTokens += round(len(enc.encode(user))*3) + outputTotalTokens += round(len(enc.encode(user))*2) return [inputTotalTokens, outputTotalTokens] @@ -608,7 +492,8 @@ def combineList(tlist, text): return tlist[0] @retry(exceptions=Exception, tries=5, delay=5) -def translateGPT(text, history, fullPromptFlag): +def translateGPT(text, history, fullPromptFlag, pbar, filename): + mismatch = False totalTokens = [0, 0] if isinstance(text, list): tList = batchList(text, BATCHSIZE) @@ -619,7 +504,7 @@ def translateGPT(text, history, fullPromptFlag): # Before sending to translation, if we have a list of items, add the formatting if isinstance(tItem, list): payload = '\n'.join([f'`{item}`' for i, item in enumerate(tItem)]) - payload = payload.replace('``', '`Placeholder Text`') + payload = re.sub(r'(<)(\/Line\d+>)', r'\1>Placeholder Text<\3', payload) varResponse = subVars(payload) subbedT = varResponse[0] else: @@ -647,16 +532,34 @@ def translateGPT(text, history, fullPromptFlag): totalTokens[1] += response.usage.completion_tokens # Formatting - translatedTextList = cleanTranslatedText(translatedText, varResponse) + translatedText = cleanTranslatedText(translatedText, varResponse) if isinstance(tItem, list): - extractedTranslations = extractTranslation(translatedTextList, True) - tList[index] = extractedTranslations - if len(tItem) != len(translatedTextList): - mismatch = True # Just here so breakpoint can be set - history = extractedTranslations[-10:] # Update history if we have a list + extractedTranslations = extractTranslation(translatedText, True) + if len(tItem) != len(extractedTranslations): + # Mismatch. Try Again + response = translateText(characters, system, user, history) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Formatting + translatedText = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedText, True) + if len(tItem) == len(extractedTranslations): + tList[index] = extractedTranslations + else: + MISMATCH.append(filename) + else: + tList[index] = extractedTranslations + + # Create History + history = tList[index] # Update history if we have a list + pbar.update(len(tList[index])) + else: # Ensure we're passing a single string to extractTranslation - extractedTranslations = extractTranslation('\n'.join(translatedTextList), False) + extractedTranslations = extractTranslation(translatedText, False) tList[index] = extractedTranslations finalList = combineList(tList, text) diff --git a/vocab.txt b/vocab.txt index f096b08..8a8fc77 100644 --- a/vocab.txt +++ b/vocab.txt @@ -44,5 +44,21 @@ ME 音量 (ME Volume) アスカロン (Ascalon) # Other -パパ (papa) +悪魔 (Devil) +上級悪魔 (Arch Devil) +歪魔 (Distorted Devil) +魔神 (Demon) +魔人 (Majin) +睡魔 (Mare) +淫魔 (Succubus) +天使 (Angel) +大天使 (Archangel) +権天使 (Ruler) +能天使 (Power) +力天使 (Virtue) +主天使 (Dominion) +智天使 (Cherub) +飛天魔 (Nephilim) +堕天使 (Fallen Angel) +鬼 (Oni) ``` \ No newline at end of file