From aebde1a8190bf4c1ff496fa2a9df4de46bd63548 Mon Sep 17 00:00:00 2001 From: DazedAnon Date: Fri, 17 May 2024 12:25:20 -0500 Subject: [PATCH] Adjust pricing for gpt4o --- modules/alice.py | 2 +- modules/anim.py | 2 +- modules/atelier.py | 2 +- modules/csv.py | 6 +- modules/irissoft.py | 706 ++++++++++++++++++++++++++++++++++++++++ modules/javascript.py | 6 +- modules/json.py | 2 +- modules/kansen.py | 2 +- modules/lune.py | 2 +- modules/main.py | 2 + modules/nscript.py | 6 +- modules/regex.py | 175 +--------- modules/rpgmakerace.py | 2 +- modules/rpgmakermvmz.py | 2 +- modules/sakuranbo.py | 2 +- modules/tyrano.py | 6 +- modules/wolf.py | 6 +- modules/wolf2.py | 6 +- 18 files changed, 753 insertions(+), 184 deletions(-) create mode 100644 modules/irissoft.py diff --git a/modules/alice.py b/modules/alice.py index 6c1d034..da644d4 100644 --- a/modules/alice.py +++ b/modules/alice.py @@ -483,7 +483,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list): diff --git a/modules/anim.py b/modules/anim.py index cb4705c..72e9601 100644 --- a/modules/anim.py +++ b/modules/anim.py @@ -443,7 +443,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list): diff --git a/modules/atelier.py b/modules/atelier.py index 30daf1b..6cfde80 100644 --- a/modules/atelier.py +++ b/modules/atelier.py @@ -301,7 +301,7 @@ def translateGPT(t, history, fullPromptFlag): # If ESTIMATE is True just count this as an execution and return. if ESTIMATE: - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') historyRaw = '' if isinstance(history, list): for line in history: diff --git a/modules/csv.py b/modules/csv.py index ef73fc4..b9a507c 100644 --- a/modules/csv.py +++ b/modules/csv.py @@ -45,8 +45,8 @@ if 'gpt-3.5' in MODEL: BATCHSIZE = 10 FREQUENCY_PENALTY = 0.2 elif 'gpt-4' in MODEL: - INPUTAPICOST = .01 - OUTPUTAPICOST = .03 + INPUTAPICOST = .005 + OUTPUTAPICOST = .015 BATCHSIZE = 40 FREQUENCY_PENALTY = 0.1 @@ -524,7 +524,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list): diff --git a/modules/irissoft.py b/modules/irissoft.py new file mode 100644 index 0000000..ee21d9f --- /dev/null +++ b/modules/irissoft.py @@ -0,0 +1,706 @@ +# Libraries +import os, re, textwrap, threading, time, traceback, tiktoken, openai +from pathlib import Path +from colorama import Fore +from dotenv import load_dotenv +from retry import retry +from tqdm import tqdm + +# Open AI +load_dotenv() +if os.getenv('api').replace(' ', '') != '': + openai.base_url = os.getenv('api') +openai.organization = os.getenv('org') +openai.api_key = os.getenv('key') + +#Globals +MODEL = os.getenv('model') +TIMEOUT = int(os.getenv('timeout')) +LANGUAGE = os.getenv('language').capitalize() +PROMPT = Path('prompt.txt').read_text(encoding='utf-8') +VOCAB = Path('vocab.txt').read_text(encoding='utf-8') +THREADS = int(os.getenv('threads')) +LOCK = threading.Lock() +WIDTH = int(os.getenv('width')) +LISTWIDTH = int(os.getenv('listWidth')) +NOTEWIDTH = 70 +MAXHISTORY = 10 +ESTIMATE = '' +TOKENS = [0, 0] +NAMESLIST = [] +NAMES = False # Output a list of all the character names found +BRFLAG = False # If the game uses
instead +FIXTEXTWRAP = True # Overwrites textwrap +IGNORETLTEXT = False # Ignores all translated text. +MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) + +#tqdm Globals +BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' +POSITION = 0 +LEAVE = False + +# Pricing - Depends on the model https://openai.com/pricing +# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request +# If you are getting a MISMATCH LENGTH error, lower the batch size. +if 'gpt-3.5' in MODEL: + INPUTAPICOST = .002 + OUTPUTAPICOST = .002 + BATCHSIZE = 10 +elif 'gpt-4' in MODEL: + INPUTAPICOST = .005 + OUTPUTAPICOST = .015 + BATCHSIZE = 40 + +def handleIris(filename, estimate): + global ESTIMATE + ESTIMATE = estimate + + if ESTIMATE: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + + # Print Total + totalString = getResultString(['', TOKENS, None], end - start, 'TOTAL') + + # Print any errors on maps + if len(MISMATCH) > 0: + return totalString + Fore.RED + f'\nMismatch Errors: {MISMATCH}' + Fore.RESET + else: + return totalString + + else: + try: + with open('translated/' + filename, 'w', encoding='cp932', errors='ignore') as outFile: + start = time.time() + translatedData = openFiles(filename) + + # Print Result + end = time.time() + outFile.writelines(translatedData[0]) + tqdm.write(getResultString(translatedData, end - start, filename)) + with LOCK: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + except Exception as e: + traceback.print_exc() + return 'Fail' + + return getResultString(['', TOKENS, None], end - start, 'TOTAL') + +def getResultString(translatedData, translationTime, filename): + # File Print String + totalTokenstring =\ + Fore.YELLOW +\ + '[Input: ' + str(translatedData[1][0]) + ']'\ + '[Output: ' + str(translatedData[1][1]) + ']'\ + '[Cost: ${:,.4f}'.format((translatedData[1][0] * .001 * INPUTAPICOST) +\ + (translatedData[1][1] * .001 * OUTPUTAPICOST)) + ']' + timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' + + if translatedData[2] == None: + # Success + return filename + ': ' + totalTokenstring + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET + + else: + # Fail + try: + raise translatedData[2] + except Exception as e: + traceback.print_exc() + errorString = str(e) + Fore.RED + return filename + ': ' + totalTokenstring + timeString + Fore.RED + u' \u2717 ' +\ + errorString + Fore.RESET + +def openFiles(filename): + with open('files/' + filename, 'r', encoding='shift_jis') as readFile: + translatedData = parseIris(readFile, filename) + + # Delete lines marked for deletion + finalData = [] + for line in translatedData[0]: + if line != '\\d\n': + finalData.append(line) + translatedData[0] = finalData + + return translatedData + +def parseIris(readFile, filename): + totalTokens = [0,0] + + # Read File into data + data = readFile.readlines() + + # Create Progress Bar + with tqdm(bar_format=BAR_FORMAT, position=POSITION, leave=LEAVE) as pbar: + pbar.desc=filename + + try: + result = translateIris(data, pbar, filename, []) + totalTokens[0] += result[0] + totalTokens[1] += result[1] + except Exception as e: + traceback.print_exc() + return [data, totalTokens, e] + return [data, totalTokens, None] + +def translateIris(data, pbar, filename, translatedList): + stringList = [] + currentGroup = [] + tokens = [0,0] + speaker = '' + voice = False + global LOCK, ESTIMATE + i = 0 + + while i < len(data): + voice = False + speaker = '' + if '#MSGVOICE' in data[i]: + i += 1 + voice = True + voiceVar = data[i] + if '#MSG,' in data[i] or '#MSG\n' in data[i] or voice == True: + i += 1 + # Speaker + if re.search(r'^ ?([^#\/."、。*!!()\(\)\[\] \n]+)\n', data[i]) and len(data[i]) < 30: + match = re.search(r'(.*)', data[i]) + if match != None: + speaker = match.group(1) + if speaker[0] == '\u3000': + speaker = speaker[1:] + response = getSpeaker(speaker, pbar, filename) + speaker = response[0] + tokens[0] += response[1][0] + tokens[1] += response[1][1] + if translatedList != []: + speaker = speaker.replace(' ', '\u3000') + data[i] = f'\u3000{speaker}\n' + else: + speaker = '' + i += 1 + + # Lines + match = re.search(r'(.*)', data[i]) + if match != None and match.group(1) != '': + # Pass 1 + if translatedList == []: + # Grab Consecutive Strings + jaString = data[i] + if data[i] != '\n': + if data[i][0] == '\u3000': + jaString = data[i][1:] + currentGroup.append(jaString) + i += 1 + while data[i] != '\n': + jaString = data[i] + if data[i] != '\n': + jaString = data[i][1:] + currentGroup.append(jaString) + i += 1 + + # Join up 401 groups for better translation. + if len(currentGroup) > 0: + jaString = ''.join(currentGroup) + currentGroup = [] + + # Remove any textwrap + jaString = jaString.replace('\n', ' ') + + # Temporarily convert spaces (For Textwrap Later) + jaString = jaString.replace('\u3000', ' ') + + # Add Speaker (If there is one) + if speaker != '': + jaString = f'{speaker}: {jaString}' + + # Add String + stringList.append(jaString.strip()) + + # Pass 2 + else: + # Insert Strings + while data[i] != '\n': + data.pop(i) + + # Get Text + if translatedList: + translatedText = translatedList[0] + translatedList.pop(0) + if len(translatedList) <= 0: + translatedList = None + + # Remove added speaker + translatedText = re.sub(r'^.+?:\s', '', translatedText) + + # Textwrap + translatedText = textwrap.fill(translatedText, width=WIDTH) + translatedText = translatedText.replace('\n', '\n\u3000') + + # Replace Whitespace and Commas + translatedText = translatedText.replace(', ', '、') + translatedText = translatedText.replace(',\u3000', '、') + translatedText = translatedText.replace(',', '、') + translatedText = translatedText.replace(' ', '\u3000') + + # Set Data + # Game crashes on more than 3 lines. Will need to create a new MSG for long translations + if translatedText.count('\n') > 2: + # Split List + translatedTextList = splitNewlines(translatedText) + + # MSG Voice + count = 0 + for text in translatedTextList: + if count != 0: + if voice == True: + #MSG for each item in the list + data.insert(i, f'#MSGVOICE,\n') + i += 1 + data.insert(i, f'{voiceVar}') + i += 1 + else: + data.insert(i, f'#MSG,\n') + i += 1 + if speaker: + data[i] = f'\u3000{speaker}\n' + i += 1 + if text[0] == '\u3000': + data.insert(i, f'{text}\n') + else: + data.insert(i, f'\u3000{text}\n') + i += 1 + count += 1 + if data[i] != '\n': + data.insert(i, '\n') + data[i] = f'\n{data[i]}' + else: + data.insert(i, f'\u3000{translatedText}\n') + i += 1 + if data[i] != '\n': + data[i] = f'\n{data[i]}' + + elif '#SELECT' in data[i] and translatedList == []: + Iris = r'(.+?) +\d$' + i += 1 + match = re.search(Iris, data[i]) + if match: + choiceList = [] + choiceList.append(match.group(1)) + i += 1 + match = re.search(Iris, data[i]) + while(match): + choiceList.append(match.group(1)) + i += 1 + match = re.search(Iris, data[i]) + + # Translate + question = stringList[len(stringList) - 1] + response = translateGPT(choiceList, f'Previous text for context: {question}\n\nThis will be a dialogue option', True, pbar, filename) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + choiceListTL = response[0] + + # Set Data + i = i - len(choiceListTL) + for j in range(len(choiceListTL)): + # Replace Whitespace and Commas + choiceListTL[j] = choiceListTL[j].replace(', ', '、') + choiceListTL[j] = choiceListTL[j].replace(',\u3000', '、') + choiceListTL[j] = choiceListTL[j].replace(',', '、') + choiceListTL[j] = choiceListTL[j].replace(' ', '\u3000') + data[i] = data[i].replace(choiceList[j], choiceListTL[j]) + i += 1 + + # Nothing relevant. Skip Line. + else: + i += 1 + else: + i += 1 + + # EOF + if len(stringList) > 0: + # Set Progress + pbar.total = len(stringList) + pbar.refresh() + + # Translate + response = translateGPT(stringList, '', True, pbar, filename) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + translatedList = response[0] + + # Set Strings + if len(stringList) == len(translatedList): + translateIris(data, pbar, filename, translatedList) + + # Mismatch + else: + with LOCK: + if filename not in MISMATCH: + MISMATCH.append(filename) + return tokens + +def splitNewlines(text): + parts = [] + newline_count = 0 # Counts the number of newline characters encountered + start_index = 0 # Start index of the current string part + + for i, char in enumerate(text): + if char == '\n': + newline_count += 1 + if newline_count == 3: + # Append the string part from start_index to current index (inclusive) + parts.append(text[start_index:i+1]) + # Reset newline count and update start_index for the next string part + newline_count = 0 + start_index = i+1 + + # Edge case: if the text does not end with a newline, we still need to append the last part + if start_index < len(text): + parts.append(text[start_index:]) + + return parts + +# Save some money and enter the character before translation +def getSpeaker(speaker, pbar, filename): + match speaker: + case 'ファイン': + return ['Fine', [0,0]] + case '': + return ['', [0,0]] + case _: + # Store Speaker + if speaker not in str(NAMESLIST): + response = translateGPT(speaker, 'Reply with only the '+ LANGUAGE +' translation of the NPC name.', False, pbar, filename) + response[0] = response[0].replace("'S", "'s") + speakerList = [speaker, response[0]] + NAMESLIST.append(speakerList) + return response + + # Find Speaker + else: + for i in range(len(NAMESLIST)): + if speaker == NAMESLIST[i][0]: + return [NAMESLIST[i][1],[0,0]] + + return [speaker,[0,0]] + +def subVars(jaString): + jaString = jaString.replace('\u3000', ' ') + + # Nested + count = 0 + nestedList = re.findall(r'[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]', jaString) + nestedList = set(nestedList) + if len(nestedList) != 0: + for icon in nestedList: + jaString = jaString.replace(icon, '[Nested_' + str(count) + ']') + count += 1 + + # Icons + count = 0 + iconList = re.findall(r'[\\]+[iIkKwWaA]+\[[0-9]+\]', jaString) + iconList = set(iconList) + if len(iconList) != 0: + for icon in iconList: + jaString = jaString.replace(icon, '[Ascii_' + str(count) + ']') + count += 1 + + # Colors + count = 0 + colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString) + colorList = set(colorList) + if len(colorList) != 0: + for color in colorList: + jaString = jaString.replace(color, '[Color_' + str(count) + ']') + count += 1 + + # Names + count = 0 + nameList = re.findall(r'[\\]+[nN]\[.+?\]+', jaString) + nameList = set(nameList) + if len(nameList) != 0: + for name in nameList: + jaString = jaString.replace(name, '[Noun_' + str(count) + ']') + count += 1 + + # Variables + count = 0 + varList = re.findall(r'[\\]+[vV]\[[0-9]+\]', jaString) + varList = set(varList) + if len(varList) != 0: + for var in varList: + jaString = jaString.replace(var, '[Var_' + str(count) + ']') + count += 1 + + # Formatting + count = 0 + formatList = re.findall(r'[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]', jaString) + formatList = set(formatList) + if len(formatList) != 0: + for var in formatList: + jaString = jaString.replace(var, '[FCode_' + str(count) + ']') + count += 1 + + # Put all lists in list and return + allList = [nestedList, iconList, colorList, nameList, varList, formatList] + return [jaString, allList] + +def resubVars(translatedText, allList): + # Fix Spacing and ChatGPT Nonsense + matchList = re.findall(r'\[\s?.+?\s?\]', translatedText) + if len(matchList) > 0: + for match in matchList: + text = match.strip() + translatedText = translatedText.replace(match, text) + + # Nested + count = 0 + if len(allList[0]) != 0: + for var in allList[0]: + translatedText = translatedText.replace('[Nested_' + str(count) + ']', var) + count += 1 + + # Icons + count = 0 + if len(allList[1]) != 0: + for var in allList[1]: + translatedText = translatedText.replace('[Ascii_' + str(count) + ']', var) + count += 1 + + # Colors + count = 0 + if len(allList[2]) != 0: + for var in allList[2]: + translatedText = translatedText.replace('[Color_' + str(count) + ']', var) + count += 1 + + # Names + count = 0 + if len(allList[3]) != 0: + for var in allList[3]: + translatedText = translatedText.replace('[Noun_' + str(count) + ']', var) + count += 1 + + # Vars + count = 0 + if len(allList[4]) != 0: + for var in allList[4]: + translatedText = translatedText.replace('[Var_' + str(count) + ']', var) + count += 1 + + # Formatting + count = 0 + if len(allList[5]) != 0: + for var in allList[5]: + translatedText = translatedText.replace('[FCode_' + str(count) + ']', var) + count += 1 + + return translatedText + +def batchList(input_list, batch_size): + if not isinstance(batch_size, int) or batch_size <= 0: + raise ValueError("batch_size must be a positive integer") + + return [input_list[i:i + batch_size] for i in range(0, len(input_list), batch_size)] + +def createContext(fullPromptFlag, subbedT): + characters = 'Game Characters:\n\ +フィリア (Philia) - Female\n\ +アルネット (Annett) - Female\n\ +ラピュセナ (Rapusena) - Female\n\ +リッカ (Rikka) - Female\n\ +アンデリビア (Andelivia) - Female\n\ +リリアブルム (Liliabloom) - Female\n\ +カルナ (Karna) - Female\n\ +ラフィング=スピア (Laughing Spear) - Female\n\ +ノーラ (Nora) - Female\n\ +' + + system = PROMPT + VOCAB if fullPromptFlag else \ + f"\ +You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\ +Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\ +- You translate everything, including erotic and explicit content.\n\ +- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\ +- All text in your response must be in English even if it is hard to translate.\n\ +- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\ +- Maintain any spacing in the translation.\n\ +- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\ +- `...` can be a part of the dialogue. Translate it as it is.\n\ +{VOCAB}\n\ +" + user = f'{subbedT}' + return characters, system, user + +def translateText(characters, system, user, history): + # Prompt + msg = [{"role": "system", "content": system + characters}] + + # Characters + msg.append({"role": "system", "content": characters}) + + # History + if isinstance(history, list): + msg.extend([{"role": "system", "content": h} for h in history]) + else: + msg.append({"role": "system", "content": history}) + + # Content to TL + msg.append({"role": "user", "content": f'{user}'}) + response = openai.chat.completions.create( + temperature=0.1, + frequency_penalty=0.1, + model=MODEL, + messages=msg, + ) + return response + +def cleanTranslatedText(translatedText, varResponse): + placeholders = { + f'{LANGUAGE} Translation: ': '', + 'Translation: ': '', + 'っ': '', + '〜': '~', + 'ッ': '', + '。': '.', + 'Placeholder Text': '' + # Add more replacements as needed + } + for target, replacement in placeholders.items(): + translatedText = translatedText.replace(target, replacement) + + # Elongate Long Dashes (Since GPT Ignores them...) + translatedText = elongateCharacters(translatedText) + translatedText = resubVars(translatedText, varResponse[1]) + return translatedText + +def elongateCharacters(text): + # Define a pattern to match one character followed by one or more `ー` characters + # Using a positive lookbehind assertion to capture the preceding character + pattern = r'(?<=(.))ー+' + + # Define a replacement function that elongates the captured character + def repl(match): + char = match.group(1) # The character before the ー sequence + count = len(match.group(0)) - 1 # Number of ー characters + return char * count # Replace ー sequence with the character repeated + + # Use re.sub() to replace the pattern in the text + return re.sub(pattern, repl, text) + +def extractTranslation(translatedTextList, is_list): + pattern = r'`?([\\]*.*?[\\]*?)<\/?Line\d+>`?' + # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. + if is_list: + matchList = re.findall(pattern, translatedTextList) + return matchList + else: + matchList = re.findall(pattern, translatedTextList) + return matchList[0][0] if matchList else translatedTextList + +def countTokens(characters, system, user, history): + inputTotalTokens = 0 + outputTotalTokens = 0 + enc = tiktoken.encoding_for_model('gpt-4') + + # Input + if isinstance(history, list): + for line in history: + inputTotalTokens += len(enc.encode(line)) + else: + inputTotalTokens += len(enc.encode(history)) + inputTotalTokens += len(enc.encode(system)) + inputTotalTokens += len(enc.encode(characters)) + inputTotalTokens += len(enc.encode(user)) + + # Output + outputTotalTokens += round(len(enc.encode(user))*2) + + return [inputTotalTokens, outputTotalTokens] + +def combineList(tlist, text): + if isinstance(text, list): + return [t for sublist in tlist for t in sublist] + return tlist[0] + +@retry(exceptions=Exception, tries=5, delay=5) +def translateGPT(text, history, fullPromptFlag, pbar, filename): + mismatch = False + totalTokens = [0, 0] + if isinstance(text, list): + tList = batchList(text, BATCHSIZE) + else: + tList = [text] + + for index, tItem in enumerate(tList): + # Before sending to translation, if we have a list of items, add the formatting + if isinstance(tItem, list): + payload = '\n'.join([f'`{item}`' for i, item in enumerate(tItem)]) + payload = re.sub(r'(<)(\/Line\d+>)', r'\1>Placeholder Text<\3', payload) + varResponse = subVars(payload) + subbedT = varResponse[0] + else: + varResponse = subVars(tItem) + subbedT = varResponse[0] + + # Things to Check before starting translation + if not re.search(r'[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+', subbedT): + continue + + # Create Message + characters, system, user = createContext(fullPromptFlag, subbedT) + + # Calculate Estimate + if ESTIMATE: + estimate = countTokens(characters, system, user, history) + totalTokens[0] += estimate[0] + totalTokens[1] += estimate[1] + continue + + # Translating + response = translateText(characters, system, user, history) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Formatting + translatedText = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedText, True) + if len(tItem) != len(extractedTranslations): + # Mismatch. Try Again + response = translateText(characters, system, user, history) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Formatting + translatedText = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedText, True) + if len(tItem) == len(extractedTranslations): + tList[index] = extractedTranslations + else: + MISMATCH.append(filename) + else: + tList[index] = extractedTranslations + + # Create History + history = tList[index] # Update history if we have a list + pbar.update(len(tList[index])) + + else: + # Ensure we're passing a single string to extractTranslation + extractedTranslations = extractTranslation(translatedText, False) + tList[index] = extractedTranslations + + finalList = combineList(tList, text) + return [finalList, totalTokens] diff --git a/modules/javascript.py b/modules/javascript.py index cfb2dfa..ce2eaa4 100644 --- a/modules/javascript.py +++ b/modules/javascript.py @@ -47,8 +47,8 @@ if 'gpt-3.5' in MODEL: OUTPUTAPICOST = .002 BATCHSIZE = 10 elif 'gpt-4' in MODEL: - INPUTAPICOST = .01 - OUTPUTAPICOST = .03 + INPUTAPICOST = .005 + OUTPUTAPICOST = .015 BATCHSIZE = 40 def handleJavascript(filename, estimate): @@ -400,7 +400,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list): diff --git a/modules/json.py b/modules/json.py index c2bb558..b41d899 100644 --- a/modules/json.py +++ b/modules/json.py @@ -487,7 +487,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list): diff --git a/modules/kansen.py b/modules/kansen.py index 050130b..ea5b937 100644 --- a/modules/kansen.py +++ b/modules/kansen.py @@ -579,7 +579,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list): diff --git a/modules/lune.py b/modules/lune.py index 99a8731..ca185a6 100644 --- a/modules/lune.py +++ b/modules/lune.py @@ -467,7 +467,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list): diff --git a/modules/main.py b/modules/main.py index 742e969..980cca3 100644 --- a/modules/main.py +++ b/modules/main.py @@ -30,6 +30,7 @@ from modules.nscript import handleNScript from modules.wolf import handleWOLF from modules.wolf2 import handleWOLF2 from modules.javascript import handleJavascript +from modules.irissoft import handleIris from modules.regex import handleRegex # For GPT4 rate limit will be hit if you have more than 1 thread. @@ -52,6 +53,7 @@ MODULES = [ ["Wolf", "json", handleWOLF], ["Wolf", "txt", handleWOLF2], ["Javascript", "js", handleJavascript], + ["Iris", "txt", handleIris], ["Regex", "txt", handleRegex], ] diff --git a/modules/nscript.py b/modules/nscript.py index eda3362..46a8469 100644 --- a/modules/nscript.py +++ b/modules/nscript.py @@ -47,8 +47,8 @@ if 'gpt-3.5' in MODEL: OUTPUTAPICOST = .002 BATCHSIZE = 10 elif 'gpt-4' in MODEL: - INPUTAPICOST = .01 - OUTPUTAPICOST = .03 + INPUTAPICOST = .005 + OUTPUTAPICOST = .015 BATCHSIZE = 40 def handleNScript(filename, estimate): @@ -585,7 +585,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list): diff --git a/modules/regex.py b/modules/regex.py index ea91114..f988019 100644 --- a/modules/regex.py +++ b/modules/regex.py @@ -47,8 +47,8 @@ if 'gpt-3.5' in MODEL: OUTPUTAPICOST = .002 BATCHSIZE = 10 elif 'gpt-4' in MODEL: - INPUTAPICOST = .01 - OUTPUTAPICOST = .03 + INPUTAPICOST = .005 + OUTPUTAPICOST = .015 BATCHSIZE = 40 def handleRegex(filename, estimate): @@ -162,161 +162,43 @@ def translateRegex(data, pbar, filename, translatedList): while i < len(data): voice = False speaker = '' - if '#MSGVOICE' in data[i]: - i += 1 - voice = True - voiceVar = data[i] - if '#MSG,' in data[i] or '#MSG\n' in data[i] or voice == True: - i += 1 - # Speaker - if re.search(r'^ ?([^#\/."、。*!!()\(\)\[\] \n]+)\n', data[i]) and len(data[i]) < 30: - match = re.search(r'(.*)', data[i]) - if match != None: - speaker = match.group(1) - if speaker[0] == '\u3000': - speaker = speaker[1:] - response = getSpeaker(speaker, pbar, filename) - speaker = response[0] - tokens[0] += response[1][0] - tokens[1] += response[1][1] - if translatedList != []: - speaker = speaker.replace(' ', '\u3000') - data[i] = f'\u3000{speaker}\n' - else: - speaker = '' - i += 1 - + if 'MSG' in data[i] or 'SYSTEM' in data[i]: # Lines - match = re.search(r'(.*)', data[i]) + match = re.search(r'.+ .+ MSG.*? .+ \u3000?(.+) .+ .+ .+ ', data[i]) + if match == None: + match = re.search(r'.+ .+ SYSTEM.*? .+ \u3000?(.+) .+ .+ .+ ', data[i]) if match != None and match.group(1) != '': + originalString = match.group(1) # Pass 1 if translatedList == []: # Grab Consecutive Strings - jaString = data[i] - if data[i] != '\n': - if data[i][0] == '\u3000': - jaString = data[i][1:] - currentGroup.append(jaString) - i += 1 - while data[i] != '\n': - jaString = data[i] - if data[i] != '\n': - jaString = data[i][1:] - currentGroup.append(jaString) - i += 1 + jaString = match.group(1) - # Join up 401 groups for better translation. - if len(currentGroup) > 0: - jaString = ''.join(currentGroup) - currentGroup = [] - - # Remove any textwrap - jaString = jaString.replace('\n', ' ') + # Remove any textwrap + jaString = jaString.replace('
', ' ') - # Temporarily convert spaces (For Textwrap Later) - jaString = jaString.replace('\u3000', ' ') - - # Add Speaker (If there is one) - if speaker != '': - jaString = f'{speaker}: {jaString}' - - # Add String - stringList.append(jaString.strip()) + # Add String + stringList.append(jaString.strip()) # Pass 2 else: - # Insert Strings - while data[i] != '\n': - data.pop(i) - # Get Text if translatedList: + # Grab and Pop translatedText = translatedList[0] translatedList.pop(0) + + # Set to None if empty list if len(translatedList) <= 0: translatedList = None - # Remove added speaker - translatedText = re.sub(r'^.+?:\s', '', translatedText) - # Textwrap translatedText = textwrap.fill(translatedText, width=WIDTH) - translatedText = translatedText.replace('\n', '\n\u3000') - - # Replace Whitespace and Commas - translatedText = translatedText.replace(', ', '、') - translatedText = translatedText.replace(',\u3000', '、') - translatedText = translatedText.replace(',', '、') - translatedText = translatedText.replace(' ', '\u3000') + translatedText = translatedText.replace('\n', '
') # Set Data - # Game crashes on more than 3 lines. Will need to create a new MSG for long translations - if translatedText.count('\n') > 2: - # Split List - translatedTextList = splitNewlines(translatedText) - - # MSG Voice - count = 0 - for text in translatedTextList: - if count != 0: - if voice == True: - #MSG for each item in the list - data.insert(i, f'#MSGVOICE,\n') - i += 1 - data.insert(i, f'{voiceVar}') - i += 1 - else: - data.insert(i, f'#MSG,\n') - i += 1 - if speaker: - data[i] = f'\u3000{speaker}\n' - i += 1 - if text[0] == '\u3000': - data.insert(i, f'{text}\n') - else: - data.insert(i, f'\u3000{text}\n') - i += 1 - count += 1 - if data[i] != '\n': - data.insert(i, '\n') - data[i] = f'\n{data[i]}' - else: - data.insert(i, f'\u3000{translatedText}\n') - i += 1 - if data[i] != '\n': - data[i] = f'\n{data[i]}' - - elif '#SELECT' in data[i] and translatedList == []: - regex = r'(.+?) +\d$' - i += 1 - match = re.search(regex, data[i]) - if match: - choiceList = [] - choiceList.append(match.group(1)) + data[i] = data[i].replace(originalString, translatedText) i += 1 - match = re.search(regex, data[i]) - while(match): - choiceList.append(match.group(1)) - i += 1 - match = re.search(regex, data[i]) - - # Translate - question = stringList[len(stringList) - 1] - response = translateGPT(choiceList, f'Previous text for context: {question}\n\nThis will be a dialogue option', True, pbar, filename) - tokens[0] += response[1][0] - tokens[1] += response[1][1] - choiceListTL = response[0] - - # Set Data - i = i - len(choiceListTL) - for j in range(len(choiceListTL)): - # Replace Whitespace and Commas - choiceListTL[j] = choiceListTL[j].replace(', ', '、') - choiceListTL[j] = choiceListTL[j].replace(',\u3000', '、') - choiceListTL[j] = choiceListTL[j].replace(',', '、') - choiceListTL[j] = choiceListTL[j].replace(' ', '\u3000') - data[i] = data[i].replace(choiceList[j], choiceListTL[j]) - i += 1 # Nothing relevant. Skip Line. else: @@ -347,27 +229,6 @@ def translateRegex(data, pbar, filename, translatedList): MISMATCH.append(filename) return tokens -def splitNewlines(text): - parts = [] - newline_count = 0 # Counts the number of newline characters encountered - start_index = 0 # Start index of the current string part - - for i, char in enumerate(text): - if char == '\n': - newline_count += 1 - if newline_count == 3: - # Append the string part from start_index to current index (inclusive) - parts.append(text[start_index:i+1]) - # Reset newline count and update start_index for the next string part - newline_count = 0 - start_index = i+1 - - # Edge case: if the text does not end with a newline, we still need to append the last part - if start_index < len(text): - parts.append(text[start_index:]) - - return parts - # Save some money and enter the character before translation def getSpeaker(speaker, pbar, filename): match speaker: @@ -609,7 +470,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list): diff --git a/modules/rpgmakerace.py b/modules/rpgmakerace.py index aa1c8b6..868134d 100644 --- a/modules/rpgmakerace.py +++ b/modules/rpgmakerace.py @@ -2239,7 +2239,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list): diff --git a/modules/rpgmakermvmz.py b/modules/rpgmakermvmz.py index 5f7fe9b..3a501a7 100644 --- a/modules/rpgmakermvmz.py +++ b/modules/rpgmakermvmz.py @@ -2139,7 +2139,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list): diff --git a/modules/sakuranbo.py b/modules/sakuranbo.py index 5ea2a22..e45c177 100644 --- a/modules/sakuranbo.py +++ b/modules/sakuranbo.py @@ -551,7 +551,7 @@ def translateGPT(t, history, fullPromptFlag): # If ESTIMATE is True just count this as an execution and return. if ESTIMATE: - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') historyRaw = "" if isinstance(history, list): for line in history: diff --git a/modules/tyrano.py b/modules/tyrano.py index 48192e5..e183cc1 100644 --- a/modules/tyrano.py +++ b/modules/tyrano.py @@ -52,8 +52,8 @@ if 'gpt-3.5' in MODEL: OUTPUTAPICOST = .002 BATCHSIZE = 10 elif 'gpt-4' in MODEL: - INPUTAPICOST = .01 - OUTPUTAPICOST = .03 + INPUTAPICOST = .005 + OUTPUTAPICOST = .015 BATCHSIZE = 40 def handleTyrano(filename, estimate): @@ -563,7 +563,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list): diff --git a/modules/wolf.py b/modules/wolf.py index e46f583..f7bc1b2 100644 --- a/modules/wolf.py +++ b/modules/wolf.py @@ -46,8 +46,8 @@ if 'gpt-3.5' in MODEL: BATCHSIZE = 10 FREQUENCY_PENALTY = 0.2 elif 'gpt-4' in MODEL: - INPUTAPICOST = .01 - OUTPUTAPICOST = .03 + INPUTAPICOST = .005 + OUTPUTAPICOST = .015 BATCHSIZE = 40 FREQUENCY_PENALTY = 0.1 @@ -1202,7 +1202,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list): diff --git a/modules/wolf2.py b/modules/wolf2.py index a5957a9..6527458 100644 --- a/modules/wolf2.py +++ b/modules/wolf2.py @@ -47,8 +47,8 @@ if 'gpt-3.5' in MODEL: OUTPUTAPICOST = .002 BATCHSIZE = 10 elif 'gpt-4' in MODEL: - INPUTAPICOST = .01 - OUTPUTAPICOST = .03 + INPUTAPICOST = .005 + OUTPUTAPICOST = .015 BATCHSIZE = 40 def handleWOLF2(filename, estimate): @@ -486,7 +486,7 @@ def extractTranslation(translatedTextList, is_list): def countTokens(characters, system, user, history): inputTotalTokens = 0 outputTotalTokens = 0 - enc = tiktoken.encoding_for_model(MODEL) + enc = tiktoken.encoding_for_model('gpt-4') # Input if isinstance(history, list):