import json import os from pathlib import Path import re import sys import textwrap import threading import time import traceback import tiktoken from colorama import Fore from dotenv import load_dotenv import openai from retry import retry from tqdm import tqdm #Globals load_dotenv() openai.api_base = os.getenv('proxy') openai.organization = os.getenv('org') openai.api_key = os.getenv('key') MODEL = os.getenv('model') TIMEOUT = int(os.getenv('timeout')) LANGUAGE=os.getenv('language').capitalize() INPUTAPICOST = .002 # Depends on the model https://openai.com/pricing OUTPUTAPICOST = .002 PROMPT = Path('prompt.txt').read_text(encoding='utf-8') THREADS = int(os.getenv('threads')) # Controls how many threads are working on a single file (May have to drop this) LOCK = threading.Lock() WIDTH = int(os.getenv('width')) LISTWIDTH = int(os.getenv('listWidth')) NOTEWIDTH = 50 MAXHISTORY = 10 ESTIMATE = '' totalTokens = [0, 0] NAMESLIST = [] #tqdm Globals BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' POSITION=0 LEAVE=False BRFLAG = False # If the game uses
instead FIXTEXTWRAP = True IGNORETLTEXT = False def handleLune(filename, estimate): global ESTIMATE, totalTokens ESTIMATE = estimate if estimate: start = time.time() translatedData = openFiles(filename) # Print Result end = time.time() tqdm.write(getResultString(translatedData, end - start, filename)) with LOCK: totalTokens[0] += translatedData[1][0] totalTokens[1] += translatedData[1][1] return getResultString(['', totalTokens, None], end - start, 'TOTAL') else: try: with open('translated/' + filename, 'w', encoding='UTF-8') as outFile: start = time.time() translatedData = openFiles(filename) # Print Result end = time.time() json.dump(translatedData[0], outFile, ensure_ascii=False) tqdm.write(getResultString(translatedData, end - start, filename)) with LOCK: totalTokens[0] += translatedData[1][0] totalTokens[1] += translatedData[1][1] except Exception as e: return 'Fail' return getResultString(['', totalTokens, None], end - start, 'TOTAL') def openFiles(filename): with open('files/' + filename, 'r', encoding='UTF-8-sig') as f: data = json.load(f) # Map Files if '.json' in filename: translatedData = parseJSON(data, filename) else: raise NameError(filename + ' Not Supported') return translatedData def getResultString(translatedData, translationTime, filename): # File Print String totalTokenstring =\ Fore.YELLOW +\ '[Input: ' + str(translatedData[1][0]) + ']'\ '[Output: ' + str(translatedData[1][1]) + ']'\ '[Cost: ${:,.4f}'.format((translatedData[1][0] * .001 * INPUTAPICOST) +\ (translatedData[1][1] * .001 * OUTPUTAPICOST)) + ']' timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' if translatedData[2] == None: # Success return filename + ': ' + totalTokenstring + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET else: # Fail try: raise translatedData[2] except Exception as e: traceback.print_exc() errorString = str(e) + Fore.RED return filename + ': ' + totalTokenstring + timeString + Fore.RED + u' \u2717 ' +\ errorString + Fore.RESET def parseJSON(data, filename): totalTokens = [0, 0] totalLines = 0 totalLines = len(data) global LOCK with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar: pbar.desc=filename pbar.total=totalLines try: result = translateJSON(data, pbar) totalTokens[0] += result[0] totalTokens[1] += result[1] except Exception as e: traceback.print_exc() return [data, totalTokens, e] return [data, totalTokens, None] def translateJSON(data, pbar): textHistory = [] maxHistory = MAXHISTORY tokens = [0, 0] speaker = 'None' for item in data: # Speaker if 'name' in item: if item['name'] not in [None, '-']: response = translateGPT(item['name'], 'Reply with only the '+ LANGUAGE +' translation of the NPC name', False) speaker = response[0] tokens[0] += response[1][0] tokens[1] += response[1][1] item['name'] = speaker else: speaker = 'None' # Text if 'message' in item: if item['message'] != None: jaString = item['message'] # Remove any textwrap if FIXTEXTWRAP == True: jaString = jaString.replace('\n', ' ') # Translate if jaString != '': response = translateGPT(f'{speaker} | {jaString}', textHistory, True) tokens[0] += response[1][0] tokens[1] += response[1][1] translatedText = response[0] textHistory.append('\"' + translatedText + '\"') else: translatedText = jaString textHistory.append('\"' + translatedText + '\"') # Remove added speaker translatedText = re.sub(r'^.+?\s\|\s?', '', translatedText) # Textwrap translatedText = textwrap.fill(translatedText, width=WIDTH) # Set Data item['message'] = translatedText # Keep textHistory list at length maxHistory if len(textHistory) > maxHistory: textHistory.pop(0) currentGroup = [] pbar.update(1) return tokens def subVars(jaString): jaString = jaString.replace('\u3000', ' ') # Icons count = 0 iconList = re.findall(r'[\\]+[iIkKwW]+\[[0-9]+\]', jaString) iconList = set(iconList) if len(iconList) != 0: for icon in iconList: jaString = jaString.replace(icon, '[Ascii_' + str(count) + ']') count += 1 # Colors count = 0 colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString) colorList = set(colorList) if len(colorList) != 0: for color in colorList: jaString = jaString.replace(color, '[Color_' + str(count) + ']') count += 1 # Names count = 0 nameList = re.findall(r'[\\]+[nN]\[.+?\]+', jaString) nameList = set(nameList) if len(nameList) != 0: for name in nameList: jaString = jaString.replace(name, '[N_' + str(count) + ']') count += 1 # Variables count = 0 varList = re.findall(r'[\\]+[vV]\[[0-9]+\]', jaString) varList = set(varList) if len(varList) != 0: for var in varList: jaString = jaString.replace(var, '[Var_' + str(count) + ']') count += 1 # Formatting count = 0 if '笑えるよね.' in jaString: print('t') formatList = re.findall(r'[\\]+CL', jaString) formatList = set(formatList) if len(formatList) != 0: for var in formatList: jaString = jaString.replace(var, '[FCode_' + str(count) + ']') count += 1 # Put all lists in list and return allList = [iconList, colorList, nameList, varList, formatList] return [jaString, allList] def resubVars(translatedText, allList): # Fix Spacing and ChatGPT Nonsense matchList = re.findall(r'\[\s?.+?\s?\]', translatedText) if len(matchList) > 0: for match in matchList: text = match.strip() translatedText = translatedText.replace(match, text) # Icons count = 0 if len(allList[0]) != 0: for var in allList[0]: translatedText = translatedText.replace('[Ascii_' + str(count) + ']', var) count += 1 # Colors count = 0 if len(allList[1]) != 0: for var in allList[1]: translatedText = translatedText.replace('[Color_' + str(count) + ']', var) count += 1 # Names count = 0 if len(allList[2]) != 0: for var in allList[2]: translatedText = translatedText.replace('[N_' + str(count) + ']', var) count += 1 # Vars count = 0 if len(allList[3]) != 0: for var in allList[3]: translatedText = translatedText.replace('[Var_' + str(count) + ']', var) count += 1 # Formatting count = 0 if len(allList[4]) != 0: for var in allList[4]: translatedText = translatedText.replace('[FCode_' + str(count) + ']', var) count += 1 # Remove Color Variables Spaces # if '\\c' in translatedText: # translatedText = re.sub(r'\s*(\\+c\[[1-9]+\])\s*', r' \1', translatedText) # translatedText = re.sub(r'\s*(\\+c\[0+\])', r'\1', translatedText) return translatedText @retry(exceptions=Exception, tries=5, delay=5) def translateGPT(t, history, fullPromptFlag): # If ESTIMATE is True just count this as an execution and return. if ESTIMATE: enc = tiktoken.encoding_for_model(MODEL) historyRaw = '' if isinstance(history, list): for line in history: historyRaw += line else: historyRaw = history inputTotalTokens = len(enc.encode(historyRaw)) + len(enc.encode(PROMPT)) outputTotalTokens = len(enc.encode(t)) * 2 # Estimating 2x the size of the original text totalTokens = [inputTotalTokens, outputTotalTokens] return (t, totalTokens) # Sub Vars varResponse = subVars(t) subbedT = varResponse[0] # If there isn't any Japanese in the text just skip if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴ]+|[\uFF00-\uFFEF]', subbedT): return(t, [0,0]) # Characters context = '```\ Game Characters:\ Character: ソル == Sol - Gender: Female\ Character: ェニ先生 == Eni-sensei - Gender: Female\ Character: 神泉 理央 == Kamiizumi Rio - Gender: Female\ Character: 吉祥寺 アリサ == Kisshouji Arisa - Gender: Female\ ```' # Prompt if fullPromptFlag: system = PROMPT user = 'Line to Translate = ' + subbedT else: system = 'Output ONLY the '+ LANGUAGE +' translation in the following format: `Translation: <'+ LANGUAGE.upper() +'_TRANSLATION>`' user = 'Line to Translate = ' + subbedT # Create Message List msg = [] msg.append({"role": "system", "content": system}) msg.append({"role": "user", "content": context}) if isinstance(history, list): for line in history: msg.append({"role": "user", "content": line}) else: msg.append({"role": "user", "content": history}) msg.append({"role": "user", "content": user}) response = openai.ChatCompletion.create( temperature=0.1, frequency_penalty=0.2, presence_penalty=0.2, model=MODEL, messages=msg, request_timeout=TIMEOUT, ) # Save Translated Text translatedText = response.choices[0].message.content totalTokens = [response.usage.prompt_tokens, response.usage.completion_tokens] # Resub Vars translatedText = resubVars(translatedText, varResponse[1]) # Remove Placeholder Text translatedText = translatedText.replace(LANGUAGE +' Translation: ', '') translatedText = translatedText.replace('Translation: ', '') translatedText = translatedText.replace('Line to Translate = ', '') translatedText = translatedText.replace('Translation = ', '') translatedText = translatedText.replace('Translate = ', '') translatedText = translatedText.replace(LANGUAGE +' Translation:', '') translatedText = translatedText.replace('Translation:', '') translatedText = translatedText.replace('Line to Translate =', '') translatedText = translatedText.replace('Translation =', '') translatedText = translatedText.replace('Translate =', '') translatedText = re.sub(r'Note:.*', '', translatedText) translatedText = translatedText.replace('っ', '') # Return Translation if len(translatedText) > 15 * len(t) or "I'm sorry, but I'm unable to assist with that translation" in translatedText: raise Exception else: return [translatedText, totalTokens]