diff --git a/modules/json.py b/modules/json.py index b5b3149..e1fc72d 100644 --- a/modules/json.py +++ b/modules/json.py @@ -1,4 +1,3 @@ -from concurrent.futures import ThreadPoolExecutor, as_completed import json import os from pathlib import Path @@ -21,16 +20,17 @@ load_dotenv() openai.organization = os.getenv('org') openai.api_key = os.getenv('key') -APICOST = .002 # Depends on the model https://openai.com/pricing +INPUTAPICOST = .002 # Depends on the model https://openai.com/pricing +OUTPUTAPICOST = .002 PROMPT = Path('prompt.txt').read_text(encoding='utf-8') -THREADS = 10 # For GPT4 rate limit will be hit if you have more than 1 thread. +THREADS = 10 # Controls how many threads are working on a single file (May have to drop this) LOCK = threading.Lock() -WIDTH = 90 -LISTWIDTH = 60 +WIDTH = 50 +LISTWIDTH = 90 +NOTEWIDTH = 50 MAXHISTORY = 10 ESTIMATE = '' -TOTALCOST = 0 -TOTALTOKENS = 0 +totalTokens = [0, 0] NAMESLIST = [] #tqdm Globals @@ -39,9 +39,10 @@ POSITION=0 LEAVE=False BRFLAG = False # If the game uses
instead FIXTEXTWRAP = True +IGNORETLTEXT = False def handleJSON(filename, estimate): - global ESTIMATE, TOKENS, TOTALTOKENS, TOTALCOST + global ESTIMATE, totalTokens ESTIMATE = estimate if estimate: @@ -52,10 +53,10 @@ def handleJSON(filename, estimate): end = time.time() tqdm.write(getResultString(translatedData, end - start, filename)) with LOCK: - TOTALCOST += translatedData[1] * .001 * APICOST - TOTALTOKENS += translatedData[1] + totalTokens[0] += translatedData[1][0] + totalTokens[1] += translatedData[1][1] - return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL') + return getResultString(['', totalTokens, None], end - start, 'TOTAL') else: try: @@ -68,20 +69,19 @@ def handleJSON(filename, estimate): json.dump(translatedData[0], outFile, ensure_ascii=False) tqdm.write(getResultString(translatedData, end - start, filename)) with LOCK: - TOTALCOST += translatedData[1] * .001 * APICOST - TOTALTOKENS += translatedData[1] + totalTokens[0] += translatedData[1][0] + totalTokens[1] += translatedData[1][1] except Exception as e: - traceback.print_exc() return 'Fail' - return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL') + return getResultString(['', totalTokens, None], end - start, 'TOTAL') def openFiles(filename): with open('files/' + filename, 'r', encoding='UTF-8-sig') as f: data = json.load(f) # Map Files - if 'test' in filename: + if '.json' in filename: translatedData = parseJSON(data, filename) else: @@ -91,25 +91,30 @@ def openFiles(filename): def getResultString(translatedData, translationTime, filename): # File Print String - tokenString = Fore.YELLOW + '[' + str(translatedData[1]) + \ - ' Tokens/${:,.4f}'.format(translatedData[1] * .001 * APICOST) + ']' + totalTokenstring =\ + Fore.YELLOW +\ + '[Input: ' + str(translatedData[1][0]) + ']'\ + '[Output: ' + str(translatedData[1][1]) + ']'\ + '[Cost: ${:,.4f}'.format((translatedData[1][0] * .001 * INPUTAPICOST) +\ + (translatedData[1][1] * .001 * OUTPUTAPICOST)) + ']' timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' if translatedData[2] == None: # Success - return filename + ': ' + tokenString + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET + return filename + ': ' + totalTokenstring + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET else: # Fail try: raise translatedData[2] except Exception as e: + traceback.print_exc() errorString = str(e) + Fore.RED - return filename + ': ' + tokenString + timeString + Fore.RED + u' \u2717 ' +\ + return filename + ': ' + totalTokenstring + timeString + Fore.RED + u' \u2717 ' +\ errorString + Fore.RESET def parseJSON(data, filename): - totalTokens = 0 + totalTokens = [0, 0] totalLines = 0 totalLines = len(data) global LOCK @@ -118,7 +123,9 @@ def parseJSON(data, filename): pbar.desc=filename pbar.total=totalLines try: - totalTokens += translateJSON(data, pbar) + result = translateJSON(data, pbar) + totalTokens[0] += result[0] + totalTokens[1] += result[1] except Exception as e: return [data, totalTokens, e] return [data, totalTokens, None] @@ -126,36 +133,55 @@ def parseJSON(data, filename): def translateJSON(data, pbar): textHistory = [] maxHistory = MAXHISTORY - tokens = 0 + tokens = [0, 0] + speaker = 'None' - for item in data: - # Remove any textwrap - if FIXTEXTWRAP == True: - jaString = item['value'] - jaString = jaString.replace('\n', '') + for item in data.items(): + # Speaker + if 'name' in item[1]: + if item[1]['name'] not in [None, '-']: + response = translateGPT(item[1]['name'], 'Reply with only the english translation of the NPC name', False) + speaker = response[0] + tokens[0] += response[1][0] + tokens[1] += response[1][1] + item[1]['name'] = speaker + else: + speaker = 'None' - # Translate - if jaString != '': - response = translateGPT(jaString, 'Past Translated Text: ' + '|\n\n'.join(textHistory), True) - tokens += response[1] - translatedText = response[0] - textHistory.append('\"' + translatedText + '\"') - else: - translatedText = jaString - textHistory.append('\"' + translatedText + '\"') + # Text + if 'text' in item[1]: + if item[1]['text'] != None: + jaString = item[1]['text'] - # Textwrap - translatedText = textwrap.fill(translatedText, width=WIDTH) - translatedText = translatedText.replace('\n', '@b') + # Remove any textwrap + if FIXTEXTWRAP == True: + jaString = jaString.replace('\n', '') - # Set Data - item['value'] = translatedText + # Translate + if jaString != '': + response = translateGPT(f'{speaker} | {jaString}', textHistory, True) + tokens[0] += response[1][0] + tokens[1] += response[1][1] + translatedText = response[0] + textHistory.append('\"' + translatedText + '\"') + else: + translatedText = jaString + textHistory.append('\"' + translatedText + '\"') - # Keep textHistory list at length maxHistory - if len(textHistory) > maxHistory: - textHistory.pop(0) - currentGroup = [] - pbar.update(1) + # Remove added speaker + translatedText = re.sub(r'^.+?\s\|\s?', '', translatedText) + + # Textwrap + translatedText = textwrap.fill(translatedText, width=WIDTH) + + # Set Data + item[1]['text'] = translatedText + + # Keep textHistory list at length maxHistory + if len(textHistory) > maxHistory: + textHistory.pop(0) + currentGroup = [] + pbar.update(1) return tokens @@ -164,11 +190,11 @@ def subVars(jaString): # Icons count = 0 - iconList = re.findall(r'[\\]+[iI]\[[0-9]+\]', jaString) + iconList = re.findall(r'[\\]+[iIkKwW]+\[[0-9]+\]', jaString) iconList = set(iconList) if len(iconList) != 0: for icon in iconList: - jaString = jaString.replace(icon, '') + jaString = jaString.replace(icon, '[Ascii_' + str(count) + ']') count += 1 # Colors @@ -177,16 +203,16 @@ def subVars(jaString): colorList = set(colorList) if len(colorList) != 0: for color in colorList: - jaString = jaString.replace(color, '') + jaString = jaString.replace(color, '[Color_' + str(count) + ']') count += 1 # Names count = 0 - nameList = re.findall(r'[\\]+[nN]\[[0-9]+\]', jaString) + nameList = re.findall(r'[\\]+[nN]\[.+?\]+', jaString) nameList = set(nameList) if len(nameList) != 0: for name in nameList: - jaString = jaString.replace(name, '') + jaString = jaString.replace(name, '[N_' + str(count) + ']') count += 1 # Variables @@ -195,16 +221,18 @@ def subVars(jaString): varList = set(varList) if len(varList) != 0: for var in varList: - jaString = jaString.replace(var, '') + jaString = jaString.replace(var, '[Var_' + str(count) + ']') count += 1 # Formatting count = 0 - formatList = re.findall(r'[\\]+[!.]', jaString) + if '笑えるよね.' in jaString: + print('t') + formatList = re.findall(r'[\\]+CL', jaString) formatList = set(formatList) if len(formatList) != 0: - for format in formatList: - jaString = jaString.replace(format, '') + for var in formatList: + jaString = jaString.replace(var, '[FCode_' + str(count) + ']') count += 1 # Put all lists in list and return @@ -213,7 +241,7 @@ def subVars(jaString): def resubVars(translatedText, allList): # Fix Spacing and ChatGPT Nonsense - matchList = re.findall(r'<\s?.+?\s?>', translatedText) + matchList = re.findall(r'\[\s?.+?\s?\]', translatedText) if len(matchList) > 0: for match in matchList: text = match.replace(' ', '') @@ -223,35 +251,35 @@ def resubVars(translatedText, allList): count = 0 if len(allList[0]) != 0: for var in allList[0]: - translatedText = translatedText.replace('', var) + translatedText = translatedText.replace('[Ascii_' + str(count) + ']', var) count += 1 # Colors count = 0 if len(allList[1]) != 0: for var in allList[1]: - translatedText = translatedText.replace('', var) + translatedText = translatedText.replace('[Color_' + str(count) + ']', var) count += 1 # Names count = 0 if len(allList[2]) != 0: for var in allList[2]: - translatedText = translatedText.replace('', var) + translatedText = translatedText.replace('[N_' + str(count) + ']', var) count += 1 # Vars count = 0 if len(allList[3]) != 0: for var in allList[3]: - translatedText = translatedText.replace('', var) + translatedText = translatedText.replace('[Var_' + str(count) + ']', var) count += 1 # Formatting count = 0 if len(allList[4]) != 0: for var in allList[4]: - translatedText = translatedText.replace('', var) + translatedText = translatedText.replace('[FCode_' + str(count) + ']', var) count += 1 # Remove Color Variables Spaces @@ -264,9 +292,18 @@ def resubVars(translatedText, allList): def translateGPT(t, history, fullPromptFlag): # If ESTIMATE is True just count this as an execution and return. if ESTIMATE: - enc = tiktoken.encoding_for_model("gpt-4") - tokens = len(enc.encode(t)) * 2 + len(enc.encode(str(history))) + len(enc.encode(PROMPT)) - return (t, tokens) + enc = tiktoken.encoding_for_model("gpt-3.5-turbo") + historyRaw = '' + if isinstance(history, list): + for line in history: + historyRaw += line + else: + historyRaw = history + + inputTotalTokens = len(enc.encode(historyRaw)) + len(enc.encode(PROMPT)) + outputTotalTokens = len(enc.encode(t)) * 2 # Estimating 2x the size of the original text + totalTokens = [inputTotalTokens, outputTotalTokens] + return (t, totalTokens) # Sub Vars varResponse = subVars(t) @@ -274,16 +311,15 @@ def translateGPT(t, history, fullPromptFlag): # If there isn't any Japanese in the text just skip if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴ]+|[\uFF00-\uFFEF]', subbedT): - return(t, 0) + return(t, [0,0]) # Characters context = '```\ Game Characters:\ - Character: 池ノ上 拓海 == Ikenoue Takumi - Gender: Male\ - Character: 福永 こはる == Fukunaga Koharu - Gender: Female\ + Character: ソル == Sol - Gender: Female\ + Character: ェニ先生 == Eni-sensei - Gender: Female\ Character: 神泉 理央 == Kamiizumi Rio - Gender: Female\ Character: 吉祥寺 アリサ == Kisshouji Arisa - Gender: Female\ - Character: 久我 友里子 == Kuga Yuriko - Gender: Female\ ```' # Prompt @@ -309,14 +345,14 @@ def translateGPT(t, history, fullPromptFlag): temperature=0.1, frequency_penalty=0.2, presence_penalty=0.2, - model="gpt-3.5-turbo", + model="gpt-3.5-turbo-1106", messages=msg, request_timeout=30, ) # Save Translated Text translatedText = response.choices[0].message.content - tokens = response.usage.total_tokens + totalTokens = [response.usage.prompt_tokens, response.usage.completion_tokens] # Resub Vars translatedText = resubVars(translatedText, varResponse[1]) @@ -339,4 +375,4 @@ def translateGPT(t, history, fullPromptFlag): if len(translatedText) > 15 * len(t) or "I'm sorry, but I'm unable to assist with that translation" in translatedText: raise Exception else: - return [translatedText, tokens] \ No newline at end of file + return [translatedText, totalTokens] \ No newline at end of file