diff --git a/.gitignore b/.gitignore index b93dd79..2ca655e 100644 --- a/.gitignore +++ b/.gitignore @@ -5,4 +5,5 @@ *.txt !requirements.txt *.csv +*.ks __pycache__ \ No newline at end of file diff --git a/modules/alltext.py b/modules/alltext.py new file mode 100644 index 0000000..424d831 --- /dev/null +++ b/modules/alltext.py @@ -0,0 +1,92 @@ +from concurrent.futures import ThreadPoolExecutor, as_completed +import json +import os +from pathlib import Path +import re +import sys +import textwrap +import threading +import time +import traceback +import tiktoken + +from colorama import Fore +from dotenv import load_dotenv +import openai +from retry import retry +from tqdm import tqdm + +#Globals +load_dotenv() +openai.organization = os.getenv('org') +openai.api_key = os.getenv('key') + +APICOST = .002 # Depends on the model https://openai.com/pricing +PROMPT = Path('prompt.txt').read_text(encoding='utf-8') +THREADS = 10 # For GPT4 rate limit will be hit if you have more than 1 thread. +LOCK = threading.Lock() +WIDTH = 60 +LISTWIDTH = 60 +MAXHISTORY = 10 +ESTIMATE = '' +TOTALCOST = 0 +TOKENS = 0 +TOTALTOKENS = 0 +NAMESLIST = [] + +#tqdm Globals +BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' +POSITION=0 +LEAVE=False + +def handleAllText(filename, estimate): + global ESTIMATE, TOKENS, TOTALTOKENS, TOTALCOST + ESTIMATE = estimate + + with open('translated/' + filename, 'w+t', newline='', encoding='utf-16-le') as writeFile: + start = time.time() + translatedData = openFiles(filename, writeFile) + + if estimate: + # Print Result + end = time.time() + tqdm.write(getResultString(['', TOKENS, None], end - start, filename)) + TOTALCOST += TOKENS * .001 * APICOST + TOTALTOKENS += TOKENS + TOKENS = 0 + os.remove('translated/' + filename) + + else: + # Print Result + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filename)) + TOTALCOST += translatedData[1] * .001 * APICOST + TOTALTOKENS += translatedData[1] + + return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL') + +def openFiles(filename, writeFile): + with open('files/' + filename, 'r', encoding='utf-16-le') as readFile, writeFile: + translatedData = parseText(readFile, writeFile, filename) + + return translatedData + +def getResultString(translatedData, translationTime, filename): + # File Print String + tokenString = Fore.YELLOW + '[' + str(translatedData[1]) + \ + ' Tokens/${:,.4f}'.format(translatedData[1] * .001 * APICOST) + ']' + timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' + + if translatedData[2] == None: + # Success + return filename + ': ' + tokenString + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET + + else: + # Fail + try: + raise translatedData[2] + except Exception as e: + errorString = str(e) + '|' + translatedData[3] + Fore.RED + return filename + ': ' + tokenString + timeString + Fore.RED + u' \u2717 ' +\ + errorString + Fore.RESET + \ No newline at end of file diff --git a/modules/main.py b/modules/main.py index 4e52e88..0206107 100644 --- a/modules/main.py +++ b/modules/main.py @@ -9,7 +9,7 @@ from modules.rpgmakerace import handleACE from modules.csvtl import handleCSV from modules.textfile import handleTextfile -THREADS = 20 # For GPT4 rate limit will be hit if you have more than 1 thread. +THREADS = 10 # For GPT4 rate limit will be hit if you have more than 1 thread. # Info Message print(Fore.LIGHTYELLOW_EX + "WARNING: Once a translation starts do not close it unless you want to lose your\ diff --git a/modules/rpgmakermvmz.py b/modules/rpgmakermvmz.py index e263694..3b3eb4b 100644 --- a/modules/rpgmakermvmz.py +++ b/modules/rpgmakermvmz.py @@ -25,7 +25,7 @@ APICOST = .002 # Depends on the model https://openai.com/pricing PROMPT = Path('prompt.txt').read_text(encoding='utf-8') THREADS = 10 # For GPT4 rate limit will be hit if you have more than 1 thread. LOCK = threading.Lock() -WIDTH = 40 +WIDTH = 60 LISTWIDTH = 60 MAXHISTORY = 10 ESTIMATE = '' @@ -54,7 +54,9 @@ CODE324 = False CODE111 = False CODE408 = False CODE108 = False -NAMES = True # Output a list of all the character names found +NAMES = True # Output a list of all the character names found +BRFLAG = False # If the game uses
instead +FIXTEXTWRAP = False def handleMVMZ(filename, estimate): global ESTIMATE, TOKENS, TOTALTOKENS, TOTALCOST @@ -189,7 +191,7 @@ def parseMap(data, filename): if 'Map' in filename: response = translateGPT(data['displayName'], 'Reply with only the english translation of the RPG location name', False) totalTokens += response[1] - data['displayName'] = response[0].strip('.\"') + data['displayName'] = response[0].replace('\"', '') # Get total for progress bar for event in events: @@ -401,14 +403,14 @@ def searchThings(name, pbar): # Set Data if 'name' in name: - name['name'] = nameResponse[0].strip('\"') + name['name'] = nameResponse[0].replace('\"', '') if 'description' in name: description = descriptionResponse[0] # Remove Textwrap description = description.replace('\n', ' ') description = textwrap.fill(descriptionResponse[0], LISTWIDTH) - name['description'] = description.strip('\"') + name['description'] = description.replace('\"', '') pbar.update(1) return tokens @@ -461,19 +463,19 @@ def searchNames(name, pbar, context): responseList[i] = responseList[i][0] # Set Data - name['name'] = responseList[0].strip('.\"') + name['name'] = responseList[0].replace('\"', '') if 'Actors' in context: translatedText = textwrap.fill(responseList[1], LISTWIDTH) - name['profile'] = translatedText.strip('\"') + name['profile'] = translatedText.replace('\"', '') translatedText = textwrap.fill(responseList[2], LISTWIDTH) - name['nickname'] = translatedText.strip('\"') + name['nickname'] = translatedText.replace('\"', '') if '<特徴1:' in name['note']: tokens += translateNote(name, r'<特徴1:([^>]*)>') if 'Armors' in context or 'Weapons' in context: translatedText = textwrap.fill(responseList[1], LISTWIDTH) if 'description' in name: - name['description'] = translatedText.strip('\"') + name['description'] = translatedText.replace('\"', '') if '\n([\s\S]*?)\n') pbar.update(1) @@ -581,10 +583,30 @@ def searchCodes(page, pbar): codeList[j]['code'] = code # Remove nametag from final string - finalJAString = finalJAString.replace(match[0], '') + finalJAString = finalJAString.replace(match[0], '') + elif '\\nc' in finalJAString: + matchList = re.findall(r'(\\+nc<(.*?)>)(.+)?', finalJAString) + if len(matchList) != 0: + # Translate Speaker + response = translateGPT(matchList[0][1], 'Reply with only the english translation of the NPC name', True) + tokens += response[1] + speaker = response[0].strip('.') + nametag = matchList[0][0].replace(matchList[0][1], speaker) + finalJAString = finalJAString.replace(matchList[0][0], '') + + # Set dialogue + codeList[j]['parameters'][0] = matchList[0][2] + codeList[j]['code'] = 401 + + # Remove nametag from final string + finalJAString = finalJAString.replace(nametag, '') # Remove any textwrap - finalJAString = re.sub(r'\n', ' ', finalJAString) + if FIXTEXTWRAP == True: + finalJAString = re.sub(r'\n', ' ', finalJAString) + finalJAString = finalJAString.replace('
', ' ') + + # Remove Extra Stuff finalJAString = finalJAString.replace('゙', '') finalJAString = finalJAString.replace('。', '.') finalJAString = finalJAString.replace('・', '.') @@ -628,11 +650,10 @@ def searchCodes(page, pbar): translatedText = finalJAString # Textwrap - textList = re.findall(r'(^.+?)(\s.+)$', translatedText) # Strip out vars - if len(textList) > 0: - translatedText = textList[0][0] + textwrap.fill(textList[0][1], width=WIDTH) - else: + if '\n' not in translatedText and '
' not in translatedText: translatedText = textwrap.fill(translatedText, width=WIDTH) + if BRFLAG == True: + translatedText = translatedText.replace('\n', '
') # Add Beginning Text translatedText = startString + translatedText @@ -660,8 +681,8 @@ def searchCodes(page, pbar): if codeList[i]['code'] == 122 and CODE122 == True: # This is going to be the var being set. (IMPORTANT) varNum = codeList[i]['parameters'][0] - if varNum != 328: - continue + # if varNum != 328: + # continue jaString = codeList[i]['parameters'][4] if type(jaString) != str: @@ -671,6 +692,10 @@ def searchCodes(page, pbar): if '■' in jaString or '_' in jaString: continue + # Definitely don't want to mess with files + if '\"' not in jaString: + continue + # Need to remove outside code and put it back later matchList = re.findall(r"[\'\"\`](.*?)[\'\"\`]", jaString) @@ -691,10 +716,10 @@ def searchCodes(page, pbar): translatedText = translatedText.replace(char, '') # Textwrap - translatedText = textwrap.fill(translatedText, width=30) - translatedText = translatedText.replace('\n', '\\n') - translatedText = translatedText.replace('\'', '\\\'') - translatedText = '\'' + translatedText + '\'' + # translatedText = textwrap.fill(translatedText, width=30) + # translatedText = translatedText.replace('\n', '\\n') + # translatedText = translatedText.replace('\'', '\\\'') + translatedText = '\"' + translatedText + '\"' # Set Data codeList[i]['parameters'][4] = translatedText @@ -827,7 +852,7 @@ def searchCodes(page, pbar): NAMESLIST.append(speaker) ## Event Code: 355 or 655 Scripts [Optional] - if (codeList[i]['code'] == 355) and CODE355655 == True: + if (codeList[i]['code'] == 355 or codeList[i]['code'] == 655) and CODE355655 == True: jaString = codeList[i]['parameters'][0] # If there isn't any Japanese in the text just skip @@ -835,11 +860,11 @@ def searchCodes(page, pbar): continue # Want to translate this script - if codeList[i]['code'] == 355 and 'BattleManager._logWindow.addText' not in jaString: + if codeList[i]['code'] == 355 and '$gameSystem.addLog' not in jaString: continue # Don't want to touch certain scripts - if codeList[i]['code'] == 655 and '.' in jaString: + if codeList[i]['code'] == 655 and '$gameSystem.addLog' not in jaString: continue # Need to remove outside code and put it back later @@ -1198,20 +1223,20 @@ def searchSS(state, pbar): # Set Data if 'name' in state: - state['name'] = nameResponse[0].strip('\"') + state['name'] = nameResponse[0].replace('\"', '') if 'description' in state: # Textwrap translatedText = descriptionResponse[0] translatedText = textwrap.fill(translatedText, width=LISTWIDTH) - state['description'] = translatedText.strip('\"') + state['description'] = translatedText.replace('\"', '') if 'message1' in state: - state['message1'] = message1Response[0].strip('\"').replace('Taro', '') + state['message1'] = message1Response[0].replace('\"', '').replace('Taro', '') if 'message2' in state: - state['message2'] = message2Response[0].strip('\"').replace('Taro', '') + state['message2'] = message2Response[0].replace('\"', '').replace('Taro', '') if 'message3' in state: - state['message3'] = message3Response[0].strip('\"').replace('Taro', '') + state['message3'] = message3Response[0].replace('\"', '').replace('Taro', '') if 'message4' in state: - state['message4'] = message4Response[0].strip('\"').replace('Taro', '') + state['message4'] = message4Response[0].replace('\"', '').replace('Taro', '') pbar.update(1) return tokens @@ -1234,35 +1259,35 @@ def searchSystem(data, pbar): if termList[i] is not None: response = translateGPT(termList[i], context, False) tokens += response[1] - termList[i] = response[0].strip('.\"') + termList[i] = response[0].replace('\"', '') pbar.update(1) # Armor Types for i in range(len(data['armorTypes'])): response = translateGPT(data['armorTypes'][i], 'Reply with only the english translation of the armor type', False) tokens += response[1] - data['armorTypes'][i] = response[0].strip('.\"') + data['armorTypes'][i] = response[0].replace('\"', '') pbar.update(1) # Skill Types for i in range(len(data['skillTypes'])): response = translateGPT(data['skillTypes'][i], 'Reply with only the english translation', False) tokens += response[1] - data['skillTypes'][i] = response[0].strip('.\"') + data['skillTypes'][i] = response[0].replace('\"', '') pbar.update(1) # Equip Types for i in range(len(data['equipTypes'])): response = translateGPT(data['equipTypes'][i], 'Reply with only the english translation of the equipment type. No disclaimers.', False) tokens += response[1] - data['equipTypes'][i] = response[0].strip('.\"') + data['equipTypes'][i] = response[0].replace('\"', '') pbar.update(1) # Variables (Optional ususally) for i in range(len(data['variables'])): response = translateGPT(data['variables'][i], 'Reply with only the english translation of the title', False) tokens += response[1] - data['variables'][i] = response[0].strip('.\"') + data['variables'][i] = response[0].replace('\"', '') pbar.update(1) # Messages @@ -1312,9 +1337,9 @@ def resubVars(translatedText, varList): count += 1 # Remove Color Variables Spaces - if '\\c' in translatedText: - translatedText = re.sub(r'\s*(\\+c\[[1-9]+\])\s*', r' \1', translatedText) - translatedText = re.sub(r'\s*(\\+c\[0+\])', r'\1', translatedText) + # if '\\c' in translatedText: + # translatedText = re.sub(r'\s*(\\+c\[[1-9]+\])\s*', r' \1', translatedText) + # translatedText = re.sub(r'\s*(\\+c\[0+\])', r'\1', translatedText) return translatedText @retry(exceptions=Exception, tries=5, delay=5) @@ -1336,7 +1361,7 @@ def translateGPT(t, history, fullPromptFlag): return(t, 0) """Translate text using GPT""" - context = 'Eroge Names Context: ボク == Boku | Male' + context = 'Eroge Names Context: 圭 == Kei | Female, レヴァンティア == Levantia | Female, カシューナッツ == Cashew Nut | Male, レイナ == Reina | Female, ピンクキャンディー == Pink Candy | Female, ルーシャーベット == Lusha Sorbet | Female, ブラックマッスル == Black Muscle | Male, ミスターヒプノシス == Mr. Hypnosis | Male, リリス == Lilith | Female, 侵食体 == Invasoid | Female' if fullPromptFlag: system = PROMPT user = 'Line to Translate: ' + subbedT