Changing some filenames around and some tweaks to a few scripts

This commit is contained in:
Dazed 2023-10-02 10:17:37 -05:00
parent b27d3da02a
commit b56f387b0e
6 changed files with 287 additions and 16 deletions

View file

@ -73,7 +73,7 @@ A breakdown of what all the different files are, this is important.
* rpgmakermvmz.py - Translation Script for the RPGMaker MV/MZ Engine.
* rpgmakerace.py - Translation Script for the RPGMaker ACE Engine. (Requires rvpacker to unpack rvdata files)
* csvtl.py - Translation Script for CSV Files. Requires at least 2 columns to work.
* textfile.py - Translation Script for Other game engines. (More of a custom script I change depending on the game)
* TXT.py - Translation Script for Other game engines. (More of a custom script I change depending on the game)
* .env.example - An example env file. This gets renamed to .env and holds your PRIVATE API and Organization key. Do not EVER upload this information.
* RPGMakerEventCodes.info - Information on the various types of event codes in RPGMaker. More on this later.
* prompt.example - Holds an example prompt ChatGPT uses to determine what to do with text you give it. Change this as you please.

255
modules/json.py Normal file
View file

@ -0,0 +1,255 @@
from concurrent.futures import ThreadPoolExecutor, as_completed
import json
import os
from pathlib import Path
import re
import sys
import textwrap
import threading
import time
import traceback
import tiktoken
from colorama import Fore
from dotenv import load_dotenv
import openai
from retry import retry
from tqdm import tqdm
#Globals
load_dotenv()
openai.organization = os.getenv('org')
openai.api_key = os.getenv('key')
APICOST = .002 # Depends on the model https://openai.com/pricing
PROMPT = Path('prompt.txt').read_text(encoding='utf-8')
THREADS = 10 # For GPT4 rate limit will be hit if you have more than 1 thread.
LOCK = threading.Lock()
WIDTH = 90
LISTWIDTH = 60
MAXHISTORY = 10
ESTIMATE = ''
TOTALCOST = 0
TOTALTOKENS = 0
NAMESLIST = []
#tqdm Globals
BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}'
POSITION=0
LEAVE=False
BRFLAG = False # If the game uses <br> instead
FIXTEXTWRAP = True
def handleJSON(filename, estimate):
global ESTIMATE, TOKENS, TOTALTOKENS, TOTALCOST
ESTIMATE = estimate
if estimate:
start = time.time()
translatedData = openFiles(filename)
# Print Result
end = time.time()
tqdm.write(getResultString(translatedData, end - start, filename))
with LOCK:
TOTALCOST += translatedData[1] * .001 * APICOST
TOTALTOKENS += translatedData[1]
return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL')
else:
try:
with open('translated/' + filename, 'w', encoding='UTF-8') as outFile:
start = time.time()
translatedData = openFiles(filename)
# Print Result
end = time.time()
json.dump(translatedData[0], outFile, ensure_ascii=False)
tqdm.write(getResultString(translatedData, end - start, filename))
with LOCK:
TOTALCOST += translatedData[1] * .001 * APICOST
TOTALTOKENS += translatedData[1]
except Exception as e:
traceback.print_exc()
return 'Fail'
return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL')
def openFiles(filename):
with open('files/' + filename, 'r', encoding='UTF-8-sig') as f:
data = json.load(f)
# Map Files
if 'script' in filename:
translatedData = parseJSON(data, filename)
else:
raise NameError(filename + ' Not Supported')
return translatedData
def getResultString(translatedData, translationTime, filename):
# File Print String
tokenString = Fore.YELLOW + '[' + str(translatedData[1]) + \
' Tokens/${:,.4f}'.format(translatedData[1] * .001 * APICOST) + ']'
timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]'
if translatedData[2] == None:
# Success
return filename + ': ' + tokenString + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET
else:
# Fail
try:
raise translatedData[2]
except Exception as e:
errorString = str(e) + Fore.RED
return filename + ': ' + tokenString + timeString + Fore.RED + u' \u2717 ' +\
errorString + Fore.RESET
def parseJSON(data, filename):
totalTokens = 0
totalLines = 0
totalLines = len(data)
global LOCK
with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar:
pbar.desc=filename
pbar.total=totalLines
try:
totalTokens += translateJSON(data, pbar)
except Exception as e:
return [data, totalTokens, e]
return [data, totalTokens, None]
def translateJSON(data, pbar):
textHistory = []
maxHistory = MAXHISTORY
tokens = 0
for key, value in data.items():
# Remove any textwrap
if FIXTEXTWRAP == True:
value = re.sub(r'@b', ' ', value)
# Translate
if value == '':
response = translateGPT(key, 'Past Translated Text: ' + '|\n\n'.join(textHistory), True)
tokens += response[1]
translatedText = response[0]
textHistory.append('\"' + translatedText + '\"')
else:
translatedText = value
textHistory.append('\"' + translatedText + '\"')
# Textwrap
translatedText = textwrap.fill(translatedText, width=WIDTH)
translatedText = translatedText.replace('\n', '@b')
# Set Data
data[key] = translatedText
# Keep textHistory list at length maxHistory
if len(textHistory) > maxHistory:
textHistory.pop(0)
currentGroup = []
pbar.update(1)
return tokens
def subVars(jaString):
jaString = jaString.replace('\u3000', ' ')
varRegex = r'[\\]+[\w.\\\s]+?\[.+?\]]?|[\\]+[\w.\\]+?\<.+?\>\>?|[\\]+[#{}<>.]'
count = 0
varList = re.findall(varRegex, jaString)
varList = set(varList)
if len(varList) != 0:
for var in varList:
jaString = jaString.replace(var, '@' + str(count) + '')
count += 1
return [jaString, varList]
def resubVars(translatedText, varList):
count = 0
# Fix Spacing and ChatGPT Nonsense
matchList = re.findall(r'@\s?[0-9]+?', translatedText)
if len(matchList) > 0:
for match in matchList:
text = match.replace(' ', '')
translatedText = translatedText.replace(match, text)
if len(varList) != 0:
for var in varList:
translatedText = translatedText.replace('@' + str(count) + '', var)
count += 1
# Remove Color Variables Spaces
# if '\\c' in translatedText:
# translatedText = re.sub(r'\s*(\\+c\[[1-9]+\])\s*', r' \1', translatedText)
# translatedText = re.sub(r'\s*(\\+c\[0+\])', r'\1', translatedText)
return translatedText
@retry(exceptions=Exception, tries=5, delay=5)
def translateGPT(t, history, fullPromptFlag):
# If ESTIMATE is True just count this as an execution and return.
if ESTIMATE:
enc = tiktoken.encoding_for_model("gpt-3.5-turbo")
tokens = len(enc.encode(t)) * 2 + len(enc.encode(history)) + len(enc.encode(PROMPT))
return (t, tokens)
# Sub Vars
varResponse = subVars(t)
subbedT = varResponse[0]
# If there isn't any Japanese in the text just skip
if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴー]+', subbedT):
return(t, 0)
"""Translate text using GPT"""
context = 'Eroge Names Context: Name: 稲盛 楓 == Inamori Kaede\nNicknames: かえちゃん == Kae-chan or かえねぇ == Kae-nee\nGender: Female,\nName: 稲盛 真守 == Inamori Mamoru\nNicknames: まーくん == Maa-kun\nGender: Male,\nName: 蓮見 雄次郎 == Hasumi Yujiro\nGender: Male,\nName: 桐谷 拓馬 == Kiriya Takuma\nNicknames: たっくん == Tak-kun\nGender: Male,\nName: 稲盛 美玖 == Inamori Yuki\nGender: Female,\nName: 奥さん == Missus\nGender: Female'
if fullPromptFlag:
system = PROMPT
user = 'Line to Translate: ' + subbedT
else:
system = 'You are an expert translator who translates everything to English. Reply with only the English Translation of the text.'
user = 'Line to Translate: ' + subbedT
response = openai.ChatCompletion.create(
temperature=0,
frequency_penalty=0.2,
presence_penalty=0.2,
model="gpt-3.5-turbo",
messages=[
{"role": "system", "content": system},
{"role": "user", "content": context},
{"role": "user", "content": history},
{"role": "user", "content": user}
],
request_timeout=30,
)
# Save Translated Text
translatedText = response.choices[0].message.content
tokens = response.usage.total_tokens
# Resub Vars
translatedText = resubVars(translatedText, varResponse[1])
# Remove Placeholder Text
translatedText = translatedText.replace('English Translation: ', '')
translatedText = translatedText.replace('Translation: ', '')
translatedText = translatedText.replace('Line to Translate: ', '')
translatedText = translatedText.replace('English Translation:', '')
translatedText = translatedText.replace('Translation:', '')
translatedText = translatedText.replace('Line to Translate:', '')
translatedText = re.sub(r'\n\nPast Translated Text:.*', '', translatedText, 0, re.DOTALL)
translatedText = re.sub(r'Note:.*', '', translatedText)
# Return Translation
if len(translatedText) > 15 * len(t) or "I'm sorry, but I'm unable to assist with that translation" in translatedText:
return [t, response.usage.total_tokens]
else:
return [translatedText, tokens]

View file

@ -6,9 +6,10 @@ import os
from modules.rpgmakermvmz import handleMVMZ
from modules.rpgmakerace import handleACE
from modules.csvtl import handleCSV
from modules.textfile import handleTextfile
from modules.csv import handleCSV
from modules.txt import handleTXT
from modules.tyrano import handleTyrano
from modules.json import handleJSON
THREADS = 10 # For GPT4 rate limit will be hit if you have more than 1 thread.
@ -30,7 +31,7 @@ def main():
totalCost = 0
version = ''
while version == '':
version = input('Select the RPGMaker Version:\n\n1. MV/MZ\n2. ACE\n3. CSV (From Translator++)\n4. Text (Custom)\n5. Tyrano\n')
version = input('Select the RPGMaker Version:\n\n1. MV/MZ\n2. ACE\n3. CSV (From Translator++)\n4. Text (Custom)\n5. Tyrano\n6. JSON\n')
match version:
case '1':
# Open File (Threads)
@ -75,7 +76,7 @@ def main():
case '4':
# Open File (Threads)
with ThreadPoolExecutor(max_workers=THREADS) as executor:
futures = [executor.submit(handleTextfile, filename, estimate) \
futures = [executor.submit(handleTXT, filename, estimate) \
for filename in os.listdir("files") if filename.endswith('txt')]
for future in as_completed(futures):
@ -100,6 +101,20 @@ def main():
tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno)
print(Fore.RED + str(e) + '|' + tracebackLineNo + Fore.RESET)
case '6':
# Open File (Threads)
with ThreadPoolExecutor(max_workers=THREADS) as executor:
futures = [executor.submit(handleJSON, filename, estimate) \
for filename in os.listdir("files") if filename.endswith('json')]
for future in as_completed(futures):
try:
totalCost = future.result()
except Exception as e:
tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno)
print(Fore.RED + str(e) + '|' + tracebackLineNo + Fore.RESET)
case _:
version = ''
@ -110,7 +125,7 @@ def main():
# Prevent immediately closing of CLI
print(totalCost)
input('Done! Press Enter to close.')
# input('Done! Press Enter to close.')
def deleteFolderFiles(folderPath):
for filename in os.listdir(folderPath):

View file

@ -54,9 +54,10 @@ CODE324 = False
CODE111 = False
CODE408 = False
CODE108 = False
NAMES = False # Output a list of all the character names found
NAMES = False # Output a list of all the character names found
BRFLAG = False # If the game uses <br> instead
FIXTEXTWRAP = True
FIXTEXTWRAP = True # Adjust wordwrap of text (IGNORETLTEXT must be False)
IGNORETLTEXT = True # Leave this False if you need to adjust the wordwrap
def handleACE(filename, estimate):
global ESTIMATE, TOTALTOKENS, TOTALCOST
@ -193,12 +194,6 @@ def parseMap(data, filename):
events = data['events']
global LOCK
# Translate displayName for Map files
if 'Map' in filename:
response = translateGPT(data['displayName'], 'Reply with only the english translation of the RPG location name', False)
totalTokens += response[1]
data['displayName'] = response[0].replace('\"', '')
# Get total for progress bar
for key in events:
if key is not None:
@ -523,6 +518,12 @@ def searchCodes(page, pbar):
jaString = codeList[i]['p'][0]
firstJAString = jaString
# If there isn't any Japanese in the text just skip
if IGNORETLTEXT == True:
if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴー]+', jaString):
textHistory.append('\"' + jaString + '\"')
continue
# Using this to keep track of 401's in a row. Throws IndexError at EndOfList (Expected Behavior)
currentGroup.append(jaString)
@ -1362,7 +1363,7 @@ def translateGPT(t, history, fullPromptFlag):
return(t, 0)
"""Translate text using GPT"""
context = 'Eroge Names Context: アサギ == Asagi | Female, ウィップ == Whip | Female, ウラ == Ura | Female, ブレイド == Blade | Female'
context = 'Eroge Character Names Context: アサギ == Asagi | Female, ウィップ == Whip | Female, ウラ == Ura | Female, ブレイド == Blade | Female'
if fullPromptFlag:
system = PROMPT
user = 'Line to Translate: ' + subbedT

View file

@ -49,7 +49,7 @@ CODE356 = False
CODE320 = False
CODE111 = False
def handleTextfile(filename, estimate):
def handleTXT(filename, estimate):
global ESTIMATE, TOKENS, TOTALTOKENS, TOTALCOST
ESTIMATE = estimate