Add module for RPGM plugins

This commit is contained in:
DazedAnon 2024-06-06 12:15:35 -05:00
parent 044ed86f6e
commit 36b7234699
3 changed files with 667 additions and 108 deletions

View file

@ -33,6 +33,7 @@ from modules.wolf2 import handleWOLF2
from modules.javascript import handleJavascript
from modules.irissoft import handleIris
from modules.regex import handleRegex
from modules.rpgmakerplugin import handlePlugin
# For GPT4 rate limit will be hit if you have more than 1 thread.
# 1 Thread for each file. Controls how many files are worked on at once.
@ -41,6 +42,7 @@ THREADS = int(os.getenv('fileThreads'))
# [Display name, file extension, handle function]
MODULES = [
["RPGMaker MV/MZ", "json", handleMVMZ],
["RPGMaker Plugins", "js", handlePlugin],
["RPGMaker ACE", "yaml", handleACE],
["CSV (From Translator++)", "csv", handleCSV],
["Eushully", "txt", handleEushully],

View file

@ -29,10 +29,11 @@ MAXHISTORY = 10
ESTIMATE = ''
TOKENS = [0, 0]
NAMESLIST = []
FIRSTLINESPEAKERS = False # If 1st line of dialogue is a speaker, set to True
NAMES = False # Output a list of all the character names found
BRFLAG = False # If the game uses <br> instead
FIXTEXTWRAP = True # Overwrites textwrap
IGNORETLTEXT = True # Ignores all translated text.
IGNORETLTEXT = False # Ignores all translated text.
MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong)
BRACKETNAMES = False
PBAR = None
@ -48,7 +49,7 @@ if 'gpt-3.5' in MODEL:
elif 'gpt-4o' in MODEL:
INPUTAPICOST = .005
OUTPUTAPICOST = .015
BATCHSIZE = 20
BATCHSIZE = 50
FREQUENCY_PENALTY = 0.1
#tqdm Globals
@ -57,12 +58,12 @@ POSITION = 0
LEAVE = False
# Dialogue / Scroll
CODE401 = True
CODE405 = True
CODE401 = False
CODE405 = False
CODE408 = False
# Choices
CODE102 = True
CODE102 = False
# Variables
CODE122 = False
@ -78,7 +79,7 @@ CODE356 = False
CODE320 = False
CODE324 = False
CODE111 = False
CODE108 = False
CODE108 = True
def handleMVMZ(filename, estimate):
global ESTIMATE, TOKENS
@ -479,7 +480,8 @@ def searchNames(data, pbar, context):
if context in ['Armors', 'Weapons', 'Items']:
if len(nameList) < BATCHSIZE:
nameList.append(data[i]['name'])
descriptionList.append(data[i]['description'].replace('\n', ' '))
if 'description' in data[i]:
descriptionList.append(data[i]['description'].replace('\n', ' '))
if '<hint:' in data[i]['note']:
tokensResponse = translateNote(data[i], r'<hint:(.*?)>')
totalTokens[0] += tokensResponse[0]
@ -551,7 +553,9 @@ def searchNames(data, pbar, context):
if context in ['Enemies', 'Classes', 'MapInfos']:
if len(nameList) < BATCHSIZE:
nameList.append(data[i]['name'])
# tokensResponse = translateNote(data[i], r'.+')
# totalTokens[0] += tokensResponse[0]
# totalTokens[1] += tokensResponse[1]
i += 1
else:
batchFull = True
@ -627,7 +631,8 @@ def searchNames(data, pbar, context):
else:
# Get Text
data[j]['name'] = translatedNameBatch[0]
data[j]['description'] = textwrap.fill(translatedDescriptionBatch[0], LISTWIDTH)
if 'description' in data[j]:
data[j]['description'] = textwrap.fill(translatedDescriptionBatch[0], LISTWIDTH)
translatedNameBatch.pop(0)
translatedDescriptionBatch.pop(0)
@ -683,12 +688,16 @@ def searchNames(data, pbar, context):
def searchCodes(page, pbar, jobList, filename):
if len(jobList) > 0:
docList = jobList[0]
scriptList = jobList[1]
list401 = jobList[0]
list122 = jobList[1]
list355655 = jobList[2]
list108 = jobList[3]
setData = True
else:
docList = []
scriptList = []
list401 = []
list122 = []
list355655 = []
list108 = []
setData = False
currentGroup = []
textHistory = []
@ -753,7 +762,7 @@ def searchCodes(page, pbar, jobList, filename):
speakerList = re.findall(r'^【(.*?)】$', jaString)
# None
if len(speakerList) == 0:
if len(speakerList) == 0 and FIRSTLINESPEAKERS is True:
if len(jaString) < 40 \
and 'code' in codeList[i+1] \
and codeList[i+1]['code'] in [401, 405, -1] \
@ -848,10 +857,6 @@ def searchCodes(page, pbar, jobList, filename):
codeList[i]['parameters'] = [finalJAString + nametag]
elif nCase == 1:
codeList[i]['parameters'] = [nametag + finalJAString]
### Brackets
matchList = re.findall\
(r'^([\\]+[cC]\[[0-9]+\]【?(.+?)】?[\\]+[cC]\[[0-9]+\])|^(【(.+)】)', finalJAString)
# Handle both cases of the regex
if len(matchList) != 0 and BRACKETNAMES is True:
@ -949,11 +954,11 @@ def searchCodes(page, pbar, jobList, filename):
# 1st Passthrough (Grabbing Data)
if setData == False:
if speaker == '' and finalJAString != '':
docList.append(finalJAString)
list401.append(finalJAString)
elif finalJAString != '':
docList.append(f'[{speaker}]: {finalJAString}')
list401.append(f'[{speaker}]: {finalJAString}')
else:
docList.append(speaker)
list401.append(speaker)
speaker = ''
match = []
currentGroup = []
@ -962,8 +967,8 @@ def searchCodes(page, pbar, jobList, filename):
# 2nd Passthrough (Setting Data)
else:
# Grab Translated String
if len(docList) > 0:
translatedText = docList[0]
if len(list401) > 0:
translatedText = list401[0]
# Remove speaker
if speaker != '':
@ -1011,12 +1016,12 @@ def searchCodes(page, pbar, jobList, filename):
match = []
currentGroup = []
syncIndex = i + 1
docList.pop(0)
list401.pop(0)
## Event Code: 122 [Set Variables]
if 'code' in codeList[i] and codeList[i]['code'] == 122 and CODE122 is True:
# This is going to be the var being set. (IMPORTANT)
if codeList[i]['parameters'][0] not in list(range(0, 20)):
if codeList[i]['parameters'][0] not in list(range(0, 100)):
i += 1
continue
@ -1039,13 +1044,13 @@ def searchCodes(page, pbar, jobList, filename):
# Pass 1
if setData == False:
scriptList.append(finalJAString)
list122.append(finalJAString)
# Pass 2
else:
if len(scriptList) > 0:
if len(list122) > 0:
# Grab and Replace
translatedText = scriptList[0]
translatedText = list122[0]
translatedText = jaString.replace(jaString, translatedText)
# Remove characters that may break scripts
@ -1060,7 +1065,7 @@ def searchCodes(page, pbar, jobList, filename):
# Set
codeList[i]['parameters'][4] = translatedText
scriptList.pop(0)
list122.pop(0)
## Event Code: 357 [Picture Text] [Optional]
if 'code' in codeList[i] and codeList[i]['code'] == 357 and CODE357 is True:
@ -1227,70 +1232,22 @@ def searchCodes(page, pbar, jobList, filename):
## Event Code: 355 or 655 Scripts [Optional]
if 'code' in codeList[i] and (codeList[i]['code'] == 355 or codeList[i]['code'] == 655) and CODE355655 is True:
matchList = []
jaString = codeList[i]['parameters'][0]
# Var Text
if 'text =' in jaString or '$gameVariables.setValue(' in jaString:
# Pass 1
if setData is False:
list355655.append(jaString)
# Skip Console Logs
if 'console.log' in jaString:
i += 1
continue
# Skip if
if 'if(' in jaString:
i += 1
continue
# Skip if
if 'list' in jaString:
i += 1
continue
# Pass 2
else:
# Grab and Replace
translatedText = list355655[0]
stringList = []
matchList = re.findall(r"this.BLogAdd\(.+?\"(.+?)\"", jaString)
if len(matchList) > 0:
for match in matchList:
# Remove Textwrap
match = match.replace('\\n', ' ')
stringList.append(match)
# If there isn't any Japanese in the text just skip
# if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴー]+', jaString):
# continue
# Skip These
# if 'this.' in jaString:
# continue
# Need to remove outside code and put it back later
# matchList = re.findall(r'.+"(.*?)".*[;,]$', jaString)
# Want to translate this script
if 'this.BLogAdd' not in jaString:
i += 1
continue
# Translate
if len(matchList) > 0:
response = translateGPT(stringList, 'Reply with the '+ LANGUAGE +' translation of the text.', True)
totalTokens[0] += response[1][0]
totalTokens[1] += response[1][1]
translatedTextList = response[0]
translatedText = jaString
# Replace Each Instance
for j in range(len(translatedTextList)):
# Remove characters that may break scripts
translatedTextList[j] = translatedTextList[j].replace('"', r'\"')
translatedTextList[j] = translatedTextList[j].replace("'", r"\'")
translatedTextList[j] = translatedTextList[j].replace(".", r"\.")
# Wordwrap
translatedTextList[j] = textwrap.fill(translatedTextList[j], width=WIDTH).replace('\n', '\\n')
# Replace Instance
translatedText = translatedText.replace(matchList[j], translatedTextList[j])
# Set Data
codeList[i]['parameters'][0] = translatedText
# Set
codeList[i]['parameters'][0] = translatedText
list355655.pop(0)
## Event Code: 408 (Script)
if 'code' in codeList[i] and (codeList[i]['code'] == 408) and CODE408 is True:
@ -1356,12 +1313,15 @@ def searchCodes(page, pbar, jobList, filename):
# Need to remove outside code and put it back later
matchList = re.findall(regex, jaString)
# Translate
if len(matchList) > 0:
response = translateGPT(matchList[0], 'Reply with the '+ LANGUAGE +' translation of the text.', False)
totalTokens[0] += response[1][0]
totalTokens[1] += response[1][1]
translatedText = response[0]
# Pass 1
if setData is False:
list108.append(matchList[0])
# Pass 2
else:
# Grab and Replace
translatedText = list108[0]
list108.pop(0)
# Remove characters that may break scripts
charList = ['.', '\"']
@ -1698,18 +1658,20 @@ def searchCodes(page, pbar, jobList, filename):
i += 1
# End of the line
docListTL = []
scriptListTL = []
list401TL = []
list122TL = []
list355655TL = []
list108TL = []
setData = False
PBAR = pbar
# 401
if len(docList) > 0:
response = translateGPT(docList, textHistory, True)
docListTL = response[0]
if len(list401) > 0:
response = translateGPT(list401, textHistory, True)
list401TL = response[0]
totalTokens[0] += response[1][0]
totalTokens[1] += response[1][1]
if len(docListTL) != len(docList):
if len(list401TL) != len(list401):
with LOCK:
if filename not in MISMATCH:
MISMATCH.append(filename)
@ -1717,12 +1679,38 @@ def searchCodes(page, pbar, jobList, filename):
setData = True
# 122
if len(scriptList) > 0:
response = translateGPT(scriptList, textHistory, True)
scriptListTL = response[0]
if len(list122) > 0:
response = translateGPT(list122, textHistory, True)
list122TL = response[0]
totalTokens[0] += response[1][0]
totalTokens[1] += response[1][1]
if len(scriptListTL) != len(scriptList):
if len(list122TL) != len(list122):
with LOCK:
if filename not in MISMATCH:
MISMATCH.append(filename)
else:
setData = True
# 355/655
if len(list355655) > 0:
response = translateGPT(list355655, textHistory, True)
list355655TL = response[0]
totalTokens[0] += response[1][0]
totalTokens[1] += response[1][1]
if len(list355655TL) != len(list355655):
with LOCK:
if filename not in MISMATCH:
MISMATCH.append(filename)
else:
setData = True
# 108
if len(list108) > 0:
response = translateGPT(list108, textHistory, True)
list108TL = response[0]
totalTokens[0] += response[1][0]
totalTokens[1] += response[1][1]
if len(list108TL) != len(list108):
with LOCK:
if filename not in MISMATCH:
MISMATCH.append(filename)
@ -1731,7 +1719,7 @@ def searchCodes(page, pbar, jobList, filename):
# Start Pass 2
if setData:
searchCodes(page, pbar, [docListTL, scriptListTL], filename)
searchCodes(page, pbar, [list401TL, list122TL, list355655TL, list108TL], filename)
# Delete all -1 codes
codeListFinal = []
@ -2063,7 +2051,8 @@ def batchList(input_list, batch_size):
def createContext(fullPromptFlag, subbedT):
characters = 'Game Characters:\n\
グレイス (Grace) - Female\n\
ティアナ (Tiana) - Female\n\
キャサリン (Catherine) - Female\n\
'
system = PROMPT + VOCAB if fullPromptFlag else \

568
modules/rpgmakerplugin.py Normal file
View file

@ -0,0 +1,568 @@
# Libraries
import os, re, textwrap, threading, time, traceback, tiktoken, openai
from pathlib import Path
from colorama import Fore
from dotenv import load_dotenv
from retry import retry
from tqdm import tqdm
# Open AI
load_dotenv()
if os.getenv('api').replace(' ', '') != '':
openai.base_url = os.getenv('api')
openai.organization = os.getenv('org')
openai.api_key = os.getenv('key')
#Globals
MODEL = os.getenv('model')
TIMEOUT = int(os.getenv('timeout'))
LANGUAGE = os.getenv('language').capitalize()
PROMPT = Path('prompt.txt').read_text(encoding='utf-8')
VOCAB = Path('vocab.txt').read_text(encoding='utf-8')
THREADS = int(os.getenv('threads'))
LOCK = threading.Lock()
WIDTH = int(os.getenv('width'))
LISTWIDTH = int(os.getenv('listWidth'))
NOTEWIDTH = 70
MAXHISTORY = 10
ESTIMATE = ''
TOKENS = [0, 0]
NAMESLIST = []
NAMES = False # Output a list of all the character names found
BRFLAG = False # If the game uses <br> instead
FIXTEXTWRAP = True # Overwrites textwrap
IGNORETLTEXT = False # Ignores all translated text.
MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong)
#tqdm Globals
BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}'
POSITION = 0
LEAVE = False
# Pricing - Depends on the model https://openai.com/pricing
# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request
# If you are getting a MISMATCH LENGTH error, lower the batch size.
if 'gpt-3.5' in MODEL:
INPUTAPICOST = .002
OUTPUTAPICOST = .002
BATCHSIZE = 10
elif 'gpt-4' in MODEL:
INPUTAPICOST = .005
OUTPUTAPICOST = .015
BATCHSIZE = 40
def handlePlugin(filename, estimate):
global ESTIMATE
ESTIMATE = estimate
if ESTIMATE:
start = time.time()
translatedData = openFiles(filename)
# Print Result
end = time.time()
tqdm.write(getResultString(translatedData, end - start, filename))
with LOCK:
TOKENS[0] += translatedData[1][0]
TOKENS[1] += translatedData[1][1]
# Print Total
totalString = getResultString(['', TOKENS, None], end - start, 'TOTAL')
# Print any errors on maps
if len(MISMATCH) > 0:
return totalString + Fore.RED + f'\nMismatch Errors: {MISMATCH}' + Fore.RESET
else:
return totalString
else:
try:
with open('translated/' + filename, 'w', encoding='utf_8', errors='ignore') as outFile:
start = time.time()
translatedData = openFiles(filename)
# Print Result
end = time.time()
outFile.writelines(translatedData[0])
tqdm.write(getResultString(translatedData, end - start, filename))
with LOCK:
TOKENS[0] += translatedData[1][0]
TOKENS[1] += translatedData[1][1]
except Exception as e:
traceback.print_exc()
return 'Fail'
return getResultString(['', TOKENS, None], end - start, 'TOTAL')
def getResultString(translatedData, translationTime, filename):
# File Print String
totalTokenstring =\
Fore.YELLOW +\
'[Input: ' + str(translatedData[1][0]) + ']'\
'[Output: ' + str(translatedData[1][1]) + ']'\
'[Cost: ${:,.4f}'.format((translatedData[1][0] * .001 * INPUTAPICOST) +\
(translatedData[1][1] * .001 * OUTPUTAPICOST)) + ']'
timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]'
if translatedData[2] == None:
# Success
return filename + ': ' + totalTokenstring + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET
else:
# Fail
try:
raise translatedData[2]
except Exception as e:
traceback.print_exc()
errorString = str(e) + Fore.RED
return filename + ': ' + totalTokenstring + timeString + Fore.RED + u' \u2717 ' +\
errorString + Fore.RESET
def openFiles(filename):
with open('files/' + filename, 'r', encoding='utf_8') as readFile:
translatedData = parsePlugin(readFile, filename)
# Delete lines marked for deletion
finalData = []
for line in translatedData[0]:
if line != '\\d\n':
finalData.append(line)
translatedData[0] = finalData
return translatedData
def parsePlugin(readFile, filename):
totalTokens = [0,0]
# Read File into data
data = readFile.readlines()
# Create Progress Bar
with tqdm(bar_format=BAR_FORMAT, position=POSITION, leave=LEAVE) as pbar:
pbar.desc=filename
try:
result = translatePlugin(data, pbar, filename, [])
totalTokens[0] += result[0]
totalTokens[1] += result[1]
except Exception as e:
traceback.print_exc()
return [data, totalTokens, e]
return [data, totalTokens, None]
def translatePlugin(data, pbar, filename, translatedList):
stringList = []
currentGroup = []
tokens = [0,0]
speaker = ''
voice = False
global LOCK, ESTIMATE
i = 0
while i < len(data):
voice = False
speaker = ''
"""
Plugin List
Quest Name: [\\]+"QuestName[\\]+":[\\]+"(.*?)[\\]+"
Quest Client: [\\]+"QuestClientName[\\]+":[\\]+"[\\]+"[\\]+"(.*?)[\\]+"
Quest Location: [\\]+"QuestionLocation[\\]+":[\\]+"[\\]+"[\\]+"(.*?)[\\]+"
Quest Targe Location: [\\]+"PlaceInformation[\\]+":[\\]+"(.*?)[\\]+"
Quest Summary: [\\]+"QuestContent[\\]+":[\\]+"(.*?)[\\]+"
Quest Goal: [\\]+"ObjectiveContent[\\]+":[\\]+"[\\]+"[\\]+"(.*?)[\\]+"
Quest Goal 2: [\\]+"ObjectiveContent[\\]+":[\\]+"[\\]+"[\\]+".*?[\\]+"(.*?)[\\]+"
TODO TL all of the above in one call instead of multiple
"""
# Lines
matchList = re.findall(r'[\\]+"PlaceInformation[\\]+":[\\]+"(.*?)[\\]+"', data[i])
if len(matchList) > 0:
for match in matchList:
# Save Original String
originalString = match
# Remove any textwrap
match = match.replace(r'\\\\\\\\n', ' ')
# Pass 1
if translatedList == []:
# Add String
stringList.append(match.strip())
# Pass 2
else:
# Get Text
if translatedList:
# Grab and Pop
translatedText = translatedList[0]
translatedList.pop(0)
# Set to None if empty list
if len(translatedList) <= 0:
translatedList = None
# Textwrap
translatedText = textwrap.fill(translatedText, width=WIDTH)
translatedText = translatedText.replace('\n', r'\\\\\\\\n')
# Replace Single Quotes
translatedText = translatedText.replace("'", "\\'")
# Set Data
data[i] = data[i].replace(originalString, translatedText)
# Next Line
i += 1
# EOF
if len(stringList) > 0:
# Set Progress
pbar.total = len(stringList)
pbar.refresh()
# Translate
response = translateGPT(stringList, 'The following lines are quest locations', True, pbar, filename)
tokens[0] += response[1][0]
tokens[1] += response[1][1]
translatedList = response[0]
# Set Strings
if len(stringList) == len(translatedList):
translatePlugin(data, pbar, filename, translatedList)
# Mismatch
else:
with LOCK:
if filename not in MISMATCH:
MISMATCH.append(filename)
return tokens
# Save some money and enter the character before translation
def getSpeaker(speaker, pbar, filename):
match speaker:
case 'ファイン':
return ['Fine', [0,0]]
case '':
return ['', [0,0]]
case _:
# Store Speaker
if speaker not in str(NAMESLIST):
response = translateGPT(speaker, 'Reply with only the '+ LANGUAGE +' translation of the NPC name.', False, pbar, filename)
response[0] = response[0].replace("'S", "'s")
speakerList = [speaker, response[0]]
NAMESLIST.append(speakerList)
return response
# Find Speaker
else:
for i in range(len(NAMESLIST)):
if speaker == NAMESLIST[i][0]:
return [NAMESLIST[i][1],[0,0]]
return [speaker,[0,0]]
def subVars(jaString):
jaString = jaString.replace('\u3000', ' ')
# Nested
count = 0
nestedList = re.findall(r'[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]', jaString)
nestedList = set(nestedList)
if len(nestedList) != 0:
for icon in nestedList:
jaString = jaString.replace(icon, '[Nested_' + str(count) + ']')
count += 1
# Icons
count = 0
iconList = re.findall(r'[\\]+[iIkKwWaA]+\[[0-9]+\]', jaString)
iconList = set(iconList)
if len(iconList) != 0:
for icon in iconList:
jaString = jaString.replace(icon, '[Ascii_' + str(count) + ']')
count += 1
# Colors
count = 0
colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString)
colorList = set(colorList)
if len(colorList) != 0:
for color in colorList:
jaString = jaString.replace(color, '[Color_' + str(count) + ']')
count += 1
# Names
count = 0
nameList = re.findall(r'[\\]+[nN]\[.+?\]+', jaString)
nameList = set(nameList)
if len(nameList) != 0:
for name in nameList:
jaString = jaString.replace(name, '[Noun_' + str(count) + ']')
count += 1
# Variables
count = 0
varList = re.findall(r'[\\]+[vV]\[[0-9]+\]', jaString)
varList = set(varList)
if len(varList) != 0:
for var in varList:
jaString = jaString.replace(var, '[Var_' + str(count) + ']')
count += 1
# Formatting
count = 0
formatList = re.findall(r'[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]', jaString)
formatList = set(formatList)
if len(formatList) != 0:
for var in formatList:
jaString = jaString.replace(var, '[FCode_' + str(count) + ']')
count += 1
# Put all lists in list and return
allList = [nestedList, iconList, colorList, nameList, varList, formatList]
return [jaString, allList]
def resubVars(translatedText, allList):
# Fix Spacing and ChatGPT Nonsense
matchList = re.findall(r'\[\s?.+?\s?\]', translatedText)
if len(matchList) > 0:
for match in matchList:
text = match.strip()
translatedText = translatedText.replace(match, text)
# Nested
count = 0
if len(allList[0]) != 0:
for var in allList[0]:
translatedText = translatedText.replace('[Nested_' + str(count) + ']', var)
count += 1
# Icons
count = 0
if len(allList[1]) != 0:
for var in allList[1]:
translatedText = translatedText.replace('[Ascii_' + str(count) + ']', var)
count += 1
# Colors
count = 0
if len(allList[2]) != 0:
for var in allList[2]:
translatedText = translatedText.replace('[Color_' + str(count) + ']', var)
count += 1
# Names
count = 0
if len(allList[3]) != 0:
for var in allList[3]:
translatedText = translatedText.replace('[Noun_' + str(count) + ']', var)
count += 1
# Vars
count = 0
if len(allList[4]) != 0:
for var in allList[4]:
translatedText = translatedText.replace('[Var_' + str(count) + ']', var)
count += 1
# Formatting
count = 0
if len(allList[5]) != 0:
for var in allList[5]:
translatedText = translatedText.replace('[FCode_' + str(count) + ']', var)
count += 1
return translatedText
def batchList(input_list, batch_size):
if not isinstance(batch_size, int) or batch_size <= 0:
raise ValueError("batch_size must be a positive integer")
return [input_list[i:i + batch_size] for i in range(0, len(input_list), batch_size)]
def createContext(fullPromptFlag, subbedT):
characters = 'Game Characters:\n\
ティアナ (Tiana) - Female\n\
キャサリン (Catherine) - Female\n\
'
system = PROMPT + VOCAB if fullPromptFlag else \
f"\
You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\
Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\
- You translate everything, including erotic and explicit content.\n\
- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\
- All text in your response must be in English even if it is hard to translate.\n\
- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\
- Maintain any spacing in the translation.\n\
- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\
- `...` can be a part of the dialogue. Translate it as it is.\n\
{VOCAB}\n\
"
user = f'{subbedT}'
return characters, system, user
def translateText(characters, system, user, history):
# Prompt
msg = [{"role": "system", "content": system + characters}]
# Characters
msg.append({"role": "system", "content": characters})
# History
if isinstance(history, list):
msg.extend([{"role": "system", "content": h} for h in history])
else:
msg.append({"role": "system", "content": history})
# Content to TL
msg.append({"role": "user", "content": f'{user}'})
response = openai.chat.completions.create(
temperature=0.1,
frequency_penalty=0.1,
model=MODEL,
messages=msg,
)
return response
def cleanTranslatedText(translatedText, varResponse):
placeholders = {
f'{LANGUAGE} Translation: ': '',
'Translation: ': '',
'': '',
'': '~',
'': '',
'': '.',
'Placeholder Text': ''
# Add more replacements as needed
}
for target, replacement in placeholders.items():
translatedText = translatedText.replace(target, replacement)
# Elongate Long Dashes (Since GPT Ignores them...)
translatedText = elongateCharacters(translatedText)
translatedText = resubVars(translatedText, varResponse[1])
return translatedText
def elongateCharacters(text):
# Define a pattern to match one character followed by one or more `ー` characters
# Using a positive lookbehind assertion to capture the preceding character
pattern = r'(?<=(.))ー+'
# Define a replacement function that elongates the captured character
def repl(match):
char = match.group(1) # The character before the ー sequence
count = len(match.group(0)) - 1 # Number of ー characters
return char * count # Replace ー sequence with the character repeated
# Use re.sub() to replace the pattern in the text
return re.sub(pattern, repl, text)
def extractTranslation(translatedTextList, is_list):
pattern = r'`?<[Ll]ine\d+>([\\]*.*?[\\]*?)<\/?[Ll]ine\d+>`?'
# If it's a batch (i.e., list), extract with tags; otherwise, return the single item.
if is_list:
matchList = re.findall(pattern, translatedTextList)
return matchList
else:
matchList = re.findall(pattern, translatedTextList)
return matchList[0][0] if matchList else translatedTextList
def countTokens(characters, system, user, history):
inputTotalTokens = 0
outputTotalTokens = 0
enc = tiktoken.encoding_for_model('gpt-4')
# Input
if isinstance(history, list):
for line in history:
inputTotalTokens += len(enc.encode(line))
else:
inputTotalTokens += len(enc.encode(history))
inputTotalTokens += len(enc.encode(system))
inputTotalTokens += len(enc.encode(characters))
inputTotalTokens += len(enc.encode(user))
# Output
outputTotalTokens += round(len(enc.encode(user))*2)
return [inputTotalTokens, outputTotalTokens]
def combineList(tlist, text):
if isinstance(text, list):
return [t for sublist in tlist for t in sublist]
return tlist[0]
@retry(exceptions=Exception, tries=5, delay=5)
def translateGPT(text, history, fullPromptFlag, pbar, filename):
mismatch = False
totalTokens = [0, 0]
if isinstance(text, list):
tList = batchList(text, BATCHSIZE)
else:
tList = [text]
for index, tItem in enumerate(tList):
# Before sending to translation, if we have a list of items, add the formatting
if isinstance(tItem, list):
payload = '\n'.join([f'`<Line{i}>{item}</Line{i}>`' for i, item in enumerate(tItem)])
payload = re.sub(r'(<Line\d+)(><)(\/Line\d+>)', r'\1>Placeholder Text<\3', payload)
varResponse = subVars(payload)
subbedT = varResponse[0]
else:
varResponse = subVars(tItem)
subbedT = varResponse[0]
# Things to Check before starting translation
if not re.search(r'[一-龠ぁ-ゔァ-ヴーa---]+', subbedT):
continue
# Create Message
characters, system, user = createContext(fullPromptFlag, subbedT)
# Calculate Estimate
if ESTIMATE:
estimate = countTokens(characters, system, user, history)
totalTokens[0] += estimate[0]
totalTokens[1] += estimate[1]
continue
# Translating
response = translateText(characters, system, user, history)
translatedText = response.choices[0].message.content
totalTokens[0] += response.usage.prompt_tokens
totalTokens[1] += response.usage.completion_tokens
# Formatting
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedText, True)
if len(tItem) != len(extractedTranslations):
# Mismatch. Try Again
response = translateText(characters, system, user, history)
translatedText = response.choices[0].message.content
totalTokens[0] += response.usage.prompt_tokens
totalTokens[1] += response.usage.completion_tokens
# Formatting
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedText, True)
if len(tItem) == len(extractedTranslations):
tList[index] = extractedTranslations
else:
MISMATCH.append(filename)
else:
tList[index] = extractedTranslations
# Create History
history = tList[index] # Update history if we have a list
pbar.update(len(tList[index]))
else:
# Ensure we're passing a single string to extractTranslation
extractedTranslations = extractTranslation(translatedText, False)
tList[index] = extractedTranslations
finalList = combineList(tList, text)
return [finalList, totalTokens]