Create Ushully script and update onscripter

This commit is contained in:
DazedAnon 2024-05-24 10:39:40 -05:00
parent 2ba6e7d9f9
commit bf9ed0bd54
4 changed files with 898 additions and 251 deletions

726
modules/eushully.py Normal file
View file

@ -0,0 +1,726 @@
# Libraries
import json, os, re, textwrap, threading, time, traceback, tiktoken, openai, csv
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
from colorama import Fore
from dotenv import load_dotenv
from retry import retry
from tqdm import tqdm
# Open AI
load_dotenv()
if os.getenv('api').replace(' ', '') != '':
openai.base_url = os.getenv('api')
openai.organization = os.getenv('org')
openai.api_key = os.getenv('key')
#Globals
MODEL = os.getenv('model')
TIMEOUT = int(os.getenv('timeout'))
LANGUAGE = os.getenv('language').capitalize()
PROMPT = Path('prompt.txt').read_text(encoding='utf-8')
VOCAB = Path('vocab.txt').read_text(encoding='utf-8')
THREADS = int(os.getenv('threads'))
LOCK = threading.Lock()
WIDTH = int(os.getenv('width'))
LISTWIDTH = int(os.getenv('listWidth'))
NOTEWIDTH = int(os.getenv('noteWidth'))
MAXHISTORY = 10
ESTIMATE = ''
TOKENS = [0, 0]
NAMESLIST = []
NAMES = False # Output a list of all the character names found
BRFLAG = False # If the game uses <br> instead
FIXTEXTWRAP = True # Overwrites textwrap
IGNORETLTEXT = True # Ignores all translated text.
MISMATCH = [] # Lists files that thdata a mismatch error (Length of GPT list response is wrong)
BRACKETNAMES = False
TOTALLINES = 0
PBAR = None
# Pricing - Depends on the model https://openai.com/pricing
# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request
# If you are getting a MISMATCH LENGTH error, lower the batch size.
if 'gpt-3.5' in MODEL:
INPUTAPICOST = .002
OUTPUTAPICOST = .002
BATCHSIZE = 10
FREQUENCY_PENALTY = 0.2
elif 'gpt-4' in MODEL:
INPUTAPICOST = .005
OUTPUTAPICOST = .015
BATCHSIZE = 20
FREQUENCY_PENALTY = 0.1
#tqdm Globals
BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}'
POSITION = 0
LEAVE = False
def handleEushully(filename, estimate):
global ESTIMATE, TOKENS
ESTIMATE = estimate
if not ESTIMATE:
with open('translated/' + filename, 'w+t', newline='', encoding='utf-8') as writeFile:
# Translate
start = time.time()
translatedData = openFiles(filename, writeFile)
# Print Result
end = time.time()
tqdm.write(getResultString(translatedData, end - start, filename))
with LOCK:
TOKENS[0] += translatedData[1][0]
TOKENS[1] += translatedData[1][1]
else:
# Translate
start = time.time()
translatedData = openFilesEstimate(filename)
# Print Result
end = time.time()
tqdm.write(getResultString(translatedData, end - start, filename))
with LOCK:
TOKENS[0] += translatedData[1][0]
TOKENS[1] += translatedData[1][1]
# Print Total
totalString = getResultString(['', TOKENS, None], end - start, 'TOTAL')
# Print any errors on maps
if len(MISMATCH) > 0:
return totalString + Fore.RED + f'\nMismatch Errors: {MISMATCH}' + Fore.RESET
else:
return totalString
def openFiles(filename, writeFile):
with open('files/' + filename, 'r', encoding='utf-8') as readFile, writeFile:
translatedData = parseCSV(readFile, writeFile, filename)
return translatedData
def openFilesEstimate(filename):
with open('files/' + filename, 'r', encoding='utf-8') as readFile:
translatedData = parseCSV(readFile, '', filename)
return translatedData
def getResultString(translatedData, translationTime, filename):
# File Print String
totalTokenstring =\
Fore.YELLOW +\
'[Input: ' + str(translatedData[1][0]) + ']'\
'[Output: ' + str(translatedData[1][1]) + ']'\
'[Lines: ' + str(TOTALLINES) + ']'\
'[Cost: ${:,.4f}'.format((translatedData[1][0] * .001 * INPUTAPICOST) +\
(translatedData[1][1] * .001 * OUTPUTAPICOST)) + ']'
timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]'
if translatedData[2] is None:
# Success
return filename + ': ' + totalTokenstring + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET
else:
# Fail
try:
raise translatedData[2]
except Exception as e:
traceback.print_exc()
errorString = str(e) + Fore.RED
return filename + ': ' + totalTokenstring + timeString + Fore.RED + u' \u2717 ' +\
errorString + Fore.RESET
def parseCSV(readFile, writeFile, filename):
totalTokens = [0,0]
totalLines = 0
textHistory = []
global LOCK
# Get total for progress bar
totalLines = len(readFile.readlines())
readFile.seek(0)
data = []
reader = csv.reader(readFile, delimiter=',',)
if not ESTIMATE:
writer = csv.writer(writeFile, delimiter=',', quoting=csv.QUOTE_ALL)
else:
writer = ''
# Write All Rows to Data
for row in reader:
data.append(row)
with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar:
pbar.desc=filename
pbar.total=totalLines
try:
if 'SC' == filename[0:2] or 'SP' == filename[0:2]:
response = translateDialogue(data, pbar, writer, format, filename, [])
totalTokens[0] = response[0]
totalTokens[1] = response[1]
elif 'UI' == filename[0:2]:
response = translateUI(data, pbar, writer, format, filename, [])
totalTokens[0] = response[0]
totalTokens[1] = response[1]
except Exception as e:
traceback.print_exc()
return [reader, totalTokens, None]
def translateDialogue(data, pbar, writer, format, filename, translatedList):
global LOCK, ESTIMATE
tokens = [0,0]
stringList = [None] * 2
i = 0
try:
# Set Variables
speakerColumn = 0
textSourceColumn = 3
textTargetColumn = 3
# Lists
dialogueList = []
setStringList = []
# Parse Data
while i in range(len(data)):
# Dialogue
if len(data[i][speakerColumn]) > 0 and data[i][speakerColumn][0].isupper() \
or 'show-text' in data[i][speakerColumn] \
or 'concat' in data[i][speakerColumn]:
# Speaker
speaker = ''
if data[i][speakerColumn][0].isupper():
if speakerColumn != None:
response = getSpeaker(data[i][speakerColumn])
tokens[0] += response[1][0]
tokens[1] += response[1][1]
speaker = response[0]
# Dialogue
jaString = data[i][textSourceColumn]
# Remove Textwrap
jaString = jaString.replace('\n', ' ')
# Replace Unicode
jaString = jaString.replace('\ue000', '...')
# Pass 1
if translatedList == []:
# Add to list
if speaker:
dialogueList.append(f'[{speaker}]: {jaString}')
else:
dialogueList.append(f'[InnerVoice]: {jaString}')
stringList[0] = dialogueList
# Pass 2
else:
if translatedList[0]:
# Grab and Pop
translatedText = translatedList[0][0]
translatedList[0].pop(0)
# Set to None if empty list
if len(translatedList[0]) <= 0:
translatedList[0] = None
# Remove speaker
translatedText = re.sub(r'^\[(.+?)\]\s?[|:]\s?', '', translatedText)
# Set Data
data[i][textTargetColumn] = f'{translatedText}'
# Set String Command
if 'set-string' in data[i][speakerColumn]:
jaString = data[i][textSourceColumn]
# Pass 1
if translatedList == []:
setStringList.append(jaString)
stringList[1] = setStringList
# Pass 2
else:
if len(translatedList) > 1 and translatedList[1]:
# Grab and Pop
translatedText = translatedList[1][0]
translatedList[1].pop(0)
# Set to None if empty list
if len(translatedList[1]) <= 0:
translatedList[1] = None
# Textwrap
translatedText = textwrap.fill(translatedText, WIDTH)
# Set Data
data[i][textTargetColumn] = f'{translatedText}'
# Iterate
i += 1
# EOF
stringList = [x for x in stringList if x is not None]
if len(stringList) > 0:
# Translate
pbar.total = 0
for i in range(len(stringList)):
# Set Progress
pbar.total += len(stringList[i])
pbar.refresh()
PBAR = pbar
response = translateGPT(stringList[i], '', True)
tokens[0] += response[1][0]
tokens[1] += response[1][1]
translatedList.append(response[0])
# Set Strings
if len(stringList) == len(translatedList):
translateDialogue(data, pbar, writer, format, filename, translatedList)
# Write all Data
with LOCK:
if not ESTIMATE:
for row in data:
writer.writerow(row)
except Exception as e:
traceback.print_exc()
return tokens
def translateUI(data, pbar, writer, format, filename, translatedList):
global LOCK, ESTIMATE
tokens = [0,0]
stringList = [None] * 1
i = 0
try:
# Lists
textList = []
# Parse Data
while i in range(len(data)):
# Text
for j in range(len(data[i])):
jaString = data[i][j]
# If Japanese Text, Translate it.
if not re.search(r'[一-龠ぁ-ゔァ-ヴーa---]+', jaString):
continue
# Replace Unicode
jaString = jaString.replace('', '...')
# Pass 1
if translatedList == []:
# Add to list
textList.append(f'{jaString}')
stringList[0] = textList
# Pass 2
else:
if translatedList[0]:
# Grab and Pop
translatedText = translatedList[0][0]
translatedList[0].pop(0)
# Set to None if empty list
if len(translatedList[0]) <= 0:
translatedList[0] = None
# Set Data
if len(data[i]) > j + 1:
data[i][j+1] = f'{translatedText}'
# Iterate
i += 1
# EOF
stringList = [x for x in stringList if x is not None]
if len(stringList) > 0:
# Translate
pbar.total = 0
for i in range(len(stringList)):
# Set Progress
pbar.total += len(stringList[i])
pbar.refresh()
response = translateGPT(stringList[i], '', True)
tokens[0] += response[1][0]
tokens[1] += response[1][1]
translatedList.append(response[0])
# Set Strings
if len(stringList) == len(translatedList):
translateUI(data, pbar, writer, format, filename, translatedList)
# Write all Data
with LOCK:
if not ESTIMATE:
for row in data:
writer.writerow(row)
except Exception as e:
traceback.print_exc()
return tokens
# Save some money and enter the character before translation
def getSpeaker(speaker):
match speaker:
case 'ファイン':
return ['Fine', [0,0]]
case '':
return ['', [0,0]]
case _:
# Store Speaker
if speaker not in str(NAMESLIST):
response = translateGPT(speaker, 'Reply with only the '+ LANGUAGE +' translation of the NPC name.', False)
response[0] = response[0].replace("'S", "'s")
speakerList = [speaker, response[0]]
NAMESLIST.append(speakerList)
return response
# Find Speaker
else:
for i in range(len(NAMESLIST)):
if speaker == NAMESLIST[i][0]:
return [NAMESLIST[i][1],[0,0]]
return [speaker,[0,0]]
def subVars(jaString):
jaString = jaString.replace('\u3000', ' ')
# Nested
count = 0
nestedList = re.findall(r'[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]', jaString)
nestedList = set(nestedList)
if len(nestedList) != 0:
for icon in nestedList:
jaString = jaString.replace(icon, '[Nested_' + str(count) + ']')
count += 1
# Icons
count = 0
iconList = re.findall(r'[\\]+[iIkKwWaA]+\[[0-9]+\]', jaString)
iconList = set(iconList)
if len(iconList) != 0:
for icon in iconList:
jaString = jaString.replace(icon, '[Ascii_' + str(count) + ']')
count += 1
# Colors
count = 0
colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString)
colorList = set(colorList)
if len(colorList) != 0:
for color in colorList:
jaString = jaString.replace(color, '[Color_' + str(count) + ']')
count += 1
# Names
count = 0
nameList = re.findall(r'[\\]+[nN]\[.+?\]+', jaString)
nameList = set(nameList)
if len(nameList) != 0:
for name in nameList:
jaString = jaString.replace(name, '[Noun_' + str(count) + ']')
count += 1
# Variables
count = 0
varList = re.findall(r'[\\]+[vV]\[[0-9]+\]', jaString)
varList = set(varList)
if len(varList) != 0:
for var in varList:
jaString = jaString.replace(var, '[Var_' + str(count) + ']')
count += 1
# Formatting
count = 0
formatList = re.findall(r'[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]', jaString)
formatList = set(formatList)
if len(formatList) != 0:
for var in formatList:
jaString = jaString.replace(var, '[FCode_' + str(count) + ']')
count += 1
# Put all lists in list and return
allList = [nestedList, iconList, colorList, nameList, varList, formatList]
return [jaString, allList]
def resubVars(translatedText, allList):
# Fix Spacing and ChatGPT Nonsense
matchList = re.findall(r'\[\s?.+?\s?\]', translatedText)
if len(matchList) > 0:
for match in matchList:
text = match.strip()
translatedText = translatedText.replace(match, text)
# Nested
count = 0
if len(allList[0]) != 0:
for var in allList[0]:
translatedText = translatedText.replace('[Nested_' + str(count) + ']', var)
count += 1
# Icons
count = 0
if len(allList[1]) != 0:
for var in allList[1]:
translatedText = translatedText.replace('[Ascii_' + str(count) + ']', var)
count += 1
# Colors
count = 0
if len(allList[2]) != 0:
for var in allList[2]:
translatedText = translatedText.replace('[Color_' + str(count) + ']', var)
count += 1
# Names
count = 0
if len(allList[3]) != 0:
for var in allList[3]:
translatedText = translatedText.replace('[Noun_' + str(count) + ']', var)
count += 1
# Vars
count = 0
if len(allList[4]) != 0:
for var in allList[4]:
translatedText = translatedText.replace('[Var_' + str(count) + ']', var)
count += 1
# Formatting
count = 0
if len(allList[5]) != 0:
for var in allList[5]:
translatedText = translatedText.replace('[FCode_' + str(count) + ']', var)
count += 1
return translatedText
def batchList(input_list, batch_size):
if not isinstance(batch_size, int) or batch_size <= 0:
raise ValueError("batch_size must be a positive integer")
return [input_list[i:i + batch_size] for i in range(0, len(input_list), batch_size)]
def createContext(fullPromptFlag, subbedT):
characters = 'Game Characters:\n\
クラウス (Klaus) - Male\n\
ベアトリース (Beatrice) - Female\n\
カミラ (Camilla) - Female\n\
セルージュ (Cerouge) - Female\n\
エルヴィール (Elvire) - Female\n\
ヘルミィナ (Helmina) - Female\n\
アンリエット (Henriette) - Female\n\
ユリアーナ (Juliana) - Female\n\
ルシエル (Luciel) - Female\n\
メイズ (Maize) - Female\n\
メイヴィスレイン (Mavislaine) - Female\n\
ラムエル (Ramiel) - Female\n\
レジーニア (Reginia) - Female\n\
リリィ (Lily) - Female\n\
エウクレイアさん (Ms. Eukleia) - Female\n\
'
system = PROMPT + VOCAB if fullPromptFlag else \
f"\
You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\
Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\
- You translate everything, including erotic and explicit content.\n\
- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\
- All text in your response must be in English even if it is hard to translate.\n\
- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\
- Maintain any spacing in the translation.\n\
- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\
- `...` can be a part of the dialogue. Translate it as it is.\n\
{VOCAB}\n\
"
user = f'{subbedT}'
return characters, system, user
def translateText(characters, system, user, history, penalty):
# Prompt
msg = [{"role": "system", "content": system + characters}]
# Characters
msg.append({"role": "system", "content": characters})
# History
if isinstance(history, list):
msg.extend([{"role": "system", "content": h} for h in history])
else:
msg.append({"role": "system", "content": history})
# Content to TL
msg.append({"role": "user", "content": f'{user}'})
response = openai.chat.completions.create(
temperature=0,
frequency_penalty=penalty,
model=MODEL,
messages=msg,
)
return response
def cleanTranslatedText(translatedText, varResponse):
placeholders = {
f'{LANGUAGE} Translation: ': '',
'Translation: ': '',
'': '',
'': '~',
'': '',
'': '.',
'< ': '<',
'</ ': '</',
' >': '>',
'Placeholder Text': '',
'- chan': '-chan',
'- kun': '-kun',
'- san': '-san',
# Add more replacements as needed
}
for target, replacement in placeholders.items():
translatedText = translatedText.replace(target, replacement)
# Elongate Long Dashes (Since GPT Ignores them...)
translatedText = elongateCharacters(translatedText)
translatedText = resubVars(translatedText, varResponse[1])
return translatedText
def elongateCharacters(text):
# Define a pattern to match one character followed by one or more `ー` characters
# Using a positive lookbehind assertion to capture the preceding character
pattern = r'(?<=(.))ー+'
# Define a replacement function that elongates the captured character
def repl(match):
char = match.group(1) # The character before the ー sequence
count = len(match.group(0)) - 1 # Number of ー characters
return char * count # Replace ー sequence with the character repeated
# Use re.sub() to replace the pattern in the text
return re.sub(pattern, repl, text)
def extractTranslation(translatedTextList, is_list):
pattern = r'`?<Line\d+>([\\]*.*?[\\]*?)<\/?Line\d+>`?'
# If it's a batch (i.e., list), extract with tags; otherwise, return the single item.
if is_list:
matchList = re.findall(pattern, translatedTextList)
return matchList
else:
matchList = re.findall(pattern, translatedTextList)
return matchList[0][0] if matchList else translatedTextList
def countTokens(characters, system, user, history):
inputTotalTokens = 0
outputTotalTokens = 0
enc = tiktoken.encoding_for_model('gpt-4')
# Input
if isinstance(history, list):
for line in history:
inputTotalTokens += len(enc.encode(line))
else:
inputTotalTokens += len(enc.encode(history))
inputTotalTokens += len(enc.encode(system))
inputTotalTokens += len(enc.encode(characters))
inputTotalTokens += len(enc.encode(user))
# Output
outputTotalTokens += round(len(enc.encode(user))*3)
return [inputTotalTokens, outputTotalTokens]
def combineList(tlist, text):
if isinstance(text, list):
return [t for sublist in tlist for t in sublist]
return tlist[0]
@retry(exceptions=Exception, tries=5, delay=5)
def translateGPT(text, history, fullPromptFlag):
global PBAR
mismatch = False
totalTokens = [0, 0]
if isinstance(text, list):
tList = batchList(text, BATCHSIZE)
else:
tList = [text]
for index, tItem in enumerate(tList):
# Before sending to translation, if we have a list of items, add the formatting
if isinstance(tItem, list):
payload = '\n'.join([f'`<Line{i}>{item}</Line{i}>`' for i, item in enumerate(tItem)])
payload = re.sub(r'(<Line\d+)(><)(\/Line\d+>)', r'\1>Placeholder Text<\3', payload)
varResponse = subVars(payload)
subbedT = varResponse[0]
else:
varResponse = subVars(tItem)
subbedT = varResponse[0]
# Things to Check before starting translation
if not re.search(r'[一-龠ぁ-ゔァ-ヴーa---]+', subbedT):
if PBAR is not None:
PBAR.update(len(tItem))
continue
# Create Message
characters, system, user = createContext(fullPromptFlag, subbedT)
# Calculate Estimate
if ESTIMATE:
estimate = countTokens(characters, system, user, history)
totalTokens[0] += estimate[0]
totalTokens[1] += estimate[1]
continue
# Translating
response = translateText(characters, system, user, history, 0.02)
translatedText = response.choices[0].message.content
totalTokens[0] += response.usage.prompt_tokens
totalTokens[1] += response.usage.completion_tokens
# Formatting
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedText, True)
tList[index] = extractedTranslations
if len(tItem) != len(extractedTranslations):
# Mismatch. Try Again
response = translateText(characters, system, user, history, 0.2)
translatedText = response.choices[0].message.content
totalTokens[0] += response.usage.prompt_tokens
totalTokens[1] += response.usage.completion_tokens
# Formatting
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedText, True)
tList[index] = extractedTranslations
if len(tItem) != len(extractedTranslations):
mismatch = True # Just here for breakpoint
# Create History
with LOCK:
if PBAR is not None:
PBAR.update(len(tItem))
if not mismatch:
history = extractedTranslations[-10:] # Update history if we have a list
else:
history = text[-10:]
else:
# Ensure we're passing a single string to extractTranslation
extractedTranslations = extractTranslation(translatedText, False)
tList[index] = extractedTranslations
finalList = combineList(tList, text)
return [finalList, totalTokens]

View file

@ -19,6 +19,7 @@ these values using an .env file, for an example see .env.example')
from modules.rpgmakermvmz import handleMVMZ
from modules.rpgmakerace import handleACE
from modules.csv import handleCSV
from modules.eushully import handleEushully
from modules.alice import handleAlice
from modules.tyrano import handleTyrano
from modules.json import handleJSON
@ -26,7 +27,7 @@ from modules.kansen import handleKansen
from modules.lune import handleLune
from modules.atelier import handleAtelier
from modules.anim import handleAnim
from modules.nscript import handleNScript
from modules.nscript import handleOnscripter
from modules.wolf import handleWOLF
from modules.wolf2 import handleWOLF2
from modules.javascript import handleJavascript
@ -42,6 +43,7 @@ MODULES = [
["RPGMaker MV/MZ", "json", handleMVMZ],
["RPGMaker ACE", "yaml", handleACE],
["CSV (From Translator++)", "csv", handleCSV],
["Eushully", "csv", handleEushully],
["Alice", "txt", handleAlice],
["Tyrano", "ks", handleTyrano],
["JSON", "json", handleJSON],
@ -49,7 +51,7 @@ MODULES = [
["Lune", "json", handleLune],
["Atelier", "txt", handleAtelier],
["Anim", "json", handleAnim],
["NScript", "txt", handleNScript],
["NScript", "txt", handleOnscripter],
["Wolf", "json", handleWOLF],
["Wolf", "txt", handleWOLF2],
["Javascript", "js", handleJavascript],

View file

@ -51,7 +51,7 @@ elif 'gpt-4' in MODEL:
OUTPUTAPICOST = .015
BATCHSIZE = 40
def handleNScript(filename, estimate):
def handleOnscripter(filename, estimate):
global ESTIMATE
ESTIMATE = estimate
@ -77,7 +77,7 @@ def handleNScript(filename, estimate):
else:
try:
with open('translated/' + filename, 'w', encoding='utf8', errors='ignore') as outFile:
with open('translated/' + filename, 'w', encoding='cp932', errors='ignore') as outFile:
start = time.time()
translatedData = openFiles(filename)
@ -120,7 +120,7 @@ def getResultString(translatedData, translationTime, filename):
def openFiles(filename):
with open('files/' + filename, 'r', encoding='cp932') as readFile:
translatedData = parseNScript(readFile, filename)
translatedData = parseOnscripter(readFile, filename)
# Delete lines marked for deletion
finalData = []
@ -131,20 +131,18 @@ def openFiles(filename):
return translatedData
def parseNScript(readFile, filename):
def parseOnscripter(readFile, filename):
totalTokens = [0,0]
totalLines = 0
# Get total for progress bar
# Read File into data
data = readFile.readlines()
totalLines = len(data)
with tqdm(bar_format=BAR_FORMAT, position=POSITION, total=totalLines, leave=LEAVE) as pbar:
# Create Progress Bar
with tqdm(bar_format=BAR_FORMAT, position=POSITION, leave=LEAVE) as pbar:
pbar.desc=filename
pbar.total=totalLines
try:
result = translateNScript(data, pbar, totalLines)
result = translateOnscripter(data, pbar, filename, [])
totalTokens[0] += result[0]
totalTokens[1] += result[1]
except Exception as e:
@ -152,240 +150,108 @@ def parseNScript(readFile, filename):
return [data, totalTokens, e]
return [data, totalTokens, None]
def translateNScript(data, pbar, totalLines):
textHistory = []
batch = []
def translateOnscripter(data, pbar, filename, translatedList):
stringList = []
currentGroup = []
maxHistory = MAXHISTORY
tokens = [0,0]
speaker = ''
insertBool = False
voice = False
global LOCK, ESTIMATE
i = 0
batchStartIndex = 0
while i < len(data):
# Speaker
matchList = re.findall(r'^【\s+(.*?)\s+】$', data[i])
if len(matchList) != 0:
response = getSpeaker(matchList[0])
speaker = response[0]
tokens[0] += response[1][0]
tokens[1] += response[1][1]
data[i] = '>[' + speaker + ']\n'
i += 1
else:
speaker = ''
# Choices
if 'select' in data[i]:
matchList = re.findall(r'\"(.*?)\"', data[i])
if len(matchList) != 0:
originalTextList = matchList
if len(textHistory) > 0:
response = translateGPT(matchList, 'Keep your translation as brief as possible. Previous text for context: ' + textHistory[len(textHistory)-1] + '\n\nReply in the style of a dialogue option.', True)
else:
response = translateGPT(matchList, '\n\nReply in the style of a dialogue option.', True)
translatedTextList = response[0]
tokens[0] += response[1][0]
tokens[1] += response[1][1]
for choice in range(len(translatedTextList)):
translatedText = translatedTextList[choice]
# Remove characters that may break scripts
charList = ['.', '\"', '\\n']
for char in charList:
translatedText = translatedText.replace(char, '')
# Escape all '
translatedText = translatedText.replace('\\', '')
translatedText = translatedText.replace(' ', ' ')
# Set Data
translatedText = data[i].replace(originalTextList[choice], translatedText)
data[i] = translatedText
pbar.update(1)
i += 1
else:
pbar.update(1)
i += 1
# Lines
matchList = re.findall(r'^[一-龠ぁ-ゔァ-ヴーa---9「」『』 >].*', data[i])
if len(matchList) > 0:
currentGroup.append(matchList[0])
if len(data) > i+1:
if speaker == '':
while '\n' != data[i+1] and '' not in data[i+1]:
if insertBool is True:
data[i] = r'\d\n'
pbar.update(1)
i += 1
matchList = re.findall(r'^[一-龠ぁ-ゔァ-ヴーa---9「」『』 >].*', data[i])
if len(matchList) > 0:
currentGroup.append(matchList[0])
else:
while ' ' in data[i+1][0] or '"' in data[i+1] or ')' in data[i+1] or '' in data[i+1]:
if insertBool is True:
data[i] = r'\d\n'
pbar.update(1)
i += 1
matchList = re.findall(r'^[一-龠ぁ-ゔァ-ヴーa---9「」『』 >].*', data[i])
if len(matchList) > 0:
currentGroup.append(matchList[0])
voice = False
speaker = ''
if re.search(r'^caption|^event|^operation_caption|^manual|^mes\s\d+', data[i]):
# Lines
match = re.search(r'="(.+?)"', data[i])
if match == None:
match = re.search(r'mes\s\d+,"(.+?)"', data[i])
if match != None and match.group(1) != '':
originalString = match.group(1)
# Pass 1
if translatedList == []:
# Grab Consecutive Strings
jaString = match.group(1)
# Join up 401 groups for better translation.
if len(currentGroup) > 0:
finalJAString = ' '.join(currentGroup)
oldjaString = finalJAString
# Remove any textwrap
jaString = jaString.replace('\\n', ' ')
# Remove any textwrap
if FIXTEXTWRAP == True:
finalJAString = finalJAString.replace('>', '')
finalJAString = finalJAString.replace('\\', ' ')
# Remove Furigana
furiMatch = re.findall(r'({(.+?)\/(.+?)})', jaString)
if furiMatch:
for match in furiMatch:
jaString = jaString.replace(match[0], match[2])
# Remove Extra Stuff bad for translation.
finalJAString = finalJAString.replace('', '')
finalJAString = finalJAString.replace('', '.')
finalJAString = finalJAString.replace('', '')
finalJAString = finalJAString.replace('', '')
finalJAString = finalJAString.replace('', '-')
finalJAString = finalJAString.replace('', '...')
finalJAString = re.sub(r'(\.{3}\.+)', '...', finalJAString)
finalJAString = finalJAString.replace(' ', ' ')
# Add String
stringList.append(jaString.strip())
# Pass 2
else:
# Get Text
if translatedList:
# Grab and Pop
translatedText = translatedList[0]
translatedList.pop(0)
# Furigana Removal
matchList = re.findall(r'\((.+)/.*?』', finalJAString)
if len(matchList) > 0:
finalJAString = finalJAString.replace(matchList[0][0], matchList[0][1])
# Set to None if empty list
if len(translatedList) <= 0:
translatedList = None
# Add Speaker (If there is one)
if speaker != '':
finalJAString = f'{speaker}: {finalJAString}'
# [Passthrough 1] Pulling From File
if insertBool is False:
# Append to List and Clear Values
batch.append(finalJAString)
speaker = ''
# Translate Batch if Full
if len(batch) == BATCHSIZE:
# Translate
response = translateGPT(batch, textHistory, True)
tokens[0] += response[1][0]
tokens[1] += response[1][1]
translatedBatch = response[0]
textHistory = translatedBatch[-10:]
# Set Values
if len(batch) == len(translatedBatch):
i = batchStartIndex
insertBool = True
# Mismatch
else:
pbar.write(f'Mismatch: {batchStartIndex} - {i}')
MISMATCH.append(batch)
batchStartIndex = i
batch.clear()
# Textwrap
translatedText = textwrap.fill(translatedText, width=WIDTH)
translatedText = translatedText.replace('\n', '\\n')
translatedText = translatedText.replace('\"', '\'')
# Set Data
data[i] = data[i].replace(originalString, translatedText)
i += 1
if insertBool is True:
pbar.update(1)
currentGroup = []
# [Passthrough 2] Setting Data
# Nothing relevant. Skip Line.
else:
# Get Text
translatedText = translatedBatch[0]
translatedText = translatedText.replace('\\"', '\"')
translatedText = translatedText.replace('[', '(')
translatedText = translatedText.replace(']', ')')
# Remove added speaker
translatedText = re.sub(r'^.+?:\s', '', translatedText)
# Textwrap
translatedText = textwrap.fill(translatedText, width=WIDTH)
textList = translatedText.split('\n')
# Set Text
data[i] = r'\d\n'
counter = 0
for line in textList:
# Wordwrap Text
line = textwrap.fill(line, width=WIDTH)
# Set
data.insert(i, '>' + line.strip() + '\n')
counter += 1
i+=1
# Go to new window if too long
if counter >= 4:
data[i-1] = data[i-1].replace('\n', '\\\n')
counter = 0
if '\\' not in data[i-1]:
data[i-1] = data[i-1].replace('\n', '\\\n')
translatedBatch.pop(0)
speaker = ''
currentGroup = []
# If Batch is empty. Move on.
if len(translatedBatch) == 0:
insertBool = False
batchStartIndex = i
batch.clear()
# Nothing relevant. Skip Line.
i += 1
else:
i += 1
if insertBool is False:
pbar.update(1)
# Translate Batch if not empty and EOF
if len(batch) != 0 and i >= len(data):
# Translate
response = translateGPT(batch, textHistory, True)
tokens[0] += response[1][0]
tokens[1] += response[1][1]
translatedBatch = response[0]
textHistory = translatedBatch[-10:]
# EOF
if len(stringList) > 0:
# Set Progress
pbar.total = len(stringList)
pbar.refresh()
# Translate
response = translateGPT(stringList, '', True, pbar, filename)
tokens[0] += response[1][0]
tokens[1] += response[1][1]
translatedList = response[0]
# Set Values
if len(batch) == len(translatedBatch):
i = batchStartIndex
insertBool = True
# Set Strings
if len(stringList) == len(translatedList):
translateOnscripter(data, pbar, filename, translatedList)
# Mismatch
else:
pbar.write(f'Mismatch: {batchStartIndex} - {i}')
MISMATCH.append(batch)
batchStartIndex = i
batch.clear()
currentGroup = []
# Mismatch
else:
with LOCK:
if filename not in MISMATCH:
MISMATCH.append(filename)
return tokens
# Save some money and enter the character before translation
def getSpeaker(speaker):
def getSpeaker(speaker, pbar, filename):
match speaker:
case 'ルイ':
return ['Rui', [0,0]]
case 'チュベロス':
return ['Tuberose', [0,0]]
case 'ファイン':
return ['Fine', [0,0]]
case '':
return ['', [0,0]]
case _:
# Store Speaker
if speaker not in str(NAMESLIST):
response = translateGPT(speaker, 'Reply with only the '+ LANGUAGE +' translation of the NPC name.', False)
response[0] = response[0].title()
response = translateGPT(speaker, 'Reply with only the '+ LANGUAGE +' translation of the NPC name.', False, pbar, filename)
response[0] = response[0].replace("'S", "'s")
speakerList = [speaker, response[0]]
NAMESLIST.append(speakerList)
return response
# Find Speaker
else:
for i in range(len(NAMESLIST)):
@ -393,7 +259,7 @@ def getSpeaker(speaker):
return [NAMESLIST[i][1],[0,0]]
return [speaker,[0,0]]
def subVars(jaString):
jaString = jaString.replace('\u3000', ' ')
@ -403,7 +269,7 @@ def subVars(jaString):
nestedList = set(nestedList)
if len(nestedList) != 0:
for icon in nestedList:
jaString = jaString.replace(icon, '{Nested_' + str(count) + '}')
jaString = jaString.replace(icon, '[Nested_' + str(count) + ']')
count += 1
# Icons
@ -412,7 +278,7 @@ def subVars(jaString):
iconList = set(iconList)
if len(iconList) != 0:
for icon in iconList:
jaString = jaString.replace(icon, '{Ascii_' + str(count) + '}')
jaString = jaString.replace(icon, '[Ascii_' + str(count) + ']')
count += 1
# Colors
@ -421,7 +287,7 @@ def subVars(jaString):
colorList = set(colorList)
if len(colorList) != 0:
for color in colorList:
jaString = jaString.replace(color, '{Color_' + str(count) + '}')
jaString = jaString.replace(color, '[Color_' + str(count) + ']')
count += 1
# Names
@ -430,7 +296,7 @@ def subVars(jaString):
nameList = set(nameList)
if len(nameList) != 0:
for name in nameList:
jaString = jaString.replace(name, '{Noun_' + str(count) + '}')
jaString = jaString.replace(name, '[Noun_' + str(count) + ']')
count += 1
# Variables
@ -439,16 +305,16 @@ def subVars(jaString):
varList = set(varList)
if len(varList) != 0:
for var in varList:
jaString = jaString.replace(var, '{Var_' + str(count) + '}')
jaString = jaString.replace(var, '[Var_' + str(count) + ']')
count += 1
# Formatting
count = 0
formatList = re.findall(r'[\\]+[\w]+\[.+?\]', jaString)
formatList = re.findall(r'[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]', jaString)
formatList = set(formatList)
if len(formatList) != 0:
for var in formatList:
jaString = jaString.replace(var, '{FCode_' + str(count) + '}')
jaString = jaString.replace(var, '[FCode_' + str(count) + ']')
count += 1
# Put all lists in list and return
@ -467,42 +333,42 @@ def resubVars(translatedText, allList):
count = 0
if len(allList[0]) != 0:
for var in allList[0]:
translatedText = translatedText.replace('{Nested_' + str(count) + '}', var)
translatedText = translatedText.replace('[Nested_' + str(count) + ']', var)
count += 1
# Icons
count = 0
if len(allList[1]) != 0:
for var in allList[1]:
translatedText = translatedText.replace('{Ascii_' + str(count) + '}', var)
translatedText = translatedText.replace('[Ascii_' + str(count) + ']', var)
count += 1
# Colors
count = 0
if len(allList[2]) != 0:
for var in allList[2]:
translatedText = translatedText.replace('{Color_' + str(count) + '}', var)
translatedText = translatedText.replace('[Color_' + str(count) + ']', var)
count += 1
# Names
count = 0
if len(allList[3]) != 0:
for var in allList[3]:
translatedText = translatedText.replace('{Noun_' + str(count) + '}', var)
translatedText = translatedText.replace('[Noun_' + str(count) + ']', var)
count += 1
# Vars
count = 0
if len(allList[4]) != 0:
for var in allList[4]:
translatedText = translatedText.replace('{Var_' + str(count) + '}', var)
translatedText = translatedText.replace('[Var_' + str(count) + ']', var)
count += 1
# Formatting
count = 0
if len(allList[5]) != 0:
for var in allList[5]:
translatedText = translatedText.replace('{FCode_' + str(count) + '}', var)
translatedText = translatedText.replace('[FCode_' + str(count) + ']', var)
count += 1
return translatedText
@ -515,19 +381,21 @@ def batchList(input_list, batch_size):
def createContext(fullPromptFlag, subbedT):
characters = 'Game Characters:\n\
水原 (Minahara Yuki) - Female\n\
黒服の男 (Man in Black) - Male\n\
駿河 京也 (Suruga Kyouya) - Male\n\
壱型02 (Type 02) - Monster\n\
エル (El) - Female\n\
'
system = PROMPT + VOCAB if fullPromptFlag else \
f"\
You are an expert Eroge Game translator who translates Japanese text to English.\n\
You are going to be translating text from a videogame.\n\
I will give you lines of text, and you must translate each line to the best of your ability.\n\
You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\
Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\
- You translate everything, including erotic and explicit content.\n\
- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\
- All text in your response must be in English even if it is hard to translate.\n\
- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\
- Maintain any spacing in the translation.\n\
- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\
- `...` can be a part of the dialogue. Translate it as it is.\n\
{VOCAB}\n\
Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\
"
user = f'{subbedT}'
return characters, system, user
@ -550,7 +418,6 @@ def translateText(characters, system, user, history):
response = openai.chat.completions.create(
temperature=0.1,
frequency_penalty=0.1,
presence_penalty=0.1,
model=MODEL,
messages=msg,
)
@ -570,17 +437,34 @@ def cleanTranslatedText(translatedText, varResponse):
for target, replacement in placeholders.items():
translatedText = translatedText.replace(target, replacement)
# Elongate Long Dashes (Since GPT Ignores them...)
translatedText = elongateCharacters(translatedText)
translatedText = resubVars(translatedText, varResponse[1])
return [line for line in translatedText.replace('\\n', '\n').split('\n') if line]
return translatedText
def elongateCharacters(text):
# Define a pattern to match one character followed by one or more `ー` characters
# Using a positive lookbehind assertion to capture the preceding character
pattern = r'(?<=(.))ー+'
# Define a replacement function that elongates the captured character
def repl(match):
char = match.group(1) # The character before the ー sequence
count = len(match.group(0)) - 1 # Number of ー characters
return char * count # Replace ー sequence with the character repeated
# Use re.sub() to replace the pattern in the text
return re.sub(pattern, repl, text)
def extractTranslation(translatedTextList, is_list):
pattern = r'`?<Line(\d+)>([\\]*.*?[\\]*?)<\/?Line\d+>`?'
pattern = r'`?<Line\d+>([\\]*.*?[\\]*?)<\/?Line\d+>`?'
# If it's a batch (i.e., list), extract with tags; otherwise, return the single item.
if is_list:
return [re.findall(pattern, line)[0][1] for line in translatedTextList if re.search(pattern, line)]
matchList = re.findall(pattern, translatedTextList)
return matchList
else:
matchList = re.findall(pattern, translatedTextList)
return matchList[0][1] if matchList else translatedTextList
return matchList[0][0] if matchList else translatedTextList
def countTokens(characters, system, user, history):
inputTotalTokens = 0
@ -598,7 +482,7 @@ def countTokens(characters, system, user, history):
inputTotalTokens += len(enc.encode(user))
# Output
outputTotalTokens += round(len(enc.encode(user))*3)
outputTotalTokens += round(len(enc.encode(user))*2)
return [inputTotalTokens, outputTotalTokens]
@ -608,7 +492,8 @@ def combineList(tlist, text):
return tlist[0]
@retry(exceptions=Exception, tries=5, delay=5)
def translateGPT(text, history, fullPromptFlag):
def translateGPT(text, history, fullPromptFlag, pbar, filename):
mismatch = False
totalTokens = [0, 0]
if isinstance(text, list):
tList = batchList(text, BATCHSIZE)
@ -619,7 +504,7 @@ def translateGPT(text, history, fullPromptFlag):
# Before sending to translation, if we have a list of items, add the formatting
if isinstance(tItem, list):
payload = '\n'.join([f'`<Line{i}>{item}</Line{i}>`' for i, item in enumerate(tItem)])
payload = payload.replace('``', '`Placeholder Text`')
payload = re.sub(r'(<Line\d+)(><)(\/Line\d+>)', r'\1>Placeholder Text<\3', payload)
varResponse = subVars(payload)
subbedT = varResponse[0]
else:
@ -647,16 +532,34 @@ def translateGPT(text, history, fullPromptFlag):
totalTokens[1] += response.usage.completion_tokens
# Formatting
translatedTextList = cleanTranslatedText(translatedText, varResponse)
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedTextList, True)
tList[index] = extractedTranslations
if len(tItem) != len(translatedTextList):
mismatch = True # Just here so breakpoint can be set
history = extractedTranslations[-10:] # Update history if we have a list
extractedTranslations = extractTranslation(translatedText, True)
if len(tItem) != len(extractedTranslations):
# Mismatch. Try Again
response = translateText(characters, system, user, history)
translatedText = response.choices[0].message.content
totalTokens[0] += response.usage.prompt_tokens
totalTokens[1] += response.usage.completion_tokens
# Formatting
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedText, True)
if len(tItem) == len(extractedTranslations):
tList[index] = extractedTranslations
else:
MISMATCH.append(filename)
else:
tList[index] = extractedTranslations
# Create History
history = tList[index] # Update history if we have a list
pbar.update(len(tList[index]))
else:
# Ensure we're passing a single string to extractTranslation
extractedTranslations = extractTranslation('\n'.join(translatedTextList), False)
extractedTranslations = extractTranslation(translatedText, False)
tList[index] = extractedTranslations
finalList = combineList(tList, text)

View file

@ -44,5 +44,21 @@ ME 音量 (ME Volume)
アスカロン (Ascalon)
# Other
パパ (papa)
悪魔 (Devil)
上級悪魔 (Arch Devil)
歪魔 (Distorted Devil)
魔神 (Demon)
魔人 (Majin)
睡魔 (Mare)
淫魔 (Succubus)
天使 (Angel)
大天使 (Archangel)
権天使 (Ruler)
能天使 (Power)
力天使 (Virtue)
主天使 (Dominion)
智天使 (Cherub)
飛天魔 (Nephilim)
堕天使 (Fallen Angel)
鬼 (Oni)
```