diff --git a/modules/kansen.py b/modules/kansen.py
index 950ffa5..0936359 100644
--- a/modules/kansen.py
+++ b/modules/kansen.py
@@ -1,55 +1,58 @@
-from concurrent.futures import ThreadPoolExecutor, as_completed
-import os
+# Libraries
+import json, os, re, textwrap, threading, time, traceback, tiktoken, openai
from pathlib import Path
-import re
-import textwrap
-import threading
-import time
-import traceback
-import tiktoken
-
from colorama import Fore
from dotenv import load_dotenv
-import openai
from retry import retry
from tqdm import tqdm
-#Globals
+# Open AI
load_dotenv()
if os.getenv('api').replace(' ', '') != '':
openai.api_base = os.getenv('api')
-
openai.organization = os.getenv('org')
openai.api_key = os.getenv('key')
+
+#Globals
MODEL = os.getenv('model')
TIMEOUT = int(os.getenv('timeout'))
-LANGUAGE=os.getenv('language').capitalize()
-
-APICOST = .002 # Depends on the model https://openai.com/pricing
+LANGUAGE = os.getenv('language').capitalize()
PROMPT = Path('prompt.txt').read_text(encoding='utf-8')
-THREADS = int(os.getenv('threads')) # For GPT4 rate limit will be hit if you have more than 1 thread.
+THREADS = int(os.getenv('threads'))
LOCK = threading.Lock()
WIDTH = int(os.getenv('width'))
LISTWIDTH = int(os.getenv('listWidth'))
+NOTEWIDTH = 70
MAXHISTORY = 10
ESTIMATE = ''
-TOTALCOST = 0
-TOKENS = 0
-TOTALTOKENS = 0
+TOKENS = [0, 0]
NAMESLIST = []
+NAMES = False # Output a list of all the character names found
+BRFLAG = False # If the game uses
instead
+FIXTEXTWRAP = True # Overwrites textwrap
+IGNORETLTEXT = False # Ignores all translated text.
+MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong)
#tqdm Globals
BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}'
-POSITION=0
-LEAVE=False
+POSITION = 0
+LEAVE = False
-# Flags
-NAMES = False # Output a list of all the character names found
-FIXTEXTWRAP = True
-IGNORETLTEXT = True
+# Pricing - Depends on the model https://openai.com/pricing
+# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request
+# If you are getting a MISMATCH LENGTH error, lower the batch size.
+if 'gpt-3.5' in MODEL:
+ INPUTAPICOST = .002
+ OUTPUTAPICOST = .002
+ BATCHSIZE = 10
+elif 'gpt-4' in MODEL:
+ INPUTAPICOST = .01
+ OUTPUTAPICOST = .03
+ BATCHSIZE = 50
def handleKansen(filename, estimate):
- global ESTIMATE, TOKENS, TOTALTOKENS, TOTALCOST
+ global ESTIMATE
+ totalTokens = [0,0]
ESTIMATE = estimate
if estimate:
@@ -60,10 +63,17 @@ def handleKansen(filename, estimate):
end = time.time()
tqdm.write(getResultString(translatedData, end - start, filename))
with LOCK:
- TOTALCOST += translatedData[1] * .001 * APICOST
- TOTALTOKENS += translatedData[1]
+ totalTokens[0] += translatedData[1][0]
+ totalTokens[1] += translatedData[1][1]
- return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL')
+ # Print Total
+ totalString = getResultString(['', totalTokens, None], end - start, 'TOTAL')
+
+ # Print any errors on maps
+ if len(MISMATCH) > 0:
+ return totalString + Fore.RED + f'\nMismatch Errors: {MISMATCH}' + Fore.RESET
+ else:
+ return totalString
else:
try:
@@ -72,17 +82,41 @@ def handleKansen(filename, estimate):
translatedData = openFiles(filename)
# Print Result
- outFile.writelines(translatedData[0])
end = time.time()
+ outFile.writelines(translatedData[0])
tqdm.write(getResultString(translatedData, end - start, filename))
with LOCK:
- TOTALCOST += translatedData[1] * .001 * APICOST
- TOTALTOKENS += translatedData[1]
+ totalTokens[0] += translatedData[1][0]
+ totalTokens[1] += translatedData[1][1]
except Exception as e:
traceback.print_exc()
return 'Fail'
- return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL')
+ return getResultString(['', totalTokens, None], end - start, 'TOTAL')
+
+def getResultString(translatedData, translationTime, filename):
+ # File Print String
+ totalTokenstring =\
+ Fore.YELLOW +\
+ '[Input: ' + str(translatedData[1][0]) + ']'\
+ '[Output: ' + str(translatedData[1][1]) + ']'\
+ '[Cost: ${:,.4f}'.format((translatedData[1][0] * .001 * INPUTAPICOST) +\
+ (translatedData[1][1] * .001 * OUTPUTAPICOST)) + ']'
+ timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]'
+
+ if translatedData[2] == None:
+ # Success
+ return filename + ': ' + totalTokenstring + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET
+
+ else:
+ # Fail
+ try:
+ raise translatedData[2]
+ except Exception as e:
+ traceback.print_exc()
+ errorString = str(e) + Fore.RED
+ return filename + ': ' + totalTokenstring + timeString + Fore.RED + u' \u2717 ' +\
+ errorString + Fore.RESET
def openFiles(filename):
with open('files/' + filename, 'r', encoding='cp932') as readFile:
@@ -98,7 +132,7 @@ def openFiles(filename):
return translatedData
def parseTyrano(readFile, filename):
- totalTokens = 0
+ totalTokens = [0,0]
totalLines = 0
# Get total for progress bar
@@ -110,25 +144,27 @@ def parseTyrano(readFile, filename):
pbar.total=totalLines
try:
- totalTokens += translateTyrano(data, pbar)
+ result = translateTyrano(data, pbar, totalLines)
+ totalTokens[0] += result[0]
+ totalTokens[1] += result[1]
except Exception as e:
traceback.print_exc()
return [data, totalTokens, e]
return [data, totalTokens, None]
-def translateTyrano(data, pbar):
+def translateTyrano(data, pbar, totalLines):
textHistory = []
- maxHistory = MAXHISTORY
- tokens = 0
+ batch = []
currentGroup = []
- syncIndex = 0
+ maxHistory = MAXHISTORY
+ tokens = [0,0]
speaker = ''
+ insertBool = False
global LOCK, ESTIMATE
+ i = 0
+ batchStartIndex = 0
- for i in range(len(data)):
- if syncIndex > i:
- i = syncIndex
-
+ while i < len(data):
# Speaker
if '[ns]' in data[i]:
matchList = re.findall(r'\[ns\](.+?)\[', data[i])
@@ -144,11 +180,11 @@ def translateTyrano(data, pbar):
elif '[eval exp="f.seltext' in data[i]:
matchList = re.findall(r'\[eval exp=.+?\'(.+)\'', data[i])
if len(matchList) != 0:
+ originalText = matchList[0]
if len(textHistory) > 0:
- originalText = matchList[0]
- response = translateGPT(matchList[0], 'Past Translated Text: ' + textHistory[len(textHistory)-1] + '\n\nReply in the style of a dialogue option.', True)
+ response = translateGPT(matchList[0], 'Keep your translation as brief as possible. Previous text for context: ' + textHistory[len(textHistory)-1] + '\n\nReply in the style of a dialogue option.', False)
else:
- response = translateGPT(matchList[0], '', False)
+ response = translateGPT(matchList[0], '\n\nReply in the style of a dialogue option.', False)
translatedText = response[0]
tokens += response[1]
@@ -166,27 +202,27 @@ def translateTyrano(data, pbar):
data[i] = translatedText
# Lines
- matchList = re.findall(r'(.+?)\[r\]$', data[i])
+ matchList = re.findall(r'(.+?)\[[rpcms]+\]$', data[i])
if len(matchList) > 0:
matchList[0] = matchList[0].replace('「', '')
matchList[0] = matchList[0].replace('」', '')
currentGroup.append(matchList[0])
if len(data) > i+1:
while '[r]' in data[i+1]:
- data[i] = '\d\n' # \d Marks line for deletion
+ if insertBool is True:
+ data[i] = '\d\n'
+ pbar.update(1)
i += 1
matchList = re.findall(r'(.+?)\[r\]', data[i])
if len(matchList) > 0:
- matchList[0] = matchList[0].replace('「', '')
- matchList[0] = matchList[0].replace('」', '')
currentGroup.append(matchList[0])
while '[pcms]' in data[i+1]:
- data[i] = '\d\n'
+ if insertBool is True:
+ data[i] = '\d\n'
+ pbar.update(1)
i += 1
matchList = re.findall(r'(.+?)\[pcms\]', data[i])
if len(matchList) > 0:
- matchList[0] = matchList[0].replace('「', '')
- matchList[0] = matchList[0].replace('」', '')
currentGroup.append(matchList[0])
# Join up 401 groups for better translation.
if len(currentGroup) > 0:
@@ -197,197 +233,123 @@ def translateTyrano(data, pbar):
if FIXTEXTWRAP == True:
finalJAString = re.sub(r'[r]', ' ', finalJAString)
- #Check Speaker
- if speaker == '':
- response = translateGPT(finalJAString, 'Previous Dialogue: ' + '\n\n'.join(textHistory), True)
- tokens += response[1]
- translatedText = response[0]
- textHistory.append('\"' + translatedText + '\"')
+ # Add Speaker (If there is one)
+ if speaker != '':
+ finalJAString = f'{speaker}: {finalJAString}'
+
+ # [Passthrough 1] Pulling From File
+ if insertBool is False:
+ # Append to List and Clear Values
+ batch.append(finalJAString)
+
+ # Translate Batch if Full
+ if len(batch) == BATCHSIZE:
+ # Translate
+ response = translateGPT(batch, textHistory, True)
+ tokens[0] += response[1][0]
+ tokens[1] += response[1][1]
+ translatedBatch = response[0]
+ textHistory = translatedBatch[-10:]
+
+ # Set Values
+ if len(batch) == len(translatedBatch):
+ i = batchStartIndex
+ insertBool = True
+
+ # Mismatch
+ else:
+ pbar.write(f'Mismatch: {batchStartIndex} - {i}')
+ MISMATCH.append(batch)
+ batchStartIndex = i
+ batch.clear()
+
+ i += 1
+ if insertBool is True:
+ pbar.update(1)
+ currentGroup = []
+
+ # [Passthrough 2] Setting Data
else:
- response = translateGPT(speaker + ': ' + finalJAString, 'Previous Dialogue: ' + '\n\n'.join(textHistory), True)
- tokens += response[1]
- translatedText = response[0]
- textHistory.append('\"' + translatedText + '\"')
+ # Get Text
+ translatedText = translatedBatch[0]
- # Remove added speaker
- translatedText = re.sub(r'^.+:\s?', '', translatedText)
+ # Remove added speaker and quotes
+ translatedText = re.sub(r'^.+?:\s', '', translatedText)
- # Set Data
- translatedText = translatedText.replace('ッ', '')
- translatedText = translatedText.replace('っ', '')
- translatedText = translatedText.replace('ー', '')
- translatedText = translatedText.replace('\"', '')
- translatedText = translatedText.replace('[', '')
- translatedText = translatedText.replace(']', '')
-
- # Format Text
- matchList = re.findall(r'(.+?[)\.\?\!)。・]+)', translatedText)
- translatedText = re.sub(r'(.+?[)\.\?\!)。・]+)', '', translatedText)
-
- # Combine Lists
- for k in range(len(matchList)):
- matchList[k] = matchList[k].strip()
- j=0
- while(len(matchList) > j+1):
- while len(matchList[j]) < 30 and len(matchList) > j:
- matchList[j:j+2] = [' '.join(matchList[j:j+2])]
- if len(matchList) == j+1:
- matchList[j] = matchList[j] + ' ' + translatedText
- translatedText = ''
- break
- j+=1
-
- if len(matchList) > 0:
+ # Textwrap
+ translatedText = translatedText.replace('\"', '\\"')
+ translatedText = textwrap.fill(translatedText, width=WIDTH)
+ textList = translatedText.split('\n')
+
+ # Set Text
data[i] = '\d\n'
- for line in matchList:
+ for line in textList:
# Wordwrap Text
if '[r]' not in line:
line = textwrap.fill(line, width=WIDTH)
line = line.replace('\n', '[r]')
# Set
- data.insert(i, line.strip() + '[l][er]\n')
+ data.insert(i, line.strip() + '[r]\n')
i+=1
- data[i-1] = data[i-1].replace('[l][er]', '[pcms]')
- # else:
- # print ('No Matches')
- if translatedText != '':
- # Wordwrap Text
- if '[r]' not in translatedText:
- translatedText = textwrap.fill(translatedText, width=WIDTH)
- translatedText = translatedText.replace('\n', '[r]')
+ data[i-1] = data[i-1].replace('[r]', '[pcms]')
+ translatedBatch.pop(0)
- # Set Backup
- data[i] = translatedText.strip() + '[l][er]\n'
+ # If Batch is empty. Move on.
+ if len(translatedBatch) == 0:
+ insertBool = False
+ batchStartIndex = i
+ batch.clear()
- # Keep textHistory list at length maxHistory
- if len(textHistory) > maxHistory:
- textHistory.pop(0)
- currentGroup = []
- speaker = ''
-
- matchList = re.findall(r'(.+?)\[pcms\]$', data[i])
- if len(matchList) > 0:
- matchList[0] = matchList[0].replace('「', '')
- matchList[0] = matchList[0].replace('」', '')
- finalJAString = matchList[0]
-
- # Remove any textwrap
- if FIXTEXTWRAP == True:
- finalJAString = finalJAString.replace('[r]', ' ')
-
- #Check Speaker
- if speaker == '':
- response = translateGPT(finalJAString, 'Previous Dialogue: ' + '\n\n'.join(textHistory), True)
- tokens += response[1]
- translatedText = response[0]
- textHistory.append('\"' + translatedText + '\"')
- else:
- response = translateGPT(speaker + ': ' + finalJAString, 'Previous Dialogue: ' + '\n\n'.join(textHistory), True)
- tokens += response[1]
- translatedText = response[0]
- textHistory.append('\"' + translatedText + '\"')
-
- # Remove added speaker
- translatedText = re.sub(r'^.+:\s?', '', translatedText)
-
- # Set Data
- translatedText = translatedText.replace('ッ', '')
- translatedText = translatedText.replace('っ', '')
- translatedText = translatedText.replace('ー', '')
- translatedText = translatedText.replace('\"', '')
- translatedText = translatedText.replace('[', '')
- translatedText = translatedText.replace(']', '')
-
- # Format Text
- matchList = re.findall(r'(.+?[)\.\?\!)。・]+)', translatedText)
- translatedText = re.sub(r'(.+?[)\.\?\!)。・]+)', '', translatedText)
-
- # Get rid of whitespace for each item and add wordwrap
- for k in range(len(matchList)):
- matchList[k] = matchList[k].strip()
-
- # Combine Sentences with a max limit (Wordwrap basically)
- j=0
- while(len(matchList) > j+1):
- while len(matchList[j]) < 30 and len(matchList) > j:
- matchList[j:j+2] = [' '.join(matchList[j:j+2])]
- if len(matchList) == j+1:
- matchList[j] = matchList[j] + ' ' + translatedText
- translatedText = ''
- break
- j+=1
-
- # Set Data
- if len(matchList) > 0:
- data[i] = '\d\n'
- for line in matchList:
- # Wordwrap Text
- if '[r]' not in line:
- line = textwrap.fill(line, width=WIDTH)
- line = line.replace('\n', '[r]')
-
- # Set
- data.insert(i, line.strip() + '[l][er]\n')
- i+=1
- # Set last line as [pcms] instead of [r]
- data[i-1] = data[i-1].replace('[l][er]', '[pcms]')
- # else:
- # print ('No Matches')
- if translatedText != '':
- # Wordwrap Text
- if '[r]' not in translatedText:
- translatedText = textwrap.fill(translatedText, width=WIDTH)
- translatedText = translatedText.replace('\n', '[r]')
-
- # Set Backup
- data[i] = translatedText.strip() + '[l][er]\n'
-
- # Keep textHistory list at length maxHistory
- if len(textHistory) > maxHistory:
- textHistory.pop(0)
- currentGroup = []
- speaker = ''
-
- currentGroup = []
- pbar.update(1)
- if len(data) > i+1:
- syncIndex = i+1
+ # Nothing relevant. Skip Line.
else:
- break
+ i += 1
+ if insertBool is True:
+ pbar.update(1)
+ # Translate Batch if not empty and EOF
+ if len(batch) != 0 and i >= len(data):
+ # Translate
+ response = translateGPT(batch, textHistory, True)
+ tokens[0] += response[1][0]
+ tokens[1] += response[1][1]
+ translatedBatch = response[0]
+ textHistory = translatedBatch[-10:]
+
+ # Set Values
+ if len(batch) == len(translatedBatch):
+ i = batchStartIndex
+ insertBool = True
+
+ # Mismatch
+ else:
+ pbar.write(f'Mismatch: {batchStartIndex} - {i}')
+ MISMATCH.append(batch)
+ batchStartIndex = i
+ batch.clear()
+
+ currentGroup = []
return tokens
-
-def getResultString(translatedData, translationTime, filename):
- # File Print String
- tokenString = Fore.YELLOW + '[' + str(translatedData[1]) + \
- ' Tokens/${:,.4f}'.format(translatedData[1] * .001 * APICOST) + ']'
- timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]'
-
- if translatedData[2] == None:
- # Success
- return filename + ': ' + tokenString + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET
-
- else:
- # Fail
- try:
- raise translatedData[2]
- except Exception as e:
- traceback.print_exc()
- errorString = str(e) + Fore.RED
- return filename + ': ' + tokenString + timeString + Fore.RED + u' \u2717 ' +\
- errorString + Fore.RESET
def subVars(jaString):
jaString = jaString.replace('\u3000', ' ')
+ # Nested
+ count = 0
+ nestedList = re.findall(r'[\\]+[\w]+\[[\\]+[\w]+\[[0-9]+\]\]', jaString)
+ nestedList = set(nestedList)
+ if len(nestedList) != 0:
+ for icon in nestedList:
+ jaString = jaString.replace(icon, '{Nested_' + str(count) + '}')
+ count += 1
+
# Icons
count = 0
- iconList = re.findall(r'[\\]+[iIkKwW]+\[[0-9]+\]', jaString)
+ iconList = re.findall(r'[\\]+[iIkKwWaA]+\[[0-9]+\]', jaString)
iconList = set(iconList)
if len(iconList) != 0:
for icon in iconList:
- jaString = jaString.replace(icon, '[Icon' + str(count) + ']')
+ jaString = jaString.replace(icon, '{Ascii_' + str(count) + '}')
count += 1
# Colors
@@ -396,16 +358,16 @@ def subVars(jaString):
colorList = set(colorList)
if len(colorList) != 0:
for color in colorList:
- jaString = jaString.replace(color, '[Color' + str(count) + ']')
+ jaString = jaString.replace(color, '{Color_' + str(count) + '}')
count += 1
# Names
count = 0
- nameList = re.findall(r'[\\]+[nN]\[[0-9]+\]', jaString)
+ nameList = re.findall(r'[\\]+[nN]\[.+?\]+', jaString)
nameList = set(nameList)
if len(nameList) != 0:
for name in nameList:
- jaString = jaString.replace(name, '[Name' + str(count) + ']')
+ jaString = jaString.replace(name, '{Noun_' + str(count) + '}')
count += 1
# Variables
@@ -414,11 +376,20 @@ def subVars(jaString):
varList = set(varList)
if len(varList) != 0:
for var in varList:
- jaString = jaString.replace(var, '[Var' + str(count) + ']')
+ jaString = jaString.replace(var, '{Var_' + str(count) + '}')
+ count += 1
+
+ # Formatting
+ count = 0
+ formatList = re.findall(r'[\\]+[\w]+\[.+?\]', jaString)
+ formatList = set(formatList)
+ if len(formatList) != 0:
+ for var in formatList:
+ jaString = jaString.replace(var, '{FCode_' + str(count) + '}')
count += 1
# Put all lists in list and return
- allList = [iconList, colorList, nameList, varList]
+ allList = [nestedList, iconList, colorList, nameList, varList, formatList]
return [jaString, allList]
def resubVars(translatedText, allList):
@@ -429,117 +400,203 @@ def resubVars(translatedText, allList):
text = match.strip()
translatedText = translatedText.replace(match, text)
- # Icons
+ # Nested
count = 0
if len(allList[0]) != 0:
for var in allList[0]:
- translatedText = translatedText.replace('[Icon' + str(count) + ']', var)
+ translatedText = translatedText.replace('{Nested_' + str(count) + '}', var)
+ count += 1
+
+ # Icons
+ count = 0
+ if len(allList[1]) != 0:
+ for var in allList[1]:
+ translatedText = translatedText.replace('{Ascii_' + str(count) + '}', var)
count += 1
# Colors
count = 0
- if len(allList[1]) != 0:
- for var in allList[1]:
- translatedText = translatedText.replace('[Color' + str(count) + ']', var)
+ if len(allList[2]) != 0:
+ for var in allList[2]:
+ translatedText = translatedText.replace('{Color_' + str(count) + '}', var)
count += 1
# Names
count = 0
- if len(allList[2]) != 0:
- for var in allList[2]:
- translatedText = translatedText.replace('[Name' + str(count) + ']', var)
+ if len(allList[3]) != 0:
+ for var in allList[3]:
+ translatedText = translatedText.replace('{Noun_' + str(count) + '}', var)
count += 1
# Vars
count = 0
- if len(allList[3]) != 0:
- for var in allList[3]:
- translatedText = translatedText.replace('[Var' + str(count) + ']', var)
+ if len(allList[4]) != 0:
+ for var in allList[4]:
+ translatedText = translatedText.replace('{Var_' + str(count) + '}', var)
+ count += 1
+
+ # Formatting
+ count = 0
+ if len(allList[5]) != 0:
+ for var in allList[5]:
+ translatedText = translatedText.replace('{FCode_' + str(count) + '}', var)
count += 1
- # Remove Color Variables Spaces
- # if '\\c' in translatedText:
- # translatedText = re.sub(r'\s*(\\+c\[[1-9]+\])\s*', r' \1', translatedText)
- # translatedText = re.sub(r'\s*(\\+c\[0+\])', r'\1', translatedText)
return translatedText
-@retry(exceptions=Exception, tries=5, delay=5)
-def translateGPT(t, history, fullPromptFlag):
- # If ESTIMATE is True just count this as an execution and return.
- if ESTIMATE:
- enc = tiktoken.encoding_for_model(MODEL)
- tokens = len(enc.encode(t)) * 2 + len(enc.encode(str(history))) + len(enc.encode(PROMPT))
- return (t, tokens)
-
- # Sub Vars
- varResponse = subVars(t)
- subbedT = varResponse[0]
+def batchList(input_list, batch_size):
+ if not isinstance(batch_size, int) or batch_size <= 0:
+ raise ValueError("batch_size must be a positive integer")
+
+ return [input_list[i:i + batch_size] for i in range(0, len(input_list), batch_size)]
- # If there isn't any Japanese in the text just skip
- if not re.search(r'[一-龠]+|[ぁ-ゔ]+|[ァ-ヴ]+|[\uFF00-\uFFEF]', subbedT):
- return(t, 0)
+def createContext(fullPromptFlag, subbedT):
+ characters = 'Game Characters:\
+ 護 == Name: Mamoru - Male\
+ 神代 一騎 == Last Name: Kamishiro, First Name: Ikki - Male\
+ 神代 琴音 == Last Name: Kamishiro, First Name: Kotone - Female\
+ 神代 莉々子 == Last Name: Kamishiro, First Name: Ririko - Female\
+ 神代 紗夜 == Last Name: Kamishiro, First Name: Saya - Female\
+ 篠原漣 == Last Name: Shinohara, First Name: Ren - Male\
+ 藪井 == Name: Yabui - Male\
+ 舟木 == Name: Funaki - Male\
+ 貞二 == Name: Jouji - Male\
+ 兼田 響子 == Last Name: Kaneda, First Name: Kyouko - Female\
+ 兼田 真人 == Last Name: Kaneda, First Name: Masato - Male\
+ 小出 == Name: Koide - Male\
+ 進士 == Name: Shinji - Male\
+ 雪乃 == Name: Yukino - Female'
+
+ system = PROMPT if fullPromptFlag else \
+ f'Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`'
+ user = f'{subbedT}'
+ return characters, system, user
+
+def translateText(characters, system, user, history):
+ # Prompt
+ msg = [{"role": "system", "content": system + characters}]
# Characters
- context = '```\
- Game Characters:\
- Character: 池ノ上 拓海 == Ikenoue Takumi - Gender: Male\
- Character: 福永 こはる == Fukunaga Koharu - Gender: Female\
- Character: 神泉 理央 == Kamiizumi Rio - Gender: Female\
- Character: 吉祥寺 アリサ == Kisshouji Arisa - Gender: Female\
- Character: 久我 友里子 == Kuga Yuriko - Gender: Female\
- ```'
+ msg.append({"role": "system", "content": characters})
- # Prompt
- if fullPromptFlag:
- system = PROMPT
- user = 'Line to Translate = ' + subbedT
- else:
- system = 'Output ONLY the '+ LANGUAGE +' translation in the following format: `Translation: <'+ LANGUAGE.upper() +'_TRANSLATION>`'
- user = 'Line to Translate = ' + subbedT
-
- # Create Message List
- msg = []
- msg.append({"role": "system", "content": system})
- msg.append({"role": "user", "content": context})
+ # History
if isinstance(history, list):
- for line in history:
- msg.append({"role": "user", "content": line})
+ msg.extend([{"role": "assistant", "content": h} for h in history])
else:
- msg.append({"role": "user", "content": history})
- msg.append({"role": "user", "content": user})
-
- response = openai.ChatCompletion.create(
+ msg.append({"role": "assistant", "content": history})
+
+ # Content to TL
+ msg.append({"role": "user", "content": f'{user}'})
+ response = openai.chat.completions.create(
temperature=0.1,
- frequency_penalty=0.2,
- presence_penalty=0.2,
+ top_p = 0.2,
+ frequency_penalty=0.1,
+ presence_penalty=0.1,
model=MODEL,
messages=msg,
- request_timeout=TIMEOUT,
)
+ return response
- # Save Translated Text
- translatedText = response.choices[0].message.content
- tokens = response.usage.total_tokens
+def cleanTranslatedText(translatedText, varResponse):
+ placeholders = {
+ f'{LANGUAGE} Translation: ': '',
+ 'Translation: ': '',
+ 'っ': '',
+ '〜': '~',
+ 'ー': '-',
+ 'ッ': ''
+ # Add more replacements as needed
+ }
+ for target, replacement in placeholders.items():
+ translatedText = translatedText.replace(target, replacement)
- # Resub Vars
translatedText = resubVars(translatedText, varResponse[1])
+ return [line for line in translatedText.split('\\n') if line]
- # Remove Placeholder Text
- translatedText = translatedText.replace(LANGUAGE +' Translation: ', '')
- translatedText = translatedText.replace('Translation: ', '')
- translatedText = translatedText.replace('Line to Translate = ', '')
- translatedText = translatedText.replace('Translation = ', '')
- translatedText = translatedText.replace('Translate = ', '')
- translatedText = translatedText.replace(LANGUAGE +' Translation:', '')
- translatedText = translatedText.replace('Translation:', '')
- translatedText = translatedText.replace('Line to Translate =', '')
- translatedText = translatedText.replace('Translation =', '')
- translatedText = translatedText.replace('Translate =', '')
- translatedText = re.sub(r'Note:.*', '', translatedText)
- translatedText = translatedText.replace('っ', '')
-
- # Return Translation
- if len(translatedText) > 15 * len(t) or "I'm sorry, but I'm unable to assist with that translation" in translatedText:
- raise Exception
+def extractTranslation(translatedTextList, is_list):
+ pattern = r'[\\]*`?(.*?)[\\]*?`?'
+ # If it's a batch (i.e., list), extract with tags; otherwise, return the single item.
+ if is_list:
+ return [re.findall(pattern, line)[0][1] for line in translatedTextList if re.search(pattern, line)]
else:
- return [translatedText, tokens]
+ matchList = re.findall(pattern, translatedTextList)
+ return matchList[0][1] if matchList else translatedTextList
+
+def countTokens(characters, system, user, history):
+ inputTotalTokens = 0
+ outputTotalTokens = 0
+ enc = tiktoken.encoding_for_model(MODEL)
+
+ # Input
+ if isinstance(history, list):
+ for line in history:
+ inputTotalTokens += len(enc.encode(line))
+ else:
+ inputTotalTokens += len(enc.encode(history))
+ inputTotalTokens += len(enc.encode(system))
+ inputTotalTokens += len(enc.encode(characters))
+ inputTotalTokens += len(enc.encode(user))
+
+ # Output
+ outputTotalTokens += round(len(enc.encode(user))/1.7)
+
+ return [inputTotalTokens, outputTotalTokens]
+
+def combineList(tlist, text):
+ if isinstance(text, list):
+ return [t for sublist in tlist for t in sublist]
+ return tlist[0]
+
+@retry(exceptions=Exception, tries=5, delay=5)
+def translateGPT(text, history, fullPromptFlag):
+ totalTokens = [0, 0]
+ if isinstance(text, list):
+ tList = batchList(text, BATCHSIZE)
+ else:
+ tList = [text]
+
+ for index, tItem in enumerate(tList):
+ # Before sending to translation, if we have a list of items, add the formatting
+ if isinstance(tItem, list):
+ payload = '\\n'.join([f'\`{item}\`' for i, item in enumerate(tItem)])
+ varResponse = subVars(payload)
+ subbedT = varResponse[0]
+ else:
+ varResponse = subVars(tItem)
+ subbedT = varResponse[0]
+
+ # Things to Check before starting translation
+ if not re.search(r'[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+', subbedT):
+ continue
+
+ # Create Message
+ characters, system, user = createContext(fullPromptFlag, subbedT)
+
+ # Calculate Estimate
+ if ESTIMATE:
+ estimate = countTokens(characters, system, user, history)
+ totalTokens[0] += estimate[0]
+ totalTokens[1] += estimate[1]
+ continue
+
+ # Translating
+ response = translateText(characters, system, user, history)
+ translatedText = response.choices[0].message.content
+ totalTokens[0] += response.usage.prompt_tokens
+ totalTokens[1] += response.usage.completion_tokens
+
+ # Formatting
+ translatedTextList = cleanTranslatedText(translatedText, varResponse)
+ if isinstance(tItem, list):
+ extractedTranslations = extractTranslation(translatedTextList, True)
+ tList[index] = extractedTranslations
+ if len(tList[index]) != len(translatedTextList):
+ print('Test')
+ history = extractedTranslations[-10:] # Update history if we have a list
+ else:
+ # Ensure we're passing a single string to extractTranslation
+ extractedTranslations = extractTranslation('\\n'.join(translatedTextList), False)
+ tList[index] = extractedTranslations
+
+ finalList = combineList(tList, text)
+ return [finalList, totalTokens]
\ No newline at end of file
diff --git a/modules/rpgmakermvmz.py b/modules/rpgmakermvmz.py
index a532e0e..d24430d 100644
--- a/modules/rpgmakermvmz.py
+++ b/modules/rpgmakermvmz.py
@@ -23,7 +23,7 @@ THREADS = int(os.getenv('threads'))
LOCK = threading.Lock()
WIDTH = int(os.getenv('width'))
LISTWIDTH = int(os.getenv('listWidth'))
-NOTEWIDTH = 70
+NOTEWIDTH = int(os.getenv('noteWidth'))
MAXHISTORY = 10
ESTIMATE = ''
TOKENS = [0, 0]
@@ -1436,6 +1436,19 @@ def searchCodes(page, pbar, fillList, filename):
codeList[i]['parameters'][0] = translatedText
else:
continue
+ if 'namePop' in jaString:
+ matchList = re.findall(r'namePop\s\d+\s(.+?)\s.+', jaString)
+ if len(matchList) > 0:
+ # Translate
+ text = matchList[0]
+ response = translateGPT(text, 'Reply with the '+ LANGUAGE +' Translation', False)
+ translatedText = response[0]
+ totalTokens[0] += response[1][0]
+ totalTokens[1] += response[1][1]
+
+ # Set Data
+ translatedText = jaString.replace(text, translatedText)
+ codeList[i]['parameters'][0] = translatedText
### Event Code: 102 Show Choice
if codeList[i]['code'] == 102 and CODE102 is True: