Add image module

This commit is contained in:
DazedAnon 2024-08-26 10:49:27 -05:00
parent db7495c03b
commit b9284d9116
5 changed files with 525 additions and 21 deletions

1
.gitignore vendored
View file

@ -9,4 +9,5 @@
*.csv *.csv
*.ks *.ks
*.cid *.cid
*.png
__pycache__ __pycache__

492
modules/images.py Normal file
View file

@ -0,0 +1,492 @@
# Libraries
from PIL import Image, ImageDraw, ImageFont
import json, os, re, textwrap, threading, time, traceback, tiktoken, openai
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
from colorama import Fore
from dotenv import load_dotenv
from retry import retry
from tqdm import tqdm
#Globals
MODEL = os.getenv('model')
TIMEOUT = int(os.getenv('timeout'))
LANGUAGE = os.getenv('language').capitalize()
PROMPT = Path('prompt.txt').read_text(encoding='utf-8')
VOCAB = Path('vocab.txt').read_text(encoding='utf-8')
THREADS = int(os.getenv('threads'))
LOCK = threading.Lock()
PBAR = None
WIDTH = int(os.getenv('width'))
LISTWIDTH = int(os.getenv('listWidth'))
NOTEWIDTH = int(os.getenv('noteWidth'))
MAXHISTORY = 10
ESTIMATE = ''
TOKENS = [0, 0]
NAMESLIST = []
MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong)
# Open AI
load_dotenv()
if os.getenv('api').replace(' ', '') != '':
openai.base_url = os.getenv('api')
openai.organization = os.getenv('org')
openai.api_key = os.getenv('key')
# Pricing - Depends on the model https://openai.com/pricing
# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request
# If you are getting a MISMATCH LENGTH error, lower the batch size.
if 'gpt-3.5' in MODEL:
INPUTAPICOST = .002
OUTPUTAPICOST = .002
BATCHSIZE = 10
FREQUENCY_PENALTY = 0.2
elif 'gpt-4' in MODEL:
INPUTAPICOST = .005
OUTPUTAPICOST = .015
BATCHSIZE = 20
FREQUENCY_PENALTY = 0.1
#tqdm Globals
BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}'
POSITION = 0
LEAVE = False
def handleImages(filepath, estimate):
global ESTIMATE, TOKENS
ESTIMATE = estimate
start = time.time()
translatedData = openFiles(filepath)
# Convert Strings to Images
for i in range(len(translatedData[0][0])):
translatedList = translatedData[0][0]
originalList = translatedData[0][1]
dimensionsList = translatedData[0][2]
image = stringToImage(translatedList[i], dimensionsList[i][0], dimensionsList[i][1])
image.save(f'translated/{originalList[i]}.png', quality=100)
# Print File
end = time.time()
tqdm.write(getResultString(translatedData, end - start, filepath))
with LOCK:
TOKENS[0] += translatedData[1][0]
TOKENS[1] += translatedData[1][1]
# Print Total
totalString = getResultString(['', TOKENS, None], end - start, 'TOTAL')
# Print any errors on maps
if len(MISMATCH) > 0:
return totalString + Fore.RED + f'\nMismatch Errors: {MISMATCH}' + Fore.RESET
else:
return totalString
def openFiles(filepath):
if os.path.isdir(filepath):
imageList = [[],[]]
imageList = processImagesDir(filepath, imageList)
translatedData = translateImages(imageList)
translatedData = [[translatedData[0], imageList[0], imageList[1]], translatedData[1], translatedData[2]]
return translatedData
else:
print("The provided directory path does not exist.")
def getResultString(translatedData, translationTime, filename):
# File Print String
totalTokenstring =\
Fore.YELLOW +\
'[Input: ' + str(translatedData[1][0]) + ']'\
'[Output: ' + str(translatedData[1][1]) + ']'\
'[Cost: ${:,.4f}'.format((translatedData[1][0] * .001 * INPUTAPICOST) +\
(translatedData[1][1] * .001 * OUTPUTAPICOST)) + ']'
timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]'
if translatedData[2] is None:
# Success
return filename + ': ' + totalTokenstring + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET
else:
# Fail
try:
raise translatedData[2]
except Exception as e:
traceback.print_exc()
errorString = str(e) + Fore.RED
return filename + ': ' + totalTokenstring + timeString + Fore.RED + u' \u2717 ' +\
errorString + Fore.RESET
def getFontSize(text, image_width, image_height, font_path):
# Start with a high font size and keep reducing it until the text fits within the image bounds
font_size = min(image_width, image_height)
while font_size > 0:
font = ImageFont.truetype(font_path, font_size)
text_bbox = ImageDraw.Draw(Image.new('RGB', (1, 1))).textbbox((0, 0), text, font=font)
text_width = text_bbox[2] - text_bbox[0]
text_height = text_bbox[3] - text_bbox[1]
if text_width <= image_width and text_height <= image_height:
return font_size
font_size -= 1
return font_size
def stringToImage(text, width, height, font_path='arial.ttf'):
# Find the appropriate font size
font_size = getFontSize(text, width, height, font_path)
if font_size == 0:
raise ValueError("Text is too long to fit in the supplied dimensions.")
# Create a new image with the specified width and height and a transparent background
image = Image.new('RGBA', (width, height), (255, 255, 255, 0))
# Create a drawing context
draw = ImageDraw.Draw(image)
# Load the appropriate font
font = ImageFont.truetype(font_path, font_size)
# Calculate the size of the text to center it
text_bbox = draw.textbbox((0, 0), text, font=font)
text_width = text_bbox[2] - text_bbox[0]
text_height = text_bbox[3] - text_bbox[1]
x = (width - text_width) // 2
y = (height - text_height) // 2
# Draw the text on the image
draw.text((x, y), text, font=font, fill=(255, 255, 255, 255))
return image
def getImageDimensions(file_path):
try:
with Image.open(file_path) as img:
width, height = img.size
return width, height
except Exception as e:
print(f"Error reading {file_path}: {e}")
return None, None
def processImagesDir(directory_path, imageList):
for file_name in os.listdir(directory_path):
if '.png' in file_name:
file_path = os.path.join(directory_path, file_name)
if os.path.isfile(file_path):
# Check if the file is an image
try:
width, height = getImageDimensions(file_path)
if width is not None and height is not None:
imageList[0].append(file_name.replace('.png', ' '))
imageList[1].append([width, height])
except Exception as e:
print(f"Error processing {file_name}: {e}")
return imageList
def translateImages(imageList):
global PBAR
totalTokens = [0,0]
with tqdm(bar_format=BAR_FORMAT, position=POSITION, leave=LEAVE, desc='Images', total=len(imageList[0])) as PBAR:
# Translate GPT
response = translateGPT(imageList[0], 'Keep the Translation brief', True)
translatedList = response[0]
totalTokens[0] += response[1][0]
totalTokens[1] += response[1][1]
return [translatedList, totalTokens, None]
# Save some money and enter the character before translation
def getSpeaker(speaker):
match speaker:
case 'ファイン':
return ['Fine', [0,0]]
case '':
return ['', [0,0]]
case _:
# Store Speaker
if speaker not in str(NAMESLIST):
response = translateGPT(speaker, 'Reply with the '+ LANGUAGE +' translation of the NPC name.', False)
response[0] = response[0].title()
response[0] = response[0].replace("'S", "'s")
# Retry if name doesn't translate for some reason
if re.search(r'([a-zA-Z?])', response[0]) == None:
response = translateGPT(speaker, 'Reply with the '+ LANGUAGE +' translation of the NPC name.', False)
response[0] = response[0].title()
response[0] = response[0].replace("'S", "'s")
speakerList = [speaker, response[0]]
NAMESLIST.append(speakerList)
return response
# Find Speaker
else:
for i in range(len(NAMESLIST)):
if speaker == NAMESLIST[i][0]:
return [NAMESLIST[i][1],[0,0]]
return [speaker,[0,0]]
def subVars(jaString):
jaString = jaString.replace('\u3000', ' ')
# Formatting
count = 0
codeList = re.findall(r'[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]', jaString)
codeList = set(codeList)
if len(codeList) != 0:
for var in codeList:
jaString = jaString.replace(var, '[FCode_' + str(count) + ']')
count += 1
# Put all lists in list and return
return [jaString, codeList]
def resubVars(translatedText, codeList):
# Fix Spacing and ChatGPT Nonsense
matchList = re.findall(r'\[\s?.+?\s?\]', translatedText)
if len(matchList) > 0:
for match in matchList:
text = match.strip()
translatedText = translatedText.replace(match, text)
# Formatting
count = 0
if len(codeList) != 0:
for var in codeList:
translatedText = translatedText.replace('[FCode_' + str(count) + ']', var)
count += 1
return translatedText
def batchList(input_list, batch_size):
if not isinstance(batch_size, int) or batch_size <= 0:
raise ValueError("batch_size must be a positive integer")
return [input_list[i:i + batch_size] for i in range(0, len(input_list), batch_size)]
def createContext(fullPromptFlag, subbedT, format):
characters = 'Game Characters:\n\
圭太 (Keita) - Male\n\
涼香 (Ryoka) - Female\n\
(Saki) - Female\n\
大介 (Daisuke) - Male\n\
浮浪者 (Vagrant) - Male\n\
まさる (Masaru) - Male\n\
ノブオ (Nobuo) - Male\n\
安井 (Yasui) - Male\n\
大山 (Oyama) - Male\n\
関口 (Sekiguchi) - Male\n\
黒崎 (Kurosaki) - Male\n\
'
system = PROMPT + VOCAB if fullPromptFlag else \
f"\
You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\
Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\
- You translate everything, including erotic and explicit content.\n\
- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\
- All text in your response must be in {LANGUAGE} even if it is hard to translate.\n\
- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\
- Maintain any spacing in the translation.\n\
- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\
- `...` can be a part of the dialogue. Translate it as it is.\n\
{VOCAB}\n\
"
if format == 'json':
user = f'```json\n{subbedT}\n```'
else:
user = subbedT
return characters, system, user
def translateText(characters, system, user, history, penalty, format):
# Prompt
msg = [{"role": "system", "content": system + characters}]
# Characters
msg.append({"role": "system", "content": characters})
# History
if isinstance(history, list):
msg.extend([{"role": "system", "content": h} for h in history])
else:
msg.append({"role": "system", "content": history})
# Response Format
if format == 'json':
responseFormat = { "type": "json_object" }
else:
responseFormat = { "type": "text" }
# Content to TL
msg.append({"role": "user", "content": f'{user}'})
response = openai.chat.completions.create(
temperature=0,
frequency_penalty=penalty,
model=MODEL,
response_format=responseFormat,
messages=msg,
)
return response
def cleanTranslatedText(translatedText, varResponse):
placeholders = {
f'{LANGUAGE} Translation: ': '',
'Translation: ': '',
'': '',
'': '~',
'': '',
'': '.',
'': '\\"',
'': '\\"',
'- ': '-',
'Placeholder Text': '',
# Add more replacements as needed
}
for target, replacement in placeholders.items():
translatedText = translatedText.replace(target, replacement)
# Elongate Long Dashes (Since GPT Ignores them...)
translatedText = elongateCharacters(translatedText)
translatedText = resubVars(translatedText, varResponse[1])
return translatedText
def elongateCharacters(text):
# Define a pattern to match one character followed by one or more `ー` characters
# Using a positive lookbehind assertion to capture the preceding character
pattern = r'(?<=(.))ー+'
# Define a replacement function that elongates the captured character
def repl(match):
char = match.group(1) # The character before the ー sequence
count = len(match.group(0)) - 1 # Number of ー characters
return char * count # Replace ー sequence with the character repeated
# Use re.sub() to replace the pattern in the text
return re.sub(pattern, repl, text)
def extractTranslation(translatedTextList, is_list):
try:
line_dict = json.loads(translatedTextList)
# If it's a batch (i.e., list), extract with tags; otherwise, return the single item.
string_list = list(line_dict.values())
if is_list:
return string_list
else:
return string_list[0]
except Exception as e:
print(f'extractTranslation Error: {e}')
return None
def countTokens(characters, system, user, history):
inputTotalTokens = 0
outputTotalTokens = 0
enc = tiktoken.encoding_for_model('gpt-4')
# Input
if isinstance(history, list):
for line in history:
inputTotalTokens += len(enc.encode(line))
else:
inputTotalTokens += len(enc.encode(history))
inputTotalTokens += len(enc.encode(system))
inputTotalTokens += len(enc.encode(characters))
inputTotalTokens += len(enc.encode(user))
# Output
outputTotalTokens += round(len(enc.encode(user))*3)
return [inputTotalTokens, outputTotalTokens]
def combineList(tlist, text):
if isinstance(text, list):
return [t for sublist in tlist for t in sublist]
return tlist[0]
@retry(exceptions=Exception, tries=5, delay=5)
def translateGPT(text, history, fullPromptFlag):
global PBAR
mismatch = False
totalTokens = [0, 0]
if isinstance(text, list):
format = 'json'
tList = batchList(text, BATCHSIZE)
else:
format = 'text'
tList = [text]
for index, tItem in enumerate(tList):
# Before sending to translation, if we have a list of items, add the formatting
if isinstance(tItem, list):
payload = {f"Line{i+1}": string for i, string in enumerate(tItem)}
payload = json.dumps(payload, indent=4, ensure_ascii=False)
varResponse = subVars(payload)
subbedT = varResponse[0]
else:
varResponse = subVars(tItem)
subbedT = varResponse[0]
# Things to Check before starting translation
if not re.search(r'[一-龠ぁ-ゔァ-ヴーa---]+', subbedT):
if PBAR is not None:
PBAR.update(len(tItem))
continue
# Create Message
characters, system, user = createContext(fullPromptFlag, subbedT, format)
# Calculate Estimate
if ESTIMATE:
estimate = countTokens(characters, system, user, history)
totalTokens[0] += estimate[0]
totalTokens[1] += estimate[1]
continue
# Translating
response = translateText(characters, system, user, history, 0.05, format)
translatedText = response.choices[0].message.content
totalTokens[0] += response.usage.prompt_tokens
totalTokens[1] += response.usage.completion_tokens
# Check Translation
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedText, True)
if extractedTranslations == None or len(tItem) != len(extractedTranslations):
# Mismatch. Try Again
response = translateText(characters, system, user, history, 0.05, format)
translatedText = response.choices[0].message.content
totalTokens[0] += response.usage.prompt_tokens
totalTokens[1] += response.usage.completion_tokens
# Formatting
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedText, True)
if extractedTranslations == None or len(tItem) != len(extractedTranslations):
mismatch = True # Just here for breakpoint
# Set if no mismatch
if mismatch == False:
tList[index] = extractedTranslations
history = extractedTranslations[-10:] # Update history if we have a list
else:
history = text[-10:]
mismatch = False
# Update Loading Bar
with LOCK:
if PBAR is not None:
PBAR.update(len(tItem))
else:
# Ensure we're passing a single string to extractTranslation
tList[index] = translatedText
finalList = combineList(tList, text)
return [finalList, totalTokens]

View file

@ -33,6 +33,7 @@ from modules.wolf2 import handleWOLF2
from modules.javascript import handleJavascript from modules.javascript import handleJavascript
from modules.irissoft import handleIris from modules.irissoft import handleIris
from modules.regex import handleRegex from modules.regex import handleRegex
from modules.images import handleImages
from modules.rpgmakerplugin import handlePlugin from modules.rpgmakerplugin import handlePlugin
# For GPT4 rate limit will be hit if you have more than 1 thread. # For GPT4 rate limit will be hit if you have more than 1 thread.
@ -59,6 +60,7 @@ MODULES = [
["Javascript", "js", handleJavascript], ["Javascript", "js", handleJavascript],
["Iris", "txt", handleIris], ["Iris", "txt", handleIris],
["Regex", "txt", handleRegex], ["Regex", "txt", handleRegex],
["Images", "png", handleImages],
] ]
# Info Message # Info Message
@ -97,9 +99,11 @@ files to translate are in the /files folder and that you picked the right game e
# Open File (Threads) # Open File (Threads)
with ThreadPoolExecutor(max_workers=THREADS) as executor: with ThreadPoolExecutor(max_workers=THREADS) as executor:
futures = [executor.submit(MODULES[version][2], filename, estimate) \ if MODULES[version][0] != 'Images':
for filename in os.listdir("files") if filename.endswith(MODULES[version][1])] futures = [executor.submit(MODULES[version][2], filename, estimate) \
for filename in os.listdir("files") if filename.endswith(MODULES[version][1])]
else:
futures = [executor.submit(MODULES[version][2], 'files', estimate)]
for future in as_completed(futures): for future in as_completed(futures):
try: try:
totalCost = future.result() totalCost = future.result()

View file

@ -1050,7 +1050,7 @@ def searchCodes(page, pbar, jobList, filename):
## Event Code: 122 [Set Variables] ## Event Code: 122 [Set Variables]
if 'code' in codeList[i] and codeList[i]['code'] == 122 and CODE122 is True: if 'code' in codeList[i] and codeList[i]['code'] == 122 and CODE122 is True:
# This is going to be the var being set. (IMPORTANT) # This is going to be the var being set. (IMPORTANT)
if codeList[i]['parameters'][0] not in list(range(0, 100)): if codeList[i]['parameters'][0] not in list(range(15, 17)):
i += 1 i += 1
continue continue
@ -1066,13 +1066,17 @@ def searchCodes(page, pbar, jobList, filename):
continue continue
# Definitely don't want to mess with files # Definitely don't want to mess with files
if 'gameV' in jaString or '_' in jaString: # if 'gameV' in jaString or '_' in jaString:
i += 1 # i += 1
continue # continue
# Need to remove outside code and put it back later # Set String
matchedText = re.search(r"[\'\"\`](.*)[\'\"\`]", jaString) if len(re.findall(r"(')", jaString)) == 2:
matchedText = re.search(r"[\'\"\`](.*)[\'\"\`]", jaString)
else:
matchedText = re.search(r'(.*)', jaString)
# Last Check
if matchedText != None: if matchedText != None:
# Remove Textwrap # Remove Textwrap
finalJAString = matchedText.group(1).replace('\\n', ' ') finalJAString = matchedText.group(1).replace('\\n', ' ')
@ -1097,7 +1101,8 @@ def searchCodes(page, pbar, jobList, filename):
# Textwrap # Textwrap
translatedText = textwrap.fill(translatedText, width=80) translatedText = textwrap.fill(translatedText, width=80)
translatedText = translatedText.replace('\n', '\\n') translatedText = translatedText.replace('\n', '\\n')
translatedText = '\"' + translatedText + '\"' if len(re.findall(r"(')", jaString)) == 2:
translatedText = '\'' + translatedText + '\''
# Set # Set
codeList[i]['parameters'][4] = translatedText codeList[i]['parameters'][4] = translatedText
@ -1515,6 +1520,8 @@ def searchCodes(page, pbar, jobList, filename):
regex = r'addLog\s(.*)' regex = r'addLog\s(.*)'
elif 'DW_' in jaString: elif 'DW_' in jaString:
regex = r'DW_.*?\s(.*)' regex = r'DW_.*?\s(.*)'
elif 'CommonPopup' in jaString:
regex = r'CommonPopup\sadd\stext:(.*?)[\\]+}'
else: else:
regex = r'' regex = r''
@ -1523,7 +1530,7 @@ def searchCodes(page, pbar, jobList, filename):
# Capture Arguments and text # Capture Arguments and text
textMatch = re.search(regex, jaString) textMatch = re.search(regex, jaString)
if textMatch.group() != '': if textMatch and textMatch.group(0) != '':
text = textMatch.group(1) text = textMatch.group(1)
# Using this to keep track of 401's in a row. Throws IndexError at EndOfList (Expected Behavior) # Using this to keep track of 401's in a row. Throws IndexError at EndOfList (Expected Behavior)
@ -2117,7 +2124,7 @@ def batchList(input_list, batch_size):
return [input_list[i:i + batch_size] for i in range(0, len(input_list), batch_size)] return [input_list[i:i + batch_size] for i in range(0, len(input_list), batch_size)]
def createContext(fullPromptFlag, subbedT): def createContext(fullPromptFlag, subbedT, format):
characters = 'Game Characters:\n\ characters = 'Game Characters:\n\
圭太 (Keita) - Male\n\ 圭太 (Keita) - Male\n\
涼香 (Ryoka) - Female\n\ 涼香 (Ryoka) - Female\n\
@ -2145,8 +2152,8 @@ Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{
- `...` can be a part of the dialogue. Translate it as it is.\n\ - `...` can be a part of the dialogue. Translate it as it is.\n\
{VOCAB}\n\ {VOCAB}\n\
" "
if isinstance(subbedT, list): if format == 'json':
user = f'```json\n{subbedT}```' user = f'```json\n{subbedT}\n```'
else: else:
user = subbedT user = subbedT
return characters, system, user return characters, system, user
@ -2288,7 +2295,7 @@ def translateGPT(text, history, fullPromptFlag):
continue continue
# Create Message # Create Message
characters, system, user = createContext(fullPromptFlag, subbedT) characters, system, user = createContext(fullPromptFlag, subbedT, format)
# Calculate Estimate # Calculate Estimate
if ESTIMATE: if ESTIMATE:

View file

@ -69,9 +69,9 @@ CODE300 = False
CODE250 = False CODE250 = False
# Database # Database
NPCFLAG = True NPCFLAG = False
SCENARIOFLAG = False SCENARIOFLAG = False
ITEMFLAG = False ITEMFLAG = True
COLLECTIONFLAG = False COLLECTIONFLAG = False
ARMORFLAG = False ARMORFLAG = False
ENEMYFLAG = False ENEMYFLAG = False
@ -701,7 +701,7 @@ def searchDB(events, pbar, jobList, filename):
# Parse # Parse
for j in range(len(dataList)): for j in range(len(dataList)):
# Name # Name
if 'NULL' in dataList[j].get('name'): if '道具名' in dataList[j].get('name'):
# Pass 1 (Grab Data) # Pass 1 (Grab Data)
if setData == False: if setData == False:
if dataList[j].get('value') != '': if dataList[j].get('value') != '':
@ -714,7 +714,7 @@ def searchDB(events, pbar, jobList, filename):
itemList[0].pop(0) itemList[0].pop(0)
# Description # Description
if '説明' in dataList[j].get('name'): if 'NULL' in dataList[j].get('name'):
# Pass 1 (Grab Data) # Pass 1 (Grab Data)
if setData == False: if setData == False:
if dataList[j].get('value') != '': if dataList[j].get('value') != '':
@ -1105,7 +1105,7 @@ def searchDB(events, pbar, jobList, filename):
translate = True translate = True
# ITEMS # ITEMS
if len(itemList[1]) > 0: if len(itemList[0]) > 0:
# Progress Bar # Progress Bar
total = 0 total = 0
for itemArray in itemList: for itemArray in itemList: