diff --git a/.gitignore b/.gitignore index 60484bb..5377046 100644 --- a/.gitignore +++ b/.gitignore @@ -9,4 +9,5 @@ *.csv *.ks *.cid +*.png __pycache__ \ No newline at end of file diff --git a/modules/images.py b/modules/images.py new file mode 100644 index 0000000..5e493e6 --- /dev/null +++ b/modules/images.py @@ -0,0 +1,492 @@ +# Libraries +from PIL import Image, ImageDraw, ImageFont +import json, os, re, textwrap, threading, time, traceback, tiktoken, openai +from concurrent.futures import ThreadPoolExecutor, as_completed +from pathlib import Path +from colorama import Fore +from dotenv import load_dotenv +from retry import retry +from tqdm import tqdm + +#Globals +MODEL = os.getenv('model') +TIMEOUT = int(os.getenv('timeout')) +LANGUAGE = os.getenv('language').capitalize() +PROMPT = Path('prompt.txt').read_text(encoding='utf-8') +VOCAB = Path('vocab.txt').read_text(encoding='utf-8') +THREADS = int(os.getenv('threads')) +LOCK = threading.Lock() +PBAR = None +WIDTH = int(os.getenv('width')) +LISTWIDTH = int(os.getenv('listWidth')) +NOTEWIDTH = int(os.getenv('noteWidth')) +MAXHISTORY = 10 +ESTIMATE = '' +TOKENS = [0, 0] +NAMESLIST = [] +MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) + +# Open AI +load_dotenv() +if os.getenv('api').replace(' ', '') != '': + openai.base_url = os.getenv('api') +openai.organization = os.getenv('org') +openai.api_key = os.getenv('key') + +# Pricing - Depends on the model https://openai.com/pricing +# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request +# If you are getting a MISMATCH LENGTH error, lower the batch size. +if 'gpt-3.5' in MODEL: + INPUTAPICOST = .002 + OUTPUTAPICOST = .002 + BATCHSIZE = 10 + FREQUENCY_PENALTY = 0.2 +elif 'gpt-4' in MODEL: + INPUTAPICOST = .005 + OUTPUTAPICOST = .015 + BATCHSIZE = 20 + FREQUENCY_PENALTY = 0.1 + +#tqdm Globals +BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' +POSITION = 0 +LEAVE = False + +def handleImages(filepath, estimate): + global ESTIMATE, TOKENS + ESTIMATE = estimate + start = time.time() + + translatedData = openFiles(filepath) + + # Convert Strings to Images + for i in range(len(translatedData[0][0])): + translatedList = translatedData[0][0] + originalList = translatedData[0][1] + dimensionsList = translatedData[0][2] + image = stringToImage(translatedList[i], dimensionsList[i][0], dimensionsList[i][1]) + image.save(f'translated/{originalList[i]}.png', quality=100) + + # Print File + end = time.time() + tqdm.write(getResultString(translatedData, end - start, filepath)) + with LOCK: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + + # Print Total + totalString = getResultString(['', TOKENS, None], end - start, 'TOTAL') + + # Print any errors on maps + if len(MISMATCH) > 0: + return totalString + Fore.RED + f'\nMismatch Errors: {MISMATCH}' + Fore.RESET + else: + return totalString + +def openFiles(filepath): + if os.path.isdir(filepath): + imageList = [[],[]] + imageList = processImagesDir(filepath, imageList) + translatedData = translateImages(imageList) + translatedData = [[translatedData[0], imageList[0], imageList[1]], translatedData[1], translatedData[2]] + + return translatedData + else: + print("The provided directory path does not exist.") + +def getResultString(translatedData, translationTime, filename): + # File Print String + totalTokenstring =\ + Fore.YELLOW +\ + '[Input: ' + str(translatedData[1][0]) + ']'\ + '[Output: ' + str(translatedData[1][1]) + ']'\ + '[Cost: ${:,.4f}'.format((translatedData[1][0] * .001 * INPUTAPICOST) +\ + (translatedData[1][1] * .001 * OUTPUTAPICOST)) + ']' + timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]' + + if translatedData[2] is None: + # Success + return filename + ': ' + totalTokenstring + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET + else: + # Fail + try: + raise translatedData[2] + except Exception as e: + traceback.print_exc() + errorString = str(e) + Fore.RED + return filename + ': ' + totalTokenstring + timeString + Fore.RED + u' \u2717 ' +\ + errorString + Fore.RESET + +def getFontSize(text, image_width, image_height, font_path): + # Start with a high font size and keep reducing it until the text fits within the image bounds + font_size = min(image_width, image_height) + + while font_size > 0: + font = ImageFont.truetype(font_path, font_size) + text_bbox = ImageDraw.Draw(Image.new('RGB', (1, 1))).textbbox((0, 0), text, font=font) + text_width = text_bbox[2] - text_bbox[0] + text_height = text_bbox[3] - text_bbox[1] + + if text_width <= image_width and text_height <= image_height: + return font_size + font_size -= 1 + + return font_size + +def stringToImage(text, width, height, font_path='arial.ttf'): + # Find the appropriate font size + font_size = getFontSize(text, width, height, font_path) + + if font_size == 0: + raise ValueError("Text is too long to fit in the supplied dimensions.") + + # Create a new image with the specified width and height and a transparent background + image = Image.new('RGBA', (width, height), (255, 255, 255, 0)) + + # Create a drawing context + draw = ImageDraw.Draw(image) + + # Load the appropriate font + font = ImageFont.truetype(font_path, font_size) + + # Calculate the size of the text to center it + text_bbox = draw.textbbox((0, 0), text, font=font) + text_width = text_bbox[2] - text_bbox[0] + text_height = text_bbox[3] - text_bbox[1] + + x = (width - text_width) // 2 + y = (height - text_height) // 2 + + # Draw the text on the image + draw.text((x, y), text, font=font, fill=(255, 255, 255, 255)) + + return image + +def getImageDimensions(file_path): + try: + with Image.open(file_path) as img: + width, height = img.size + return width, height + except Exception as e: + print(f"Error reading {file_path}: {e}") + return None, None + +def processImagesDir(directory_path, imageList): + for file_name in os.listdir(directory_path): + if '.png' in file_name: + file_path = os.path.join(directory_path, file_name) + if os.path.isfile(file_path): + # Check if the file is an image + try: + width, height = getImageDimensions(file_path) + if width is not None and height is not None: + imageList[0].append(file_name.replace('.png', ' ')) + imageList[1].append([width, height]) + except Exception as e: + print(f"Error processing {file_name}: {e}") + + return imageList + +def translateImages(imageList): + global PBAR + totalTokens = [0,0] + with tqdm(bar_format=BAR_FORMAT, position=POSITION, leave=LEAVE, desc='Images', total=len(imageList[0])) as PBAR: + # Translate GPT + response = translateGPT(imageList[0], 'Keep the Translation brief', True) + translatedList = response[0] + totalTokens[0] += response[1][0] + totalTokens[1] += response[1][1] + + return [translatedList, totalTokens, None] + +# Save some money and enter the character before translation +def getSpeaker(speaker): + match speaker: + case 'ファイン': + return ['Fine', [0,0]] + case '': + return ['', [0,0]] + case _: + # Store Speaker + if speaker not in str(NAMESLIST): + response = translateGPT(speaker, 'Reply with the '+ LANGUAGE +' translation of the NPC name.', False) + response[0] = response[0].title() + response[0] = response[0].replace("'S", "'s") + + # Retry if name doesn't translate for some reason + if re.search(r'([a-zA-Z??])', response[0]) == None: + response = translateGPT(speaker, 'Reply with the '+ LANGUAGE +' translation of the NPC name.', False) + response[0] = response[0].title() + response[0] = response[0].replace("'S", "'s") + + speakerList = [speaker, response[0]] + NAMESLIST.append(speakerList) + return response + # Find Speaker + else: + for i in range(len(NAMESLIST)): + if speaker == NAMESLIST[i][0]: + return [NAMESLIST[i][1],[0,0]] + + return [speaker,[0,0]] + +def subVars(jaString): + jaString = jaString.replace('\u3000', ' ') + + # Formatting + count = 0 + codeList = re.findall(r'[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]', jaString) + codeList = set(codeList) + if len(codeList) != 0: + for var in codeList: + jaString = jaString.replace(var, '[FCode_' + str(count) + ']') + count += 1 + + # Put all lists in list and return + return [jaString, codeList] + +def resubVars(translatedText, codeList): + # Fix Spacing and ChatGPT Nonsense + matchList = re.findall(r'\[\s?.+?\s?\]', translatedText) + if len(matchList) > 0: + for match in matchList: + text = match.strip() + translatedText = translatedText.replace(match, text) + + # Formatting + count = 0 + if len(codeList) != 0: + for var in codeList: + translatedText = translatedText.replace('[FCode_' + str(count) + ']', var) + count += 1 + + return translatedText + +def batchList(input_list, batch_size): + if not isinstance(batch_size, int) or batch_size <= 0: + raise ValueError("batch_size must be a positive integer") + + return [input_list[i:i + batch_size] for i in range(0, len(input_list), batch_size)] + +def createContext(fullPromptFlag, subbedT, format): + characters = 'Game Characters:\n\ +圭太 (Keita) - Male\n\ +涼香 (Ryoka) - Female\n\ +咲 (Saki) - Female\n\ +大介 (Daisuke) - Male\n\ +浮浪者 (Vagrant) - Male\n\ +まさる (Masaru) - Male\n\ +ノブオ (Nobuo) - Male\n\ +安井 (Yasui) - Male\n\ +大山 (Oyama) - Male\n\ +関口 (Sekiguchi) - Male\n\ +黒崎 (Kurosaki) - Male\n\ +' + + system = PROMPT + VOCAB if fullPromptFlag else \ + f"\ +You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\ +Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\ +- You translate everything, including erotic and explicit content.\n\ +- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\ +- All text in your response must be in {LANGUAGE} even if it is hard to translate.\n\ +- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\ +- Maintain any spacing in the translation.\n\ +- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\ +- `...` can be a part of the dialogue. Translate it as it is.\n\ +{VOCAB}\n\ +" + if format == 'json': + user = f'```json\n{subbedT}\n```' + else: + user = subbedT + return characters, system, user + +def translateText(characters, system, user, history, penalty, format): + # Prompt + msg = [{"role": "system", "content": system + characters}] + + # Characters + msg.append({"role": "system", "content": characters}) + + # History + if isinstance(history, list): + msg.extend([{"role": "system", "content": h} for h in history]) + else: + msg.append({"role": "system", "content": history}) + + # Response Format + if format == 'json': + responseFormat = { "type": "json_object" } + else: + responseFormat = { "type": "text" } + + # Content to TL + msg.append({"role": "user", "content": f'{user}'}) + response = openai.chat.completions.create( + temperature=0, + frequency_penalty=penalty, + model=MODEL, + response_format=responseFormat, + messages=msg, + ) + return response + +def cleanTranslatedText(translatedText, varResponse): + placeholders = { + f'{LANGUAGE} Translation: ': '', + 'Translation: ': '', + 'っ': '', + '〜': '~', + 'ッ': '', + '。': '.', + '「': '\\"', + '」': '\\"', + '- ': '-', + 'Placeholder Text': '', + # Add more replacements as needed + } + for target, replacement in placeholders.items(): + translatedText = translatedText.replace(target, replacement) + + # Elongate Long Dashes (Since GPT Ignores them...) + translatedText = elongateCharacters(translatedText) + translatedText = resubVars(translatedText, varResponse[1]) + return translatedText + +def elongateCharacters(text): + # Define a pattern to match one character followed by one or more `ー` characters + # Using a positive lookbehind assertion to capture the preceding character + pattern = r'(?<=(.))ー+' + + # Define a replacement function that elongates the captured character + def repl(match): + char = match.group(1) # The character before the ー sequence + count = len(match.group(0)) - 1 # Number of ー characters + return char * count # Replace ー sequence with the character repeated + + # Use re.sub() to replace the pattern in the text + return re.sub(pattern, repl, text) + +def extractTranslation(translatedTextList, is_list): + try: + line_dict = json.loads(translatedTextList) + # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. + string_list = list(line_dict.values()) + if is_list: + return string_list + else: + return string_list[0] + + except Exception as e: + print(f'extractTranslation Error: {e}') + return None + + +def countTokens(characters, system, user, history): + inputTotalTokens = 0 + outputTotalTokens = 0 + enc = tiktoken.encoding_for_model('gpt-4') + + # Input + if isinstance(history, list): + for line in history: + inputTotalTokens += len(enc.encode(line)) + else: + inputTotalTokens += len(enc.encode(history)) + inputTotalTokens += len(enc.encode(system)) + inputTotalTokens += len(enc.encode(characters)) + inputTotalTokens += len(enc.encode(user)) + + # Output + outputTotalTokens += round(len(enc.encode(user))*3) + + return [inputTotalTokens, outputTotalTokens] + +def combineList(tlist, text): + if isinstance(text, list): + return [t for sublist in tlist for t in sublist] + return tlist[0] + +@retry(exceptions=Exception, tries=5, delay=5) +def translateGPT(text, history, fullPromptFlag): + global PBAR + + mismatch = False + totalTokens = [0, 0] + if isinstance(text, list): + format = 'json' + tList = batchList(text, BATCHSIZE) + else: + format = 'text' + tList = [text] + + for index, tItem in enumerate(tList): + # Before sending to translation, if we have a list of items, add the formatting + if isinstance(tItem, list): + payload = {f"Line{i+1}": string for i, string in enumerate(tItem)} + payload = json.dumps(payload, indent=4, ensure_ascii=False) + varResponse = subVars(payload) + subbedT = varResponse[0] + else: + varResponse = subVars(tItem) + subbedT = varResponse[0] + + # Things to Check before starting translation + if not re.search(r'[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+', subbedT): + if PBAR is not None: + PBAR.update(len(tItem)) + continue + + # Create Message + characters, system, user = createContext(fullPromptFlag, subbedT, format) + + # Calculate Estimate + if ESTIMATE: + estimate = countTokens(characters, system, user, history) + totalTokens[0] += estimate[0] + totalTokens[1] += estimate[1] + continue + + # Translating + response = translateText(characters, system, user, history, 0.05, format) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Check Translation + translatedText = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedText, True) + if extractedTranslations == None or len(tItem) != len(extractedTranslations): + # Mismatch. Try Again + response = translateText(characters, system, user, history, 0.05, format) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Formatting + translatedText = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedText, True) + if extractedTranslations == None or len(tItem) != len(extractedTranslations): + mismatch = True # Just here for breakpoint + + # Set if no mismatch + if mismatch == False: + tList[index] = extractedTranslations + history = extractedTranslations[-10:] # Update history if we have a list + else: + history = text[-10:] + mismatch = False + + # Update Loading Bar + with LOCK: + if PBAR is not None: + PBAR.update(len(tItem)) + else: + # Ensure we're passing a single string to extractTranslation + tList[index] = translatedText + + finalList = combineList(tList, text) + return [finalList, totalTokens] diff --git a/modules/main.py b/modules/main.py index 2f11848..d5f5f21 100644 --- a/modules/main.py +++ b/modules/main.py @@ -33,6 +33,7 @@ from modules.wolf2 import handleWOLF2 from modules.javascript import handleJavascript from modules.irissoft import handleIris from modules.regex import handleRegex +from modules.images import handleImages from modules.rpgmakerplugin import handlePlugin # For GPT4 rate limit will be hit if you have more than 1 thread. @@ -59,6 +60,7 @@ MODULES = [ ["Javascript", "js", handleJavascript], ["Iris", "txt", handleIris], ["Regex", "txt", handleRegex], + ["Images", "png", handleImages], ] # Info Message @@ -97,9 +99,11 @@ files to translate are in the /files folder and that you picked the right game e # Open File (Threads) with ThreadPoolExecutor(max_workers=THREADS) as executor: - futures = [executor.submit(MODULES[version][2], filename, estimate) \ - for filename in os.listdir("files") if filename.endswith(MODULES[version][1])] - + if MODULES[version][0] != 'Images': + futures = [executor.submit(MODULES[version][2], filename, estimate) \ + for filename in os.listdir("files") if filename.endswith(MODULES[version][1])] + else: + futures = [executor.submit(MODULES[version][2], 'files', estimate)] for future in as_completed(futures): try: totalCost = future.result() diff --git a/modules/rpgmakermvmz.py b/modules/rpgmakermvmz.py index 8a47a65..8ae68a8 100644 --- a/modules/rpgmakermvmz.py +++ b/modules/rpgmakermvmz.py @@ -1050,7 +1050,7 @@ def searchCodes(page, pbar, jobList, filename): ## Event Code: 122 [Set Variables] if 'code' in codeList[i] and codeList[i]['code'] == 122 and CODE122 is True: # This is going to be the var being set. (IMPORTANT) - if codeList[i]['parameters'][0] not in list(range(0, 100)): + if codeList[i]['parameters'][0] not in list(range(15, 17)): i += 1 continue @@ -1066,13 +1066,17 @@ def searchCodes(page, pbar, jobList, filename): continue # Definitely don't want to mess with files - if 'gameV' in jaString or '_' in jaString: - i += 1 - continue - - # Need to remove outside code and put it back later - matchedText = re.search(r"[\'\"\`](.*)[\'\"\`]", jaString) + # if 'gameV' in jaString or '_' in jaString: + # i += 1 + # continue + + # Set String + if len(re.findall(r"(')", jaString)) == 2: + matchedText = re.search(r"[\'\"\`](.*)[\'\"\`]", jaString) + else: + matchedText = re.search(r'(.*)', jaString) + # Last Check if matchedText != None: # Remove Textwrap finalJAString = matchedText.group(1).replace('\\n', ' ') @@ -1097,7 +1101,8 @@ def searchCodes(page, pbar, jobList, filename): # Textwrap translatedText = textwrap.fill(translatedText, width=80) translatedText = translatedText.replace('\n', '\\n') - translatedText = '\"' + translatedText + '\"' + if len(re.findall(r"(')", jaString)) == 2: + translatedText = '\'' + translatedText + '\'' # Set codeList[i]['parameters'][4] = translatedText @@ -1515,6 +1520,8 @@ def searchCodes(page, pbar, jobList, filename): regex = r'addLog\s(.*)' elif 'DW_' in jaString: regex = r'DW_.*?\s(.*)' + elif 'CommonPopup' in jaString: + regex = r'CommonPopup\sadd\stext:(.*?)[\\]+}' else: regex = r'' @@ -1523,7 +1530,7 @@ def searchCodes(page, pbar, jobList, filename): # Capture Arguments and text textMatch = re.search(regex, jaString) - if textMatch.group() != '': + if textMatch and textMatch.group(0) != '': text = textMatch.group(1) # Using this to keep track of 401's in a row. Throws IndexError at EndOfList (Expected Behavior) @@ -2117,7 +2124,7 @@ def batchList(input_list, batch_size): return [input_list[i:i + batch_size] for i in range(0, len(input_list), batch_size)] -def createContext(fullPromptFlag, subbedT): +def createContext(fullPromptFlag, subbedT, format): characters = 'Game Characters:\n\ 圭太 (Keita) - Male\n\ 涼香 (Ryoka) - Female\n\ @@ -2145,8 +2152,8 @@ Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{ - `...` can be a part of the dialogue. Translate it as it is.\n\ {VOCAB}\n\ " - if isinstance(subbedT, list): - user = f'```json\n{subbedT}```' + if format == 'json': + user = f'```json\n{subbedT}\n```' else: user = subbedT return characters, system, user @@ -2288,7 +2295,7 @@ def translateGPT(text, history, fullPromptFlag): continue # Create Message - characters, system, user = createContext(fullPromptFlag, subbedT) + characters, system, user = createContext(fullPromptFlag, subbedT, format) # Calculate Estimate if ESTIMATE: diff --git a/modules/wolf.py b/modules/wolf.py index 091e4e3..b70b610 100644 --- a/modules/wolf.py +++ b/modules/wolf.py @@ -69,9 +69,9 @@ CODE300 = False CODE250 = False # Database -NPCFLAG = True +NPCFLAG = False SCENARIOFLAG = False -ITEMFLAG = False +ITEMFLAG = True COLLECTIONFLAG = False ARMORFLAG = False ENEMYFLAG = False @@ -701,7 +701,7 @@ def searchDB(events, pbar, jobList, filename): # Parse for j in range(len(dataList)): # Name - if 'NULL' in dataList[j].get('name'): + if '道具名' in dataList[j].get('name'): # Pass 1 (Grab Data) if setData == False: if dataList[j].get('value') != '': @@ -714,7 +714,7 @@ def searchDB(events, pbar, jobList, filename): itemList[0].pop(0) # Description - if '説明' in dataList[j].get('name'): + if 'NULL' in dataList[j].get('name'): # Pass 1 (Grab Data) if setData == False: if dataList[j].get('value') != '': @@ -1105,7 +1105,7 @@ def searchDB(events, pbar, jobList, filename): translate = True # ITEMS - if len(itemList[1]) > 0: + if len(itemList[0]) > 0: # Progress Bar total = 0 for itemArray in itemList: