diff --git a/modules/alice.py b/modules/alice.py index 22da635..b907b37 100644 --- a/modules/alice.py +++ b/modules/alice.py @@ -46,6 +46,9 @@ BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" POSITION = 0 LEAVE = False +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. diff --git a/modules/anim.py b/modules/anim.py index 186695c..63aff73 100644 --- a/modules/anim.py +++ b/modules/anim.py @@ -47,6 +47,9 @@ BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" POSITION = 0 LEAVE = False +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. diff --git a/modules/csv.py b/modules/csv.py index bf026ed..de85a4b 100644 --- a/modules/csv.py +++ b/modules/csv.py @@ -44,6 +44,9 @@ IGNORETLTEXT = True # Ignores all translated text. MISMATCH = [] # Lists files that thdata a mismatch error (Length of GPT list response is wrong) BRACKETNAMES = False +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. @@ -429,7 +432,7 @@ def getSpeaker(speaker): return [NAMESLIST[i][1], [0, 0]] # If there isn't any Japanese in the text just skip - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", speaker): + if not re.search(LANGREGEX, speaker): return [speaker, [0, 0]] # Translate and Store Speaker @@ -628,7 +631,7 @@ def translateGPT(text, history, fullPromptFlag): subbedT = varResponse[0] # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", subbedT): + if not re.search(LANGREGEX, subbedT): if PBAR is not None: PBAR.update(len(tItem)) history = tItem[-MAXHISTORY:] diff --git a/modules/eushully.py b/modules/eushully.py index 40391de..0892f4b 100644 --- a/modules/eushully.py +++ b/modules/eushully.py @@ -47,6 +47,9 @@ POSITION = 0 LEAVE = False PBAR = None +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. diff --git a/modules/images.py b/modules/images.py index 987a6b0..cd2d453 100644 --- a/modules/images.py +++ b/modules/images.py @@ -1,619 +1,622 @@ -# Libraries -from PIL import Image, ImageDraw, ImageFont -import json -import os -import re -import threading -import time -import traceback -import tiktoken -import openai -from pathlib import Path -from colorama import Fore -from dotenv import load_dotenv -from retry import retry -from tqdm import tqdm - -# Globals -MODEL = os.getenv("model") -TIMEOUT = int(os.getenv("timeout")) -LANGUAGE = os.getenv("language").capitalize() -PROMPT = Path("prompt.txt").read_text(encoding="utf-8") -VOCAB = Path("vocab.txt").read_text(encoding="utf-8") -THREADS = int(os.getenv("threads")) -LOCK = threading.Lock() -PBAR = None -WIDTH = int(os.getenv("width")) -LISTWIDTH = int(os.getenv("listWidth")) -NOTEWIDTH = int(os.getenv("noteWidth")) -MAXHISTORY = 10 -ESTIMATE = "" -TOKENS = [0, 0] -NAMESLIST = [] -MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) - -# Open AI -load_dotenv() -if os.getenv("api").replace(" ", "") != "": - openai.base_url = os.getenv("api") -openai.organization = os.getenv("org") -openai.api_key = os.getenv("key") - -# Pricing - Depends on the model https://openai.com/pricing -# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request -# If you are getting a MISMATCH LENGTH error, lower the batch size. -if "gpt-3.5" in MODEL: - INPUTAPICOST = 0.002 - OUTPUTAPICOST = 0.002 - BATCHSIZE = 10 - FREQUENCY_PENALTY = 0.2 -elif "gpt-4" in MODEL: - INPUTAPICOST = 0.0025 - OUTPUTAPICOST = 0.01 - BATCHSIZE = 20 - FREQUENCY_PENALTY = 0.1 -else: - INPUTAPICOST = float(os.getenv("input_cost")) - OUTPUTAPICOST = float(os.getenv("output_cost")) - BATCHSIZE = int(os.getenv("batchsize")) - FREQUENCY_PENALTY = float(os.getenv("frequency_penalty")) - -# tqdm Globals -BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" -POSITION = 0 -LEAVE = False - - -def handleImages(folderName, estimate): - global ESTIMATE, TOKENS - ESTIMATE = estimate - start = time.time() - - # Translate Strings - translatedData = openFiles(f"files/{folderName}") - - # Custom Names - customList = [[], []] - customList = processImagesDir("Custom", customList) - - # Write Strings to Images - if not ESTIMATE: - if not os.path.exists(f"translated/{folderName}"): - os.mkdir(f"translated/{folderName}") - for i in range(len(translatedData[0][0])): - try: - translatedList = translatedData[0][0] - originalList = translatedData[0][1] - dimensionsList = translatedData[0][2] - image = stringToImage(translatedList[i], dimensionsList[i][0], dimensionsList[i][1]) - image.save(rf"translated/{folderName}/{customList[0][0]}.png", quality=100) - customList[0].pop(0) - except Exception as e: - PBAR.write(f"{translatedList[i]}: {str(e)}") - # Ignore Error - - # Print File - end = time.time() - tqdm.write(getResultString(translatedData, end - start, folderName)) - with LOCK: - TOKENS[0] += translatedData[1][0] - TOKENS[1] += translatedData[1][1] - - # Print Total - totalString = getResultString(["", TOKENS, None], end - start, "TOTAL") - - # Print any errors on maps - if len(MISMATCH) > 0: - return totalString + Fore.RED + f"\nMismatch Errors: {MISMATCH}" + Fore.RESET - else: - return totalString - - -def openFiles(folderName): - global PBAR - - if os.path.isdir(folderName): - imageList = [[], []] - imageList = processImagesDir(folderName, imageList) - - # Start Translation - with tqdm( - bar_format=BAR_FORMAT, - position=POSITION, - leave=LEAVE, - desc=folderName, - total=len(imageList[0]), - ) as PBAR: - translatedData = translateImages(imageList) - translatedData = [ - [translatedData[0], imageList[0], imageList[1]], - translatedData[1], - translatedData[2], - ] - - return translatedData - else: - print("The provided directory path does not exist.") - - -def getResultString(translatedData, translationTime, filename): - # File Print String - totalTokenstring = ( - Fore.YELLOW + "[Input: " + str(translatedData[1][0]) + "]" - "[Output: " - + str(translatedData[1][1]) - + "]" "[Cost: ${:,.4f}".format((translatedData[1][0] * 0.001 * INPUTAPICOST) + (translatedData[1][1] * 0.001 * OUTPUTAPICOST)) - + "]" - ) - timeString = Fore.BLUE + "[" + str(round(translationTime, 1)) + "s]" - - if translatedData[2] is None: - # Success - return filename + ": " + totalTokenstring + timeString + Fore.GREEN + " \u2713 " + Fore.RESET - else: - # Fail - try: - raise translatedData[2] - except Exception as e: - traceback.print_exc() - errorString = str(e) + Fore.RED - return filename + ": " + totalTokenstring + timeString + Fore.RED + " \u2717 " + errorString + Fore.RESET - - -def getFontSize(text, image_width, image_height, font_path): - # Start with a high font size and keep reducing it until the text fits within the image bounds - font_size = min(image_width, image_height) - - while font_size > 0: - font = ImageFont.truetype(font_path, font_size) - text_bbox = ImageDraw.Draw(Image.new("RGB", (1, 1))).textbbox((0, 0), text, font=font) - text_width = text_bbox[2] - text_bbox[0] - text_height = text_bbox[3] - text_bbox[1] - - if text_width <= image_width and text_height <= image_height: - return font_size - font_size -= 1 - - return font_size - - -def stringToImage(text, width, height, font_path="fonts/TsunagiGothic.ttf", scale_factor=4): - # Increase the resolution - scaled_width = int(width * scale_factor) - scaled_height = int(height * scale_factor) - - # Find the appropriate font size for the scaled up image - font_size = getFontSize(text, scaled_width, scaled_height, font_path) - if font_size == 0: - raise ValueError("Text is too long to fit in the supplied dimensions.") - - # Create a new image with the scaled width and height and a transparent background - image = Image.new("RGBA", (scaled_width, scaled_height), (255, 255, 255, 0)) - - # Create a drawing context - draw = ImageDraw.Draw(image) - - # Load the appropriate font - font = ImageFont.truetype(font_path, font_size) - - # Calculate the size of the text to center it - text_bbox = draw.textbbox((0, 0), text, font=font) - text_width = text_bbox[2] - text_bbox[0] - text_height = text_bbox[3] - text_bbox[1] + 20 - x = 0 - y = (scaled_height - text_height) // 2 - - # Draw the text on the image - draw.text((x, y), text, font=font, fill=(255, 255, 255, 255)) - - # Resize back to the original dimensions to get a clearer text rendering - image = image.resize( - (width, height), - Image.LANCZOS, - ) - - return image - - -def getImageDimensions(file_path): - try: - with Image.open(file_path) as img: - width, height = img.size - return width, height - except Exception as e: - print(f"Error reading {file_path}: {e}") - return None, None - - -def processImagesDir(directory_path, imageList): - for file_name in os.listdir(directory_path): - # .png and Japanese - if ".png" in file_name: - file_path = os.path.join(directory_path, file_name) - if os.path.isfile(file_path): - # Check if the file is an image - try: - width, height = getImageDimensions(file_path) - if width is not None and height is not None: - placeholders = { - ".png": "", - } - for target, replacement in placeholders.items(): - file_name = file_name.replace(target, replacement) - imageList[0].append(file_name) - imageList[1].append([width, height]) - except Exception as e: - print(f"Error processing {file_name}: {e}") - - if ".txt" in file_name: - try: - with open(f"{directory_path}/{file_name}", "r", encoding="utf8") as file: - for line in file: - line = line.strip() - line = line.replace(":", ":") - line = line.replace("/", "/") - line = line.replace("?", "?") - imageList[0].append(line) # Using strip() to remove any extra newlines or spaces - imageList[1].append([104, 15]) - except FileNotFoundError: - print(f"The file at {file_path} was not found.") - except IOError: - print(f"An error occurred while reading the file at {file_path}.") - return imageList - - -def translateImages(imageList): - totalTokens = [0, 0] - - # Translate GPT - response = translateGPT(imageList[0], "Keep the Translation as brief as possible", True) - translatedList = response[0] - totalTokens[0] += response[1][0] - totalTokens[1] += response[1][1] - - return [translatedList, totalTokens, None] - - -# Save some money and enter the character before translation -def getSpeaker(speaker): - match speaker: - case "ファイン": - return ["Fine", [0, 0]] - case "": - return ["", [0, 0]] - case _: - # Find Speaker - for i in range(len(NAMESLIST)): - if speaker == NAMESLIST[i][0]: - return [NAMESLIST[i][1], [0, 0]] - - # Translate and Store Speaker - response = translateGPT( - f"{speaker}", - "Reply with the " + LANGUAGE + " translation of the NPC name.", - True, - ) - response[0] = response[0].title() - response[0] = response[0].replace("'S", "'s") - response[0] = response[0].replace("Speaker: ", "") - - # Retry if name doesn't translate for some reason - if re.search(r"([a-zA-Z??])", response[0]) == None: - response = translateGPT( - f"{speaker}", - "Reply with the " + LANGUAGE + " translation of the NPC name.", - False, - ) - response[0] = response[0].title() - response[0] = response[0].replace("'S", "'s") - - speakerList = [speaker, response[0]] - NAMESLIST.append(speakerList) - return response - return [speaker, [0, 0]] - - -def subVars(jaString): - jaString = jaString.replace("\u3000", " ") - - # Formatting - count = 0 - codeList = re.findall(r"[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]", jaString) - codeList = set(codeList) - if len(codeList) != 0: - for var in codeList: - jaString = jaString.replace(var, "[FCode_" + str(count) + "]") - count += 1 - - # Put all lists in list and return - return [jaString, codeList] - - -def resubVars(translatedText, codeList): - # Fix Spacing and ChatGPT Nonsense - matchList = re.findall(r"\[\s?.+?\s?\]", translatedText) - if len(matchList) > 0: - for match in matchList: - text = match.strip() - translatedText = translatedText.replace(match, text) - - # Formatting - count = 0 - if len(codeList) != 0: - for var in codeList: - translatedText = translatedText.replace("[FCode_" + str(count) + "]", var) - count += 1 - - return translatedText - - -def batchList(input_list, batch_size): - if not isinstance(batch_size, int) or batch_size <= 0: - raise ValueError("batch_size must be a positive integer") - - return [input_list[i : i + batch_size] for i in range(0, len(input_list), batch_size)] - - -def createContext(fullPromptFlag, subbedT, format): - characters = "Game Characters:\n\ -ロラン (Roland) - Male\n\ -リュカ (Ryuka) - Male\n\ -レックス (Rex) - Male\n\ -タバサ (Tabasa) - Female\n\ -アルス (Ars) - Male\n\ -アマカラ (Amakara) - Male\n\ -エリー (Eri) - Female\n\ -リオ (Rio) - Female\n\ -サマル (Samal) - Male\n\ -ムーン (Moon) - Female\n\ -アリーナ (Arina) - Female\n\ -クリフト (Cliff) - Male\n\ -マーニャ (Manya) - Female\n\ -ミネア (Minea) - Female\n\ -デボラ (Debora) - Female\n\ -ビアンカ (Bianca) - Female\n\ -フローラ (Flora) - Female\n\ -バーバラ (Barbara) - Female\n\ -ミレーユ (Mireyu) - Female\n\ -アイラ (Aira) - Female\n\ -フォズ (Foz) - Female\n\ -マリベル (Maribel) - Female\n\ -ククール (Kukool) - Male\n\ -ゲルダ (Gerda) - Female\n\ -ゼシカ (Jessica) - Female\n\ -ヤンガス (Yangus) - Male\n\ -ラヴィエル (Raviel) - Female\n\ -セティア (Setia) - Female\n\ -ダイ (Dai) - Male\n\ -ヒュンケル (Hyunckel) - Male\n\ -ポップ (Pop) - Male\n\ -マァム (Maam) - Female\n\ -レオナ (Leona) - Female\n\ -アステア (Astea) - Female\n\ -イヨ (Iyo) - Female\n\ -ジャガン (Jagan) - Male\n\ -ヤオ (Yao) - Female\n\ -デイジィ (Daisy) - Female\n\ -バイシュン (Baishun) - Male\n\ -ブライ (Buraimu) - Male\n\ -ハッサン (Hassan) - Male\n\ -アロマ (Aroma) - Female\n\ -" - - system = ( - PROMPT + VOCAB - if fullPromptFlag - else f"\ -You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\ -Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\ -- You translate everything, including erotic and explicit content.\n\ -- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\ -- All text in your response must be in {LANGUAGE} even if it is hard to translate.\n\ -- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\ -- Maintain any spacing in the translation.\n\ -- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\ -- `...` can be a part of the dialogue. Translate it as it is.\n\ -{VOCAB}\n\ -" - ) - if format == "json": - user = f"```json\n{subbedT}\n```" - else: - user = subbedT - return characters, system, user - - -def translateText(characters, system, user, history, penalty, format): - # Prompt - msg = [{"role": "system", "content": system + characters}] - - # Characters - msg.append({"role": "system", "content": characters}) - - # History - if isinstance(history, list): - msg.extend([{"role": "system", "content": h} for h in history]) - else: - msg.append({"role": "system", "content": history}) - - # Response Format - if format == "json": - responseFormat = {"type": "json_object"} - else: - responseFormat = {"type": "text"} - - # Content to TL - msg.append({"role": "user", "content": f"{user}"}) - response = openai.chat.completions.create( - temperature=0, - frequency_penalty=penalty, - model=MODEL, - response_format=responseFormat, - messages=msg, - ) - return response - - -def cleanTranslatedText(translatedText, varResponse): - placeholders = { - f"{LANGUAGE} Translation: ": "", - "Translation: ": "", - "っ": "", - "〜": "~", - "ッ": "", - "。": ".", - "「": '\\"', - "」": '\\"', - "- ": "-", - "Placeholder Text": "", - # Add more replacements as needed - } - for target, replacement in placeholders.items(): - translatedText = translatedText.replace(target, replacement) - - # Elongate Long Dashes (Since GPT Ignores them...) - translatedText = elongateCharacters(translatedText) - translatedText = resubVars(translatedText, varResponse[1]) - return translatedText - - -def elongateCharacters(text): - # Define a pattern to match one character followed by one or more `ー` characters - # Using a positive lookbehind assertion to capture the preceding character - pattern = r"(?<=(.))ー+" - - # Define a replacement function that elongates the captured character - def repl(match): - char = match.group(1) # The character before the ー sequence - count = len(match.group(0)) - 1 # Number of ー characters - return char * count # Replace ー sequence with the character repeated - - # Use re.sub() to replace the pattern in the text - return re.sub(pattern, repl, text) - - -def extractTranslation(translatedTextList, is_list): - try: - line_dict = json.loads(translatedTextList) - # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. - string_list = list(line_dict.values()) - if is_list: - return string_list - else: - return string_list[0] - - except Exception as e: - print(f"extractTranslation Error: {e}") - return None - - -def countTokens(characters, system, user, history): - inputTotalTokens = 0 - outputTotalTokens = 0 - enc = tiktoken.encoding_for_model("gpt-4") - - # Input - if isinstance(history, list): - for line in history: - inputTotalTokens += len(enc.encode(line)) - else: - inputTotalTokens += len(enc.encode(history)) - inputTotalTokens += len(enc.encode(system)) - inputTotalTokens += len(enc.encode(characters)) - inputTotalTokens += len(enc.encode(user)) - - # Output - outputTotalTokens += round(len(enc.encode(user)) * 3) - - return [inputTotalTokens, outputTotalTokens] - - -@retry(exceptions=Exception, tries=5, delay=5) -def translateGPT(text, history, fullPromptFlag): - global PBAR - - mismatch = False - totalTokens = [0, 0] - if isinstance(text, list): - format = "json" - tList = batchList(text, BATCHSIZE) - else: - format = "text" - tList = [text] - - for index, tItem in enumerate(tList): - # Before sending to translation, if we have a list of items, add the formatting - if isinstance(tItem, list): - payload = {f"Line{i+1}": string for i, string in enumerate(tItem)} - payload = json.dumps(payload, indent=4, ensure_ascii=False) - varResponse = subVars(payload) - subbedT = varResponse[0] - else: - varResponse = subVars(tItem) - subbedT = varResponse[0] - - # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+", subbedT): - if PBAR is not None: - PBAR.update(len(tItem)) - continue - - # Create Message - characters, system, user = createContext(fullPromptFlag, subbedT, format) - - # Calculate Estimate - if ESTIMATE: - estimate = countTokens(characters, system, user, history) - totalTokens[0] += estimate[0] - totalTokens[1] += estimate[1] - continue - - # Translating - response = translateText(characters, system, user, history, 0.05, format) - translatedText = response.choices[0].message.content - totalTokens[0] += response.usage.prompt_tokens - totalTokens[1] += response.usage.completion_tokens - - # Check Translation - translatedText = cleanTranslatedText(translatedText, varResponse) - if isinstance(tItem, list): - extractedTranslations = extractTranslation(translatedText, True) - if extractedTranslations == None or len(tItem) != len(extractedTranslations): - # Mismatch. Try Again - response = translateText(characters, system, user, history, 0.05, format) - translatedText = response.choices[0].message.content - totalTokens[0] += response.usage.prompt_tokens - totalTokens[1] += response.usage.completion_tokens - - # Formatting - translatedText = cleanTranslatedText(translatedText, varResponse) - if isinstance(tItem, list): - extractedTranslations = extractTranslation(translatedText, True) - if extractedTranslations == None or len(tItem) != len(extractedTranslations): - mismatch = True # Just here for breakpoint - - # Set if no mismatch - if mismatch == False: - tList[index] = extractedTranslations - history = extractedTranslations[-10:] # Update history if we have a list - else: - history = text[-10:] - mismatch = False - - # Update Loading Bar - with LOCK: - if PBAR is not None: - PBAR.update(len(tItem)) - else: - # Ensure we're passing a single string to extractTranslation - tList[index] = translatedText.replace("Placeholder Text", "") - - # Combine if multilist - if isinstance(tList[0], list): - tList = [t for sublist in tList for t in sublist] - - # Return - if format == "json": - return [tList, totalTokens] - else: - return [tList[0], totalTokens] +# Libraries +from PIL import Image, ImageDraw, ImageFont +import json +import os +import re +import threading +import time +import traceback +import tiktoken +import openai +from pathlib import Path +from colorama import Fore +from dotenv import load_dotenv +from retry import retry +from tqdm import tqdm + +# Globals +MODEL = os.getenv("model") +TIMEOUT = int(os.getenv("timeout")) +LANGUAGE = os.getenv("language").capitalize() +PROMPT = Path("prompt.txt").read_text(encoding="utf-8") +VOCAB = Path("vocab.txt").read_text(encoding="utf-8") +THREADS = int(os.getenv("threads")) +LOCK = threading.Lock() +PBAR = None +WIDTH = int(os.getenv("width")) +LISTWIDTH = int(os.getenv("listWidth")) +NOTEWIDTH = int(os.getenv("noteWidth")) +MAXHISTORY = 10 +ESTIMATE = "" +TOKENS = [0, 0] +NAMESLIST = [] +MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) + +# Open AI +load_dotenv() +if os.getenv("api").replace(" ", "") != "": + openai.base_url = os.getenv("api") +openai.organization = os.getenv("org") +openai.api_key = os.getenv("key") + +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + +# Pricing - Depends on the model https://openai.com/pricing +# Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request +# If you are getting a MISMATCH LENGTH error, lower the batch size. +if "gpt-3.5" in MODEL: + INPUTAPICOST = 0.002 + OUTPUTAPICOST = 0.002 + BATCHSIZE = 10 + FREQUENCY_PENALTY = 0.2 +elif "gpt-4" in MODEL: + INPUTAPICOST = 0.0025 + OUTPUTAPICOST = 0.01 + BATCHSIZE = 20 + FREQUENCY_PENALTY = 0.1 +else: + INPUTAPICOST = float(os.getenv("input_cost")) + OUTPUTAPICOST = float(os.getenv("output_cost")) + BATCHSIZE = int(os.getenv("batchsize")) + FREQUENCY_PENALTY = float(os.getenv("frequency_penalty")) + +# tqdm Globals +BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" +POSITION = 0 +LEAVE = False + + +def handleImages(folderName, estimate): + global ESTIMATE, TOKENS + ESTIMATE = estimate + start = time.time() + + # Translate Strings + translatedData = openFiles(f"files/{folderName}") + + # Custom Names + customList = [[], []] + customList = processImagesDir("Custom", customList) + + # Write Strings to Images + if not ESTIMATE: + if not os.path.exists(f"translated/{folderName}"): + os.mkdir(f"translated/{folderName}") + for i in range(len(translatedData[0][0])): + try: + translatedList = translatedData[0][0] + originalList = translatedData[0][1] + dimensionsList = translatedData[0][2] + image = stringToImage(translatedList[i], dimensionsList[i][0], dimensionsList[i][1]) + image.save(rf"translated/{folderName}/{customList[0][0]}.png", quality=100) + customList[0].pop(0) + except Exception as e: + PBAR.write(f"{translatedList[i]}: {str(e)}") + # Ignore Error + + # Print File + end = time.time() + tqdm.write(getResultString(translatedData, end - start, folderName)) + with LOCK: + TOKENS[0] += translatedData[1][0] + TOKENS[1] += translatedData[1][1] + + # Print Total + totalString = getResultString(["", TOKENS, None], end - start, "TOTAL") + + # Print any errors on maps + if len(MISMATCH) > 0: + return totalString + Fore.RED + f"\nMismatch Errors: {MISMATCH}" + Fore.RESET + else: + return totalString + + +def openFiles(folderName): + global PBAR + + if os.path.isdir(folderName): + imageList = [[], []] + imageList = processImagesDir(folderName, imageList) + + # Start Translation + with tqdm( + bar_format=BAR_FORMAT, + position=POSITION, + leave=LEAVE, + desc=folderName, + total=len(imageList[0]), + ) as PBAR: + translatedData = translateImages(imageList) + translatedData = [ + [translatedData[0], imageList[0], imageList[1]], + translatedData[1], + translatedData[2], + ] + + return translatedData + else: + print("The provided directory path does not exist.") + + +def getResultString(translatedData, translationTime, filename): + # File Print String + totalTokenstring = ( + Fore.YELLOW + "[Input: " + str(translatedData[1][0]) + "]" + "[Output: " + + str(translatedData[1][1]) + + "]" "[Cost: ${:,.4f}".format((translatedData[1][0] * 0.001 * INPUTAPICOST) + (translatedData[1][1] * 0.001 * OUTPUTAPICOST)) + + "]" + ) + timeString = Fore.BLUE + "[" + str(round(translationTime, 1)) + "s]" + + if translatedData[2] is None: + # Success + return filename + ": " + totalTokenstring + timeString + Fore.GREEN + " \u2713 " + Fore.RESET + else: + # Fail + try: + raise translatedData[2] + except Exception as e: + traceback.print_exc() + errorString = str(e) + Fore.RED + return filename + ": " + totalTokenstring + timeString + Fore.RED + " \u2717 " + errorString + Fore.RESET + + +def getFontSize(text, image_width, image_height, font_path): + # Start with a high font size and keep reducing it until the text fits within the image bounds + font_size = min(image_width, image_height) + + while font_size > 0: + font = ImageFont.truetype(font_path, font_size) + text_bbox = ImageDraw.Draw(Image.new("RGB", (1, 1))).textbbox((0, 0), text, font=font) + text_width = text_bbox[2] - text_bbox[0] + text_height = text_bbox[3] - text_bbox[1] + + if text_width <= image_width and text_height <= image_height: + return font_size + font_size -= 1 + + return font_size + + +def stringToImage(text, width, height, font_path="fonts/TsunagiGothic.ttf", scale_factor=4): + # Increase the resolution + scaled_width = int(width * scale_factor) + scaled_height = int(height * scale_factor) + + # Find the appropriate font size for the scaled up image + font_size = getFontSize(text, scaled_width, scaled_height, font_path) + if font_size == 0: + raise ValueError("Text is too long to fit in the supplied dimensions.") + + # Create a new image with the scaled width and height and a transparent background + image = Image.new("RGBA", (scaled_width, scaled_height), (255, 255, 255, 0)) + + # Create a drawing context + draw = ImageDraw.Draw(image) + + # Load the appropriate font + font = ImageFont.truetype(font_path, font_size) + + # Calculate the size of the text to center it + text_bbox = draw.textbbox((0, 0), text, font=font) + text_width = text_bbox[2] - text_bbox[0] + text_height = text_bbox[3] - text_bbox[1] + 20 + x = 0 + y = (scaled_height - text_height) // 2 + + # Draw the text on the image + draw.text((x, y), text, font=font, fill=(255, 255, 255, 255)) + + # Resize back to the original dimensions to get a clearer text rendering + image = image.resize( + (width, height), + Image.LANCZOS, + ) + + return image + + +def getImageDimensions(file_path): + try: + with Image.open(file_path) as img: + width, height = img.size + return width, height + except Exception as e: + print(f"Error reading {file_path}: {e}") + return None, None + + +def processImagesDir(directory_path, imageList): + for file_name in os.listdir(directory_path): + # .png and Japanese + if ".png" in file_name: + file_path = os.path.join(directory_path, file_name) + if os.path.isfile(file_path): + # Check if the file is an image + try: + width, height = getImageDimensions(file_path) + if width is not None and height is not None: + placeholders = { + ".png": "", + } + for target, replacement in placeholders.items(): + file_name = file_name.replace(target, replacement) + imageList[0].append(file_name) + imageList[1].append([width, height]) + except Exception as e: + print(f"Error processing {file_name}: {e}") + + if ".txt" in file_name: + try: + with open(f"{directory_path}/{file_name}", "r", encoding="utf8") as file: + for line in file: + line = line.strip() + line = line.replace(":", ":") + line = line.replace("/", "/") + line = line.replace("?", "?") + imageList[0].append(line) # Using strip() to remove any extra newlines or spaces + imageList[1].append([104, 15]) + except FileNotFoundError: + print(f"The file at {file_path} was not found.") + except IOError: + print(f"An error occurred while reading the file at {file_path}.") + return imageList + + +def translateImages(imageList): + totalTokens = [0, 0] + + # Translate GPT + response = translateGPT(imageList[0], "Keep the Translation as brief as possible", True) + translatedList = response[0] + totalTokens[0] += response[1][0] + totalTokens[1] += response[1][1] + + return [translatedList, totalTokens, None] + + +# Save some money and enter the character before translation +def getSpeaker(speaker): + match speaker: + case "ファイン": + return ["Fine", [0, 0]] + case "": + return ["", [0, 0]] + case _: + # Find Speaker + for i in range(len(NAMESLIST)): + if speaker == NAMESLIST[i][0]: + return [NAMESLIST[i][1], [0, 0]] + + # Translate and Store Speaker + response = translateGPT( + f"{speaker}", + "Reply with the " + LANGUAGE + " translation of the NPC name.", + True, + ) + response[0] = response[0].title() + response[0] = response[0].replace("'S", "'s") + response[0] = response[0].replace("Speaker: ", "") + + # Retry if name doesn't translate for some reason + if re.search(r"([a-zA-Z??])", response[0]) == None: + response = translateGPT( + f"{speaker}", + "Reply with the " + LANGUAGE + " translation of the NPC name.", + False, + ) + response[0] = response[0].title() + response[0] = response[0].replace("'S", "'s") + + speakerList = [speaker, response[0]] + NAMESLIST.append(speakerList) + return response + return [speaker, [0, 0]] + + +def subVars(jaString): + jaString = jaString.replace("\u3000", " ") + + # Formatting + count = 0 + codeList = re.findall(r"[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]", jaString) + codeList = set(codeList) + if len(codeList) != 0: + for var in codeList: + jaString = jaString.replace(var, "[FCode_" + str(count) + "]") + count += 1 + + # Put all lists in list and return + return [jaString, codeList] + + +def resubVars(translatedText, codeList): + # Fix Spacing and ChatGPT Nonsense + matchList = re.findall(r"\[\s?.+?\s?\]", translatedText) + if len(matchList) > 0: + for match in matchList: + text = match.strip() + translatedText = translatedText.replace(match, text) + + # Formatting + count = 0 + if len(codeList) != 0: + for var in codeList: + translatedText = translatedText.replace("[FCode_" + str(count) + "]", var) + count += 1 + + return translatedText + + +def batchList(input_list, batch_size): + if not isinstance(batch_size, int) or batch_size <= 0: + raise ValueError("batch_size must be a positive integer") + + return [input_list[i : i + batch_size] for i in range(0, len(input_list), batch_size)] + + +def createContext(fullPromptFlag, subbedT, format): + characters = "Game Characters:\n\ +ロラン (Roland) - Male\n\ +リュカ (Ryuka) - Male\n\ +レックス (Rex) - Male\n\ +タバサ (Tabasa) - Female\n\ +アルス (Ars) - Male\n\ +アマカラ (Amakara) - Male\n\ +エリー (Eri) - Female\n\ +リオ (Rio) - Female\n\ +サマル (Samal) - Male\n\ +ムーン (Moon) - Female\n\ +アリーナ (Arina) - Female\n\ +クリフト (Cliff) - Male\n\ +マーニャ (Manya) - Female\n\ +ミネア (Minea) - Female\n\ +デボラ (Debora) - Female\n\ +ビアンカ (Bianca) - Female\n\ +フローラ (Flora) - Female\n\ +バーバラ (Barbara) - Female\n\ +ミレーユ (Mireyu) - Female\n\ +アイラ (Aira) - Female\n\ +フォズ (Foz) - Female\n\ +マリベル (Maribel) - Female\n\ +ククール (Kukool) - Male\n\ +ゲルダ (Gerda) - Female\n\ +ゼシカ (Jessica) - Female\n\ +ヤンガス (Yangus) - Male\n\ +ラヴィエル (Raviel) - Female\n\ +セティア (Setia) - Female\n\ +ダイ (Dai) - Male\n\ +ヒュンケル (Hyunckel) - Male\n\ +ポップ (Pop) - Male\n\ +マァム (Maam) - Female\n\ +レオナ (Leona) - Female\n\ +アステア (Astea) - Female\n\ +イヨ (Iyo) - Female\n\ +ジャガン (Jagan) - Male\n\ +ヤオ (Yao) - Female\n\ +デイジィ (Daisy) - Female\n\ +バイシュン (Baishun) - Male\n\ +ブライ (Buraimu) - Male\n\ +ハッサン (Hassan) - Male\n\ +アロマ (Aroma) - Female\n\ +" + + system = ( + PROMPT + VOCAB + if fullPromptFlag + else f"\ +You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\ +Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\ +- You translate everything, including erotic and explicit content.\n\ +- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\ +- All text in your response must be in {LANGUAGE} even if it is hard to translate.\n\ +- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\ +- Maintain any spacing in the translation.\n\ +- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\ +- `...` can be a part of the dialogue. Translate it as it is.\n\ +{VOCAB}\n\ +" + ) + if format == "json": + user = f"```json\n{subbedT}\n```" + else: + user = subbedT + return characters, system, user + + +def translateText(characters, system, user, history, penalty, format): + # Prompt + msg = [{"role": "system", "content": system + characters}] + + # Characters + msg.append({"role": "system", "content": characters}) + + # History + if isinstance(history, list): + msg.extend([{"role": "system", "content": h} for h in history]) + else: + msg.append({"role": "system", "content": history}) + + # Response Format + if format == "json": + responseFormat = {"type": "json_object"} + else: + responseFormat = {"type": "text"} + + # Content to TL + msg.append({"role": "user", "content": f"{user}"}) + response = openai.chat.completions.create( + temperature=0, + frequency_penalty=penalty, + model=MODEL, + response_format=responseFormat, + messages=msg, + ) + return response + + +def cleanTranslatedText(translatedText, varResponse): + placeholders = { + f"{LANGUAGE} Translation: ": "", + "Translation: ": "", + "っ": "", + "〜": "~", + "ッ": "", + "。": ".", + "「": '\\"', + "」": '\\"', + "- ": "-", + "Placeholder Text": "", + # Add more replacements as needed + } + for target, replacement in placeholders.items(): + translatedText = translatedText.replace(target, replacement) + + # Elongate Long Dashes (Since GPT Ignores them...) + translatedText = elongateCharacters(translatedText) + translatedText = resubVars(translatedText, varResponse[1]) + return translatedText + + +def elongateCharacters(text): + # Define a pattern to match one character followed by one or more `ー` characters + # Using a positive lookbehind assertion to capture the preceding character + pattern = r"(?<=(.))ー+" + + # Define a replacement function that elongates the captured character + def repl(match): + char = match.group(1) # The character before the ー sequence + count = len(match.group(0)) - 1 # Number of ー characters + return char * count # Replace ー sequence with the character repeated + + # Use re.sub() to replace the pattern in the text + return re.sub(pattern, repl, text) + + +def extractTranslation(translatedTextList, is_list): + try: + line_dict = json.loads(translatedTextList) + # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. + string_list = list(line_dict.values()) + if is_list: + return string_list + else: + return string_list[0] + + except Exception as e: + print(f"extractTranslation Error: {e}") + return None + + +def countTokens(characters, system, user, history): + inputTotalTokens = 0 + outputTotalTokens = 0 + enc = tiktoken.encoding_for_model("gpt-4") + + # Input + if isinstance(history, list): + for line in history: + inputTotalTokens += len(enc.encode(line)) + else: + inputTotalTokens += len(enc.encode(history)) + inputTotalTokens += len(enc.encode(system)) + inputTotalTokens += len(enc.encode(characters)) + inputTotalTokens += len(enc.encode(user)) + + # Output + outputTotalTokens += round(len(enc.encode(user)) * 3) + + return [inputTotalTokens, outputTotalTokens] + + +@retry(exceptions=Exception, tries=5, delay=5) +def translateGPT(text, history, fullPromptFlag): + global PBAR + + mismatch = False + totalTokens = [0, 0] + if isinstance(text, list): + format = "json" + tList = batchList(text, BATCHSIZE) + else: + format = "text" + tList = [text] + + for index, tItem in enumerate(tList): + # Before sending to translation, if we have a list of items, add the formatting + if isinstance(tItem, list): + payload = {f"Line{i+1}": string for i, string in enumerate(tItem)} + payload = json.dumps(payload, indent=4, ensure_ascii=False) + varResponse = subVars(payload) + subbedT = varResponse[0] + else: + varResponse = subVars(tItem) + subbedT = varResponse[0] + + # Things to Check before starting translation + if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+", subbedT): + if PBAR is not None: + PBAR.update(len(tItem)) + continue + + # Create Message + characters, system, user = createContext(fullPromptFlag, subbedT, format) + + # Calculate Estimate + if ESTIMATE: + estimate = countTokens(characters, system, user, history) + totalTokens[0] += estimate[0] + totalTokens[1] += estimate[1] + continue + + # Translating + response = translateText(characters, system, user, history, 0.05, format) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Check Translation + translatedText = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedText, True) + if extractedTranslations == None or len(tItem) != len(extractedTranslations): + # Mismatch. Try Again + response = translateText(characters, system, user, history, 0.05, format) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Formatting + translatedText = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedText, True) + if extractedTranslations == None or len(tItem) != len(extractedTranslations): + mismatch = True # Just here for breakpoint + + # Set if no mismatch + if mismatch == False: + tList[index] = extractedTranslations + history = extractedTranslations[-10:] # Update history if we have a list + else: + history = text[-10:] + mismatch = False + + # Update Loading Bar + with LOCK: + if PBAR is not None: + PBAR.update(len(tItem)) + else: + # Ensure we're passing a single string to extractTranslation + tList[index] = translatedText.replace("Placeholder Text", "") + + # Combine if multilist + if isinstance(tList[0], list): + tList = [t for sublist in tList for t in sublist] + + # Return + if format == "json": + return [tList, totalTokens] + else: + return [tList[0], totalTokens] diff --git a/modules/irissoft.py b/modules/irissoft.py index e82e858..033ec43 100644 --- a/modules/irissoft.py +++ b/modules/irissoft.py @@ -46,6 +46,9 @@ BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" POSITION = 0 LEAVE = False +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. diff --git a/modules/javascript.py b/modules/javascript.py index bc7e2b3..36f3e52 100644 --- a/modules/javascript.py +++ b/modules/javascript.py @@ -46,6 +46,9 @@ BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" POSITION = 0 LEAVE = False +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. diff --git a/modules/json.py b/modules/json.py index 006f9df..7441533 100644 --- a/modules/json.py +++ b/modules/json.py @@ -47,6 +47,9 @@ BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" POSITION = 0 LEAVE = False +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. @@ -496,7 +499,7 @@ def translateGPT(text, history, fullPromptFlag): subbedT = varResponse[0] # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", subbedT): + if not re.search(LANGREGEX, subbedT): if PBAR is not None: PBAR.update(len(tItem)) history = tItem[-MAXHISTORY:] diff --git a/modules/kansen.py b/modules/kansen.py index f14c76a..ecc40a3 100644 --- a/modules/kansen.py +++ b/modules/kansen.py @@ -46,6 +46,9 @@ BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" POSITION = 0 LEAVE = False +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. diff --git a/modules/kirikiri.py b/modules/kirikiri.py index 35b3feb..61ed6e6 100644 --- a/modules/kirikiri.py +++ b/modules/kirikiri.py @@ -54,6 +54,10 @@ SPEAKERS = True CHOICES = True DIALOGUE = True + +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. @@ -539,7 +543,7 @@ def translateGPT(text, history, fullPromptFlag): subbedT = varResponse[0] # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", subbedT): + if not re.search(LANGREGEX, subbedT): if PBAR is not None: PBAR.update(len(tItem)) history = tItem[-MAXHISTORY:] diff --git a/modules/lune.py b/modules/lune.py index acefc9e..6e6bb2b 100644 --- a/modules/lune.py +++ b/modules/lune.py @@ -49,6 +49,9 @@ BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" POSITION = 0 LEAVE = False +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. diff --git a/modules/nscript.py b/modules/nscript.py index 0424b57..6f92b27 100644 --- a/modules/nscript.py +++ b/modules/nscript.py @@ -55,6 +55,9 @@ ascii_to_wide.update({0x20: "\u3000", 0x2D: "\u2212"}) # space and minus wide_to_ascii = dict((i, chr(i - 0xFEE0)) for i in range(0xFF01, 0xFF5F)) wide_to_ascii.update({0x3000: " ", 0x2212: "-"}) # space and minus +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. @@ -555,7 +558,7 @@ def translateGPT(text, history, fullPromptFlag): subbedT = varResponse[0] # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", subbedT): + if not re.search(LANGREGEX, subbedT): if PBAR is not None: PBAR.update(len(tItem)) history = tItem[-MAXHISTORY:] diff --git a/modules/regex.py b/modules/regex.py index 278109e..85038d2 100644 --- a/modules/regex.py +++ b/modules/regex.py @@ -49,6 +49,9 @@ POSITION = 0 LEAVE = False PBAR = None +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. @@ -550,7 +553,7 @@ def translateGPT(text, history, fullPromptFlag): subbedT = varResponse[0] # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", subbedT): + if not re.search(LANGREGEX, subbedT): if PBAR is not None: PBAR.update(len(tItem)) history = tItem[-MAXHISTORY:] diff --git a/modules/renpy.py b/modules/renpy.py index d21aa6f..6e7abf0 100644 --- a/modules/renpy.py +++ b/modules/renpy.py @@ -49,6 +49,9 @@ POSITION = 0 LEAVE = False PBAR = None +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. @@ -479,7 +482,7 @@ def translateGPT(text, history, fullPromptFlag): subbedT = varResponse[0] # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", subbedT): + if not re.search(LANGREGEX, subbedT): if PBAR is not None: PBAR.update(len(tItem)) history = tItem[-MAXHISTORY:] diff --git a/modules/rpgmakerace.py b/modules/rpgmakerace.py index c2a43d3..19726fd 100644 --- a/modules/rpgmakerace.py +++ b/modules/rpgmakerace.py @@ -49,6 +49,9 @@ BRACKETNAMES = False PBAR = None FILENAME = None +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. @@ -1276,7 +1279,7 @@ def searchCodes(page, pbar, jobList, filename): # If there isn't any Japanese in the text just skip if not re.search( - r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", + LANGREGEX, jaString, ): i += 1 @@ -1302,7 +1305,7 @@ def searchCodes(page, pbar, jobList, filename): # If there isn't any Japanese in the text just skip if not re.search( - r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", + LANGREGEX, jaString, ): i += 1 @@ -1456,7 +1459,7 @@ def searchCodes(page, pbar, jobList, filename): continue # If there isn't any Japanese in the text just skip - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", jaString): + if not re.search(LANGREGEX, jaString): i += 1 continue @@ -1590,7 +1593,7 @@ def searchCodes(page, pbar, jobList, filename): jaString = codeList[i]["p"][0] # If there isn't any Japanese in the text just skip - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", jaString): + if not re.search(LANGREGEX, jaString): i += 1 continue @@ -1635,7 +1638,7 @@ def searchCodes(page, pbar, jobList, filename): jaString = codeList[i]["p"][0] # If there isn't any Japanese in the text just skip - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", jaString): + if not re.search(LANGREGEX, jaString): i += 1 continue @@ -1955,7 +1958,7 @@ def searchCodes(page, pbar, jobList, filename): continue # If there isn't any Japanese in the text just skip - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", jaString): + if not re.search(LANGREGEX, jaString): i += 1 continue @@ -2562,7 +2565,7 @@ def translateGPT(text, history, fullPromptFlag): subbedT = varResponse[0] # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", subbedT): + if not re.search(LANGREGEX, subbedT): if PBAR is not None: PBAR.update(len(tItem)) history = tItem[-MAXHISTORY:] diff --git a/modules/rpgmakermvmz.py b/modules/rpgmakermvmz.py index 9fc0d69..d9b70c4 100644 --- a/modules/rpgmakermvmz.py +++ b/modules/rpgmakermvmz.py @@ -46,7 +46,7 @@ MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list res PBAR = None FILENAME = None -# Regex - Need to change this if you want to translate from/to other languages +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" # Pricing - Depends on the model https://openai.com/pricing @@ -83,11 +83,11 @@ CODE101 = False # Turn this one when names exist in 101 CODE408 = False # Warning, translates comments and can inflate costs. # Variables -CODE122 = False +CODE122 = True # Other CODE355655 = False -CODE357 = True +CODE357 = False CODE657 = False CODE356 = False CODE320 = False @@ -1148,7 +1148,7 @@ def searchCodes(page, pbar, jobList, filename): ## Event Code: 122 [Set Variables] if "code" in codeList[i] and codeList[i]["code"] == 122 and CODE122 is True: # This is going to be the var being set. (IMPORTANT) - if codeList[i]["parameters"][0] not in list(range(42, 45)): + if codeList[i]["parameters"][0] not in list(range(95, 96)): i += 1 continue @@ -1549,25 +1549,26 @@ def searchCodes(page, pbar, jobList, filename): ## Event Code: 355 or 655 Scripts [Optional] if "code" in codeList[i] and (codeList[i]["code"] == 355 or codeList[i]["code"] == 655) and CODE355655 is True: jaString = codeList[i]["parameters"][0] - regex = r'.*subject=(.*?)"' + regexPatterns = [r'.*subject=(.*?)"', r"テキスト-(.*)"] - # Var Text - match = re.search(regex, jaString) - if re.search(regex, jaString): - finalJAString = match.group(1) - # Pass 1 - if setData is False: - list355655.append(finalJAString) + # Iterate over the list of regex patterns + for regex in regexPatterns: + match = re.search(regex, jaString) + if re.search(regex, jaString): + finalJAString = match.group(1) + # Pass 1 + if setData is False: + list355655.append(finalJAString) - # Pass 2 - else: - # Grab and Replace - translatedText = list355655[0] - translatedText = translatedText.replace("'", "\\'") + # Pass 2 + else: + # Grab and Replace + translatedText = list355655[0] + translatedText = translatedText.replace("'", "\\'") - # Set - codeList[i]["parameters"][0] = codeList[i]["parameters"][0].replace(finalJAString, translatedText) - list355655.pop(0) + # Set + codeList[i]["parameters"][0] = codeList[i]["parameters"][0].replace(finalJAString, translatedText) + list355655.pop(0) ## Event Code: 408 (Script) if "code" in codeList[i] and (codeList[i]["code"] == 408) and CODE408 is True: diff --git a/modules/rpgmakerplugin.py b/modules/rpgmakerplugin.py index 270d167..8b01579 100644 --- a/modules/rpgmakerplugin.py +++ b/modules/rpgmakerplugin.py @@ -48,6 +48,9 @@ POSITION = 0 LEAVE = False PBAR = None +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. @@ -470,7 +473,7 @@ def translateGPT(text, history, fullPromptFlag): subbedT = varResponse[0] # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", subbedT): + if not re.search(LANGREGEX, subbedT): if PBAR is not None: PBAR.update(len(tItem)) history = tItem[-MAXHISTORY:] diff --git a/modules/text.py b/modules/text.py index c259869..b5aeefb 100644 --- a/modules/text.py +++ b/modules/text.py @@ -49,6 +49,9 @@ POSITION = 0 LEAVE = False PBAR = None +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. @@ -468,7 +471,7 @@ def translateGPT(text, history, fullPromptFlag): subbedT = varResponse[0] # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", subbedT): + if not re.search(LANGREGEX, subbedT): if PBAR is not None: PBAR.update(len(tItem)) history = tItem[-MAXHISTORY:] diff --git a/modules/tyrano.py b/modules/tyrano.py index 0519f90..0406695 100644 --- a/modules/tyrano.py +++ b/modules/tyrano.py @@ -49,6 +49,9 @@ POSITION = 0 LEAVE = False PBAR = None +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. @@ -542,7 +545,7 @@ def translateGPT(text, history, fullPromptFlag): subbedT = varResponse[0] # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", subbedT): + if not re.search(LANGREGEX, subbedT): if PBAR is not None: PBAR.update(len(tItem)) history = tItem[-MAXHISTORY:] diff --git a/modules/unity.py b/modules/unity.py index 83e3016..5b8d71d 100644 --- a/modules/unity.py +++ b/modules/unity.py @@ -55,6 +55,9 @@ ascii_to_wide.update({0x20: "\u3000", 0x2D: "\u2212"}) # space and minus wide_to_ascii = dict((i, chr(i - 0xFEE0)) for i in range(0xFF01, 0xFF5F)) wide_to_ascii.update({0x3000: " ", 0x2212: "-"}) # space and minus +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. @@ -482,7 +485,7 @@ def translateGPT(text, history, fullPromptFlag): subbedT = varResponse[0] # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", subbedT): + if not re.search(LANGREGEX, subbedT): if PBAR is not None: PBAR.update(len(tItem)) continue diff --git a/modules/wolf.py b/modules/wolf.py index 4f5c48f..6222369 100644 --- a/modules/wolf.py +++ b/modules/wolf.py @@ -46,6 +46,9 @@ MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list res FILENAME = None BRACKETNAMES = False +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. @@ -2344,7 +2347,7 @@ def translateGPT(text, history, fullPromptFlag): subbedT = varResponse[0] # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", subbedT): + if not re.search(LANGREGEX, subbedT): if PBAR is not None: PBAR.update(len(tItem)) history = tItem[-MAXHISTORY:] diff --git a/modules/wolf2.py b/modules/wolf2.py index e56aa1c..39fbaba 100644 --- a/modules/wolf2.py +++ b/modules/wolf2.py @@ -47,6 +47,9 @@ BAR_FORMAT = "{l_bar}{bar:10}{r_bar}{bar:-10b}" POSITION = 0 LEAVE = False +# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex +LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+" + # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request # If you are getting a MISMATCH LENGTH error, lower the batch size. @@ -507,7 +510,7 @@ def translateGPT(text, history, fullPromptFlag): subbedT = varResponse[0] # Things to Check before starting translation - if not re.search(r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+", subbedT): + if not re.search(LANGREGEX, subbedT): if PBAR is not None: PBAR.update(len(tItem)) history = tItem[-MAXHISTORY:]