From 17c81bf5d7b907dc2174f662ca5f34845c08d305 Mon Sep 17 00:00:00 2001 From: Dazed Date: Thu, 2 Nov 2023 13:36:34 -0500 Subject: [PATCH] Update Generic Tyrano script --- modules/tyrano.py | 187 +++++++++++++++++++++++++++++++++++++--------- 1 file changed, 152 insertions(+), 35 deletions(-) diff --git a/modules/tyrano.py b/modules/tyrano.py index b5a4308..471782f 100644 --- a/modules/tyrano.py +++ b/modules/tyrano.py @@ -1,9 +1,7 @@ from concurrent.futures import ThreadPoolExecutor, as_completed -import json import os from pathlib import Path import re -import sys import textwrap import threading import time @@ -25,7 +23,7 @@ APICOST = .002 # Depends on the model https://openai.com/pricing PROMPT = Path('prompt.txt').read_text(encoding='utf-8') THREADS = 10 # For GPT4 rate limit will be hit if you have more than 1 thread. LOCK = threading.Lock() -WIDTH = 60 +WIDTH = 80 LISTWIDTH = 60 MAXHISTORY = 10 ESTIMATE = '' @@ -41,8 +39,8 @@ LEAVE=False # Flags NAMES = False # Output a list of all the character names found -BRFLAG = False # If the game uses
instead -FIXTEXTWRAP = False +FIXTEXTWRAP = True +IGNORETLTEXT = True def handleTyrano(filename, estimate): global ESTIMATE, TOKENS, TOTALTOKENS, TOTALCOST @@ -63,7 +61,7 @@ def handleTyrano(filename, estimate): else: try: - with open('translated/' + filename, 'w', encoding='UTF-8') as outFile: + with open('translated/' + filename, 'w', encoding='utf-8', errors='ignore') as outFile: start = time.time() translatedData = openFiles(filename) @@ -126,21 +124,22 @@ def translateTyrano(data, pbar): i = syncIndex # Speaker - if '#' in data[i]: - matchList = re.findall(r'#(.+)', data[i]) + if '[ns]' in data[i]: + matchList = re.findall(r'\[ns\](.+?)\[', data[i]) if len(matchList) != 0: response = translateGPT(matchList[0], 'Reply with only the english translation of the NPC name', True) speaker = response[0] tokens += response[1] - data[i] = '#' + speaker + '\n' + data[i] = '[ns]' + speaker + '[nse]\n' else: speaker = '' # Choices - elif 'glink' in data[i]: - matchList = re.findall(r'text=\"(.+?)\"', data[i]) + elif '[eval exp="f.seltext' in data[i]: + matchList = re.findall(r'\[eval exp=.+?\'(.+)\'', data[i]) if len(matchList) != 0: if len(textHistory) > 0: + originalText = matchList[0] response = translateGPT(matchList[0], 'Past Translated Text: ' + textHistory[len(textHistory)-1] + '\n\nReply in the style of a dialogue option.', True) else: response = translateGPT(matchList[0], '', False) @@ -152,31 +151,46 @@ def translateTyrano(data, pbar): for char in charList: translatedText = translatedText.replace(char, '') + # Escape all ' + translatedText = translatedText.replace('\\', '') + translatedText = translatedText.replace("'", "\\\'") + # Set Data - translatedText = 'text=\"' + translatedText.replace(' ', ' ') + '\"' - data[i] = re.sub(r'text=\"(.+?)\"', translatedText, data[i]) + translatedText = data[i].replace(originalText, translatedText) + data[i] = translatedText # Lines - elif '[p]' in data[i]: - matchList = re.findall(r'(.+?)\[p\]', data[i]) - if len(matchList) > 0: - matchList[0] = matchList[0].replace('「', '') - matchList[0] = matchList[0].replace('」', '') - currentGroup.append(matchList[0]) - if len(data) > i+1: - while '[p]' in data[i+1]: - data[i] = '\d\n' - i += 1 - matchList = re.findall(r'(.+?)\[p\]', data[i]) - if len(matchList) > 0: - matchList[0] = matchList[0].replace('「', '') - matchList[0] = matchList[0].replace('」', '') - currentGroup.append(matchList[0]) + matchList = re.findall(r'(.+?)\[r\]$', data[i]) + if len(matchList) > 0: + matchList[0] = matchList[0].replace('「', '') + matchList[0] = matchList[0].replace('」', '') + currentGroup.append(matchList[0]) + if len(data) > i+1: + while '[r]' in data[i+1]: + data[i] = '\d\n' # \d Marks line for deletion + i += 1 + matchList = re.findall(r'(.+?)\[r\]', data[i]) + if len(matchList) > 0: + matchList[0] = matchList[0].replace('「', '') + matchList[0] = matchList[0].replace('」', '') + currentGroup.append(matchList[0]) + while '[pcm]' in data[i+1]: + data[i] = '\d\n' + i += 1 + matchList = re.findall(r'(.+?)\[pcm\]', data[i]) + if len(matchList) > 0: + matchList[0] = matchList[0].replace('「', '') + matchList[0] = matchList[0].replace('」', '') + currentGroup.append(matchList[0]) # Join up 401 groups for better translation. if len(currentGroup) > 0: - finalJAString = ''.join(currentGroup) + finalJAString = ' '.join(currentGroup) oldjaString = finalJAString + # Remove any textwrap + if FIXTEXTWRAP == True: + finalJAString = finalJAString.replace('[r]', ' ') + #Check Speaker if speaker == '': response = translateGPT(finalJAString, 'Previous Dialogue: ' + '\n\n'.join(textHistory), True) @@ -197,8 +211,10 @@ def translateTyrano(data, pbar): translatedText = translatedText.replace('っ', '') translatedText = translatedText.replace('ー', '') translatedText = translatedText.replace('\"', '') + translatedText = translatedText.replace('[', '') + translatedText = translatedText.replace(']', '') - # Format Text + # Split final string into full sentences. matchList = re.findall(r'(.+?[)\.\?\!)。・]+)', translatedText) translatedText = re.sub(r'(.+?[)\.\?\!)。・]+)', '', translatedText) @@ -207,7 +223,7 @@ def translateTyrano(data, pbar): matchList[k] = matchList[k].strip() j=0 while(len(matchList) > j+1): - while len(matchList[j]) < 100 and len(matchList) > j: + while len(matchList[j]) < 30 and len(matchList) > j: matchList[j:j+2] = [' '.join(matchList[j:j+2])] if len(matchList) == j+1: matchList[j] = matchList[j] + ' ' + translatedText @@ -215,20 +231,121 @@ def translateTyrano(data, pbar): break j+=1 + # Normal Lines if len(matchList) > 0: data[i] = '\d\n' for line in matchList: - data.insert(i, line.strip() + '[p]\n') + # Wordwrap Text + if '[r]' not in line: + line = textwrap.fill(line, width=WIDTH) + line = line.replace('\n', '[r]') + + # Set + data.insert(i, line.strip() + '[l][er]\n') i+=1 + data[i-1] = data[i-1].replace('[l][er]', '[pcm]') # else: # print ('No Matches') + + # Backup TL if translatedText != '': - data[i] = translatedText.strip() + '[p]\n' + # Wordwrap Text + if '[r]' not in translatedText: + translatedText = textwrap.fill(translatedText, width=WIDTH) + translatedText = translatedText.replace('\n', '[r]') + + # Set Last Line + data[i] = translatedText.strip() + '[pcm]\n' # Keep textHistory list at length maxHistory if len(textHistory) > maxHistory: textHistory.pop(0) currentGroup = [] + speaker = '' + + # pcm Line + matchList = re.findall(r'(.+?)\[pcm\]$', data[i]) + if len(matchList) > 0: + matchList[0] = matchList[0].replace('「', '') + matchList[0] = matchList[0].replace('」', '') + finalJAString = matchList[0] + + # Remove any textwrap + if FIXTEXTWRAP == True: + finalJAString = finalJAString.replace('[r]', ' ') + + #Check Speaker + if speaker == '': + response = translateGPT(finalJAString, 'Previous Dialogue: ' + '\n\n'.join(textHistory), True) + tokens += response[1] + translatedText = response[0] + textHistory.append('\"' + translatedText + '\"') + else: + response = translateGPT(speaker + ': ' + finalJAString, 'Previous Dialogue: ' + '\n\n'.join(textHistory), True) + tokens += response[1] + translatedText = response[0] + textHistory.append('\"' + translatedText + '\"') + + # Remove added speaker + translatedText = re.sub(r'^.+:\s?', '', translatedText) + + # Set Data + translatedText = translatedText.replace('ッ', '') + translatedText = translatedText.replace('っ', '') + translatedText = translatedText.replace('ー', '') + translatedText = translatedText.replace('\"', '') + translatedText = translatedText.replace('[', '') + translatedText = translatedText.replace(']', '') + + # Format Text + matchList = re.findall(r'(.+?[)\.\?\!)。・]+)', translatedText) + translatedText = re.sub(r'(.+?[)\.\?\!)。・]+)', '', translatedText) + + # Get rid of whitespace for each item and add wordwrap + for k in range(len(matchList)): + matchList[k] = matchList[k].strip() + + # Combine Sentences with a max limit (Wordwrap basically) + j=0 + while(len(matchList) > j+1): + while len(matchList[j]) < 30 and len(matchList) > j: + matchList[j:j+2] = [' '.join(matchList[j:j+2])] + if len(matchList) == j+1: + matchList[j] = matchList[j] + ' ' + translatedText + translatedText = '' + break + j+=1 + + # Set Data + if len(matchList) > 0: + data[i] = '\d\n' + for line in matchList: + # Wordwrap Text + if '[r]' not in line: + line = textwrap.fill(line, width=WIDTH) + line = line.replace('\n', '[r]') + + # Set + data.insert(i, line.strip() + '[l][er]\n') + i+=1 + # Set last line as [pcm] instead of [r] + data[i-1] = data[i-1].replace('[l][er]', '[pcm]') + # else: + # print ('No Matches') + if translatedText != '': + # Wordwrap Text + if '[r]' not in translatedText: + translatedText = textwrap.fill(translatedText, width=WIDTH) + translatedText = translatedText.replace('\n', '[r]') + + # Set Backup + data[i] = translatedText.strip() + '[pcm]' + + # Keep textHistory list at length maxHistory + if len(textHistory) > maxHistory: + textHistory.pop(0) + currentGroup = [] + speaker = '' currentGroup = [] pbar.update(1) @@ -361,10 +478,10 @@ def translateGPT(t, history, fullPromptFlag): return(t, 0) """Translate text using GPT""" - context = 'Eroge Names Context: (Name: サクラ == Sakura\nGender: Female,\n\nName: ルリカ == Rurika\nGender: Female,\n\nName: ケンイチ == Kenichi\nGender: Male)' + context = 'Eroge Names Context: (Name: 佐伯 瞳 == Saeki Hitomi\nGender: Female, Name: 山岸 優 == Yamagishi Yuu Gender: Female, Name: 五十嵐 朋美 == Igarashi Tomomi Gender: Female, Name: 新道 リサ == Shindou Risa Gender: Female, Name: 岸田 聡 == Kishida Satoshi Gender: Male, Name: 竹内 真也 == Takeuchi Shinya Gender: Male, Name: 田中 祐二 == Tanaka Yuuji Gender: Male)' if fullPromptFlag: system = PROMPT - user = 'Line to Translate: ' + subbedT + user = 'Line to Translate = ' + subbedT else: system = 'You are an expert translator who translates everything to English. Reply with only the English Translation of the text.' user = 'Line to Translate: ' + subbedT