From 79d159192b0c6a96ad8da64243b819f5f6e33022 Mon Sep 17 00:00:00 2001 From: DazedAnon Date: Tue, 2 Jul 2024 11:58:48 -0500 Subject: [PATCH] Fix error --- modules/eushully.py | 2 +- modules/nscript.py | 165 ++++++++++++++++++++++---------------- modules/rpgmakermvmz.py | 2 +- modules/rpgmakerplugin.py | 2 +- modules/wolf.py | 80 +++++++++++++++--- 5 files changed, 167 insertions(+), 84 deletions(-) diff --git a/modules/eushully.py b/modules/eushully.py index 38dc7a2..e2879bc 100644 --- a/modules/eushully.py +++ b/modules/eushully.py @@ -559,7 +559,7 @@ def cleanTranslatedText(translatedText, varResponse): '': '>', '「': '\"', - ' 」': '\"', + '」': '\"', 'Placeholder Text': '', '- chan': '-chan', '- kun': '-kun', diff --git a/modules/nscript.py b/modules/nscript.py index ff4224f..a74b235 100644 --- a/modules/nscript.py +++ b/modules/nscript.py @@ -38,6 +38,8 @@ MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list re BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' POSITION = 0 LEAVE = False +PBAR = None +FILENAME = None # Pricing - Depends on the model https://openai.com/pricing # Batch Size - GPT 3.5 Struggles past 15 lines per request. GPT4 struggles past 50 lines per request @@ -52,8 +54,9 @@ elif 'gpt-4' in MODEL: BATCHSIZE = 40 def handleOnscripter(filename, estimate): - global ESTIMATE + global ESTIMATE, FILENAME ESTIMATE = estimate + FILENAME = filename if ESTIMATE: start = time.time() @@ -156,60 +159,63 @@ def translateOnscripter(data, pbar, filename, translatedList): tokens = [0,0] speaker = '' voice = False - global LOCK, ESTIMATE + global LOCK, ESTIMATE, PBAR + PBAR = pbar i = 0 while i < len(data): voice = False speaker = '' - if re.search(r'^caption|^event|^operation_caption|^manual|^mes\s\d+', data[i]): - # Lines - match = re.search(r'="(.+?)"', data[i]) - if match == None: - match = re.search(r'mes\s\d+,"(.+?)"', data[i]) - if match != None and match.group(1) != '': - originalString = match.group(1) - # Pass 1 - if translatedList == []: - # Grab Consecutive Strings - jaString = match.group(1) - - # Remove any textwrap - jaString = jaString.replace('\\n', ' ') + regex = r'^([\u3000「(【[][^\n]+)' + # Lines + match = re.search(regex, data[i]) + if match != None and match.group(1) != '': + originalString = match.group(1) + # Pass 1 + if translatedList == []: + # Grab Consecutive Strings + jaString = match.group(1) + while len(data) > i+1 and re.match(regex, data[i+1]): + i += 1 + jaString = f'{jaString} {data[i]}' + + # Remove any textwrap and \u3000 and \ + jaString = jaString.replace('\n', ' ') + jaString = jaString.replace('\u3000', '') + jaString = jaString.replace('\\', '') - # Remove Furigana - furiMatch = re.findall(r'({(.+?)\/(.+?)})', jaString) - if furiMatch: - for match in furiMatch: - jaString = jaString.replace(match[0], match[2]) + # Remove Furigana + furiMatch = re.findall(r'({(.+?)\/(.+?)})', jaString) + if furiMatch: + for match in furiMatch: + jaString = jaString.replace(match[0], match[2]) - # Add String - stringList.append(jaString.strip()) - - # Pass 2 - else: - # Get Text - if translatedList: - # Grab and Pop - translatedText = translatedList[0] - translatedList.pop(0) - - # Set to None if empty list - if len(translatedList) <= 0: - translatedList = None - - # Textwrap - translatedText = textwrap.fill(translatedText, width=WIDTH) - translatedText = translatedText.replace('\n', '\\n') - translatedText = translatedText.replace('\"', '\'') - - # Set Data - data[i] = data[i].replace(originalString, translatedText) - i += 1 - - # Nothing relevant. Skip Line. + # Add String + stringList.append(jaString.strip()) + + # Pass 2 else: - i += 1 + # Get Text + if translatedList: + # Grab and Pop + translatedText = translatedList[0] + translatedList.pop(0) + + # Set to None if empty list + if len(translatedList) <= 0: + translatedList = None + + # Textwrap & Other Text + translatedText = textwrap.fill(translatedText, width=WIDTH) + translatedText = translatedText.replace('\n', '\n\u3000') + translatedText = translatedText.replace('\"', '\'') + translatedText = f'\u3000{translatedText}\\' + + # Set Data + data[i] = data[i].replace(originalString, translatedText) + i += 1 + + # Nothing relevant. Skip Line. else: i += 1 @@ -220,7 +226,7 @@ def translateOnscripter(data, pbar, filename, translatedList): pbar.refresh() # Translate - response = translateGPT(stringList, '', True, pbar, filename) + response = translateGPT(stringList, '', True) tokens[0] += response[1][0] tokens[1] += response[1][1] translatedList = response[0] @@ -237,7 +243,7 @@ def translateOnscripter(data, pbar, filename, translatedList): return tokens # Save some money and enter the character before translation -def getSpeaker(speaker, pbar, filename): +def getSpeaker(speaker): match speaker: case 'ファイン': return ['Fine', [0,0]] @@ -246,12 +252,19 @@ def getSpeaker(speaker, pbar, filename): case _: # Store Speaker if speaker not in str(NAMESLIST): - response = translateGPT(speaker, 'Reply with only the '+ LANGUAGE +' translation of the NPC name.', False, pbar, filename) + response = translateGPT(speaker, 'Reply with the '+ LANGUAGE +' translation of the NPC name.', False) + response[0] = response[0].title() response[0] = response[0].replace("'S", "'s") + + # Retry if name doesn't translate for some reason + if re.search(r'([a-zA-Z??])', response[0]) == None: + response = translateGPT(speaker, 'Reply with the '+ LANGUAGE +' translation of the NPC name.', False) + response[0] = response[0].title() + response[0] = response[0].replace("'S", "'s") + speakerList = [speaker, response[0]] NAMESLIST.append(speakerList) return response - # Find Speaker else: for i in range(len(NAMESLIST)): @@ -381,7 +394,7 @@ def batchList(input_list, batch_size): def createContext(fullPromptFlag, subbedT): characters = 'Game Characters:\n\ -エル (El) - Female\n\ +カエデ (Kaede) - Female\n\ ' system = PROMPT + VOCAB if fullPromptFlag else \ @@ -400,7 +413,7 @@ Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{ user = f'{subbedT}' return characters, system, user -def translateText(characters, system, user, history): +def translateText(characters, system, user, history, penalty): # Prompt msg = [{"role": "system", "content": system + characters}] @@ -416,8 +429,8 @@ def translateText(characters, system, user, history): # Content to TL msg.append({"role": "user", "content": f'{user}'}) response = openai.chat.completions.create( - temperature=0.1, - frequency_penalty=0.1, + temperature=0, + frequency_penalty=penalty, model=MODEL, messages=msg, ) @@ -431,7 +444,13 @@ def cleanTranslatedText(translatedText, varResponse): '〜': '~', 'ッ': '', '。': '.', - 'Placeholder Text': '' + '< ': '<', + '': '>', + '「': '\"', + '」': '\"', + '- ': '-', + 'Placeholder Text': '', # Add more replacements as needed } for target, replacement in placeholders.items(): @@ -460,7 +479,7 @@ def extractTranslation(translatedTextList, is_list): pattern = r'`?<[Ll]ine\d+>([\\]*.*?[\\]*?)<\/?[Ll]ine\d+>`?' # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. if is_list: - matchList = re.findall(pattern, translatedTextList) + matchList = re.findall(pattern, translatedTextList, flags=re.DOTALL) return matchList else: matchList = re.findall(pattern, translatedTextList) @@ -482,7 +501,7 @@ def countTokens(characters, system, user, history): inputTotalTokens += len(enc.encode(user)) # Output - outputTotalTokens += round(len(enc.encode(user))*2) + outputTotalTokens += round(len(enc.encode(user))*3) return [inputTotalTokens, outputTotalTokens] @@ -492,7 +511,9 @@ def combineList(tlist, text): return tlist[0] @retry(exceptions=Exception, tries=5, delay=5) -def translateGPT(text, history, fullPromptFlag, pbar, filename): +def translateGPT(text, history, fullPromptFlag): + global PBAR + mismatch = False totalTokens = [0, 0] if isinstance(text, list): @@ -513,6 +534,8 @@ def translateGPT(text, history, fullPromptFlag, pbar, filename): # Things to Check before starting translation if not re.search(r'[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+', subbedT): + if PBAR is not None: + PBAR.update(len(tItem)) continue # Create Message @@ -526,7 +549,7 @@ def translateGPT(text, history, fullPromptFlag, pbar, filename): continue # Translating - response = translateText(characters, system, user, history) + response = translateText(characters, system, user, history, 0.02) translatedText = response.choices[0].message.content totalTokens[0] += response.usage.prompt_tokens totalTokens[1] += response.usage.completion_tokens @@ -535,9 +558,10 @@ def translateGPT(text, history, fullPromptFlag, pbar, filename): translatedText = cleanTranslatedText(translatedText, varResponse) if isinstance(tItem, list): extractedTranslations = extractTranslation(translatedText, True) + tList[index] = extractedTranslations if len(tItem) != len(extractedTranslations): # Mismatch. Try Again - response = translateText(characters, system, user, history) + response = translateText(characters, system, user, history, 0.2) translatedText = response.choices[0].message.content totalTokens[0] += response.usage.prompt_tokens totalTokens[1] += response.usage.completion_tokens @@ -546,17 +570,18 @@ def translateGPT(text, history, fullPromptFlag, pbar, filename): translatedText = cleanTranslatedText(translatedText, varResponse) if isinstance(tItem, list): extractedTranslations = extractTranslation(translatedText, True) - if len(tItem) == len(extractedTranslations): - tList[index] = extractedTranslations - else: - MISMATCH.append(filename) - else: - tList[index] = extractedTranslations + tList[index] = extractedTranslations + if len(tItem) != len(extractedTranslations): + mismatch = True # Just here for breakpoint # Create History - history = tList[index] # Update history if we have a list - pbar.update(len(tList[index])) - + with LOCK: + if PBAR is not None: + PBAR.update(len(tItem)) + if not mismatch: + history = extractedTranslations[-10:] # Update history if we have a list + else: + history = text[-10:] else: # Ensure we're passing a single string to extractTranslation extractedTranslations = extractTranslation(translatedText, False) diff --git a/modules/rpgmakermvmz.py b/modules/rpgmakermvmz.py index dab0b00..c8a0e29 100644 --- a/modules/rpgmakermvmz.py +++ b/modules/rpgmakermvmz.py @@ -2155,7 +2155,7 @@ def cleanTranslatedText(translatedText, varResponse): '': '>', '「': '\"', - ' 」': '\"', + '」': '\"', '- ': '-', 'Placeholder Text': '', # Add more replacements as needed diff --git a/modules/rpgmakerplugin.py b/modules/rpgmakerplugin.py index 50ad2a2..359b131 100644 --- a/modules/rpgmakerplugin.py +++ b/modules/rpgmakerplugin.py @@ -439,7 +439,7 @@ def cleanTranslatedText(translatedText, varResponse): '': '>', '「': '\"', - ' 」': '\"', + '」': '\"', '- ': '-', 'Placeholder Text': '', # Add more replacements as needed diff --git a/modules/wolf.py b/modules/wolf.py index cec6a47..25a82c9 100644 --- a/modules/wolf.py +++ b/modules/wolf.py @@ -64,7 +64,7 @@ CODE122 = True # Other CODE210 = True CODE300 = True -CODE250 = True +CODE250 = False # Database NPCFLAG = False @@ -521,7 +521,7 @@ def searchCodes(events, pbar, translatedList, filename): return totalTokens -# Database +# DatabaseDatabase def searchDB(events, pbar, jobList, filename): # Set Lists if len(jobList) > 0: @@ -536,7 +536,7 @@ def searchDB(events, pbar, jobList, filename): else: scenarioList = [[],[],[]] NPCList = [[],[],[],[]] - itemList = [[],[],[]] + itemList = [[],[],[],[]] armorList = [[],[]] enemyList = [[],[]] weaponsList = [[],[],[],[]] @@ -716,6 +716,58 @@ def searchDB(events, pbar, jobList, filename): dataList[j].update({'value': translatedText}) itemList[1].pop(0) + # Description 2 + if '使用後文章[移動]' in dataList[j].get('name'): + # Pass 1 (Grab Data) + if setData == False: + if dataList[j].get('value') != '': + # Remove Textwrap + jaString = dataList[j].get('value') + jaString = jaString.replace('\n', ' ') + jaString = jaString.replace('\r', '') + jaString = re.sub(r'[\\]+f\[\d+\]', '', jaString) + + # Append Data + itemList[2].append(jaString) + + # Pass 2 (Set Data) + else: + if dataList[j].get('value') != '': + # Textwrap + translatedText = itemList[2][0] + translatedText = textwrap.fill(translatedText, LISTWIDTH) + translatedText = font + translatedText + + # Set Data + dataList[j].update({'value': translatedText}) + itemList[2].pop(0) + + # Description 3 + if '使用時文章[戦]' in dataList[j].get('name'): + # Pass 1 (Grab Data) + if setData == False: + if dataList[j].get('value') != '': + # Remove Textwrap + jaString = dataList[j].get('value') + jaString = jaString.replace('\n', ' ') + jaString = jaString.replace('\r', '') + jaString = re.sub(r'[\\]+f\[\d+\]', '', jaString) + + # Append Data + itemList[3].append(jaString) + + # Pass 2 (Set Data) + else: + if dataList[j].get('value') != '': + # Textwrap + translatedText = itemList[3][0] + translatedText = textwrap.fill(translatedText, LISTWIDTH) + translatedText = font + translatedText + + # Set Data + dataList[j].update({'value': translatedText}) + itemList[3].pop(0) + # Grab Armors if table['name'] == '防具' and ARMORFLAG == True: for armor in table['data']: @@ -852,14 +904,14 @@ def searchDB(events, pbar, jobList, filename): weaponsList[1].pop(0) # Grab Collection - if table['name'] == '属性名の設定' and COLLECTIONFLAG == True: + if table['name'] == '鍛冶師用DB' and COLLECTIONFLAG == True: for object in table['data']: dataList = object['data'] # Parse for j in range(len(dataList)): # Name - if '表示名' in dataList[j].get('name'): + if '作る装備' in dataList[j].get('name'): # Pass 1 (Grab Data) if setData == False: if dataList[j].get('value') != '': @@ -877,7 +929,7 @@ def searchDB(events, pbar, jobList, filename): collectionList[0].pop(0) # Description - if 'NULL' in dataList[j].get('name'): + if '品物の解説' in dataList[j].get('name'): # Pass 1 (Grab Data) if setData == False: if dataList[j].get('value') != '': @@ -943,7 +995,7 @@ def searchDB(events, pbar, jobList, filename): # Translation scenarioListTL = [[],[],[]] NPCListTL = [] - itemListTL = [[],[],[]] + itemListTL = [[],[],[],[]] collectionListTL = [[],[],[],[]] armorListTL = [[],[]] enemyListTL = [[],[]] @@ -1011,16 +1063,22 @@ def searchDB(events, pbar, jobList, filename): descListTL2 = response[0] totalTokens[0] += response[1][0] totalTokens[1] += response[1][1] + # Desc 3 + response = translateGPT(itemList[3], 'Reply with only the '+ LANGUAGE +' translation', True, pbar, filename) + descListTL3 = response[0] + totalTokens[0] += response[1][0] + totalTokens[1] += response[1][1] # Check Mismatch if len(nameListTL) != len(itemList[0]) or\ len(descListTL1) != len(itemList[1]) or\ - len(descListTL2) != len(itemList[2]): + len(descListTL2) != len(itemList[2])or\ + len(descListTL3) != len(itemList[3]): with LOCK: if filename not in MISMATCH: MISMATCH.append(filename) else: - itemListTL = [nameListTL, descListTL1, descListTL2] + itemListTL = [nameListTL, descListTL1, descListTL2, descListTL3] translate = True # Armor @@ -1253,7 +1311,7 @@ def subVars(jaString): # Formatting count = 0 - formatList = re.findall(r'[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]', jaString) + formatList = re.findall(r'[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_:,\s-]+\]', jaString) formatList = set(formatList) if len(formatList) != 0: for var in formatList: @@ -1394,7 +1452,7 @@ def cleanTranslatedText(translatedText, varResponse): '': '>', '「': '\"', - ' 」': '\"', + '」': '\"', '- ': '-', 'Placeholder Text': '', # Add more replacements as needed