diff --git a/modules/csv.py b/modules/csv.py index b9a507c..ea8135e 100644 --- a/modules/csv.py +++ b/modules/csv.py @@ -54,6 +54,7 @@ elif 'gpt-4' in MODEL: BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}' POSITION = 0 LEAVE = False +PBAR = None def handleCSV(filename, estimate): global ESTIMATE, TOKENS @@ -136,12 +137,10 @@ def parseCSV(readFile, writeFile, filename): format = '' while format == '': - format = input('\n\nSelect the CSV Format:\n\n1. Translator++\n2. Translate All (Depreciated)\n') + format = input('\n\nSelect the CSV Format:\n\n1. Translator++') match format: case '1': format = '1' - case '2': - format = '2' # Get total for progress bar totalLines = len(readFile.readlines()) @@ -165,10 +164,11 @@ def parseCSV(readFile, writeFile, filename): return [reader, totalTokens, None] def translateCSV(reader, pbar, writer, textHistory, format): + global LOCK, ESTIMATE, PBAR + PBAR = pbar translatedText = '' maxHistory = MAXHISTORY totalTokens = [0,0] - global LOCK, ESTIMATE data = [] batch = [] i = 0 @@ -194,12 +194,10 @@ def translateCSV(reader, pbar, writer, textHistory, format): for row in batch: if row[1] == "": jaString = row[0] - # else: - # jaString = row[] - # Remove Textwrap - jaString = jaString.replace('\n', ' ') - payload.append(jaString) + # Remove Textwrap + jaString = jaString.replace('\n', ' ') + payload.append(jaString) # Translate response = translateGPT(payload, textHistory, True) @@ -216,69 +214,11 @@ def translateCSV(reader, pbar, writer, textHistory, format): # Set Data j = i - BATCHSIZE for row in translatedTextList: - row = row.replace('"', '\\"') - row = row.replace(',', '\,') + row = row.replace('"', r'\\"') + row = row.replace(',', r'\\,') data[j][1] = row j += 1 batch.clear() - - # Translate Everything - case '2': - for i in range(len(data)): - # This will allow you to ignore certain columns - if i not in [1]: - continue - jaString = data[i] - matchList = re.findall(r':name\[(.+?),.+?\](.+?[」)\"。]+)', jaString) - - # Start Translation - if len(matchList) > 0: - for match in matchList: - speaker = match[0] - text = match[1] - - # Translate Speaker - response = translateGPT (speaker, 'Reply with the '+ LANGUAGE +' translation of the NPC name.', True) - translatedSpeaker = response[0] - totalTokens += response[1][0] - totalTokens += response[1][1] - - # Translate Line - jaText = re.sub(r'([\u3000-\uffef])\1{3,}', r'\1\1\1', text) - response = translateGPT(translatedSpeaker + ': ' + jaText, 'Previous Translated Text: ' + '|'.join(textHistory), True) - translatedText = response[0] - totalTokens[0] += response[1][0] - totalTokens[1] += response[1][1] - - # TextHistory is what we use to give GPT Context, so thats appended here. - textHistory.append(translatedText) - - # Remove Speaker from translated text - translatedText = re.sub(r'.+?: ', '', translatedText) - - # Set Data - translatedSpeaker = translatedSpeaker.replace('\"', '') - translatedText = translatedText.replace('\"', '') - translatedText = translatedText.replace('「', '') - translatedText = translatedText.replace('」', '') - data[i] = data[i].replace('\n', ' ') - - # Textwrap - translatedText = textwrap.fill(translatedText, width=WIDTH) - - translatedText = '「' + translatedText + '」' - data[i] = re.sub(rf':name\[({re.escape(speaker)}),', f':name[{translatedSpeaker},', data[i]) - data[i] = data[i].replace(text, translatedText) - - # Keep History at fixed length. - with LOCK: - if len(textHistory) > maxHistory: - textHistory.pop(0) - - with LOCK: - if not ESTIMATE: - writer.writerow(data) - pbar.update(1) # Leftovers if format == '1': @@ -308,8 +248,8 @@ def translateCSV(reader, pbar, writer, textHistory, format): # Set Data j = i - len(batch) for row in translatedTextList: - row = row.replace('"', '\\"') - row = row.replace(',', '\,') + row = row.replace('"', r'\"') + row = row.replace(',', r'\,') data[j][1] = row j += 1 batch.clear() @@ -320,12 +260,49 @@ def translateCSV(reader, pbar, writer, textHistory, format): for row in data: writer.writerow(row) - except Exception as e: + except Exception: traceback.print_exc() + # Write all Data + with LOCK: + if not ESTIMATE: + for row in data: + writer.writerow(row) + return totalTokens return totalTokens +# Save some money and enter the character before translation +def getSpeaker(speaker): + match speaker: + case 'ファイン': + return ['Fine', [0,0]] + case '': + return ['', [0,0]] + case _: + # Store Speaker + if speaker not in str(NAMESLIST): + response = translateGPT(speaker, 'Reply with the '+ LANGUAGE +' translation of the NPC name.', False) + response[0] = response[0].title() + response[0] = response[0].replace("'S", "'s") + + # Retry if name doesn't translate for some reason + if re.search(r'([a-zA-Z??])', response[0]) == None: + response = translateGPT(speaker, 'Reply with the '+ LANGUAGE +' translation of the NPC name.', False) + response[0] = response[0].title() + response[0] = response[0].replace("'S", "'s") + + speakerList = [speaker, response[0]] + NAMESLIST.append(speakerList) + return response + # Find Speaker + else: + for i in range(len(NAMESLIST)): + if speaker == NAMESLIST[i][0]: + return [NAMESLIST[i][1],[0,0]] + + return [speaker,[0,0]] + def subVars(jaString): jaString = jaString.replace('\u3000', ' ') @@ -335,7 +312,7 @@ def subVars(jaString): nestedList = set(nestedList) if len(nestedList) != 0: for icon in nestedList: - jaString = jaString.replace(icon, '{Nested_' + str(count) + '}') + jaString = jaString.replace(icon, '[Nested_' + str(count) + ']') count += 1 # Icons @@ -344,16 +321,16 @@ def subVars(jaString): iconList = set(iconList) if len(iconList) != 0: for icon in iconList: - jaString = jaString.replace(icon, '{Ascii_' + str(count) + '}') + jaString = jaString.replace(icon, '[Ascii_' + str(count) + ']') count += 1 # Colors count = 0 - colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString) + colorList = re.findall(r'([\\]+c\[\d+\][\\]+c|[\\]+c\[\d+\])', jaString) colorList = set(colorList) if len(colorList) != 0: for color in colorList: - jaString = jaString.replace(color, '{Color_' + str(count) + '}') + jaString = jaString.replace(color, '[Color_' + str(count) + ']') count += 1 # Names @@ -362,7 +339,7 @@ def subVars(jaString): nameList = set(nameList) if len(nameList) != 0: for name in nameList: - jaString = jaString.replace(name, '{Noun_' + str(count) + '}') + jaString = jaString.replace(name, '[Noun_' + str(count) + ']') count += 1 # Variables @@ -371,16 +348,16 @@ def subVars(jaString): varList = set(varList) if len(varList) != 0: for var in varList: - jaString = jaString.replace(var, '{Var_' + str(count) + '}') + jaString = jaString.replace(var, '[Var_' + str(count) + ']') count += 1 # Formatting count = 0 - formatList = re.findall(r'[\\]+[\w]+\[.+?\]', jaString) + formatList = re.findall(r'[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]', jaString) formatList = set(formatList) if len(formatList) != 0: for var in formatList: - jaString = jaString.replace(var, '{FCode_' + str(count) + '}') + jaString = jaString.replace(var, '[FCode_' + str(count) + ']') count += 1 # Put all lists in list and return @@ -399,42 +376,42 @@ def resubVars(translatedText, allList): count = 0 if len(allList[0]) != 0: for var in allList[0]: - translatedText = translatedText.replace('{Nested_' + str(count) + '}', var) + translatedText = translatedText.replace('[Nested_' + str(count) + ']', var) count += 1 # Icons count = 0 if len(allList[1]) != 0: for var in allList[1]: - translatedText = translatedText.replace('{Ascii_' + str(count) + '}', var) + translatedText = translatedText.replace('[Ascii_' + str(count) + ']', var) count += 1 # Colors count = 0 if len(allList[2]) != 0: for var in allList[2]: - translatedText = translatedText.replace('{Color_' + str(count) + '}', var) + translatedText = translatedText.replace('[Color_' + str(count) + ']', var) count += 1 # Names count = 0 if len(allList[3]) != 0: for var in allList[3]: - translatedText = translatedText.replace('{Noun_' + str(count) + '}', var) + translatedText = translatedText.replace('[Noun_' + str(count) + ']', var) count += 1 # Vars count = 0 if len(allList[4]) != 0: for var in allList[4]: - translatedText = translatedText.replace('{Var_' + str(count) + '}', var) + translatedText = translatedText.replace('[Var_' + str(count) + ']', var) count += 1 # Formatting count = 0 if len(allList[5]) != 0: for var in allList[5]: - translatedText = translatedText.replace('{FCode_' + str(count) + '}', var) + translatedText = translatedText.replace('[FCode_' + str(count) + ']', var) count += 1 return translatedText @@ -447,31 +424,52 @@ def batchList(input_list, batch_size): def createContext(fullPromptFlag, subbedT): characters = 'Game Characters:\n\ -ミオリ (Miori) - Female\n\ +レナリス (Renalith) - Female\n\ +スクルー (Sukuru) - Female\n\ +シスターミサ (Sister Misa) - Female\n\ +オリン (Orin) - Female\n\ +プローテ (Prote) - Female\n\ +夜霧 (Night Fog) - Female\n\ +ワウ (Wao) - Female\n\ +ファンナ (Fanna) - Female\n\ +精霊主スクルド (Spirit God Skuld) - Female\n\ +エキドナ (Echnida) - Female\n\ +マルス (Mars) - Male\n\ +ラヴィー (Lavi) - Unknown\n\ +魅音 (Mion) - Female\n\ +ヴィオラ (Viola) - Female\n\ +リンメイ (Lin Mei) - Female\n\ +リネット (Lynette) - Female\n\ +チェロル (Cheryl) - Female\n\ +カルーア姫 (Princess Karua) - Female\n\ +田姫 (Tajirme) - Female\n\ +リュート (Luto) - Male\n\ +ホルン (Horn) - Female\n\ +ルメラ (Lumera) - Female\n\ +末嬉 (Sueki) - Female\n\ +モニカ姫 (Princess Monica) - Female\n\ +エメルーラ (Emerald) - Female\n\ +フンシス (Funsis) - Male \n\ +バゼット (Bazzet) - Female\n\ ' - system = PROMPT if fullPromptFlag else \ + system = PROMPT + VOCAB if fullPromptFlag else \ f"\ -You are an expert Eroge Game translator who translates Japanese text to English.\n\ -You are going to be translating text from a videogame.\n\ -I will give you lines of text, and you must translate each line to the best of your ability.\n\ -- Translate 'マンコ' as 'pussy'\n\ -- Translate 'おまんこ' as 'pussy'\n\ -- Translate 'お尻' as 'butt'\n\ -- Translate '尻' as 'ass'\n\ -- Translate 'お股' as 'crotch'\n\ -- Translate '秘部' as 'genitals'\n\ -- Translate 'チンポ' as 'dick'\n\ -- Translate 'チンコ' as 'cock'\n\ -- Translate 'ショーツ' as 'panties\n\ -- Translate 'おねショタ' as 'Onee-shota'\n\ -- Translate 'よかった' as 'thank goodness'\n\ -Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\ +You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\ +Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\ +- You translate everything, including erotic and explicit content.\n\ +- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\ +- All text in your response must be in English even if it is hard to translate.\n\ +- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\ +- Maintain any spacing in the translation.\n\ +- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\ +- `...` can be a part of the dialogue. Translate it as it is.\n\ +{VOCAB}\n\ " - user = f'{subbedT}' + user = f'```json\n{subbedT}```' return characters, system, user -def translateText(characters, system, user, history): +def translateText(characters, system, user, history, penalty): # Prompt msg = [{"role": "system", "content": system + characters}] @@ -480,17 +478,17 @@ def translateText(characters, system, user, history): # History if isinstance(history, list): - msg.extend([{"role": "assistant", "content": h} for h in history]) + msg.extend([{"role": "system", "content": h} for h in history]) else: - msg.append({"role": "assistant", "content": history}) + msg.append({"role": "system", "content": history}) # Content to TL msg.append({"role": "user", "content": f'{user}'}) response = openai.chat.completions.create( - temperature=0.1, - frequency_penalty=0.1, - presence_penalty=0.1, + temperature=0, + frequency_penalty=penalty, model=MODEL, + response_format={ "type": "json_object" }, messages=msg, ) return response @@ -503,23 +501,44 @@ def cleanTranslatedText(translatedText, varResponse): '〜': '~', 'ッ': '', '。': '.', - 'Placeholder Text': '' + '「': '\\"', + '」': '\\"', + '- ': '-', + 'Placeholder Text': '', # Add more replacements as needed } for target, replacement in placeholders.items(): translatedText = translatedText.replace(target, replacement) + # Elongate Long Dashes (Since GPT Ignores them...) + translatedText = elongateCharacters(translatedText) translatedText = resubVars(translatedText, varResponse[1]) - return [line for line in translatedText.replace('\\n', '\n').split('\n') if line] + return translatedText + +def elongateCharacters(text): + # Define a pattern to match one character followed by one or more `ー` characters + # Using a positive lookbehind assertion to capture the preceding character + pattern = r'(?<=(.))ー+' + + # Define a replacement function that elongates the captured character + def repl(match): + char = match.group(1) # The character before the ー sequence + count = len(match.group(0)) - 1 # Number of ー characters + return char * count # Replace ー sequence with the character repeated + + # Use re.sub() to replace the pattern in the text + return re.sub(pattern, repl, text) def extractTranslation(translatedTextList, is_list): - pattern = r'`?([\\]*.*?[\\]*?)<\/?Line\d+>`?' - # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. - if is_list: - return [re.findall(pattern, line)[0][1] for line in translatedTextList if re.search(pattern, line)] - else: - matchList = re.findall(pattern, translatedTextList) - return matchList[0][1] if matchList else translatedTextList + try: + line_dict = json.loads(translatedTextList) + # If it's a batch (i.e., list), extract with tags; otherwise, return the single item. + if is_list: + string_list = list(line_dict.values()) + return string_list + except Exception as e: + print(e) + return translatedTextList def countTokens(characters, system, user, history): inputTotalTokens = 0 @@ -548,6 +567,9 @@ def combineList(tlist, text): @retry(exceptions=Exception, tries=5, delay=5) def translateGPT(text, history, fullPromptFlag): + global PBAR + + mismatch = False totalTokens = [0, 0] if isinstance(text, list): tList = batchList(text, BATCHSIZE) @@ -557,17 +579,19 @@ def translateGPT(text, history, fullPromptFlag): for index, tItem in enumerate(tList): # Before sending to translation, if we have a list of items, add the formatting if isinstance(tItem, list): - payload = '\n'.join([f'`{item}`' for i, item in enumerate(tItem)]) - payload = payload.replace('``', '`Placeholder Text`') + payload = {f"Line{i+1}": string for i, string in enumerate(tItem)} + payload = json.dumps(payload, indent=4, ensure_ascii=False) varResponse = subVars(payload) subbedT = varResponse[0] else: varResponse = subVars(tItem) subbedT = varResponse[0] - # Things to Check before starting translation - if not re.search(r'[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+', subbedT): - continue + # # Things to Check before starting translation + # if not re.search(r'[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9]+', subbedT): + # if PBAR is not None: + # PBAR.update(len(tItem)) + # continue # Create Message characters, system, user = createContext(fullPromptFlag, subbedT) @@ -580,23 +604,45 @@ def translateGPT(text, history, fullPromptFlag): continue # Translating - response = translateText(characters, system, user, history) + response = translateText(characters, system, user, history, 0.02) translatedText = response.choices[0].message.content totalTokens[0] += response.usage.prompt_tokens totalTokens[1] += response.usage.completion_tokens - # Formatting - translatedTextList = cleanTranslatedText(translatedText, varResponse) + # Check Translation + translatedText = cleanTranslatedText(translatedText, varResponse) if isinstance(tItem, list): - extractedTranslations = extractTranslation(translatedTextList, True) - tList[index] = extractedTranslations - if len(tItem) != len(translatedTextList): - mismatch = True # Just here so breakpoint can be set - history = extractedTranslations[-10:] # Update history if we have a list + extractedTranslations = extractTranslation(translatedText, True) + if len(tItem) != len(extractedTranslations): + # Mismatch. Try Again + response = translateText(characters, system, user, history, 0.2) + translatedText = response.choices[0].message.content + totalTokens[0] += response.usage.prompt_tokens + totalTokens[1] += response.usage.completion_tokens + + # Formatting + translatedText = cleanTranslatedText(translatedText, varResponse) + if isinstance(tItem, list): + extractedTranslations = extractTranslation(translatedText, True) + if len(tItem) != len(extractedTranslations): + mismatch = True # Just here for breakpoint + + # Set if no mismatch + if mismatch == False: + tList[index] = extractedTranslations + history = extractedTranslations[-10:] # Update history if we have a list + else: + history = text[-10:] + mismatch = False + + # Update Loading Bar + with LOCK: + if PBAR is not None: + PBAR.update(len(tItem)) else: # Ensure we're passing a single string to extractTranslation - extractedTranslations = extractTranslation('\n'.join(translatedTextList), False) + extractedTranslations = extractTranslation(translatedText, False) tList[index] = extractedTranslations finalList = combineList(tList, text) - return [finalList, totalTokens] \ No newline at end of file + return [finalList, totalTokens]