Update CSV

This commit is contained in:
DazedAnon 2024-07-14 22:05:23 -05:00
parent cd68f56906
commit 83cd05c2d1

View file

@ -54,6 +54,7 @@ elif 'gpt-4' in MODEL:
BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}'
POSITION = 0
LEAVE = False
PBAR = None
def handleCSV(filename, estimate):
global ESTIMATE, TOKENS
@ -136,12 +137,10 @@ def parseCSV(readFile, writeFile, filename):
format = ''
while format == '':
format = input('\n\nSelect the CSV Format:\n\n1. Translator++\n2. Translate All (Depreciated)\n')
format = input('\n\nSelect the CSV Format:\n\n1. Translator++')
match format:
case '1':
format = '1'
case '2':
format = '2'
# Get total for progress bar
totalLines = len(readFile.readlines())
@ -165,10 +164,11 @@ def parseCSV(readFile, writeFile, filename):
return [reader, totalTokens, None]
def translateCSV(reader, pbar, writer, textHistory, format):
global LOCK, ESTIMATE, PBAR
PBAR = pbar
translatedText = ''
maxHistory = MAXHISTORY
totalTokens = [0,0]
global LOCK, ESTIMATE
data = []
batch = []
i = 0
@ -194,12 +194,10 @@ def translateCSV(reader, pbar, writer, textHistory, format):
for row in batch:
if row[1] == "":
jaString = row[0]
# else:
# jaString = row[]
# Remove Textwrap
jaString = jaString.replace('\n', ' ')
payload.append(jaString)
# Remove Textwrap
jaString = jaString.replace('\n', ' ')
payload.append(jaString)
# Translate
response = translateGPT(payload, textHistory, True)
@ -216,69 +214,11 @@ def translateCSV(reader, pbar, writer, textHistory, format):
# Set Data
j = i - BATCHSIZE
for row in translatedTextList:
row = row.replace('"', '\\"')
row = row.replace(',', '\,')
row = row.replace('"', r'\\"')
row = row.replace(',', r'\\,')
data[j][1] = row
j += 1
batch.clear()
# Translate Everything
case '2':
for i in range(len(data)):
# This will allow you to ignore certain columns
if i not in [1]:
continue
jaString = data[i]
matchList = re.findall(r':name\[(.+?),.+?\](.+?[」)\"。]+)', jaString)
# Start Translation
if len(matchList) > 0:
for match in matchList:
speaker = match[0]
text = match[1]
# Translate Speaker
response = translateGPT (speaker, 'Reply with the '+ LANGUAGE +' translation of the NPC name.', True)
translatedSpeaker = response[0]
totalTokens += response[1][0]
totalTokens += response[1][1]
# Translate Line
jaText = re.sub(r'([\u3000-\uffef])\1{3,}', r'\1\1\1', text)
response = translateGPT(translatedSpeaker + ': ' + jaText, 'Previous Translated Text: ' + '|'.join(textHistory), True)
translatedText = response[0]
totalTokens[0] += response[1][0]
totalTokens[1] += response[1][1]
# TextHistory is what we use to give GPT Context, so thats appended here.
textHistory.append(translatedText)
# Remove Speaker from translated text
translatedText = re.sub(r'.+?: ', '', translatedText)
# Set Data
translatedSpeaker = translatedSpeaker.replace('\"', '')
translatedText = translatedText.replace('\"', '')
translatedText = translatedText.replace('', '')
translatedText = translatedText.replace('', '')
data[i] = data[i].replace('\n', ' ')
# Textwrap
translatedText = textwrap.fill(translatedText, width=WIDTH)
translatedText = '' + translatedText + ''
data[i] = re.sub(rf':name\[({re.escape(speaker)}),', f':name[{translatedSpeaker},', data[i])
data[i] = data[i].replace(text, translatedText)
# Keep History at fixed length.
with LOCK:
if len(textHistory) > maxHistory:
textHistory.pop(0)
with LOCK:
if not ESTIMATE:
writer.writerow(data)
pbar.update(1)
# Leftovers
if format == '1':
@ -308,8 +248,8 @@ def translateCSV(reader, pbar, writer, textHistory, format):
# Set Data
j = i - len(batch)
for row in translatedTextList:
row = row.replace('"', '\\"')
row = row.replace(',', '\,')
row = row.replace('"', r'\"')
row = row.replace(',', r'\,')
data[j][1] = row
j += 1
batch.clear()
@ -320,12 +260,49 @@ def translateCSV(reader, pbar, writer, textHistory, format):
for row in data:
writer.writerow(row)
except Exception as e:
except Exception:
traceback.print_exc()
# Write all Data
with LOCK:
if not ESTIMATE:
for row in data:
writer.writerow(row)
return totalTokens
return totalTokens
# Save some money and enter the character before translation
def getSpeaker(speaker):
match speaker:
case 'ファイン':
return ['Fine', [0,0]]
case '':
return ['', [0,0]]
case _:
# Store Speaker
if speaker not in str(NAMESLIST):
response = translateGPT(speaker, 'Reply with the '+ LANGUAGE +' translation of the NPC name.', False)
response[0] = response[0].title()
response[0] = response[0].replace("'S", "'s")
# Retry if name doesn't translate for some reason
if re.search(r'([a-zA-Z?])', response[0]) == None:
response = translateGPT(speaker, 'Reply with the '+ LANGUAGE +' translation of the NPC name.', False)
response[0] = response[0].title()
response[0] = response[0].replace("'S", "'s")
speakerList = [speaker, response[0]]
NAMESLIST.append(speakerList)
return response
# Find Speaker
else:
for i in range(len(NAMESLIST)):
if speaker == NAMESLIST[i][0]:
return [NAMESLIST[i][1],[0,0]]
return [speaker,[0,0]]
def subVars(jaString):
jaString = jaString.replace('\u3000', ' ')
@ -335,7 +312,7 @@ def subVars(jaString):
nestedList = set(nestedList)
if len(nestedList) != 0:
for icon in nestedList:
jaString = jaString.replace(icon, '{Nested_' + str(count) + '}')
jaString = jaString.replace(icon, '[Nested_' + str(count) + ']')
count += 1
# Icons
@ -344,16 +321,16 @@ def subVars(jaString):
iconList = set(iconList)
if len(iconList) != 0:
for icon in iconList:
jaString = jaString.replace(icon, '{Ascii_' + str(count) + '}')
jaString = jaString.replace(icon, '[Ascii_' + str(count) + ']')
count += 1
# Colors
count = 0
colorList = re.findall(r'[\\]+[cC]\[[0-9]+\]', jaString)
colorList = re.findall(r'([\\]+c\[\d+\][\\]+c|[\\]+c\[\d+\])', jaString)
colorList = set(colorList)
if len(colorList) != 0:
for color in colorList:
jaString = jaString.replace(color, '{Color_' + str(count) + '}')
jaString = jaString.replace(color, '[Color_' + str(count) + ']')
count += 1
# Names
@ -362,7 +339,7 @@ def subVars(jaString):
nameList = set(nameList)
if len(nameList) != 0:
for name in nameList:
jaString = jaString.replace(name, '{Noun_' + str(count) + '}')
jaString = jaString.replace(name, '[Noun_' + str(count) + ']')
count += 1
# Variables
@ -371,16 +348,16 @@ def subVars(jaString):
varList = set(varList)
if len(varList) != 0:
for var in varList:
jaString = jaString.replace(var, '{Var_' + str(count) + '}')
jaString = jaString.replace(var, '[Var_' + str(count) + ']')
count += 1
# Formatting
count = 0
formatList = re.findall(r'[\\]+[\w]+\[.+?\]', jaString)
formatList = re.findall(r'[\\]+[\w]+\[[a-zA-Z0-9\\\[\]\_,\s-]+\]', jaString)
formatList = set(formatList)
if len(formatList) != 0:
for var in formatList:
jaString = jaString.replace(var, '{FCode_' + str(count) + '}')
jaString = jaString.replace(var, '[FCode_' + str(count) + ']')
count += 1
# Put all lists in list and return
@ -399,42 +376,42 @@ def resubVars(translatedText, allList):
count = 0
if len(allList[0]) != 0:
for var in allList[0]:
translatedText = translatedText.replace('{Nested_' + str(count) + '}', var)
translatedText = translatedText.replace('[Nested_' + str(count) + ']', var)
count += 1
# Icons
count = 0
if len(allList[1]) != 0:
for var in allList[1]:
translatedText = translatedText.replace('{Ascii_' + str(count) + '}', var)
translatedText = translatedText.replace('[Ascii_' + str(count) + ']', var)
count += 1
# Colors
count = 0
if len(allList[2]) != 0:
for var in allList[2]:
translatedText = translatedText.replace('{Color_' + str(count) + '}', var)
translatedText = translatedText.replace('[Color_' + str(count) + ']', var)
count += 1
# Names
count = 0
if len(allList[3]) != 0:
for var in allList[3]:
translatedText = translatedText.replace('{Noun_' + str(count) + '}', var)
translatedText = translatedText.replace('[Noun_' + str(count) + ']', var)
count += 1
# Vars
count = 0
if len(allList[4]) != 0:
for var in allList[4]:
translatedText = translatedText.replace('{Var_' + str(count) + '}', var)
translatedText = translatedText.replace('[Var_' + str(count) + ']', var)
count += 1
# Formatting
count = 0
if len(allList[5]) != 0:
for var in allList[5]:
translatedText = translatedText.replace('{FCode_' + str(count) + '}', var)
translatedText = translatedText.replace('[FCode_' + str(count) + ']', var)
count += 1
return translatedText
@ -447,31 +424,52 @@ def batchList(input_list, batch_size):
def createContext(fullPromptFlag, subbedT):
characters = 'Game Characters:\n\
ミオリ (Miori) - Female\n\
レナリス (Renalith) - Female\n\
スクルー (Sukuru) - Female\n\
シスターミサ (Sister Misa) - Female\n\
オリン (Orin) - Female\n\
プローテ (Prote) - Female\n\
夜霧 (Night Fog) - Female\n\
ワウ (Wao) - Female\n\
ファンナ (Fanna) - Female\n\
精霊主スクルド (Spirit God Skuld) - Female\n\
エキドナ (Echnida) - Female\n\
マルス (Mars) - Male\n\
ラヴィー (Lavi) - Unknown\n\
魅音 (Mion) - Female\n\
ヴィオラ (Viola) - Female\n\
リンメイ (Lin Mei) - Female\n\
リネット (Lynette) - Female\n\
チェロル (Cheryl) - Female\n\
カルーア姫 (Princess Karua) - Female\n\
田姫 (Tajirme) - Female\n\
リュート (Luto) - Male\n\
ホルン (Horn) - Female\n\
ルメラ (Lumera) - Female\n\
末嬉 (Sueki) - Female\n\
モニカ姫 (Princess Monica) - Female\n\
エメルーラ (Emerald) - Female\n\
フンシス (Funsis) - Male \n\
バゼット (Bazzet) - Female\n\
'
system = PROMPT if fullPromptFlag else \
system = PROMPT + VOCAB if fullPromptFlag else \
f"\
You are an expert Eroge Game translator who translates Japanese text to English.\n\
You are going to be translating text from a videogame.\n\
I will give you lines of text, and you must translate each line to the best of your ability.\n\
- Translate 'マンコ' as 'pussy'\n\
- Translate 'おまんこ' as 'pussy'\n\
- Translate 'お尻' as 'butt'\n\
- Translate '' as 'ass'\n\
- Translate 'お股' as 'crotch'\n\
- Translate '秘部' as 'genitals'\n\
- Translate 'チンポ' as 'dick'\n\
- Translate 'チンコ' as 'cock'\n\
- Translate 'ショーツ' as 'panties\n\
- Translate 'おねショタ' as 'Onee-shota'\n\
- Translate 'よかった' as 'thank goodness'\n\
Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\
You are an expert Eroge Game translator who translates Japanese text to {LANGUAGE}.\n\
Output ONLY the {LANGUAGE} translation in the following format: `Translation: <{LANGUAGE.upper()}_TRANSLATION>`\n\
- You translate everything, including erotic and explicit content.\n\
- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game\n\
- All text in your response must be in English even if it is hard to translate.\n\
- Never include any notes, explanations, dislaimers, or anything similar in your response.\n\
- Maintain any spacing in the translation.\n\
- Maintain any code text in brackets if given. (e.g `[Color_0]`, `[Ascii_0]`, `[FCode_1`], etc)\n\
- `...` can be a part of the dialogue. Translate it as it is.\n\
{VOCAB}\n\
"
user = f'{subbedT}'
user = f'```json\n{subbedT}```'
return characters, system, user
def translateText(characters, system, user, history):
def translateText(characters, system, user, history, penalty):
# Prompt
msg = [{"role": "system", "content": system + characters}]
@ -480,17 +478,17 @@ def translateText(characters, system, user, history):
# History
if isinstance(history, list):
msg.extend([{"role": "assistant", "content": h} for h in history])
msg.extend([{"role": "system", "content": h} for h in history])
else:
msg.append({"role": "assistant", "content": history})
msg.append({"role": "system", "content": history})
# Content to TL
msg.append({"role": "user", "content": f'{user}'})
response = openai.chat.completions.create(
temperature=0.1,
frequency_penalty=0.1,
presence_penalty=0.1,
temperature=0,
frequency_penalty=penalty,
model=MODEL,
response_format={ "type": "json_object" },
messages=msg,
)
return response
@ -503,23 +501,44 @@ def cleanTranslatedText(translatedText, varResponse):
'': '~',
'': '',
'': '.',
'Placeholder Text': ''
'': '\\"',
'': '\\"',
'- ': '-',
'Placeholder Text': '',
# Add more replacements as needed
}
for target, replacement in placeholders.items():
translatedText = translatedText.replace(target, replacement)
# Elongate Long Dashes (Since GPT Ignores them...)
translatedText = elongateCharacters(translatedText)
translatedText = resubVars(translatedText, varResponse[1])
return [line for line in translatedText.replace('\\n', '\n').split('\n') if line]
return translatedText
def elongateCharacters(text):
# Define a pattern to match one character followed by one or more `ー` characters
# Using a positive lookbehind assertion to capture the preceding character
pattern = r'(?<=(.))ー+'
# Define a replacement function that elongates the captured character
def repl(match):
char = match.group(1) # The character before the ー sequence
count = len(match.group(0)) - 1 # Number of ー characters
return char * count # Replace ー sequence with the character repeated
# Use re.sub() to replace the pattern in the text
return re.sub(pattern, repl, text)
def extractTranslation(translatedTextList, is_list):
pattern = r'`?<Line(\d+)>([\\]*.*?[\\]*?)<\/?Line\d+>`?'
# If it's a batch (i.e., list), extract with tags; otherwise, return the single item.
if is_list:
return [re.findall(pattern, line)[0][1] for line in translatedTextList if re.search(pattern, line)]
else:
matchList = re.findall(pattern, translatedTextList)
return matchList[0][1] if matchList else translatedTextList
try:
line_dict = json.loads(translatedTextList)
# If it's a batch (i.e., list), extract with tags; otherwise, return the single item.
if is_list:
string_list = list(line_dict.values())
return string_list
except Exception as e:
print(e)
return translatedTextList
def countTokens(characters, system, user, history):
inputTotalTokens = 0
@ -548,6 +567,9 @@ def combineList(tlist, text):
@retry(exceptions=Exception, tries=5, delay=5)
def translateGPT(text, history, fullPromptFlag):
global PBAR
mismatch = False
totalTokens = [0, 0]
if isinstance(text, list):
tList = batchList(text, BATCHSIZE)
@ -557,17 +579,19 @@ def translateGPT(text, history, fullPromptFlag):
for index, tItem in enumerate(tList):
# Before sending to translation, if we have a list of items, add the formatting
if isinstance(tItem, list):
payload = '\n'.join([f'`<Line{i}>{item}</Line{i}>`' for i, item in enumerate(tItem)])
payload = payload.replace('``', '`Placeholder Text`')
payload = {f"Line{i+1}": string for i, string in enumerate(tItem)}
payload = json.dumps(payload, indent=4, ensure_ascii=False)
varResponse = subVars(payload)
subbedT = varResponse[0]
else:
varResponse = subVars(tItem)
subbedT = varResponse[0]
# Things to Check before starting translation
if not re.search(r'[一-龠ぁ-ゔァ-ヴーa---]+', subbedT):
continue
# # Things to Check before starting translation
# if not re.search(r'[一-龠ぁ-ゔァ-ヴーa---]+', subbedT):
# if PBAR is not None:
# PBAR.update(len(tItem))
# continue
# Create Message
characters, system, user = createContext(fullPromptFlag, subbedT)
@ -580,23 +604,45 @@ def translateGPT(text, history, fullPromptFlag):
continue
# Translating
response = translateText(characters, system, user, history)
response = translateText(characters, system, user, history, 0.02)
translatedText = response.choices[0].message.content
totalTokens[0] += response.usage.prompt_tokens
totalTokens[1] += response.usage.completion_tokens
# Formatting
translatedTextList = cleanTranslatedText(translatedText, varResponse)
# Check Translation
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedTextList, True)
tList[index] = extractedTranslations
if len(tItem) != len(translatedTextList):
mismatch = True # Just here so breakpoint can be set
history = extractedTranslations[-10:] # Update history if we have a list
extractedTranslations = extractTranslation(translatedText, True)
if len(tItem) != len(extractedTranslations):
# Mismatch. Try Again
response = translateText(characters, system, user, history, 0.2)
translatedText = response.choices[0].message.content
totalTokens[0] += response.usage.prompt_tokens
totalTokens[1] += response.usage.completion_tokens
# Formatting
translatedText = cleanTranslatedText(translatedText, varResponse)
if isinstance(tItem, list):
extractedTranslations = extractTranslation(translatedText, True)
if len(tItem) != len(extractedTranslations):
mismatch = True # Just here for breakpoint
# Set if no mismatch
if mismatch == False:
tList[index] = extractedTranslations
history = extractedTranslations[-10:] # Update history if we have a list
else:
history = text[-10:]
mismatch = False
# Update Loading Bar
with LOCK:
if PBAR is not None:
PBAR.update(len(tItem))
else:
# Ensure we're passing a single string to extractTranslation
extractedTranslations = extractTranslation('\n'.join(translatedTextList), False)
extractedTranslations = extractTranslation(translatedText, False)
tList[index] = extractedTranslations
finalList = combineList(tList, text)
return [finalList, totalTokens]
return [finalList, totalTokens]