172 lines
No EOL
5.7 KiB
Python
172 lines
No EOL
5.7 KiB
Python
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||
import json
|
||
import os
|
||
from pathlib import Path
|
||
import re
|
||
import sys
|
||
import textwrap
|
||
import threading
|
||
import time
|
||
import traceback
|
||
import tiktoken
|
||
|
||
from colorama import Fore
|
||
from dotenv import load_dotenv
|
||
import openai
|
||
from retry import retry
|
||
from tqdm import tqdm
|
||
|
||
#Globals
|
||
load_dotenv()
|
||
openai.organization = os.getenv('org')
|
||
openai.api_key = os.getenv('key')
|
||
|
||
APICOST = .002 # Depends on the model https://openai.com/pricing
|
||
PROMPT = Path('prompt.txt').read_text(encoding='utf-8')
|
||
THREADS = 10 # For GPT4 rate limit will be hit if you have more than 1 thread.
|
||
LOCK = threading.Lock()
|
||
WIDTH = 60
|
||
LISTWIDTH = 60
|
||
MAXHISTORY = 10
|
||
ESTIMATE = ''
|
||
TOTALCOST = 0
|
||
TOKENS = 0
|
||
TOTALTOKENS = 0
|
||
NAMESLIST = []
|
||
|
||
#tqdm Globals
|
||
BAR_FORMAT='{l_bar}{bar:10}{r_bar}{bar:-10b}'
|
||
POSITION=0
|
||
LEAVE=False
|
||
|
||
def handleAllText(filename, estimate):
|
||
global ESTIMATE, TOKENS, TOTALTOKENS, TOTALCOST
|
||
ESTIMATE = estimate
|
||
|
||
with open('translated/' + filename, 'w+t', newline='', encoding='utf-16-le') as writeFile:
|
||
start = time.time()
|
||
translatedData = openFiles(filename, writeFile)
|
||
|
||
if estimate:
|
||
# Print Result
|
||
end = time.time()
|
||
tqdm.write(getResultString(['', TOKENS, None], end - start, filename))
|
||
TOTALCOST += TOKENS * .001 * APICOST
|
||
TOTALTOKENS += TOKENS
|
||
TOKENS = 0
|
||
os.remove('translated/' + filename)
|
||
|
||
else:
|
||
# Print Result
|
||
end = time.time()
|
||
tqdm.write(getResultString(translatedData, end - start, filename))
|
||
TOTALCOST += translatedData[1] * .001 * APICOST
|
||
TOTALTOKENS += translatedData[1]
|
||
|
||
return getResultString(['', TOTALTOKENS, None], end - start, 'TOTAL')
|
||
|
||
def openFiles(filename, writeFile):
|
||
with open('files/' + filename, 'r', encoding='utf-16-le') as readFile, writeFile:
|
||
translatedData = parseText(readFile, writeFile, filename)
|
||
|
||
return translatedData
|
||
|
||
def getResultString(translatedData, translationTime, filename):
|
||
# File Print String
|
||
tokenString = Fore.YELLOW + '[' + str(translatedData[1]) + \
|
||
' Tokens/${:,.4f}'.format(translatedData[1] * .001 * APICOST) + ']'
|
||
timeString = Fore.BLUE + '[' + str(round(translationTime, 1)) + 's]'
|
||
|
||
if translatedData[2] == None:
|
||
# Success
|
||
return filename + ': ' + tokenString + timeString + Fore.GREEN + u' \u2713 ' + Fore.RESET
|
||
|
||
else:
|
||
# Fail
|
||
try:
|
||
raise translatedData[2]
|
||
except Exception as e:
|
||
errorString = str(e) + '|' + translatedData[3] + Fore.RED
|
||
return filename + ': ' + tokenString + timeString + Fore.RED + u' \u2717 ' +\
|
||
errorString + Fore.RESET
|
||
|
||
@retry(exceptions=Exception, tries=5, delay=5)
|
||
def translateGPT(t, history, fullPromptFlag):
|
||
# If ESTIMATE is True just count this as an execution and return.
|
||
if ESTIMATE:
|
||
enc = tiktoken.encoding_for_model("gpt-4")
|
||
tokens = len(enc.encode(t)) * 2 + len(enc.encode(str(history))) + len(enc.encode(PROMPT))
|
||
return (t, tokens)
|
||
|
||
# Sub Vars
|
||
varResponse = subVars(t)
|
||
subbedT = varResponse[0]
|
||
|
||
# If there isn't any Japanese in the text just skip
|
||
if not re.search(r'[<5B><>-<2D><>]+|[<5B><>-?]+|[<5B>@-<2D><>]+|[\uFF00-\uFFEF]', subbedT):
|
||
return(t, 0)
|
||
|
||
# Characters
|
||
context = '```\
|
||
Game Characters:\
|
||
Character: <20>r<EFBFBD>m<EFBFBD><6D> <20><><EFBFBD>C == Ikenoue Takumi - Gender: Male\
|
||
Character: <20><><EFBFBD>i <20><><EFBFBD>͂<EFBFBD> == Fukunaga Koharu - Gender: Female\
|
||
Character: <20>_<EFBFBD><5F> <20><><EFBFBD><EFBFBD> == Kamiizumi Rio - Gender: Female\
|
||
Character: <20>g<EFBFBD>ˎ<EFBFBD> <20>A<EFBFBD><41><EFBFBD>T == Kisshouji Arisa - Gender: Female\
|
||
Character: <20>v<EFBFBD><76> <20>F<EFBFBD><46><EFBFBD>q == Kuga Yuriko - Gender: Female\
|
||
```'
|
||
|
||
# Prompt
|
||
if fullPromptFlag:
|
||
system = PROMPT
|
||
user = 'Line to Translate = ' + subbedT
|
||
else:
|
||
system = 'Output ONLY the english translation in the following format: `Translation: <ENGLISH_TRANSLATION>`'
|
||
user = 'Line to Translate = ' + subbedT
|
||
|
||
# Create Message List
|
||
msg = []
|
||
msg.append({"role": "system", "content": system})
|
||
msg.append({"role": "user", "content": context})
|
||
if isinstance(history, list):
|
||
for line in history:
|
||
msg.append({"role": "user", "content": line})
|
||
else:
|
||
msg.append({"role": "user", "content": history})
|
||
msg.append({"role": "user", "content": user})
|
||
|
||
response = openai.ChatCompletion.create(
|
||
temperature=0.1,
|
||
frequency_penalty=0.2,
|
||
presence_penalty=0.2,
|
||
model="gpt-3.5-turbo",
|
||
messages=msg,
|
||
request_timeout=30,
|
||
)
|
||
|
||
# Save Translated Text
|
||
translatedText = response.choices[0].message.content
|
||
tokens = response.usage.total_tokens
|
||
|
||
# Resub Vars
|
||
translatedText = resubVars(translatedText, varResponse[1])
|
||
|
||
# Remove Placeholder Text
|
||
translatedText = translatedText.replace('English Translation: ', '')
|
||
translatedText = translatedText.replace('Translation: ', '')
|
||
translatedText = translatedText.replace('Line to Translate = ', '')
|
||
translatedText = translatedText.replace('Translation = ', '')
|
||
translatedText = translatedText.replace('Translate = ', '')
|
||
translatedText = translatedText.replace('English Translation:', '')
|
||
translatedText = translatedText.replace('Translation:', '')
|
||
translatedText = translatedText.replace('Line to Translate =', '')
|
||
translatedText = translatedText.replace('Translation =', '')
|
||
translatedText = translatedText.replace('Translate =', '')
|
||
translatedText = re.sub(r'Note:.*', '', translatedText)
|
||
translatedText = translatedText.replace('<EFBFBD><EFBFBD>', '')
|
||
|
||
# Return Translation
|
||
if len(translatedText) > 15 * len(t) or "I'm sorry, but I'm unable to assist with that translation" in translatedText:
|
||
raise Exception
|
||
else:
|
||
return [translatedText, tokens] |