feat: add option to grab only speakers
This commit is contained in:
parent
156ac2b5b7
commit
6458aa97d8
2 changed files with 176 additions and 45 deletions
|
|
@ -32,7 +32,7 @@ if envMissing:
|
||||||
these values using an .env file, for an example see .env.example"
|
these values using an .env file, for an example see .env.example"
|
||||||
)
|
)
|
||||||
|
|
||||||
from modules.rpgmakermvmz import handleMVMZ
|
from modules.rpgmakermvmz import handleMVMZ, setSpeakerParseMode, finalizeSpeakerParse
|
||||||
from modules.rpgmakerace import handleACE
|
from modules.rpgmakerace import handleACE
|
||||||
from modules.csv import handleCSV
|
from modules.csv import handleCSV
|
||||||
from modules.tyrano import handleTyrano
|
from modules.tyrano import handleTyrano
|
||||||
|
|
@ -87,8 +87,9 @@ to worry about being charged twice. You can simply copy the file generated in /t
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
estimate = ""
|
estimate = ""
|
||||||
|
speaker_parse = False # Deferred until after engine select
|
||||||
while estimate == "":
|
while estimate == "":
|
||||||
estimate = input("Select Translation or Cost Estimation:\n\n 1. Translate\n 2. Estimate\n")
|
estimate = input("Select Mode:\n\n 1. Translate\n 2. Estimate\n")
|
||||||
match estimate:
|
match estimate:
|
||||||
case "1":
|
case "1":
|
||||||
estimate = False
|
estimate = False
|
||||||
|
|
@ -116,14 +117,28 @@ def main():
|
||||||
files to translate are in the /files folder and that you picked the right game engine."
|
files to translate are in the /files folder and that you picked the right game engine."
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# If translating RPGMaker MV/MZ (index 0) prompt for speaker parse mode
|
||||||
|
if version == 0 and not estimate:
|
||||||
|
sub = ""
|
||||||
|
while sub == "":
|
||||||
|
sub = input("RPGMaker MV/MZ options:\n\n 1. Standard Translate\n 2. Parse Speakers (collect speaker names only)\n")
|
||||||
|
match sub:
|
||||||
|
case "1":
|
||||||
|
speaker_parse = False
|
||||||
|
case "2":
|
||||||
|
speaker_parse = True
|
||||||
|
case _:
|
||||||
|
sub = ""
|
||||||
|
if speaker_parse:
|
||||||
|
setSpeakerParseMode(True)
|
||||||
|
|
||||||
# Open File (Threads)
|
# Open File (Threads)
|
||||||
with ThreadPoolExecutor(max_workers=THREADS) as executor:
|
with ThreadPoolExecutor(max_workers=THREADS) as executor:
|
||||||
futures = [
|
futures = []
|
||||||
executor.submit(MODULES[version][2], filename, estimate)
|
for filename in os.listdir("files"):
|
||||||
for filename in os.listdir("files")
|
for m in MODULES[version][1]:
|
||||||
for m in MODULES[version][1]
|
if filename.endswith(m) and filename != ".gitkeep":
|
||||||
if filename.endswith(m) and filename != ".gitkeep"
|
futures.append(executor.submit(MODULES[version][2], filename, estimate))
|
||||||
]
|
|
||||||
for future in as_completed(futures):
|
for future in as_completed(futures):
|
||||||
try:
|
try:
|
||||||
totalCost = future.result()
|
totalCost = future.result()
|
||||||
|
|
@ -131,6 +146,10 @@ files to translate are in the /files folder and that you picked the right game e
|
||||||
tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno)
|
tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno)
|
||||||
tqdm.write(Fore.RED + str(e) + "|" + tracebackLineNo + Fore.RESET)
|
tqdm.write(Fore.RED + str(e) + "|" + tracebackLineNo + Fore.RESET)
|
||||||
|
|
||||||
|
# Finalize speaker parse mode by writing collected speakers to vocab
|
||||||
|
if speaker_parse:
|
||||||
|
finalizeSpeakerParse()
|
||||||
|
|
||||||
# Delete Tmp Files
|
# Delete Tmp Files
|
||||||
if os.path.isfile("csv.tmp"):
|
if os.path.isfile("csv.tmp"):
|
||||||
os.remove("csv.tmp")
|
os.remove("csv.tmp")
|
||||||
|
|
|
||||||
|
|
@ -8,7 +8,6 @@ import time
|
||||||
import traceback
|
import traceback
|
||||||
import openai
|
import openai
|
||||||
import copy
|
import copy
|
||||||
# Removed concurrent.futures usage for simplicity; running synchronously
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
import shutil
|
import shutil
|
||||||
from colorama import Fore
|
from colorama import Fore
|
||||||
|
|
@ -32,7 +31,6 @@ PROMPT = Path("prompt.txt").read_text(encoding="utf-8")
|
||||||
VOCAB = Path("vocab.txt").read_text(encoding="utf-8")
|
VOCAB = Path("vocab.txt").read_text(encoding="utf-8")
|
||||||
THREADS = int(os.getenv("threads"))
|
THREADS = int(os.getenv("threads"))
|
||||||
LOCK = threading.Lock()
|
LOCK = threading.Lock()
|
||||||
# Thread-local context to carry per-thread filename safely
|
|
||||||
THREAD_CTX = threading.local()
|
THREAD_CTX = threading.local()
|
||||||
WIDTH = int(os.getenv("width"))
|
WIDTH = int(os.getenv("width"))
|
||||||
LISTWIDTH = int(os.getenv("listWidth"))
|
LISTWIDTH = int(os.getenv("listWidth"))
|
||||||
|
|
@ -40,14 +38,18 @@ NOTEWIDTH = int(os.getenv("noteWidth"))
|
||||||
MAXHISTORY = 10
|
MAXHISTORY = 10
|
||||||
ESTIMATE = ""
|
ESTIMATE = ""
|
||||||
TOKENS = [0, 0]
|
TOKENS = [0, 0]
|
||||||
NAMESLIST = []
|
|
||||||
MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong)
|
MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong)
|
||||||
PBAR = None
|
PBAR = None
|
||||||
FILENAME = None
|
FILENAME = None
|
||||||
TIMETOTAL = 0 # Total Time Taken for all translations
|
TIMETOTAL = 0 # Total Time Taken for all translations
|
||||||
# Dedicated lock for vocab file updates to avoid races when translating multiple files concurrently
|
|
||||||
VOCAB_LOCK = threading.Lock()
|
VOCAB_LOCK = threading.Lock()
|
||||||
|
|
||||||
|
# Speakers
|
||||||
|
NAMESLIST = []
|
||||||
|
SPEAKER_PARSE_MODE = False
|
||||||
|
_speakerCache = {}
|
||||||
|
_speakerCacheLock = threading.Lock()
|
||||||
|
|
||||||
# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex
|
# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex
|
||||||
LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+"
|
LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa-zA-Z0-9\uFF61-\uFF9F]+"
|
||||||
|
|
||||||
|
|
@ -2907,42 +2909,67 @@ def searchSystem(data, pbar):
|
||||||
return totalTokens
|
return totalTokens
|
||||||
|
|
||||||
# Save some money and enter the character before translation
|
# Save some money and enter the character before translation
|
||||||
def getSpeaker(speaker):
|
def getSpeaker(speaker: str):
|
||||||
match speaker:
|
"""Translate a speaker name with caching (thread-safe).
|
||||||
case "ファイン":
|
|
||||||
return ["Fine", [0, 0]]
|
|
||||||
case "":
|
|
||||||
return ["", [0, 0]]
|
|
||||||
case _:
|
|
||||||
# Find Speaker
|
|
||||||
for i in range(len(NAMESLIST)):
|
|
||||||
if speaker == NAMESLIST[i][0]:
|
|
||||||
return [NAMESLIST[i][1], [0, 0]]
|
|
||||||
|
|
||||||
# Translate and Store Speaker
|
Behavior:
|
||||||
response = translateAI(
|
- Empty string returns immediately.
|
||||||
f"{speaker}",
|
- Hard-coded JP -> EN fast map for known names.
|
||||||
"Reply with the " + LANGUAGE + " translation of the NPC name.",
|
- Uses a global dict `_speakerCache` protected by `_speakerCacheLock` to prevent duplicate API calls.
|
||||||
False,
|
- Maintains legacy `NAMESLIST` (list of [jp, en]) for any downstream code expecting order; first insertion order preserved.
|
||||||
)
|
- In speaker-parse mode all speakers are still translated (non-speaker text skipped elsewhere).
|
||||||
response[0] = response[0].title()
|
Returns: [translated_name, [in_tokens, out_tokens]] like legacy translateAI results.
|
||||||
response[0] = response[0].replace("'S", "'s")
|
"""
|
||||||
response[0] = response[0].replace("Speaker: ", "")
|
if speaker == "":
|
||||||
|
return ["", [0, 0]]
|
||||||
|
|
||||||
# Retry if name doesn't translate for some reason
|
# Fast dictionary check under lock
|
||||||
if re.search(r"([a-zA-Z??])", response[0]) == None:
|
with _speakerCacheLock:
|
||||||
response = translateAI(
|
cached = _speakerCache.get(speaker)
|
||||||
f"{speaker}",
|
if cached is not None:
|
||||||
"Reply with the " + LANGUAGE + " translation of the NPC name.",
|
return [cached, [0, 0]]
|
||||||
False,
|
|
||||||
)
|
|
||||||
response[0] = response[0].title()
|
|
||||||
response[0] = response[0].replace("'S", "'s")
|
|
||||||
|
|
||||||
speakerList = [speaker, response[0]]
|
# Need to translate; mark context to force translation even in parse mode
|
||||||
NAMESLIST.append(speakerList)
|
try:
|
||||||
return response
|
THREAD_CTX.in_speaker = True
|
||||||
return [speaker, [0, 0]]
|
except Exception:
|
||||||
|
pass
|
||||||
|
response = translateAI(
|
||||||
|
speaker,
|
||||||
|
"Reply with the " + LANGUAGE + " translation of the NPC name.",
|
||||||
|
False,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
THREAD_CTX.in_speaker = False
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
translated = response[0].title().replace("'S", "'s").replace("Speaker: ", "")
|
||||||
|
|
||||||
|
# Retry if translation looks empty of latin / punctuation (heuristic)
|
||||||
|
if re.search(r"([a-zA-Z??])", translated) is None:
|
||||||
|
try:
|
||||||
|
THREAD_CTX.in_speaker = True
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
response = translateAI(
|
||||||
|
speaker,
|
||||||
|
"Reply with the " + LANGUAGE + " translation of the NPC name.",
|
||||||
|
False,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
THREAD_CTX.in_speaker = False
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
translated = response[0].title().replace("'S", "'s")
|
||||||
|
|
||||||
|
# Store in cache (double-checked lock)
|
||||||
|
with _speakerCacheLock:
|
||||||
|
if speaker not in _speakerCache:
|
||||||
|
_speakerCache[speaker] = translated
|
||||||
|
NAMESLIST.append([speaker, translated]) # Maintain legacy structure & order
|
||||||
|
|
||||||
|
return [translated, response[1]]
|
||||||
|
|
||||||
def translateAI(text, history, fullPromptFlag):
|
def translateAI(text, history, fullPromptFlag):
|
||||||
"""
|
"""
|
||||||
|
|
@ -2961,6 +2988,11 @@ def translateAI(text, history, fullPromptFlag):
|
||||||
except Exception:
|
except Exception:
|
||||||
tl_filename = FILENAME
|
tl_filename = FILENAME
|
||||||
|
|
||||||
|
# Speaker-parse mode: bypass all non-speaker translations to save tokens
|
||||||
|
if SPEAKER_PARSE_MODE and not getattr(THREAD_CTX, "in_speaker", False):
|
||||||
|
# Return original text unmodified with zero tokens
|
||||||
|
return [text, [0, 0]]
|
||||||
|
|
||||||
return sharedtranslateAI(
|
return sharedtranslateAI(
|
||||||
text=text,
|
text=text,
|
||||||
history=history,
|
history=history,
|
||||||
|
|
@ -2971,3 +3003,83 @@ def translateAI(text, history, fullPromptFlag):
|
||||||
lock=LOCK,
|
lock=LOCK,
|
||||||
mismatchList=MISMATCH
|
mismatchList=MISMATCH
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def setSpeakerParseMode(flag: bool):
|
||||||
|
"""Enable/disable speaker-only parse mode."""
|
||||||
|
global SPEAKER_PARSE_MODE
|
||||||
|
SPEAKER_PARSE_MODE = bool(flag)
|
||||||
|
|
||||||
|
def finalizeSpeakerParse():
|
||||||
|
"""Finalize speaker parse by writing a fresh # Speakers section.
|
||||||
|
Rules:
|
||||||
|
- Always REPLACE existing # Speakers section (not additive).
|
||||||
|
- Insert the section right after the # Game Characters section (before the next header, typically # Lewd Terms).
|
||||||
|
- If # Game Characters not found, prepend near top.
|
||||||
|
"""
|
||||||
|
if not SPEAKER_PARSE_MODE:
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
vocab_path = Path("vocab.txt")
|
||||||
|
if not vocab_path.exists():
|
||||||
|
return
|
||||||
|
content = vocab_path.read_text(encoding="utf-8")
|
||||||
|
|
||||||
|
# Collect and dedupe speakers preserving first-seen order
|
||||||
|
seen = set()
|
||||||
|
lines = []
|
||||||
|
for orig, tl in NAMESLIST:
|
||||||
|
if not orig or not tl:
|
||||||
|
continue
|
||||||
|
if orig in seen:
|
||||||
|
continue
|
||||||
|
seen.add(orig)
|
||||||
|
lines.append(f"{orig} ({tl})")
|
||||||
|
if not lines:
|
||||||
|
return
|
||||||
|
|
||||||
|
section_block = "# Speakers\n" + "\n".join(lines) + "\n\n"
|
||||||
|
|
||||||
|
# Remove any existing # Speakers section anywhere in file
|
||||||
|
speakers_pattern = re.compile(r"^[\t ]*#+\s*Speakers\s*$\r?\n.*?(?=^[\t ]*#|\Z)", re.MULTILINE | re.DOTALL)
|
||||||
|
content = speakers_pattern.sub("", content)
|
||||||
|
|
||||||
|
# Find # Game Characters section end (blank line after its block) and insert after it
|
||||||
|
game_char_header = re.compile(r"^[\t ]*#\s*Game Characters\s*$", re.MULTILINE)
|
||||||
|
match_gc = game_char_header.search(content)
|
||||||
|
insert_index = 0
|
||||||
|
if match_gc:
|
||||||
|
# Find end of that section: next header or double newline after header lines without starting '#'
|
||||||
|
# Simplest: locate first header after match_gc
|
||||||
|
subsequent_headers = list(re.finditer(r"^[\t ]*#\s+.*$", content[match_gc.end():], re.MULTILINE))
|
||||||
|
if subsequent_headers:
|
||||||
|
# Insert before the first header that isn't the same line (which should be # Lewd Terms)
|
||||||
|
first_header_rel = subsequent_headers[0].start()
|
||||||
|
# Walk backwards from that header start to remove leading blank lines for clean insertion
|
||||||
|
insert_index = match_gc.end() + first_header_rel
|
||||||
|
else:
|
||||||
|
insert_index = len(content)
|
||||||
|
else:
|
||||||
|
# Prepend below initial intro line if any
|
||||||
|
insert_index = 0
|
||||||
|
|
||||||
|
# Ensure exactly one blank line before section
|
||||||
|
before = content[:insert_index]
|
||||||
|
after = content[insert_index:]
|
||||||
|
if not before.endswith("\n\n"):
|
||||||
|
if not before.endswith("\n"):
|
||||||
|
before += "\n"
|
||||||
|
before += "\n"
|
||||||
|
new_content = before + section_block + after.lstrip("\n")
|
||||||
|
|
||||||
|
# Write atomically
|
||||||
|
tmp_path = vocab_path.with_suffix(vocab_path.suffix + f".{os.getpid()}.{threading.get_ident()}.tmp")
|
||||||
|
tmp_path.write_text(new_content, encoding="utf-8")
|
||||||
|
try:
|
||||||
|
os.replace(tmp_path, vocab_path)
|
||||||
|
except Exception:
|
||||||
|
try:
|
||||||
|
shutil.move(str(tmp_path), str(vocab_path))
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
except Exception:
|
||||||
|
traceback.print_exc()
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue