feat: add option to grab only speakers

This commit is contained in:
dazedanon 2025-09-11 12:36:29 -05:00
parent 156ac2b5b7
commit 6458aa97d8
2 changed files with 176 additions and 45 deletions

View file

@ -32,7 +32,7 @@ if envMissing:
these values using an .env file, for an example see .env.example" these values using an .env file, for an example see .env.example"
) )
from modules.rpgmakermvmz import handleMVMZ from modules.rpgmakermvmz import handleMVMZ, setSpeakerParseMode, finalizeSpeakerParse
from modules.rpgmakerace import handleACE from modules.rpgmakerace import handleACE
from modules.csv import handleCSV from modules.csv import handleCSV
from modules.tyrano import handleTyrano from modules.tyrano import handleTyrano
@ -87,8 +87,9 @@ to worry about being charged twice. You can simply copy the file generated in /t
def main(): def main():
estimate = "" estimate = ""
speaker_parse = False # Deferred until after engine select
while estimate == "": while estimate == "":
estimate = input("Select Translation or Cost Estimation:\n\n 1. Translate\n 2. Estimate\n") estimate = input("Select Mode:\n\n 1. Translate\n 2. Estimate\n")
match estimate: match estimate:
case "1": case "1":
estimate = False estimate = False
@ -116,14 +117,28 @@ def main():
files to translate are in the /files folder and that you picked the right game engine." files to translate are in the /files folder and that you picked the right game engine."
) )
# If translating RPGMaker MV/MZ (index 0) prompt for speaker parse mode
if version == 0 and not estimate:
sub = ""
while sub == "":
sub = input("RPGMaker MV/MZ options:\n\n 1. Standard Translate\n 2. Parse Speakers (collect speaker names only)\n")
match sub:
case "1":
speaker_parse = False
case "2":
speaker_parse = True
case _:
sub = ""
if speaker_parse:
setSpeakerParseMode(True)
# Open File (Threads) # Open File (Threads)
with ThreadPoolExecutor(max_workers=THREADS) as executor: with ThreadPoolExecutor(max_workers=THREADS) as executor:
futures = [ futures = []
executor.submit(MODULES[version][2], filename, estimate) for filename in os.listdir("files"):
for filename in os.listdir("files") for m in MODULES[version][1]:
for m in MODULES[version][1] if filename.endswith(m) and filename != ".gitkeep":
if filename.endswith(m) and filename != ".gitkeep" futures.append(executor.submit(MODULES[version][2], filename, estimate))
]
for future in as_completed(futures): for future in as_completed(futures):
try: try:
totalCost = future.result() totalCost = future.result()
@ -131,6 +146,10 @@ files to translate are in the /files folder and that you picked the right game e
tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno) tracebackLineNo = str(traceback.extract_tb(sys.exc_info()[2])[-1].lineno)
tqdm.write(Fore.RED + str(e) + "|" + tracebackLineNo + Fore.RESET) tqdm.write(Fore.RED + str(e) + "|" + tracebackLineNo + Fore.RESET)
# Finalize speaker parse mode by writing collected speakers to vocab
if speaker_parse:
finalizeSpeakerParse()
# Delete Tmp Files # Delete Tmp Files
if os.path.isfile("csv.tmp"): if os.path.isfile("csv.tmp"):
os.remove("csv.tmp") os.remove("csv.tmp")

View file

@ -8,7 +8,6 @@ import time
import traceback import traceback
import openai import openai
import copy import copy
# Removed concurrent.futures usage for simplicity; running synchronously
from pathlib import Path from pathlib import Path
import shutil import shutil
from colorama import Fore from colorama import Fore
@ -32,7 +31,6 @@ PROMPT = Path("prompt.txt").read_text(encoding="utf-8")
VOCAB = Path("vocab.txt").read_text(encoding="utf-8") VOCAB = Path("vocab.txt").read_text(encoding="utf-8")
THREADS = int(os.getenv("threads")) THREADS = int(os.getenv("threads"))
LOCK = threading.Lock() LOCK = threading.Lock()
# Thread-local context to carry per-thread filename safely
THREAD_CTX = threading.local() THREAD_CTX = threading.local()
WIDTH = int(os.getenv("width")) WIDTH = int(os.getenv("width"))
LISTWIDTH = int(os.getenv("listWidth")) LISTWIDTH = int(os.getenv("listWidth"))
@ -40,14 +38,18 @@ NOTEWIDTH = int(os.getenv("noteWidth"))
MAXHISTORY = 10 MAXHISTORY = 10
ESTIMATE = "" ESTIMATE = ""
TOKENS = [0, 0] TOKENS = [0, 0]
NAMESLIST = []
MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong) MISMATCH = [] # Lists files that throw a mismatch error (Length of GPT list response is wrong)
PBAR = None PBAR = None
FILENAME = None FILENAME = None
TIMETOTAL = 0 # Total Time Taken for all translations TIMETOTAL = 0 # Total Time Taken for all translations
# Dedicated lock for vocab file updates to avoid races when translating multiple files concurrently
VOCAB_LOCK = threading.Lock() VOCAB_LOCK = threading.Lock()
# Speakers
NAMESLIST = []
SPEAKER_PARSE_MODE = False
_speakerCache = {}
_speakerCacheLock = threading.Lock()
# Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex # Regex - Need to change this if you want to translate from/to other languages. Default is Japanese Regex
LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa---\uFF61-\uFF9F]+" LANGREGEX = r"[一-龠ぁ-ゔァ-ヴーa---\uFF61-\uFF9F]+"
@ -2907,42 +2909,67 @@ def searchSystem(data, pbar):
return totalTokens return totalTokens
# Save some money and enter the character before translation # Save some money and enter the character before translation
def getSpeaker(speaker): def getSpeaker(speaker: str):
match speaker: """Translate a speaker name with caching (thread-safe).
case "ファイン":
return ["Fine", [0, 0]]
case "":
return ["", [0, 0]]
case _:
# Find Speaker
for i in range(len(NAMESLIST)):
if speaker == NAMESLIST[i][0]:
return [NAMESLIST[i][1], [0, 0]]
# Translate and Store Speaker Behavior:
response = translateAI( - Empty string returns immediately.
f"{speaker}", - Hard-coded JP -> EN fast map for known names.
"Reply with the " + LANGUAGE + " translation of the NPC name.", - Uses a global dict `_speakerCache` protected by `_speakerCacheLock` to prevent duplicate API calls.
False, - Maintains legacy `NAMESLIST` (list of [jp, en]) for any downstream code expecting order; first insertion order preserved.
) - In speaker-parse mode all speakers are still translated (non-speaker text skipped elsewhere).
response[0] = response[0].title() Returns: [translated_name, [in_tokens, out_tokens]] like legacy translateAI results.
response[0] = response[0].replace("'S", "'s") """
response[0] = response[0].replace("Speaker: ", "") if speaker == "":
return ["", [0, 0]]
# Retry if name doesn't translate for some reason # Fast dictionary check under lock
if re.search(r"([a-zA-Z?])", response[0]) == None: with _speakerCacheLock:
response = translateAI( cached = _speakerCache.get(speaker)
f"{speaker}", if cached is not None:
"Reply with the " + LANGUAGE + " translation of the NPC name.", return [cached, [0, 0]]
False,
)
response[0] = response[0].title()
response[0] = response[0].replace("'S", "'s")
speakerList = [speaker, response[0]] # Need to translate; mark context to force translation even in parse mode
NAMESLIST.append(speakerList) try:
return response THREAD_CTX.in_speaker = True
return [speaker, [0, 0]] except Exception:
pass
response = translateAI(
speaker,
"Reply with the " + LANGUAGE + " translation of the NPC name.",
False,
)
try:
THREAD_CTX.in_speaker = False
except Exception:
pass
translated = response[0].title().replace("'S", "'s").replace("Speaker: ", "")
# Retry if translation looks empty of latin / punctuation (heuristic)
if re.search(r"([a-zA-Z?])", translated) is None:
try:
THREAD_CTX.in_speaker = True
except Exception:
pass
response = translateAI(
speaker,
"Reply with the " + LANGUAGE + " translation of the NPC name.",
False,
)
try:
THREAD_CTX.in_speaker = False
except Exception:
pass
translated = response[0].title().replace("'S", "'s")
# Store in cache (double-checked lock)
with _speakerCacheLock:
if speaker not in _speakerCache:
_speakerCache[speaker] = translated
NAMESLIST.append([speaker, translated]) # Maintain legacy structure & order
return [translated, response[1]]
def translateAI(text, history, fullPromptFlag): def translateAI(text, history, fullPromptFlag):
""" """
@ -2961,6 +2988,11 @@ def translateAI(text, history, fullPromptFlag):
except Exception: except Exception:
tl_filename = FILENAME tl_filename = FILENAME
# Speaker-parse mode: bypass all non-speaker translations to save tokens
if SPEAKER_PARSE_MODE and not getattr(THREAD_CTX, "in_speaker", False):
# Return original text unmodified with zero tokens
return [text, [0, 0]]
return sharedtranslateAI( return sharedtranslateAI(
text=text, text=text,
history=history, history=history,
@ -2971,3 +3003,83 @@ def translateAI(text, history, fullPromptFlag):
lock=LOCK, lock=LOCK,
mismatchList=MISMATCH mismatchList=MISMATCH
) )
def setSpeakerParseMode(flag: bool):
"""Enable/disable speaker-only parse mode."""
global SPEAKER_PARSE_MODE
SPEAKER_PARSE_MODE = bool(flag)
def finalizeSpeakerParse():
"""Finalize speaker parse by writing a fresh # Speakers section.
Rules:
- Always REPLACE existing # Speakers section (not additive).
- Insert the section right after the # Game Characters section (before the next header, typically # Lewd Terms).
- If # Game Characters not found, prepend near top.
"""
if not SPEAKER_PARSE_MODE:
return
try:
vocab_path = Path("vocab.txt")
if not vocab_path.exists():
return
content = vocab_path.read_text(encoding="utf-8")
# Collect and dedupe speakers preserving first-seen order
seen = set()
lines = []
for orig, tl in NAMESLIST:
if not orig or not tl:
continue
if orig in seen:
continue
seen.add(orig)
lines.append(f"{orig} ({tl})")
if not lines:
return
section_block = "# Speakers\n" + "\n".join(lines) + "\n\n"
# Remove any existing # Speakers section anywhere in file
speakers_pattern = re.compile(r"^[\t ]*#+\s*Speakers\s*$\r?\n.*?(?=^[\t ]*#|\Z)", re.MULTILINE | re.DOTALL)
content = speakers_pattern.sub("", content)
# Find # Game Characters section end (blank line after its block) and insert after it
game_char_header = re.compile(r"^[\t ]*#\s*Game Characters\s*$", re.MULTILINE)
match_gc = game_char_header.search(content)
insert_index = 0
if match_gc:
# Find end of that section: next header or double newline after header lines without starting '#'
# Simplest: locate first header after match_gc
subsequent_headers = list(re.finditer(r"^[\t ]*#\s+.*$", content[match_gc.end():], re.MULTILINE))
if subsequent_headers:
# Insert before the first header that isn't the same line (which should be # Lewd Terms)
first_header_rel = subsequent_headers[0].start()
# Walk backwards from that header start to remove leading blank lines for clean insertion
insert_index = match_gc.end() + first_header_rel
else:
insert_index = len(content)
else:
# Prepend below initial intro line if any
insert_index = 0
# Ensure exactly one blank line before section
before = content[:insert_index]
after = content[insert_index:]
if not before.endswith("\n\n"):
if not before.endswith("\n"):
before += "\n"
before += "\n"
new_content = before + section_block + after.lstrip("\n")
# Write atomically
tmp_path = vocab_path.with_suffix(vocab_path.suffix + f".{os.getpid()}.{threading.get_ident()}.tmp")
tmp_path.write_text(new_content, encoding="utf-8")
try:
os.replace(tmp_path, vocab_path)
except Exception:
try:
shutil.move(str(tmp_path), str(vocab_path))
except Exception:
pass
except Exception:
traceback.print_exc()