fix(speakers): clarify parse progress and shorten nameplates

- Mark harvested files as Scanned until vocab write finishes
- Prompt and normalize speakers into short title-case nameplates
- Keep speaker finalize chunking on the Settings batch size
This commit is contained in:
DazedAnon 2026-07-25 13:46:57 -05:00
parent ba8b0e3bf6
commit 38c4f5350d
4 changed files with 148 additions and 17 deletions

View file

@ -3,6 +3,7 @@
"names": {
"npc": "Reply with the {language} translation of the NPC name.",
"npc_only": "Reply with only the {language} translation of the NPC name",
"speaker": "Reply with only a short {language} dialogue nameplate for this speaker (e.g. Clerk, Townsman, Electrician). Prefer 1-3 words with title capitalization. Never write a sentence, role description, or explanation.",
"enemy": "Reply with only the {language} translation of the enemy NPC name",
"location": "Reply with only the {language} translation of the RPG location name",
"location_short": "Reply with the {language} translation of the location name.",

View file

@ -199,6 +199,7 @@ class TranslationWorker(QThread):
progress_signal = pyqtSignal(int, int, str) # current_file, total_files, filename
item_progress_signal = pyqtSignal(str, int, int) # filename, current_item, total_items (for tqdm within file)
file_error_signal = pyqtSignal(str, str) # filename, error_message
status_signal = pyqtSignal(str) # updates the top translating_label from the worker
finished_signal = pyqtSignal(bool, str)
batch_phase_signal = pyqtSignal(str, object) # phase name, optional payload
@ -519,7 +520,7 @@ class TranslationWorker(QThread):
try:
from modules.rpgmakermvmz import (
handleMVMZ as handler, setSpeakerParseMode, finalizeSpeakerParse,
resetSpeakerState, TOKENS, calculateCost, MODEL,
resetSpeakerState, collectedSpeakerCount, TOKENS, calculateCost, MODEL,
)
except Exception as e:
self.emit_log(f"❌ Could not import rpgmakermvmz for speaker-parse: {e}")
@ -550,13 +551,33 @@ class TranslationWorker(QThread):
completed_count += 1
self.emit_progress(completed_count, total_files, filename)
try:
speaker_n = collectedSpeakerCount()
except Exception:
speaker_n = 0
if self.should_stop:
self.emit_log("⏹ Parse Speakers stopped before translating collected names.")
elif speaker_n:
self.status_signal.emit(f"Translating {speaker_n} speakers…")
self.emit_log(
f"🔤 File scan complete ({completed_count}/{total_files}). "
f"Translating {speaker_n} unique speakers into vocab.txt…"
)
else:
self.status_signal.emit("No speakers found")
self.emit_log(
f"🔤 File scan complete ({completed_count}/{total_files}). "
"No speaker names collected."
)
try:
before_in, before_out = (0, 0)
try:
before_in, before_out = (int(TOKENS[0]), int(TOKENS[1]))
except Exception:
pass
finalizeSpeakerParse()
if not self.should_stop:
finalizeSpeakerParse()
after_in, after_out = (0, 0)
try:
after_in, after_out = (int(TOKENS[0]), int(TOKENS[1]))
@ -2486,7 +2507,7 @@ class TranslationTab(QWidget):
counts[st] = counts.get(st, 0) + 1
parts = [f"{total} file(s)"]
for key in (
"Done", "Collected", "Writing", "Scanning", "Translating",
"Done", "Scanned", "Collected", "Writing", "Scanning", "Translating",
"Failed", "Skipped", "Unsupported", "Waiting",
):
if counts.get(key):
@ -2581,10 +2602,15 @@ class TranslationTab(QWidget):
except Exception:
pass
phase = getattr(self, "_batch_ui_phase", None)
parse_speakers = bool(
getattr(getattr(self, "translation_worker", None), "parse_speakers", False)
)
if getattr(self, "_batch_active", False) and phase == "collect":
status, color = "Scanning", "#007acc"
elif getattr(self, "_batch_active", False) and phase == "consume":
status, color = "Writing", "#007acc"
elif parse_speakers:
status, color = "Scanning", "#007acc"
else:
status, color = "Translating", "#007acc"
self._set_progress_row(
@ -2967,6 +2993,7 @@ class TranslationTab(QWidget):
self.translation_worker.log_signal.connect(self.append_log)
self.translation_worker.progress_signal.connect(self.update_file_progress)
self.translation_worker.item_progress_signal.connect(self.update_item_progress)
self.translation_worker.status_signal.connect(self.translating_label.setText)
self.translation_worker.file_error_signal.connect(self.on_file_error)
self.translation_worker.finished_signal.connect(self.on_translation_finished)
self.translation_worker.batch_phase_signal.connect(self._on_batch_phase)
@ -3192,6 +3219,9 @@ class TranslationTab(QWidget):
"""Update the file-level progress."""
batch_active = getattr(self, "_batch_active", False)
phase = getattr(self, "_batch_ui_phase", None) if batch_active else None
parse_speakers = bool(
getattr(getattr(self, "translation_worker", None), "parse_speakers", False)
)
if batch_active:
if phase == "collect":
@ -3208,14 +3238,32 @@ class TranslationTab(QWidget):
return
self.files_completed = current_file
self.files_translated_label.setText(f"{current_file}/{total_files}")
if parse_speakers:
self.files_translated_label.setText(f"{current_file}/{total_files} scanned")
else:
self.files_translated_label.setText(f"{current_file}/{total_files}")
# Row details (tokens/cost) come from parsed stdout via append_log. If that
# did not run (older modules / parse miss), finalize so the row is not stuck.
if filename in self.file_progress_items:
sl = self.file_progress_items[filename].get("status_label")
if not sl or not sl.text():
self.mark_file_complete(filename, success=True)
if parse_speakers:
# Harvest only - translation of nameplates happens after all files.
self.mark_file_queued(filename)
meta = self.file_progress_items[filename]
try:
meta["label"].setText("Scanned")
except Exception:
pass
self._set_progress_row(
filename,
status="Scanned",
status_color="#d4a017",
progress="names",
)
else:
sl = self.file_progress_items[filename].get("status_label")
if not sl or not sl.text():
self.mark_file_complete(filename, success=True)
# Clear current_translating_file if it was the same file
if self.current_translating_file == filename:
@ -3232,10 +3280,15 @@ class TranslationTab(QWidget):
)
self.batch_live_status.setText(f"Current file: {filename}")
if current_file < total_files:
self.translating_label.setText("Translating...")
self.translating_label.setText(
"Scanning speakers..." if parse_speakers else "Translating..."
)
else:
if batch_active and phase == "consume":
self.translating_label.setText("Finishing batch…")
elif parse_speakers:
# Keep a clear post-scan state until status_signal / finished.
self.translating_label.setText("Translating speakers…")
else:
self.translating_label.setText("")
@ -3322,6 +3375,15 @@ class TranslationTab(QWidget):
"""Apply UI changes for a finished translation run."""
if getattr(self, "_batch_active", False):
self._on_batch_phase("done", None)
# Parse Speakers: promote Scanned rows to Done only after vocab write finishes.
try:
worker = getattr(self, "translation_worker", None)
if success and worker and getattr(worker, "parse_speakers", False):
for fname, meta in (getattr(self, "file_progress_items", {}) or {}).items():
if (meta.get("status_text") or "") == "Scanned":
self.mark_file_complete(fname, success=True)
except Exception:
pass
# Hide the stop button and show the reset/back button
try:
self.stop_button.setVisible(False)

View file

@ -4762,6 +4762,35 @@ def _is_plausible_speaker(name: str) -> bool:
return True
def _normalize_speaker_nameplate(translated: str) -> str:
"""Normalize a model TL into a short dialogue nameplate."""
out = (translated or "").strip()
out = re.sub(r"^(speaker:\s*)", "", out, flags=re.IGNORECASE)
out = out.strip("'\"“”‘’「」")
# If the model ignored instructions and wrote a sentence, keep the lead clause.
if re.search(r"[.!?]", out):
out = re.split(r"[.!?]", out, maxsplit=1)[0].strip()
# Drop trailing commas / "the kind you'd..." style appositives.
if "," in out:
out = out.split(",", 1)[0].strip()
out = re.sub(r"\s+", " ", out).strip()
words = out.split()
# Descriptive expansions often start with filler; keep the trailing role words.
if len(words) > 3 and re.match(
r"^(JUST|ONLY|THE|A|AN|YOUR|SOME|ANY|THIS|THAT)\b", words[0], re.I
):
words = words[-2:]
elif len(words) > 4:
words = words[:3]
out = " ".join(words).title().replace("'S", "'s")
out = re.sub(
r"(\d)(St|Nd|Rd|Th)\b",
lambda m: m.group(1) + m.group(2).lower(),
out,
)
return out
# Save some money and enter the character before translation
def getSpeaker(speaker: str):
"""Return (and possibly collect) speaker name.
@ -4799,15 +4828,14 @@ def getSpeaker(speaker: str):
pass
response = translateAI(
speaker,
ctx("names.npc"),
ctx("names.speaker"),
False,
)
try:
THREAD_CTX.in_speaker = False
except Exception:
pass
translated = response[0].strip().title().replace("'S", "'s").replace("Speaker: ", "")
translated = re.sub(r'(\d)(St|Nd|Rd|Th)\b', lambda m: m.group(1) + m.group(2).lower(), translated)
translated = _normalize_speaker_nameplate(response[0])
if re.search(r"([a-zA-Z?])", translated) is None:
try:
@ -4816,15 +4844,14 @@ def getSpeaker(speaker: str):
pass
response = translateAI(
speaker,
ctx("names.npc"),
ctx("names.speaker"),
False,
)
try:
THREAD_CTX.in_speaker = False
except Exception:
pass
translated = response[0].strip().title().replace("'S", "'s")
translated = re.sub(r'(\d)(St|Nd|Rd|Th)\b', lambda m: m.group(1) + m.group(2).lower(), translated)
translated = _normalize_speaker_nameplate(response[0])
with _speakerCacheLock:
if speaker not in _speakerCache:
@ -4889,6 +4916,11 @@ def translateAI(text, history, history_ctx=None):
# Update config estimate mode based on global ESTIMATE
TRANSLATION_CONFIG.estimateMode = bool(ESTIMATE)
# Keep batch size in sync with Settings / .env (module import can be stale).
try:
TRANSLATION_CONFIG.batchSize = getPricingConfig(MODEL)["batchSize"]
except Exception:
pass
# Call the new shared translation function
# Prefer thread-local filename for logging; fall back to global
@ -5009,6 +5041,12 @@ def setSpeakerParseMode(flag: bool):
global SPEAKER_PARSE_MODE
SPEAKER_PARSE_MODE = bool(flag)
def collectedSpeakerCount() -> int:
"""Unique speaker names waiting for (or already in) the parse-mode glossary."""
with _speakerCacheLock:
return len(SPEAKER_COLLECTED)
def finalizeSpeakerParse():
"""Batch translate collected speakers and write fresh # Speakers section."""
if not SPEAKER_PARSE_MODE:
@ -5025,9 +5063,10 @@ def finalizeSpeakerParse():
THREAD_CTX.in_speaker = True
except Exception:
pass
# Full list; translateAI chunks with the Settings batch size.
resp = translateAI(
to_translate,
ctx("names.npc"),
ctx("names.speaker"),
True,
)
try:
@ -5057,9 +5096,9 @@ def finalizeSpeakerParse():
tl_list = resp[0]
with _speakerCacheLock:
for orig, tl in zip(to_translate, tl_list):
norm = tl.title().replace("'S", "'s").replace("Speaker: ", "")
norm = _normalize_speaker_nameplate(tl)
if re.search(r"([a-zA-Z?])", norm) is None:
norm = tl # keep raw if heuristic fails
norm = (tl or "").strip() or orig
if orig not in _speakerCache:
_speakerCache[orig] = norm
NAMESLIST.append([orig, norm])

View file

@ -0,0 +1,29 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""Tests for RPG Maker speaker nameplate normalization."""
import unittest
from modules.rpgmakermvmz import _normalize_speaker_nameplate
class SpeakerNameplateNormalizeTests(unittest.TestCase):
def test_title_case_short_name(self):
self.assertEqual(_normalize_speaker_nameplate("clerk"), "Clerk")
def test_strips_speaker_prefix(self):
self.assertEqual(_normalize_speaker_nameplate("Speaker: Townsman"), "Townsman")
def test_collapses_descriptive_sentence(self):
raw = "Just your average electrician guy, the kind you'd find anywhere."
self.assertEqual(_normalize_speaker_nameplate(raw), "Electrician Guy")
def test_truncates_long_phrase(self):
self.assertEqual(
_normalize_speaker_nameplate("ELECTRICIAN GUY FROM NEXT DOOR SHOP"),
"Electrician Guy From",
)
if __name__ == "__main__":
unittest.main()