From f25ce9f990f1e238be156b8f2b6fe41611afe2c7 Mon Sep 17 00:00:00 2001 From: DazedAnon Date: Thu, 9 Jul 2026 16:09:20 -0500 Subject: [PATCH] get Speaker for wolf --- modules/wolfdawn.py | 220 ++++++++++++++++++++++++++++++++++++----- tests/test_wolfdawn.py | 104 ++++++++++++++++--- util/speakers.py | 7 +- 3 files changed, 293 insertions(+), 38 deletions(-) diff --git a/modules/wolfdawn.py b/modules/wolfdawn.py index 2a0d29e..6a449b3 100644 --- a/modules/wolfdawn.py +++ b/modules/wolfdawn.py @@ -35,9 +35,11 @@ When unset, all database lines are translated. Speakers: WolfDawn tags each line with ``speaker`` / ``speaker_src``. For the first-line formats (``literal_line1`` / ``literal_line1_lowconf``) the speaker -name is baked into line 1 of ``source``. Those lines are reshaped into the shared -``[Speaker]: line`` convention (which the prompt already translates) and restored -to WOLF's native ``Speaker\nline`` layout on write-back. See ``util.speakers``. +name is baked into line 1 of ``source``. Those names are resolved to English first +(vocab hit or a cached short-string ``getSpeaker`` call, same pattern as the other +engines), then the line is reshaped into ``[Speaker]: line`` with the English tag +and restored to WOLF's native ``Speaker\nline`` layout on write-back using that +resolved name (not whatever the dialogue model echoed). See ``util.speakers``. Detection is WolfDawn's, so the reliable nameplate (``literal_line1``) is always reshaped; only the low-confidence guess (``literal_line1_lowconf``) is gated by a per-game, AI-recommended setting from the workflow. @@ -62,6 +64,7 @@ from util.translation import ( translateAI as sharedtranslateAI, getPricingConfig, calculateCost, + parseVocabWithCategories, ) from util import speakers as wolf_speakers from util import vocab as wolf_vocab @@ -98,11 +101,15 @@ _WOLF_CODE_RE = re.compile( r"\\(?:r\[[^\]]*\]|c(?:self)?\[[^\]]*\]|[A-Za-z]+\[[^\]]*\]|[A-Za-z])" ) -# Speaker handling: for first-line-speaker formats, reshape the line into the -# shared "[Speaker]: line" transport before translating and restore WOLF's -# native "Speaker\nline" layout on write-back. Which formats are reshaped is -# configurable from the workflow (data/wolf_speakers.json). +# Speaker handling: for first-line-speaker formats, resolve the nameplate to +# English (vocab / cached short-string translation), reshape into the shared +# "[Speaker]: line" transport, then restore WOLF's native "Speaker\nline" +# layout on write-back. Which formats are reshaped is configurable from the +# workflow (data/wolf_speakers.json). SPEAKER_CONFIG = wolf_speakers.load_config() +NAMESLIST = [] # [[jp, en], ...] session glossary of resolved speaker names +_speakerCache = {} +_speakerCacheLock = threading.Lock() # Pricing / batching from the configured model PRICING_CONFIG = getPricingConfig(MODEL) @@ -250,9 +257,10 @@ def collectEntries(data): def _text_check_body(text: str, speaker_src: str = "") -> str: """Body text used for skip-translated checks (nameplates / codes ignored). - Already-translated WOLF dialogue often keeps a Japanese speaker name on + Already-translated WOLF dialogue may still have a Japanese speaker name on line 1 (``司祭\\nSorry to keep you...``) and may embed Japanese only inside - control codes like ``\\r[我,わ]``. Those lines are finished translations. + control codes like ``\\r[我,わ]``. Body-only checks treat those as finished + dialogue; ``_maybe_fix_nameplate`` then swaps the nameplate to English. """ if not isinstance(text, str): return "" @@ -285,6 +293,149 @@ def _text_still_needs_translation(entry) -> bool: return bool(re.search(LANGREGEX, _text_check_body(txt, entry.get("speaker_src", "")))) +def _vocab_speaker_lookup(speaker: str) -> str | None: + """Return an English gloss for *speaker* from vocab.txt, or None.""" + if not speaker or not VOCAB: + return None + try: + for item in parseVocabWithCategories(VOCAB): + # parseVocabWithCategories yields ((jp, en), line, category) or (term, ...) + term = item[0] + if not isinstance(term, tuple) or len(term) != 2: + continue + jp, en = term + if jp == speaker and isinstance(en, str) and en.strip(): + # Prefer a gloss that is not still Japanese. + if not re.search(LANGREGEX, en): + return en.strip() + except Exception: + pass + return None + + +def _normalize_speaker_name(translated: str) -> str: + """Title-case / tidy a short NPC-name translation (mirrors other engines).""" + out = translated.strip().title().replace("'S", "'s").replace("Speaker: ", "") + return re.sub( + r"(\d)(St|Nd|Rd|Th)\b", + lambda m: m.group(1) + m.group(2).lower(), + out, + ) + + +def getSpeaker(speaker: str): + """Resolve a WolfDawn nameplate to English with caching. + + Order: session cache / NAMESLIST -> vocab.txt -> live short-string + translation (same prompt as the other engines). Single-string calls go + through the live API even during batch collect, so later dialogue batches + embed a stable English tag. + """ + global NAMESLIST + if not speaker: + return ["", [0, 0]] + # Already English / no target-language characters - keep as-is. + if not re.search(LANGREGEX, speaker): + return [speaker, [0, 0]] + + with _speakerCacheLock: + cached = _speakerCache.get(speaker) + if cached is not None: + return [cached, [0, 0]] + for jp, en in NAMESLIST: + if jp == speaker and en: + _speakerCache[speaker] = en + return [en, [0, 0]] + + vocab_hit = _vocab_speaker_lookup(speaker) + if vocab_hit is not None: + with _speakerCacheLock: + _speakerCache[speaker] = vocab_hit + if not any(jp == speaker for jp, _en in NAMESLIST): + NAMESLIST.append([speaker, vocab_hit]) + return [vocab_hit, [0, 0]] + + # Estimate / preflight: do not spend tokens; leave Japanese for a real run. + if ESTIMATE: + return [speaker, [0, 0]] + + response = translateAI( + speaker, + "Reply with the " + LANGUAGE + " translation of the NPC name.", + ) + translated = _normalize_speaker_name( + response[0] if isinstance(response[0], str) else str(response[0]) + ) + + # Retry once if the model returned something with no Latin / ? characters. + if re.search(r"([a-zA-Z??])", translated) is None: + response = translateAI( + speaker, + "Reply with the " + LANGUAGE + " translation of the NPC name.", + ) + translated = _normalize_speaker_name( + response[0] if isinstance(response[0], str) else str(response[0]) + ) + + # Still Japanese after retries - keep original rather than inventing junk. + if re.search(LANGREGEX, translated) and not re.search(r"[A-Za-z]", translated): + translated = speaker + + with _speakerCacheLock: + if speaker not in _speakerCache: + _speakerCache[speaker] = translated + if not any(jp == speaker for jp, _en in NAMESLIST): + NAMESLIST.append([speaker, translated]) + return [translated, response[1]] + + +def _resolve_nameplate(speaker: str, total_tokens: list) -> str: + """Translate *speaker* and accumulate token cost into *total_tokens*.""" + en, tokens = getSpeaker(speaker) + total_tokens[0] += tokens[0] + total_tokens[1] += tokens[1] + return en + + +def _body_after_dropped_tag(text: str, original_speaker: str) -> str: + """When the model drops ``[Speaker]:``, peel a leading JP nameplate if present.""" + if not isinstance(text, str) or "\n" not in text: + return text + line1, rest = text.split("\n", 1) + if not line1.strip(): + return text + if original_speaker and line1 == original_speaker: + return rest + # Short Japanese first line - treat as an echoed nameplate. + if 0 < len(line1.strip()) <= 20 and re.search(LANGREGEX, line1): + return rest + return text + + +def _maybe_fix_nameplate(entry, total_tokens: list) -> None: + """Swap a Japanese first-line nameplate for English on an otherwise-done line.""" + if ESTIMATE or _batch_phase() == "collect": + return + speaker_src = entry.get("speaker_src", "") + if not wolf_speakers.is_firstline_enabled(speaker_src, SPEAKER_CONFIG): + return + txt = entry.get("text") + if not isinstance(txt, str) or not txt.strip(): + return + # Only touch finished dialogue (body already English / skip-translated). + if _text_still_needs_translation(entry): + return + split = wolf_speakers.split_source(txt, speaker_src, SPEAKER_CONFIG) + if split is None: + return + prefix, speaker, body = split + if not re.search(LANGREGEX, speaker): + return + speaker_en = _resolve_nameplate(speaker, total_tokens) + if speaker_en and speaker_en != speaker: + entry["text"] = wolf_speakers.restore_source(prefix, speaker_en, body) + + def parseDocument(data, filename): """Translate every translatable leaf entry and return [data, tokens, error].""" global PBAR @@ -306,6 +457,16 @@ def parseDocument(data, filename): return False return True + # Fix Japanese nameplates on lines whose dialogue body is already done, so + # resume / re-wrap runs do not leave ``セルリア\\nPlease hold on...`` behind. + if IGNORETLTEXT and not ESTIMATE and _batch_phase() != "collect": + for entry in entries: + if _translatable(entry): + continue + src = entry.get("source") + if isinstance(src, str) and re.search(LANGREGEX, src): + _maybe_fix_nameplate(entry, totalTokens) + # Only translate entries that still need work; names.json also requires a # safe badge. Untouched leaves keep ``text == source`` so WolfDawn # treats them as no-ops on inject. @@ -316,10 +477,10 @@ def parseDocument(data, filename): PBAR = pbar if translatable: # Reshape first-line-speaker lines into the shared "[Speaker]: line" - # transport format. plans[i] carries what is needed to restore each - # entry after translation. + # transport format with a pre-resolved English nameplate. plans[i] + # carries what is needed to restore each entry after translation. sources = [] - plans = [] # (entry, prefix, has_speaker, is_firstline, code_map) + plans = [] # (entry, prefix, has_speaker, is_firstline, code_map, speaker_en) for entry in translatable: src = entry["source"] protected_src, code_map = wolf_codes.protect_wolf_codes(src) @@ -329,11 +490,12 @@ def parseDocument(data, filename): ) if split is not None: prefix, speaker, body = split - sources.append(wolf_speakers.to_prefixed(speaker, body)) - plans.append((entry, prefix, True, is_firstline, code_map)) + speaker_en = _resolve_nameplate(speaker, totalTokens) + sources.append(wolf_speakers.to_prefixed(speaker_en, body)) + plans.append((entry, prefix, True, is_firstline, code_map, speaker_en)) else: sources.append(protected_src) - plans.append((entry, "", False, is_firstline, code_map)) + plans.append((entry, "", False, is_firstline, code_map, "")) try: response = translateAI(sources, []) @@ -354,7 +516,7 @@ def parseDocument(data, filename): and isinstance(translated, list) and len(translated) == len(plans) ): - for (entry, prefix, has_speaker, is_firstline, code_map), text, src in zip( + for (entry, prefix, has_speaker, is_firstline, code_map, speaker_en), text, src in zip( plans, translated, sources ): if not isinstance(text, str): @@ -364,13 +526,27 @@ def parseDocument(data, filename): continue text = wolf_codes.restore_wolf_code_placeholders(text, code_map) if has_speaker: - speaker_en, body_en = wolf_speakers.parse_prefixed(text) - if speaker_en is not None: - entry["text"] = wolf_speakers.restore_source( - prefix, speaker_en, body_en + model_speaker, body_en = wolf_speakers.parse_prefixed(text) + if model_speaker is None: + body_en = _body_after_dropped_tag( + text, entry.get("speaker") or "" ) - else: - entry["text"] = prefix + text + # Always prefer the pre-resolved English nameplate. Only + # take the model's tag when ours is still Japanese and + # the model produced a Latin gloss. + final_speaker = speaker_en or model_speaker or ( + entry.get("speaker") or "" + ) + if ( + re.search(LANGREGEX, final_speaker) + and isinstance(model_speaker, str) + and model_speaker + and not re.search(LANGREGEX, model_speaker) + ): + final_speaker = model_speaker + entry["text"] = wolf_speakers.restore_source( + prefix, final_speaker, body_en + ) else: entry["text"] = text wolf_codes.repair_entry(entry) diff --git a/tests/test_wolfdawn.py b/tests/test_wolfdawn.py index 3ab781a..c02e5a0 100644 --- a/tests/test_wolfdawn.py +++ b/tests/test_wolfdawn.py @@ -52,12 +52,18 @@ class _WolfTranslateHarness: orig_t = wd.translateAI orig_estimate = wd.ESTIMATE orig_ignore = wd.IGNORETLTEXT + orig_vocab = wd.VOCAB orig_update = wd.wolf_vocab.update_vocab_section orig_labels = wd.wolf_names.derive_db_labels orig_db_filter = wd.wolf_db.load_db_filter_config + orig_names = list(wd.NAMESLIST) + orig_cache = dict(wd._speakerCache) wd.translateAI = translate wd.ESTIMATE = estimate wd.IGNORETLTEXT = ignore_tl_text + wd.VOCAB = "" # isolate speaker lookup from the real glossary + wd.NAMESLIST = [] + wd._speakerCache.clear() # Never touch the real glossary / DB files during tests. wd.wolf_vocab.update_vocab_section = capture_vocab wd.wolf_names.derive_db_labels = lambda _p: {} @@ -70,9 +76,13 @@ class _WolfTranslateHarness: wd.translateAI = orig_t wd.ESTIMATE = orig_estimate wd.IGNORETLTEXT = orig_ignore + wd.VOCAB = orig_vocab wd.wolf_vocab.update_vocab_section = orig_update wd.wolf_names.derive_db_labels = orig_labels wd.wolf_db.load_db_filter_config = orig_db_filter + wd.NAMESLIST = orig_names + wd._speakerCache.clear() + wd._speakerCache.update(orig_cache) MAP_DOC = { @@ -453,9 +463,12 @@ class TestTranslationWriteback(unittest.TestCase): (data, _t, err), captured = _WolfTranslateHarness().run(doc, "nameplate.mps.json") self.assertIsNone(err) lines = data["scenes"][0]["lines"] - self.assertEqual(lines[0]["text"], "司祭\nSorry to keep you all waiting......") + # Dialogue body is skipped; Japanese nameplate is fixed via getSpeaker + # (mock returns EN_*, then title-cased like the other engines). + self.assertEqual(lines[0]["text"], "En_司祭\nSorry to keep you all waiting......") self.assertEqual(lines[1]["text"], "EN_まだだ") - self.assertEqual(captured, [["まだだ"]]) + # Short-string speaker resolve, then the remaining dialogue batch. + self.assertEqual(captured, ["司祭", ["まだだ"]]) def test_ignore_tl_text_false_retranslates(self): doc = { @@ -627,15 +640,23 @@ class _SpeakerHarness: self.captured.append(copy.deepcopy(text)) return _mock_translate_speaker(text, history, history_ctx) - orig = (wd.translateAI, wd.ESTIMATE, wd.SPEAKER_CONFIG) + orig = (wd.translateAI, wd.ESTIMATE, wd.SPEAKER_CONFIG, wd.VOCAB) + orig_names = list(wd.NAMESLIST) + orig_cache = dict(wd._speakerCache) wd.translateAI = translate wd.ESTIMATE = False wd.SPEAKER_CONFIG = self.config + wd.VOCAB = "" + wd.NAMESLIST = [] + wd._speakerCache.clear() try: result = wd.parseDocument(copy.deepcopy(data), filename) return result, self.captured finally: - (wd.translateAI, wd.ESTIMATE, wd.SPEAKER_CONFIG) = orig + (wd.translateAI, wd.ESTIMATE, wd.SPEAKER_CONFIG, wd.VOCAB) = orig + wd.NAMESLIST = orig_names + wd._speakerCache.clear() + wd._speakerCache.update(orig_cache) class TestSpeakerReshaping(unittest.TestCase): @@ -643,15 +664,18 @@ class TestSpeakerReshaping(unittest.TestCase): cfg = {"literal_line1": True, "literal_line1_lowconf": True} (data, _t, err), captured = _SpeakerHarness(cfg).run(SPEAKER_MAP_DOC) self.assertIsNone(err) - # Model saw the [Speaker]: transport for the two nameplate lines. + # Speakers are resolved first (live short-string), then the batch uses + # English nameplates in the [Speaker]: transport. + self.assertEqual(captured[0], "市民") + self.assertEqual(captured[1], "セルリア") self.assertEqual( - captured[0], - ["[市民]: おはよう\n元気?", "[セルリア]: ふふふ", "むかしむかし"], + captured[2], + ["[En_市民]: おはよう\n元気?", "[En_セルリア]: ふふふ", "むかしむかし"], ) lines = data["scenes"][0]["lines"] - # Restored to WOLF's native Speaker\nbody layout. - self.assertEqual(lines[0]["text"], "EN_市民\nEN_おはよう\n元気?") - self.assertEqual(lines[1]["text"], "EN_セルリア\nEN_ふふふ") + # Restored with the pre-resolved English nameplate (not the model's tag). + self.assertEqual(lines[0]["text"], "En_市民\nEN_おはよう\n元気?") + self.assertEqual(lines[1]["text"], "En_セルリア\nEN_ふふふ") # Narration was translated as a plain blob. self.assertEqual(lines[2]["text"], "EN_むかしむかし") # Sources are preserved for the inject drift guard. @@ -663,12 +687,66 @@ class TestSpeakerReshaping(unittest.TestCase): cfg = {"literal_line1": True, "literal_line1_lowconf": False} (data, _t, err), captured = _SpeakerHarness(cfg).run(SPEAKER_MAP_DOC) self.assertIsNone(err) - # Low-confidence line is sent as the raw source (no reshaping). - self.assertIn("市民\nおはよう\n元気?", captured[0]) - self.assertIn("[セルリア]: ふふふ", captured[0]) + # High-confidence speaker is resolved; low-confidence line stays raw. + self.assertEqual(captured[0], "セルリア") + batch = captured[1] + self.assertIn("市民\nおはよう\n元気?", batch) + self.assertIn("[En_セルリア]: ふふふ", batch) lines = data["scenes"][0]["lines"] self.assertEqual(lines[0]["text"], "EN_市民\nおはよう\n元気?") + def test_writeback_keeps_preresolved_speaker_when_model_leaves_japanese(self): + """Model echoing a JP tag must not overwrite the pre-resolved English name.""" + cfg = {"literal_line1": True, "literal_line1_lowconf": True} + doc = { + "kind": "map", + "scenes": [ + { + "event": 0, + "name": "ev", + "lines": [ + { + "cmd": 59, + "str": 0, + "speaker": "セルリア", + "speaker_src": "literal_line1_lowconf", + "source": "セルリア\nも、もう少し、耐えてください!", + "text": "セルリア\nも、もう少し、耐えてください!", + }, + ], + } + ], + } + + def bad_model(text, history=None, history_ctx=None): + # Speakers resolve normally; dialogue batch keeps the JP tag. + if isinstance(text, str): + return [f"EN_{text}", [1, 1]] + return [["[セルリア]: Please hold on just a little longer!"], [1, 1]] + + orig = (wd.translateAI, wd.ESTIMATE, wd.SPEAKER_CONFIG, wd.VOCAB) + orig_names = list(wd.NAMESLIST) + orig_cache = dict(wd._speakerCache) + wd.translateAI = bad_model + wd.ESTIMATE = False + wd.SPEAKER_CONFIG = cfg + wd.VOCAB = "" + wd.NAMESLIST = [] + wd._speakerCache.clear() + try: + data, _t, err = wd.parseDocument(copy.deepcopy(doc), "bad.mps.json") + finally: + (wd.translateAI, wd.ESTIMATE, wd.SPEAKER_CONFIG, wd.VOCAB) = orig + wd.NAMESLIST = orig_names + wd._speakerCache.clear() + wd._speakerCache.update(orig_cache) + + self.assertIsNone(err) + self.assertEqual( + data["scenes"][0]["lines"][0]["text"], + "En_セルリア\nPlease hold on just a little longer!", + ) + class TestOpenFiles(unittest.TestCase): def test_rejects_unknown_kind(self): diff --git a/util/speakers.py b/util/speakers.py index 26439de..1bb60f8 100644 --- a/util/speakers.py +++ b/util/speakers.py @@ -346,9 +346,10 @@ def _has_japanese(text: str) -> bool: # "市民\nおぉっ!来た!帰ってきたぞ!" speaker_src = literal_line1_lowconf # "セルリア\nほーら、ローザも手を振って。" speaker_src = literal_line1 # -# When translating we reshape those into ``[Speaker]: body`` (the prompt already -# knows to translate the tag) and, on write-back, restore WOLF's native -# ``Speaker\nbody`` layout so injection stays byte-faithful. +# When translating we resolve the nameplate to English first (vocab / +# ``getSpeaker``), reshape into ``[Speaker]: body`` with that English tag, and +# on write-back restore WOLF's native ``Speaker\nbody`` layout using the +# pre-resolved name so a model that echoes Japanese cannot poison ``text``. # # WolfDawn does the detection, so there is nothing to configure for the reliable # format: ``literal_line1`` is a real nameplate (a face window precedes the line),