Reconcile Names

This commit is contained in:
DazedAnon 2026-07-09 08:04:53 -05:00
parent 519b09fad2
commit a733cff6b7
3 changed files with 341 additions and 1 deletions

View file

@ -2493,9 +2493,20 @@ class WolfWorkflowTab(QWidget):
) )
layout.addWidget(self.drift_cb) layout.addWidget(self.drift_cb)
check_row = QHBoxLayout()
check_btn = self._register(_make_btn("Check name consistency", "#3a3a3a")) check_btn = self._register(_make_btn("Check name consistency", "#3a3a3a"))
check_btn.clicked.connect(self._names_check) check_btn.clicked.connect(self._names_check)
layout.addWidget(check_btn) reconcile_btn = self._register(_make_btn("Reconcile names → glossary", "#007acc"))
reconcile_btn.setToolTip(
"Rewrite extract lines so each names.json source uses one English value. "
"Translated names.json entries win; if the glossary is still Japanese "
"(refs/verify), divergent extracts are unified to the majority spelling."
)
reconcile_btn.clicked.connect(self._names_reconcile)
check_row.addWidget(check_btn)
check_row.addWidget(reconcile_btn)
check_row.addStretch()
layout.addLayout(check_row)
layout.addWidget(_make_hr()) layout.addWidget(_make_hr())
layout.addWidget(self._subheading("Translated files")) layout.addWidget(self._subheading("Translated files"))
@ -3751,6 +3762,40 @@ class WolfWorkflowTab(QWidget):
self._run_task(task) self._run_task(task)
def _names_reconcile(self):
"""Force extract JSON to one English per glossary name (names.json wins)."""
translated = self._tool_root() / "translated"
if not (translated / "names.json").is_file():
QMessageBox.warning(
self,
"Reconcile names",
"translated/names.json not found. Finish Step 3 first.",
)
return
def task(log, progress=None):
from util.wolfdawn import names as wolf_names
report = wolf_names.reconcile_translated_dir(translated, dry_run=False)
summary = wolf_names.format_reconcile_summary(report)
log(summary)
for change in report.changes[:40]:
log(
f" {change.json_file}: {change.source!r} "
f"{change.old_text!r}{change.new_text!r} ({change.reason})"
)
if len(report.changes) > 40:
log(f" … and {len(report.changes) - 40} more")
return True, summary
def _after(ok: bool, msg: str):
if ok:
QMessageBox.information(self, "Reconcile names", msg)
else:
QMessageBox.warning(self, "Reconcile names", msg or "Reconcile failed.")
self._run_task(task, on_done=_after)
# ── Step 5: Package ──────────────────────────────────────────────────────── # ── Step 5: Package ────────────────────────────────────────────────────────
def _build_step5_package(self, layout: QVBoxLayout): def _build_step5_package(self, layout: QVBoxLayout):

View file

@ -148,5 +148,68 @@ class TestNameWrapRoles(unittest.TestCase):
self.assertEqual(role["font"], 14) self.assertEqual(role["font"], 14)
class TestNameReconcile(unittest.TestCase):
def test_glossary_wins_over_divergent_extracts(self):
names = {
"kind": "names",
"names": [{"source": "牧場主", "text": "Rancher", "safety": "safe"}],
}
db = {
"kind": "db",
"groups": [
{
"type": 1,
"lines": [
{"source": "牧場主", "text": "Ranch Owner"},
{"source": "牧場主", "text": "Rancher"},
{"source": "牧場主", "text": "牧場主"},
],
}
],
}
report = wn.reconcile_names_to_glossary(names, {"DataBase.project.json": db})
self.assertEqual(report.changed, 2) # Owner + identity → Rancher
texts = [ln["text"] for ln in db["groups"][0]["lines"]]
self.assertEqual(texts, ["Rancher", "Rancher", "Rancher"])
self.assertTrue(all(c.reason == "glossary" for c in report.changes))
def test_majority_when_glossary_still_japanese(self):
names = {
"kind": "names",
"names": [{"source": "魔獣マニア", "text": "魔獣マニア", "safety": "refs"}],
}
db = {
"kind": "db",
"groups": [
{
"type": 1,
"lines": [
{"source": "魔獣マニア", "text": "Beast Maniac"},
{"source": "魔獣マニア", "text": "Monster Beast Maniac"},
{"source": "魔獣マニア", "text": "Monster Beast Maniac"},
],
}
],
}
report = wn.reconcile_names_to_glossary(names, {"DataBase.project.json": db})
self.assertEqual(report.changed, 1)
self.assertEqual(report.changes[0].new_text, "Monster Beast Maniac")
self.assertEqual(report.changes[0].reason, "majority")
texts = [ln["text"] for ln in db["groups"][0]["lines"]]
self.assertEqual(
texts,
["Monster Beast Maniac", "Monster Beast Maniac", "Monster Beast Maniac"],
)
def test_tie_breaks_lexicographically(self):
chosen, reason = wn.choose_canonical(
"牧場主",
"牧場主",
{"Ranch Owner": 4, "Rancher": 4},
)
self.assertEqual(reason, "majority")
self.assertEqual(chosen, "Ranch Owner")
if __name__ == "__main__": if __name__ == "__main__":
unittest.main() unittest.main()

View file

@ -22,6 +22,7 @@ from __future__ import annotations
import json import json
import re import re
from dataclasses import dataclass, field
from pathlib import Path from pathlib import Path
from typing import Any, Union from typing import Any, Union
@ -376,3 +377,234 @@ def upsert_name_wrap_role(
encoding="utf-8", encoding="utf-8",
) )
return path return path
# ── Name-consistency reconcile (names.json wins) ─────────────────────────────
@dataclass
class NameReconcileChange:
"""One extract line rewritten to match the chosen canonical translation."""
json_file: str
source: str
old_text: str
new_text: str
reason: str # glossary | majority
@dataclass
class NameReconcileReport:
changes: list[NameReconcileChange] = field(default_factory=list)
skipped_identity_glossary: list[str] = field(default_factory=list)
unresolved: list[str] = field(default_factory=list)
files_written: list[str] = field(default_factory=list)
@property
def changed(self) -> int:
return len(self.changes)
def glossary_map(names_doc: dict[str, Any]) -> dict[str, str]:
"""``source -> text`` from names.json (last entry wins if duplicates)."""
out: dict[str, str] = {}
for entry in parse_names_entries(names_doc):
src = entry.get("source")
txt = entry.get("text")
if isinstance(src, str) and src and isinstance(txt, str):
out[src] = txt
return out
def _walk_source_text_leaves(obj: Any, visit) -> None:
"""Call ``visit(leaf_dict)`` for every object with string source+text."""
if isinstance(obj, dict):
src, txt = obj.get("source"), obj.get("text")
if isinstance(src, str) and isinstance(txt, str):
visit(obj)
for value in obj.values():
_walk_source_text_leaves(value, visit)
elif isinstance(obj, list):
for item in obj:
_walk_source_text_leaves(item, visit)
def _collect_extract_variants(
extract_docs: dict[str, dict[str, Any]],
glossary_sources: set[str],
) -> dict[str, dict[str, int]]:
"""source -> {text: count} for non-identity glossary-name lines in extracts."""
from collections import Counter
variants: dict[str, Counter[str]] = {}
for doc in extract_docs.values():
def visit(leaf: dict[str, Any], _doc=doc) -> None:
src = leaf["source"]
txt = leaf["text"]
if src not in glossary_sources or txt == src:
return
variants.setdefault(src, Counter())[txt] += 1
_walk_source_text_leaves(doc, visit)
return {src: dict(counts) for src, counts in variants.items()}
def choose_canonical(
source: str,
glossary_text: str,
extract_variants: dict[str, int],
) -> tuple[str | None, str]:
"""Pick the canonical English for *source*.
Priority:
1. Translated ``names.json`` entry (``glossary_text != source``)
2. Majority among extract variants (tie lexicographically first)
3. None if glossary is still JP and there are no extract variants
"""
if glossary_text != source:
return glossary_text, "glossary"
if not extract_variants:
return None, "identity_glossary"
best_count = max(extract_variants.values())
candidates = sorted(t for t, n in extract_variants.items() if n == best_count)
return candidates[0], "majority"
def reconcile_names_to_glossary(
names_doc: dict[str, Any],
extract_docs: dict[str, dict[str, Any]],
*,
dry_run: bool = False,
) -> NameReconcileReport:
"""Rewrite extract lines so each glossary name uses one canonical translation.
*names_doc* is not modified. Only exact full-string ``source`` matches are
touched (same rule as WolfDawn ``names-check``).
"""
report = NameReconcileReport()
gloss = glossary_map(names_doc)
if not gloss:
return report
variants = _collect_extract_variants(extract_docs, set(gloss))
canonical: dict[str, tuple[str, str]] = {}
for source, gtext in gloss.items():
extract_texts = variants.get(source, {})
chosen, reason = choose_canonical(source, gtext, extract_texts)
if chosen is None:
continue
if reason == "majority" and len(extract_texts) < 2 and gtext == source:
# Glossary still JP and extracts already agree - nothing to reconcile.
continue
if reason == "glossary":
# Always apply glossary EN onto extracts (identity or wrong EN).
canonical[source] = (chosen, reason)
elif len(extract_texts) >= 2:
# Divergent extracts, glossary still JP - unify to majority.
canonical[source] = (chosen, reason)
# else: glossary JP, single extract variant - leave alone
for json_name, doc in extract_docs.items():
changed_here = 0
def visit(leaf: dict[str, Any], _name=json_name) -> None:
nonlocal changed_here
src = leaf["source"]
if src not in canonical:
return
chosen, reason = canonical[src]
old = leaf["text"]
if old == chosen:
return
# When unifying via majority (glossary still JP), do not invent EN onto
# identity lines - only collapse divergent translations.
if reason == "majority" and old == src:
return
report.changes.append(
NameReconcileChange(
json_file=_name,
source=src,
old_text=old,
new_text=chosen,
reason=reason,
)
)
if not dry_run:
leaf["text"] = chosen
changed_here += 1
_walk_source_text_leaves(doc, visit)
if changed_here and not dry_run:
report.files_written.append(json_name)
return report
def format_reconcile_summary(report: NameReconcileReport) -> str:
"""Short human summary for the GUI / log."""
if not report.changes:
return "Name reconcile: nothing to change."
parts = [f"Name reconcile: {report.changed} line(s) updated"]
by_reason: dict[str, int] = {}
for c in report.changes:
by_reason[c.reason] = by_reason.get(c.reason, 0) + 1
if by_reason:
detail = ", ".join(f"{n} via {r}" for r, n in sorted(by_reason.items()))
parts.append(f"({detail})")
if report.files_written:
parts.append(f"in {len(report.files_written)} file(s)")
majority_sources = sorted({c.source for c in report.changes if c.reason == "majority"})
if majority_sources:
parts.append(
f"; {len(majority_sources)} name(s) used majority because names.json "
"is still Japanese (refs/verify) - translate them in names.json to lock the winner"
)
return " ".join(parts)
def reconcile_translated_dir(
translated_dir: PathLike,
*,
dry_run: bool = False,
) -> NameReconcileReport:
"""Load ``translated/names.json`` + other JSON extracts and reconcile in place."""
base = Path(translated_dir)
names_path = base / "names.json"
if not names_path.is_file():
report = NameReconcileReport()
report.unresolved.append("names.json missing")
return report
names_doc = json.loads(names_path.read_text(encoding="utf-8-sig"))
if not isinstance(names_doc, dict):
report = NameReconcileReport()
report.unresolved.append("names.json invalid")
return report
extract_docs: dict[str, dict[str, Any]] = {}
for path in sorted(base.glob("*.json")):
if path.name in ("names.json", "wolfdawn-roles.json", "wrap_profile.json"):
continue
try:
data = json.loads(path.read_text(encoding="utf-8-sig"))
except Exception:
continue
if isinstance(data, dict) and data.get("kind") in (
"db",
"map",
"common",
"gamedat",
"txt",
"txt-dir",
):
extract_docs[path.name] = data
report = reconcile_names_to_glossary(names_doc, extract_docs, dry_run=dry_run)
if not dry_run:
for name in report.files_written:
path = base / name
path.write_text(
json.dumps(extract_docs[name], ensure_ascii=False, indent=4) + "\n",
encoding="utf-8",
)
return report