From c8b9352bc0c32aec58aa56bb34f642fa2e058918 Mon Sep 17 00:00:00 2001 From: dazedanon Date: Mon, 16 Mar 2026 10:53:28 -0500 Subject: [PATCH] GUI estimate wasn't properly counting cache writes --- gui/translation_tab.py | 14 +++ util/actor_substitutor.py | 248 -------------------------------------- util/translation.py | 58 +++++++-- 3 files changed, 64 insertions(+), 256 deletions(-) delete mode 100644 util/actor_substitutor.py diff --git a/gui/translation_tab.py b/gui/translation_tab.py index 5bc7b95..7868b02 100644 --- a/gui/translation_tab.py +++ b/gui/translation_tab.py @@ -335,6 +335,15 @@ class TranslationWorker(QThread): old_cwd = os.getcwd() os.chdir(str(self.project_root)) + # Reset estimate written-sizes file at the start of each run so the + # warm-cache window is recalculated fresh across all subprocesses. + if self.estimate_only: + try: + from util.translation import clear_estimate_written_sizes + clear_estimate_written_sizes() + except Exception: + pass + try: if self.parse_speakers and (is_mvmz or is_ace): # Run handlers sequentially in this worker process so globals are shared @@ -517,6 +526,11 @@ class TranslationWorker(QThread): self.emit_log("✅ Translation completed successfully!") else: self.emit_log("✅ Estimation completed!") + try: + from util.translation import clear_estimate_written_sizes + clear_estimate_written_sizes() + except Exception: + pass self.finished_signal.emit(True, str(total_cost)) else: if not self.should_stop: diff --git a/util/actor_substitutor.py b/util/actor_substitutor.py deleted file mode 100644 index b3663ed..0000000 --- a/util/actor_substitutor.py +++ /dev/null @@ -1,248 +0,0 @@ -""" -Actor Variable Substitutor - -RPGMaker uses \\n[X] (stored as \\n[X] in JSON, single backslash after json.load) -to pull actor names dynamically at runtime. This module: - - 1. Builds a map of actor_id -> name from Actors.json - (prefers translated/Actors.json over files/Actors.json so the AI - receives the target-language name in context) - - 2. substitute_in_files() – walks every .json file inside `files/`, - replaces \\n[X] with the actor's name, writes the file back. - Returns per-file statistics. - - 3. restore_in_translated() – walks every .json file inside `translated/`, - replaces the substituted name back with \\n[X]. - Returns per-file statistics. - -The rename-able protagonist case is handled automatically: substitute the -default/translated name for context during translation, then restore the -\\n[X] variable so the player's custom name still works at runtime. -""" - -from __future__ import annotations - -import json -import re -from pathlib import Path -from typing import Optional - -# Pattern that matches \n[X] after json.load() (single literal backslash) -_VAR_RE = re.compile(r"\\n\[(\d+)\]", re.IGNORECASE) - -# --------------------------------------------------------------------------- -# Public API -# --------------------------------------------------------------------------- - -def load_actor_map( - files_dir: str | Path = "files", - translated_dir: str | Path = "translated", -) -> dict[int, str]: - """Return {actor_id: name}. - - Prefers translated/Actors.json (English names) over files/Actors.json. - """ - for candidates in ( - Path(translated_dir) / "Actors.json", - Path(files_dir) / "Actors.json", - ): - if candidates.is_file(): - try: - with open(candidates, "r", encoding="utf-8-sig") as fh: - data = json.load(fh) - actor_map: dict[int, str] = {} - for entry in data: - if not entry or not isinstance(entry, dict): - continue - actor_id = entry.get("id") - name = (entry.get("name") or "").strip() - if actor_id is not None and name: - actor_map[int(actor_id)] = name - if actor_map: - return actor_map - except Exception: - continue - return {} - - -def substitute_in_files( - files_dir: str | Path = "files", - actor_map: Optional[dict[int, str]] = None, - translated_dir: str | Path = "translated", -) -> dict: - """Replace \\n[X] with actor names in all .json files inside *files_dir*. - - If *actor_map* is None it is built automatically via load_actor_map(). - Returns a summary dict: - { - "files_modified": int, - "total_substitutions": int, - "variables_found": {actor_id: name, ...}, - "errors": [str, ...] - } - """ - files_dir = Path(files_dir) - if actor_map is None: - actor_map = load_actor_map(files_dir, translated_dir) - - stats = { - "files_modified": 0, - "total_substitutions": 0, - "variables_found": {}, - "errors": [], - } - - for fp in sorted(files_dir.glob("*.json")): - try: - original = fp.read_text(encoding="utf-8-sig") - data = json.loads(original) - count = [0] - found: dict[int, str] = {} - - new_data = _walk(data, actor_map, count, found, substitute=True) - - if count[0] > 0: - stats["total_substitutions"] += count[0] - stats["files_modified"] += 1 - stats["variables_found"].update(found) - out = json.dumps(new_data, ensure_ascii=False, indent=4) - fp.write_text(out + "\n", encoding="utf-8") - except Exception as exc: - stats["errors"].append(f"{fp.name}: {exc}") - - return stats - - -def restore_in_translated( - translated_dir: str | Path = "translated", - actor_map: Optional[dict[int, str]] = None, - files_dir: str | Path = "files", -) -> dict: - """Replace actor names back with \\n[X] in all .json files inside *translated_dir*. - - If *actor_map* is None it is built via load_actor_map() which prefers - translated/Actors.json (the already-translated names). - - Returns a summary dict with the same shape as substitute_in_files(). - """ - translated_dir = Path(translated_dir) - if actor_map is None: - actor_map = load_actor_map(files_dir, translated_dir) - - # Build reverse map: name (lowercased) → "\\n[X]" - # We match case-insensitively so the AI's casing doesn't break restore. - # We keep the original-case name for display in stats. - reverse: dict[str, tuple[str, int]] = {} # lower_name -> (var_string, actor_id) - for aid, name in actor_map.items(): - if name: - reverse[name.lower()] = (f"\\n[{aid}]", aid) - - stats = { - "files_modified": 0, - "total_substitutions": 0, - "variables_found": {}, - "errors": [], - } - - for fp in sorted(translated_dir.glob("*.json")): - # Never restore inside Actors.json itself — that file has the names - # as first-class translated data, not as variable stand-ins. - if fp.name == "Actors.json": - continue - try: - data = json.loads(fp.read_text(encoding="utf-8-sig")) - count = [0] - found: dict[int, str] = {} - - new_data = _walk_restore(data, reverse, count, found) - - if count[0] > 0: - stats["total_substitutions"] += count[0] - stats["files_modified"] += 1 - stats["variables_found"].update(found) - out = json.dumps(new_data, ensure_ascii=False, indent=4) - fp.write_text(out + "\n", encoding="utf-8") - except Exception as exc: - stats["errors"].append(f"{fp.name}: {exc}") - - return stats - - -def scan_variables( - files_dir: str | Path = "files", -) -> dict[int, int]: - """Scan *files_dir* and return {actor_id: occurrence_count} without modifying files.""" - counts: dict[int, int] = {} - for fp in Path(files_dir).glob("*.json"): - try: - text = fp.read_text(encoding="utf-8-sig") - # Search in raw JSON text (\\n[ in file = single \n[ after parse) - # The raw JSON file stores it as \\n[X] so we search that directly: - for m in re.finditer(r"\\\\n\[(\d+)\]", text): - aid = int(m.group(1)) - counts[aid] = counts.get(aid, 0) + 1 - except Exception: - pass - return counts - - -# --------------------------------------------------------------------------- -# Internal helpers -# --------------------------------------------------------------------------- - -def _sub_string(s: str, actor_map: dict[int, str], count: list, found: dict) -> str: - def replace(m: re.Match) -> str: - aid = int(m.group(1)) - name = actor_map.get(aid) - if name: - count[0] += 1 - found[aid] = name - return name - return m.group(0) - return _VAR_RE.sub(replace, s) - - -def _restore_string(s: str, reverse: dict[str, tuple[str, int]], count: list, found: dict) -> str: - """Replace actor names with \\n[X] in a translated string.""" - # We need to handle the fact that names can be substrings of other words. - # Build a single regex alternation sorted by length (longest first) to - # prevent short names like "Al" matching inside "Alicia". - if not reverse: - return s - pattern = re.compile( - "|".join(re.escape(name) for name in sorted(reverse.keys(), key=len, reverse=True)), - re.IGNORECASE, - ) - - def replace(m: re.Match) -> str: - lower_matched = m.group(0).lower() - entry = reverse.get(lower_matched) - if entry: - var_str, aid = entry - count[0] += 1 - found[aid] = m.group(0) - return var_str - return m.group(0) - - return pattern.sub(replace, s) - - -def _walk(obj, actor_map, count, found, substitute: bool): - if isinstance(obj, str): - return _sub_string(obj, actor_map, count, found) - if isinstance(obj, list): - return [_walk(item, actor_map, count, found, substitute) for item in obj] - if isinstance(obj, dict): - return {k: _walk(v, actor_map, count, found, substitute) for k, v in obj.items()} - return obj - - -def _walk_restore(obj, reverse, count, found): - if isinstance(obj, str): - return _restore_string(obj, reverse, count, found) - if isinstance(obj, list): - return [_walk_restore(item, reverse, count, found) for item in obj] - if isinstance(obj, dict): - return {k: _walk_restore(v, reverse, count, found) for k, v in obj.items()} - return obj diff --git a/util/translation.py b/util/translation.py index 4b9c24d..e34be9a 100644 --- a/util/translation.py +++ b/util/translation.py @@ -30,8 +30,40 @@ _thread_local = threading.local() _global_accurate_cost = 0.0 _global_accurate_cost_lock = threading.Lock() -# Global batch counter for estimate mode — tracks total batches across all files. -_global_estimate_batch_offset = 0 +# Tracks which distinct batch sizes have already been cache-written during this estimate run. +# Each unique numLines value maps to a distinct output_config schema → one write per size. +# Persisted to disk so sequential GUI subprocesses share state. +_estimate_written_sizes: set = set() +_ESTIMATE_SIZES_FILE = Path("log/estimate_written_sizes.json") + +def _load_estimate_written_sizes(): + """Load persisted written-sizes set from disk (for GUI subprocess sharing).""" + global _estimate_written_sizes + try: + if _ESTIMATE_SIZES_FILE.exists(): + with open(_ESTIMATE_SIZES_FILE, "r", encoding="utf-8") as f: + _estimate_written_sizes = set(json.load(f)) + except Exception: + _estimate_written_sizes = set() + +def _save_estimate_written_sizes(): + """Persist written-sizes set to disk.""" + try: + _ESTIMATE_SIZES_FILE.parent.mkdir(parents=True, exist_ok=True) + with open(_ESTIMATE_SIZES_FILE, "w", encoding="utf-8") as f: + json.dump(list(_estimate_written_sizes), f) + except Exception: + pass + +def clear_estimate_written_sizes(): + """Reset the written-sizes file at the start of a new estimate run.""" + global _estimate_written_sizes + _estimate_written_sizes = set() + try: + if _ESTIMATE_SIZES_FILE.exists(): + _ESTIMATE_SIZES_FILE.unlink() + except Exception: + pass # ===== Placeholder Protection System ===== @@ -1252,17 +1284,22 @@ def calculateCost(inputTokens, outputTokens, model): _thread_local.estimate_regular_tokens = 0 _thread_local.estimate_batch_count = 0 # If cache is disabled, every batch is a write (2x) — no reads ever. - # Otherwise assume first batches globally are writes, rest are reads (0.10x). - global _global_estimate_batch_offset + # Otherwise: each distinct batch size (= distinct output_config schema) gets exactly + # one cache write on first use; all subsequent batches of that size are reads (0.10x). + # Load from disk first so GUI subprocesses (one per file) share warm-cache state. + global _estimate_written_sizes if DISABLE_CACHE: write_batches = batch_count read_batches = 0 else: - cache_write_window = pricing["batchSize"] - offset = _global_estimate_batch_offset - _global_estimate_batch_offset += batch_count - write_batches = max(0, min(cache_write_window, offset + batch_count) - offset) + _load_estimate_written_sizes() + seen_sizes = getattr(_thread_local, 'estimate_seen_sizes', set()) + new_sizes = seen_sizes - _estimate_written_sizes + write_batches = len(new_sizes) # one write per newly-seen size read_batches = batch_count - write_batches + _estimate_written_sizes.update(new_sizes) + _save_estimate_written_sizes() + _thread_local.estimate_seen_sizes = set() write_cost = (write_batches * static_tok / 1_000_000) * pricing["inputAPICost"] * 2.0 read_cost = (read_batches * static_tok / 1_000_000) * pricing["inputAPICost"] * 0.10 regular_cost = (regular_tok / 1_000_000) * pricing["inputAPICost"] @@ -1499,6 +1536,11 @@ def translateAI(text, history, config, filename=None, pbar=None, lock=None, mism regular_tok = max(0, estimate[0] - getattr(_thread_local, 'estimate_static_tokens', 0)) _thread_local.estimate_regular_tokens = getattr(_thread_local, 'estimate_regular_tokens', 0) + regular_tok _thread_local.estimate_batch_count = getattr(_thread_local, 'estimate_batch_count', 0) + 1 + # Track unique batch sizes seen this file (each maps to a distinct schema) + _size = len(clean_tItem) if isinstance(clean_tItem, list) else 1 + _seen = getattr(_thread_local, 'estimate_seen_sizes', set()) + _seen.add(_size) + _thread_local.estimate_seen_sizes = _seen # Cache the payload with original text as placeholder for future estimates if isinstance(tItem, list):