GUI estimate wasn't properly counting cache writes
This commit is contained in:
parent
f27603ce53
commit
c8b9352bc0
3 changed files with 64 additions and 256 deletions
|
|
@ -335,6 +335,15 @@ class TranslationWorker(QThread):
|
||||||
old_cwd = os.getcwd()
|
old_cwd = os.getcwd()
|
||||||
os.chdir(str(self.project_root))
|
os.chdir(str(self.project_root))
|
||||||
|
|
||||||
|
# Reset estimate written-sizes file at the start of each run so the
|
||||||
|
# warm-cache window is recalculated fresh across all subprocesses.
|
||||||
|
if self.estimate_only:
|
||||||
|
try:
|
||||||
|
from util.translation import clear_estimate_written_sizes
|
||||||
|
clear_estimate_written_sizes()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
try:
|
try:
|
||||||
if self.parse_speakers and (is_mvmz or is_ace):
|
if self.parse_speakers and (is_mvmz or is_ace):
|
||||||
# Run handlers sequentially in this worker process so globals are shared
|
# Run handlers sequentially in this worker process so globals are shared
|
||||||
|
|
@ -517,6 +526,11 @@ class TranslationWorker(QThread):
|
||||||
self.emit_log("✅ Translation completed successfully!")
|
self.emit_log("✅ Translation completed successfully!")
|
||||||
else:
|
else:
|
||||||
self.emit_log("✅ Estimation completed!")
|
self.emit_log("✅ Estimation completed!")
|
||||||
|
try:
|
||||||
|
from util.translation import clear_estimate_written_sizes
|
||||||
|
clear_estimate_written_sizes()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
self.finished_signal.emit(True, str(total_cost))
|
self.finished_signal.emit(True, str(total_cost))
|
||||||
else:
|
else:
|
||||||
if not self.should_stop:
|
if not self.should_stop:
|
||||||
|
|
|
||||||
|
|
@ -1,248 +0,0 @@
|
||||||
"""
|
|
||||||
Actor Variable Substitutor
|
|
||||||
|
|
||||||
RPGMaker uses \\n[X] (stored as \\n[X] in JSON, single backslash after json.load)
|
|
||||||
to pull actor names dynamically at runtime. This module:
|
|
||||||
|
|
||||||
1. Builds a map of actor_id -> name from Actors.json
|
|
||||||
(prefers translated/Actors.json over files/Actors.json so the AI
|
|
||||||
receives the target-language name in context)
|
|
||||||
|
|
||||||
2. substitute_in_files() – walks every .json file inside `files/`,
|
|
||||||
replaces \\n[X] with the actor's name, writes the file back.
|
|
||||||
Returns per-file statistics.
|
|
||||||
|
|
||||||
3. restore_in_translated() – walks every .json file inside `translated/`,
|
|
||||||
replaces the substituted name back with \\n[X].
|
|
||||||
Returns per-file statistics.
|
|
||||||
|
|
||||||
The rename-able protagonist case is handled automatically: substitute the
|
|
||||||
default/translated name for context during translation, then restore the
|
|
||||||
\\n[X] variable so the player's custom name still works at runtime.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
|
||||||
import re
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Optional
|
|
||||||
|
|
||||||
# Pattern that matches \n[X] after json.load() (single literal backslash)
|
|
||||||
_VAR_RE = re.compile(r"\\n\[(\d+)\]", re.IGNORECASE)
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# Public API
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
def load_actor_map(
|
|
||||||
files_dir: str | Path = "files",
|
|
||||||
translated_dir: str | Path = "translated",
|
|
||||||
) -> dict[int, str]:
|
|
||||||
"""Return {actor_id: name}.
|
|
||||||
|
|
||||||
Prefers translated/Actors.json (English names) over files/Actors.json.
|
|
||||||
"""
|
|
||||||
for candidates in (
|
|
||||||
Path(translated_dir) / "Actors.json",
|
|
||||||
Path(files_dir) / "Actors.json",
|
|
||||||
):
|
|
||||||
if candidates.is_file():
|
|
||||||
try:
|
|
||||||
with open(candidates, "r", encoding="utf-8-sig") as fh:
|
|
||||||
data = json.load(fh)
|
|
||||||
actor_map: dict[int, str] = {}
|
|
||||||
for entry in data:
|
|
||||||
if not entry or not isinstance(entry, dict):
|
|
||||||
continue
|
|
||||||
actor_id = entry.get("id")
|
|
||||||
name = (entry.get("name") or "").strip()
|
|
||||||
if actor_id is not None and name:
|
|
||||||
actor_map[int(actor_id)] = name
|
|
||||||
if actor_map:
|
|
||||||
return actor_map
|
|
||||||
except Exception:
|
|
||||||
continue
|
|
||||||
return {}
|
|
||||||
|
|
||||||
|
|
||||||
def substitute_in_files(
|
|
||||||
files_dir: str | Path = "files",
|
|
||||||
actor_map: Optional[dict[int, str]] = None,
|
|
||||||
translated_dir: str | Path = "translated",
|
|
||||||
) -> dict:
|
|
||||||
"""Replace \\n[X] with actor names in all .json files inside *files_dir*.
|
|
||||||
|
|
||||||
If *actor_map* is None it is built automatically via load_actor_map().
|
|
||||||
Returns a summary dict:
|
|
||||||
{
|
|
||||||
"files_modified": int,
|
|
||||||
"total_substitutions": int,
|
|
||||||
"variables_found": {actor_id: name, ...},
|
|
||||||
"errors": [str, ...]
|
|
||||||
}
|
|
||||||
"""
|
|
||||||
files_dir = Path(files_dir)
|
|
||||||
if actor_map is None:
|
|
||||||
actor_map = load_actor_map(files_dir, translated_dir)
|
|
||||||
|
|
||||||
stats = {
|
|
||||||
"files_modified": 0,
|
|
||||||
"total_substitutions": 0,
|
|
||||||
"variables_found": {},
|
|
||||||
"errors": [],
|
|
||||||
}
|
|
||||||
|
|
||||||
for fp in sorted(files_dir.glob("*.json")):
|
|
||||||
try:
|
|
||||||
original = fp.read_text(encoding="utf-8-sig")
|
|
||||||
data = json.loads(original)
|
|
||||||
count = [0]
|
|
||||||
found: dict[int, str] = {}
|
|
||||||
|
|
||||||
new_data = _walk(data, actor_map, count, found, substitute=True)
|
|
||||||
|
|
||||||
if count[0] > 0:
|
|
||||||
stats["total_substitutions"] += count[0]
|
|
||||||
stats["files_modified"] += 1
|
|
||||||
stats["variables_found"].update(found)
|
|
||||||
out = json.dumps(new_data, ensure_ascii=False, indent=4)
|
|
||||||
fp.write_text(out + "\n", encoding="utf-8")
|
|
||||||
except Exception as exc:
|
|
||||||
stats["errors"].append(f"{fp.name}: {exc}")
|
|
||||||
|
|
||||||
return stats
|
|
||||||
|
|
||||||
|
|
||||||
def restore_in_translated(
|
|
||||||
translated_dir: str | Path = "translated",
|
|
||||||
actor_map: Optional[dict[int, str]] = None,
|
|
||||||
files_dir: str | Path = "files",
|
|
||||||
) -> dict:
|
|
||||||
"""Replace actor names back with \\n[X] in all .json files inside *translated_dir*.
|
|
||||||
|
|
||||||
If *actor_map* is None it is built via load_actor_map() which prefers
|
|
||||||
translated/Actors.json (the already-translated names).
|
|
||||||
|
|
||||||
Returns a summary dict with the same shape as substitute_in_files().
|
|
||||||
"""
|
|
||||||
translated_dir = Path(translated_dir)
|
|
||||||
if actor_map is None:
|
|
||||||
actor_map = load_actor_map(files_dir, translated_dir)
|
|
||||||
|
|
||||||
# Build reverse map: name (lowercased) → "\\n[X]"
|
|
||||||
# We match case-insensitively so the AI's casing doesn't break restore.
|
|
||||||
# We keep the original-case name for display in stats.
|
|
||||||
reverse: dict[str, tuple[str, int]] = {} # lower_name -> (var_string, actor_id)
|
|
||||||
for aid, name in actor_map.items():
|
|
||||||
if name:
|
|
||||||
reverse[name.lower()] = (f"\\n[{aid}]", aid)
|
|
||||||
|
|
||||||
stats = {
|
|
||||||
"files_modified": 0,
|
|
||||||
"total_substitutions": 0,
|
|
||||||
"variables_found": {},
|
|
||||||
"errors": [],
|
|
||||||
}
|
|
||||||
|
|
||||||
for fp in sorted(translated_dir.glob("*.json")):
|
|
||||||
# Never restore inside Actors.json itself — that file has the names
|
|
||||||
# as first-class translated data, not as variable stand-ins.
|
|
||||||
if fp.name == "Actors.json":
|
|
||||||
continue
|
|
||||||
try:
|
|
||||||
data = json.loads(fp.read_text(encoding="utf-8-sig"))
|
|
||||||
count = [0]
|
|
||||||
found: dict[int, str] = {}
|
|
||||||
|
|
||||||
new_data = _walk_restore(data, reverse, count, found)
|
|
||||||
|
|
||||||
if count[0] > 0:
|
|
||||||
stats["total_substitutions"] += count[0]
|
|
||||||
stats["files_modified"] += 1
|
|
||||||
stats["variables_found"].update(found)
|
|
||||||
out = json.dumps(new_data, ensure_ascii=False, indent=4)
|
|
||||||
fp.write_text(out + "\n", encoding="utf-8")
|
|
||||||
except Exception as exc:
|
|
||||||
stats["errors"].append(f"{fp.name}: {exc}")
|
|
||||||
|
|
||||||
return stats
|
|
||||||
|
|
||||||
|
|
||||||
def scan_variables(
|
|
||||||
files_dir: str | Path = "files",
|
|
||||||
) -> dict[int, int]:
|
|
||||||
"""Scan *files_dir* and return {actor_id: occurrence_count} without modifying files."""
|
|
||||||
counts: dict[int, int] = {}
|
|
||||||
for fp in Path(files_dir).glob("*.json"):
|
|
||||||
try:
|
|
||||||
text = fp.read_text(encoding="utf-8-sig")
|
|
||||||
# Search in raw JSON text (\\n[ in file = single \n[ after parse)
|
|
||||||
# The raw JSON file stores it as \\n[X] so we search that directly:
|
|
||||||
for m in re.finditer(r"\\\\n\[(\d+)\]", text):
|
|
||||||
aid = int(m.group(1))
|
|
||||||
counts[aid] = counts.get(aid, 0) + 1
|
|
||||||
except Exception:
|
|
||||||
pass
|
|
||||||
return counts
|
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# Internal helpers
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
def _sub_string(s: str, actor_map: dict[int, str], count: list, found: dict) -> str:
|
|
||||||
def replace(m: re.Match) -> str:
|
|
||||||
aid = int(m.group(1))
|
|
||||||
name = actor_map.get(aid)
|
|
||||||
if name:
|
|
||||||
count[0] += 1
|
|
||||||
found[aid] = name
|
|
||||||
return name
|
|
||||||
return m.group(0)
|
|
||||||
return _VAR_RE.sub(replace, s)
|
|
||||||
|
|
||||||
|
|
||||||
def _restore_string(s: str, reverse: dict[str, tuple[str, int]], count: list, found: dict) -> str:
|
|
||||||
"""Replace actor names with \\n[X] in a translated string."""
|
|
||||||
# We need to handle the fact that names can be substrings of other words.
|
|
||||||
# Build a single regex alternation sorted by length (longest first) to
|
|
||||||
# prevent short names like "Al" matching inside "Alicia".
|
|
||||||
if not reverse:
|
|
||||||
return s
|
|
||||||
pattern = re.compile(
|
|
||||||
"|".join(re.escape(name) for name in sorted(reverse.keys(), key=len, reverse=True)),
|
|
||||||
re.IGNORECASE,
|
|
||||||
)
|
|
||||||
|
|
||||||
def replace(m: re.Match) -> str:
|
|
||||||
lower_matched = m.group(0).lower()
|
|
||||||
entry = reverse.get(lower_matched)
|
|
||||||
if entry:
|
|
||||||
var_str, aid = entry
|
|
||||||
count[0] += 1
|
|
||||||
found[aid] = m.group(0)
|
|
||||||
return var_str
|
|
||||||
return m.group(0)
|
|
||||||
|
|
||||||
return pattern.sub(replace, s)
|
|
||||||
|
|
||||||
|
|
||||||
def _walk(obj, actor_map, count, found, substitute: bool):
|
|
||||||
if isinstance(obj, str):
|
|
||||||
return _sub_string(obj, actor_map, count, found)
|
|
||||||
if isinstance(obj, list):
|
|
||||||
return [_walk(item, actor_map, count, found, substitute) for item in obj]
|
|
||||||
if isinstance(obj, dict):
|
|
||||||
return {k: _walk(v, actor_map, count, found, substitute) for k, v in obj.items()}
|
|
||||||
return obj
|
|
||||||
|
|
||||||
|
|
||||||
def _walk_restore(obj, reverse, count, found):
|
|
||||||
if isinstance(obj, str):
|
|
||||||
return _restore_string(obj, reverse, count, found)
|
|
||||||
if isinstance(obj, list):
|
|
||||||
return [_walk_restore(item, reverse, count, found) for item in obj]
|
|
||||||
if isinstance(obj, dict):
|
|
||||||
return {k: _walk_restore(v, reverse, count, found) for k, v in obj.items()}
|
|
||||||
return obj
|
|
||||||
|
|
@ -30,8 +30,40 @@ _thread_local = threading.local()
|
||||||
_global_accurate_cost = 0.0
|
_global_accurate_cost = 0.0
|
||||||
_global_accurate_cost_lock = threading.Lock()
|
_global_accurate_cost_lock = threading.Lock()
|
||||||
|
|
||||||
# Global batch counter for estimate mode — tracks total batches across all files.
|
# Tracks which distinct batch sizes have already been cache-written during this estimate run.
|
||||||
_global_estimate_batch_offset = 0
|
# Each unique numLines value maps to a distinct output_config schema → one write per size.
|
||||||
|
# Persisted to disk so sequential GUI subprocesses share state.
|
||||||
|
_estimate_written_sizes: set = set()
|
||||||
|
_ESTIMATE_SIZES_FILE = Path("log/estimate_written_sizes.json")
|
||||||
|
|
||||||
|
def _load_estimate_written_sizes():
|
||||||
|
"""Load persisted written-sizes set from disk (for GUI subprocess sharing)."""
|
||||||
|
global _estimate_written_sizes
|
||||||
|
try:
|
||||||
|
if _ESTIMATE_SIZES_FILE.exists():
|
||||||
|
with open(_ESTIMATE_SIZES_FILE, "r", encoding="utf-8") as f:
|
||||||
|
_estimate_written_sizes = set(json.load(f))
|
||||||
|
except Exception:
|
||||||
|
_estimate_written_sizes = set()
|
||||||
|
|
||||||
|
def _save_estimate_written_sizes():
|
||||||
|
"""Persist written-sizes set to disk."""
|
||||||
|
try:
|
||||||
|
_ESTIMATE_SIZES_FILE.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
with open(_ESTIMATE_SIZES_FILE, "w", encoding="utf-8") as f:
|
||||||
|
json.dump(list(_estimate_written_sizes), f)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
def clear_estimate_written_sizes():
|
||||||
|
"""Reset the written-sizes file at the start of a new estimate run."""
|
||||||
|
global _estimate_written_sizes
|
||||||
|
_estimate_written_sizes = set()
|
||||||
|
try:
|
||||||
|
if _ESTIMATE_SIZES_FILE.exists():
|
||||||
|
_ESTIMATE_SIZES_FILE.unlink()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
# ===== Placeholder Protection System =====
|
# ===== Placeholder Protection System =====
|
||||||
|
|
@ -1252,17 +1284,22 @@ def calculateCost(inputTokens, outputTokens, model):
|
||||||
_thread_local.estimate_regular_tokens = 0
|
_thread_local.estimate_regular_tokens = 0
|
||||||
_thread_local.estimate_batch_count = 0
|
_thread_local.estimate_batch_count = 0
|
||||||
# If cache is disabled, every batch is a write (2x) — no reads ever.
|
# If cache is disabled, every batch is a write (2x) — no reads ever.
|
||||||
# Otherwise assume first <batchSize> batches globally are writes, rest are reads (0.10x).
|
# Otherwise: each distinct batch size (= distinct output_config schema) gets exactly
|
||||||
global _global_estimate_batch_offset
|
# one cache write on first use; all subsequent batches of that size are reads (0.10x).
|
||||||
|
# Load from disk first so GUI subprocesses (one per file) share warm-cache state.
|
||||||
|
global _estimate_written_sizes
|
||||||
if DISABLE_CACHE:
|
if DISABLE_CACHE:
|
||||||
write_batches = batch_count
|
write_batches = batch_count
|
||||||
read_batches = 0
|
read_batches = 0
|
||||||
else:
|
else:
|
||||||
cache_write_window = pricing["batchSize"]
|
_load_estimate_written_sizes()
|
||||||
offset = _global_estimate_batch_offset
|
seen_sizes = getattr(_thread_local, 'estimate_seen_sizes', set())
|
||||||
_global_estimate_batch_offset += batch_count
|
new_sizes = seen_sizes - _estimate_written_sizes
|
||||||
write_batches = max(0, min(cache_write_window, offset + batch_count) - offset)
|
write_batches = len(new_sizes) # one write per newly-seen size
|
||||||
read_batches = batch_count - write_batches
|
read_batches = batch_count - write_batches
|
||||||
|
_estimate_written_sizes.update(new_sizes)
|
||||||
|
_save_estimate_written_sizes()
|
||||||
|
_thread_local.estimate_seen_sizes = set()
|
||||||
write_cost = (write_batches * static_tok / 1_000_000) * pricing["inputAPICost"] * 2.0
|
write_cost = (write_batches * static_tok / 1_000_000) * pricing["inputAPICost"] * 2.0
|
||||||
read_cost = (read_batches * static_tok / 1_000_000) * pricing["inputAPICost"] * 0.10
|
read_cost = (read_batches * static_tok / 1_000_000) * pricing["inputAPICost"] * 0.10
|
||||||
regular_cost = (regular_tok / 1_000_000) * pricing["inputAPICost"]
|
regular_cost = (regular_tok / 1_000_000) * pricing["inputAPICost"]
|
||||||
|
|
@ -1499,6 +1536,11 @@ def translateAI(text, history, config, filename=None, pbar=None, lock=None, mism
|
||||||
regular_tok = max(0, estimate[0] - getattr(_thread_local, 'estimate_static_tokens', 0))
|
regular_tok = max(0, estimate[0] - getattr(_thread_local, 'estimate_static_tokens', 0))
|
||||||
_thread_local.estimate_regular_tokens = getattr(_thread_local, 'estimate_regular_tokens', 0) + regular_tok
|
_thread_local.estimate_regular_tokens = getattr(_thread_local, 'estimate_regular_tokens', 0) + regular_tok
|
||||||
_thread_local.estimate_batch_count = getattr(_thread_local, 'estimate_batch_count', 0) + 1
|
_thread_local.estimate_batch_count = getattr(_thread_local, 'estimate_batch_count', 0) + 1
|
||||||
|
# Track unique batch sizes seen this file (each maps to a distinct schema)
|
||||||
|
_size = len(clean_tItem) if isinstance(clean_tItem, list) else 1
|
||||||
|
_seen = getattr(_thread_local, 'estimate_seen_sizes', set())
|
||||||
|
_seen.add(_size)
|
||||||
|
_thread_local.estimate_seen_sizes = _seen
|
||||||
|
|
||||||
# Cache the payload with original text as placeholder for future estimates
|
# Cache the payload with original text as placeholder for future estimates
|
||||||
if isinstance(tItem, list):
|
if isinstance(tItem, list):
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue