From ccdbbf6e624dc48db7fed85b2b2a503d6f8adec2 Mon Sep 17 00:00:00 2001 From: dazedanon Date: Thu, 12 Mar 2026 21:17:31 -0500 Subject: [PATCH] i cant code anymore so let ai do it --- gameupdate/.gitignore | 121 ++++++++++++ gameupdate/gameupdate/patch-config.txt | 3 + gameupdate/gameupdate/vocab.txt | 212 +++++++++++++++++++++ gui/workflow_tab.py | 253 ++++++++++++++++++++----- modules/rpgmakermvmz.py | 2 +- prompt.txt | 7 +- util/speaker_detector.py | 243 +++++++++++++++--------- util/translation.py | 53 +++++- 8 files changed, 745 insertions(+), 149 deletions(-) create mode 100644 gameupdate/.gitignore create mode 100644 gameupdate/gameupdate/patch-config.txt create mode 100644 gameupdate/gameupdate/vocab.txt diff --git a/gameupdate/.gitignore b/gameupdate/.gitignore new file mode 100644 index 0000000..271a5f7 --- /dev/null +++ b/gameupdate/.gitignore @@ -0,0 +1,121 @@ +# Ignore all files +*.* + +# File Types +!*.mps +!*.json +!*.txt +!*.project +!*.js +!*.7z +!*.csv +!*.ain +!*.fnl +!*.ks +!*.tjs +!*.yaml +!*.rb +!*.rvdata2 + +# Other Needed Files +!.gitignore +!README.md +!patch-config.txt +!GameUpdate* +!patch* +!Game.dat +!bsxx.dat* +!game.ini +!package.nw +!SRPG_Unpacker.exe +!(SRPG_Unpacker Patcher).bat + +# Ignore +previous_patch_sha.txt +kabe3_save.dat +kabe3_system.dat +psbpack.dat +Save* +cg.dat +scene.dat +BSXScript_* + +# Images +!Layout - コピー (2).png_ +!battleUI_sperm.png_ +!QTE_avoid.png_ +!QTE_avoid2.png_ +!QTE_miss.png_ +!QTE_miss2.png_ +!election.png_ +!name.png_ +!preg_def.png_ +!seiyoku.png_ +!father001_tinpira.png_ +!father002_tinpiraboss.png_ +!father003_goblin.png_ +!father004_ork.png_ +!father005_charao.png_ +!father006_tenta.png_ +!father007_name.png_ +!father008_student.png_ +!father009_wolf.png_ +!father010_tanemaki.png_ +!father011_dog.png_ +!father012_mayor.png_ +!father013_forceman.png_ +!father014_unknown.png_ +!father015_father.png_ +!father016_saimin.png_ +!runa_parents.png_ +!logo2.png_ +!save_window.png_ +!title1.png_ +!title2.png_ +!title3.png_ +!title3_1.png_ +!title4.png_ +!title4_1.png_ +!title4_2.png_ +!title5.png_ +!title5_1.png_ +!trial_end.png_ +!tutorial_01.png_ +!tutorial_02.png_ +!tutorial_02_00.png_ +!tutorial_02_01.png_ +!tutorial_02_02.png_ +!tutorial_02_03.png_ +!tutorial_03.png_ +!tutorial_04.png_ +!tutorial_05.png_ +!tutorial_06.png_ +!tutorial_07.png_ +!tutorial_08.png_ +!tutorial_09.png_ +!tutorial_09_01.png_ +!tutorial_10_00.png_ +!tutorial_10_01.png_ +!tutorial_11_00.png_ +!tutorial_11_01.png_ +!tutorial_12_01.png_ +!tutorial_12_02.png_ +!tutorial_12_03.png_ +!tutorial_13_00.png_ +!tutorial_13_01.png_ +!tutorial_13_02.png_ +!tutorial_14_01.png_ +!tutorial_14_02.png_ +!tutorial_21_01.png_ +!tutorial_21_02.png_ +!tutorial_21_03.png_ +!bar_hentai.png_ +!bar_org.png_ +!bar_preg.png_ +!bar_sex.png_ +!start.png_ +!title_01.png_ +!title_02.png_ +!Command_0.png_ +!Command_1.png_ +!Command_2.png_ diff --git a/gameupdate/gameupdate/patch-config.txt b/gameupdate/gameupdate/patch-config.txt new file mode 100644 index 0000000..6798552 --- /dev/null +++ b/gameupdate/gameupdate/patch-config.txt @@ -0,0 +1,3 @@ +username=dazed-translations +repo=magical-girls-runa-and-nanami +branch=main diff --git a/gameupdate/gameupdate/vocab.txt b/gameupdate/gameupdate/vocab.txt new file mode 100644 index 0000000..fd02673 --- /dev/null +++ b/gameupdate/gameupdate/vocab.txt @@ -0,0 +1,212 @@ +# MAIN CHARACTERS +エルミナ (Elmina) - Female; protagonist; elf mage +フェリス (Feris) - Female; spirit entity; Elmina's guardian & companion + +# ALLIES / COMPANIONS +アンナ (Anna) - Female; witch; lives in 魔女の森; ally +メープル (Maple) - Female; companion; timid +ダイヤ (Daiya) - Female; companion; diligent +トルテ (Torte) - Female; companion; cook/baker +クレア (Claire) - Female; full name クレアリス・デカート (Clairis Decart); guild-affiliated +ルイン (Ruin) - Female; Elmina's younger sister (妹); key story motivation + +# SIDE / RECURRING CHARACTERS +バブンスキー (Babunsky) - Male; antagonistic early on, later redeems himself +ドンフシ (Donfushi) - Male; バブンスキー's younger brother; from プライア村 +ツギカ (Tsugika) - Male; young villager; has feelings for Elmina +ミゲル (Miguel) - Male; minor NPC; innkeeper or medical role + +# ANTAGONISTS +アナゼマ (Anazema) - Spirit entity; primary antagonist; former co-guardian of 魔界 +ロザリー (Rosalie) - Female; ruler of カロン; arc boss +キングディブロ (King Diabro) - Boss; Dwarf dungeon; defeated for 水の精霊石 +プーペ (Pupe) - Female; spider-type monster girl (クモ型魔物娘); recurring mini-boss +魔神 (Demon God) - Entity; final-area antagonist; gave power to アナゼマ +ビャクダン (Byakudan) - Named boss; Slum dungeon +# Note: 魔神 = "Demon God" do NOT render as 'Majin' + +# ENEMY / MONSTER TYPES +スライム (Slime) +オーク (Orc) +ゴブリン (Goblin) +ゴブリン兄弟 (Goblin Brothers) +こうもり (Bat) +ミノタウロス (Minotaur) +魔物娘 (Monster Girl) +クモ型魔物娘 (Spider-type Monster Girl) プーペ's species + +# LOCATIONS +## STARTING AREA +プライア村 (Puraia Village) - Elmina's starting village +ノストゥスの森 (Nostus Forest) - Forest east of Puraia Village; early dungeon + +## KARON CITY +カロン (Karon) - Major city; mid-game hub; ruled by ロザリー様 +カロン街道 (Karon Road) - Road connecting Puraia Village to Karon +オプス (Opus) - Karon's high-end inn + +## WITCH'S FOREST +魔女の森 (Witch's Forest) - Anna's domain +魔女道 (Witch's Road) - Path leading to 魔女の森 + +## DUNGEONS +オーク洞窟 (Orc Cave) - Multi-floor orc dungeon +ロザリー屋敷 (Rosalie's Mansion) - ロザリー's residence; 14+ floors + +## DWARF AREA +ドワーフ街道 (Dwarf Road) - Road from Karon toward Dwarf territory +ドワーフ隠れ里 (Dwarf Hidden Village) - Dwarf settlement +ドワーフダンジョン (Dwarf Dungeon) - キングディブロ boss dungeon + +## SLUMS +スラム (Slums) - Poverty district off the Dwarf Road +娼館 (Brothel / Pleasure House) - Located in the Slums + +## ELF VILLAGE +エルフ里 (Elf Village) - Elmina's homeland; destroyed before the story begins + +## SPIRIT STONE DUNGEONS +精霊石ダンジョン (Spirit Stone Dungeon) - Entrance to three elemental paths +火 (Fire) - Fire dungeon path +風 (Wind) - Wind dungeon path +水 (Water) - Water dungeon path + +## DEMON WORLD +魔界 (Demon World) - The demonic realm; final area +# Note: render as "Demon World" NOT "Netherworld" or "Makai" + +# WORLD-BUILDING TERMS +淫紋 (Ero Brand / Lewd Mark) - Contract seal on Elmina's belly; converts desire into magic +精霊石 (Spirit Stone) - Three elemental key items needed to grant Elmina's wish + 炎の精霊石 (Fire Spirit Stone) + 水の精霊石 (Water Spirit Stone) + 風の精霊石 (Wind Spirit Stone) +魔力 (Mana / Magical Power) +ノロイ (Curse) - アナゼマ's seal placed on the Spirit Stones +冒険者 (Adventurer) - profession/rank system +ギルド (Guild) - the Adventurers' Guild +ギルド長 (Guild Master) +調合師 (Alchemist / Apothecary) - crafts 回復ビン from ingredients +チャームリンク (Charm Link) - magical bond item between Elmina and a companion +協会 (Temple / Shrine) - religious institution enshrining the 女神達 +女神 (Goddess) - deities who once saved the continent +女神達 (Goddesses) - plural; enshrined in the 協会 +魔女 (Witch) - female magic practitioner; distinct from 魔法使い (mage) +ドワーフ族 (Dwarf Tribe / Dwarf Clan) +エルフ (Elf) - Elmina's race +アニキ (Aniki) - "big bro"; ドンフシ's term for バブンスキー + +# WEAPON TYPES (Elmina's magic) +炎の魔法 (Fire Magic) - status ailments; sustained damage +水の魔法 (Water Magic) - single-hit damage; raises HP and Defense +風の魔法 (Wind Magic) - multi-hit combos; gains debuffs on growth +盾の魔法 (Shield Magic) - raises Defense; beginner-recommended + +# ARMORS / OUTFITS +エルミナの普段着 (Elmina's Everyday Clothes) +絹の肌着 (Silk Undershirt) +魅惑のバニー (Alluring Bunny) +メイド服 (Maid Outfit) +ワンダーウルフ (Wonder Wolf) +調教服 (Training Outfit / Leash Outfit) + +# BATTLE STATES +戦闘不能 (Incapacitated / KO) +防御 (Guard / Defending) +不死身 (Immortal) +毒 (Poison) +暗闇 (Blind / Darkness) +沈黙 (Silence) +激昂 (Enrage / Fury) +混乱 (Confusion) +魅了 (Charm / Enthralled) +睡眠 (Sleep) + +# KEY ITEMS +怪しげな薬 (Suspicious Medicine) +冒険者登録証 (Adventurer's Registration Card) +トルテのお弁当 (Torte's Lunchbox) +回復ビン (Recovery Vial) - upgrades +1 through +15 +屋敷のカギ (Mansion Key) +力のグローブ (Power Gloves) +爆音袋 (Sonic Bag / Bang Bag) +子猫の入ったカバン (Bag Containing a Kitten) +チャームリンク・メープル (Charm Link: Maple) - +Defense +チャームリンク・アンナ (Charm Link: Anna) - +Magic +チャームリンク・ダイヤ (Charm Link: Daiya) - +Defense + +## Crafting Ingredients (素材) +緑草 (Green Grass) +黄草 (Yellow Grass) +赤草 (Red Grass) +黒草 (Black Grass) +ロゼの花 (Roze Flower) +オルラーヤ (Orlaya) +リンドウ (Gentian) +ベニダケ (Red Mushroom) +サボテンの果実 (Cactus Fruit) +輝くゼンマイ (Glowing Bracken) +触手の若芽 (Tentacle Sprout) +水 (Water) +純水 (Pure Water) +魔力水 (Mana Water) + +## Guild Exchange Drops (ギルド換金) +スライムの核 (Slime Core) +スライムの巨大核 (Slime Giant Core) +スライムの赤核 (Slime Red Core) +スライムの赤粘膜 (Slime Red Membrane) +オークの牙 (Orc Fang) +獣の毛 (Beast Fur) +焔心核 (Flame Heart Core) +蜘蛛の糸 (Spider Thread) +悪魔の角 (Demon Horn) +海魔の墨 (Kraken Ink) +悪魔の羽根 (Demon Feather) +死霊の灰 (Wraith Ash) + +## Mana Crystal EXP Items +小さな魔力結晶 (Small Mana Crystal) - EXP 10 +魔力結晶 (Mana Crystal) - EXP 100 +大きな魔力結晶 (Large Mana Crystal) - EXP 1,000 +巨大な魔力結晶 (Giant Mana Crystal) - EXP 10,000 + +## Boss Mana Cores (素材) +プーペの魔力核 (Pupe's Mana Core) +ビャクダンの魔力核 (Byakudan's Mana Core) +キングディブロの魔力核 (King Diabro's Mana Core) +ロザリーの魔力核 (Rosalie's Mana Core) +タイタンワームの魔力結晶 (Titan Worm Mana Crystal) + +# ROMANISATION CONVENTIONS +エルミナ (Elmina) - not Erumina; matches title +フェリス (Feris) - matches sprite file "feris1" +アンナ (Anna) +メープル (Maple) +ダイヤ (Daiya) - or Dia; avoid "Diamond" in dialogue +バブンスキー (Babunsky) +ドンフシ (Donfushi) +ツギカ (Tsugika) +トルテ (Torte) - German: layer-cake +クレア (Claire) +クレアリス・デカート (Clairis Decart) - both parts are proper nouns +アナゼマ (Anazema) +ルイン (Ruin) - or Luin +プーペ (Pupe) - or Poupée; French: doll +ミゲル (Miguel) +ロザリー (Rosalie) +キングディブロ (King Diabro) +ビャクダン (Byakudan) +カロン (Karon) +プライア村 (Puraia Village) - keep "Puraia" per map file names +ノストゥスの森 (Nostus Forest) +ドワーフ (Dwarf) +エルフ (Elf) +魔界 (Demon World) - NOT "Netherworld"; keep distinct from 地獄 "Hell" +淫紋 (Ero Brand) - or "Lewd Mark"; keep consistent within project +精霊石 (Spirit Stone) +ノロイ (Curse) +魔神 (Demon God) - NOT "Majin"; translate fully +調合師 (Alchemist) - or Apothecary; be consistent +ギルド長 (Guild Master) +アニキ (Aniki) - or "Big Bro"; keep consistent per character \ No newline at end of file diff --git a/gui/workflow_tab.py b/gui/workflow_tab.py index ee5f754..a480fd5 100644 --- a/gui/workflow_tab.py +++ b/gui/workflow_tab.py @@ -1,14 +1,15 @@ -""" +""" RPGMaker Workflow Tab - Automation hub for the full translation pipeline. Provides a guided, step-by-step interface: Step 0 – Select game project folder and import data files into files/ - Step 1 – Edit vocab.txt (glossary) inline - Step 2 – Actor variable substitution (\\n[X] ↔ names) - Step 3 – Auto-detect speaker format and apply to module settings - Step 4 – Phase 1 (safe dialogue codes) and Phase 2 (risky codes) translation - Step 5 – Export translated/ back to the game folder + Step 1 – (Optional) Pre-process game files + Step 2 – Auto-detect speaker format and apply to module settings + Step 3 – Build glossary: parse speakers, then enrich with AI prompt + Step 4 – Actor variable substitution (\\n[X] ↔ names) + Step 5 – Phase 1 (safe dialogue codes) and Phase 2 (risky codes) translation + Step 6 – Export translated/ back to the game folder """ from __future__ import annotations @@ -417,12 +418,12 @@ class WorkflowTab(QWidget): vbox.addWidget(_make_hr()) self._build_preprocess(vbox) vbox.addWidget(_make_hr()) + self._build_step3(vbox) + vbox.addWidget(_make_hr()) self._build_step1(vbox) vbox.addWidget(_make_hr()) self._build_step2(vbox) vbox.addWidget(_make_hr()) - self._build_step3(vbox) - vbox.addWidget(_make_hr()) self._build_step4(vbox) vbox.addWidget(_make_hr()) self._build_step6(vbox) @@ -543,29 +544,54 @@ class WorkflowTab(QWidget): _PROMPT_GLOSSARY = ( "You are helping me build a complete translation glossary for a Japanese RPGMaker game.\n" "\n" - "Please scan ALL of these game files (Actors.json, CommonEvents.json, Map*.json, " - "Troops.json, and any other available files) and produce TWO sections:\n" + "⚠️ FILE SIZE WARNING: Map files and CommonEvents.json can be extremely large " + "(hundreds of thousands of lines). Do NOT attempt to read them in full — you will " + "hit context limits and miss content. Instead, use this strategy:\n" + "\n" + " 1. Read the small structured DB files IN FULL first — these are the richest " + "source of names and are always small:\n" + " Actors.json, Classes.json, Troops.json, Skills.json, Items.json, " + "Armors.json, Weapons.json, States.json, System.json\n" + "\n" + " 2. For large files (CommonEvents.json, Map*.json), do NOT read sequentially. " + "Instead, SEARCH (grep) for:\n" + " - Named character patterns: 【, \\\\n<, \\\\k< at the start of dialogue lines\n" + " - Unique proper nouns: capitalised katakana clusters or kanji compound nouns " + "in \"message\" or \"parameters\" fields\n" + " Scan Map001.json through Map010.json at most — early maps have the most " + "story-critical dialogue.\n" + "\n" + " 3. Stop adding entries once you stop finding new names/terms — do not pad.\n" + "\n" + "Produce TWO sections of output.\n" + "\n" + "Output the results EXACTLY in the format shown below, including the category headers " + "starting with #. Do not add any other text, preamble, or explanation outside the entries.\n" "\n" "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" - "SECTION 1 — NAMED CHARACTERS\n" + "# Game Characters\n" "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" "Identify every NAMED CHARACTER that appears in dialogue or descriptions.\n" "For each character provide:\n" " - Japanese name (katakana/kanji as it appears in-game)\n" " - English transliteration or translation\n" " - Gender (Male / Female / Unknown) — infer from speech patterns or pronouns\n" - " - Brief role note (e.g. protagonist, antagonist, NPC, merchant)\n" + " - Role (protagonist, antagonist, NPC, etc.)\n" + " - Speech register and personality notes — how they speak, their tone, any nicknames, " + "whether their name is player-chosen, etc.\n" "\n" - "Format — one entry per line:\n" - "山田太郎 (Yamada Taro) - Male; protagonist; high-school student\n" - "浅井花子 (Asai Hanako) - Female; childhood friend; lives in 栄町\n" + "Format — the header line, then one entry per line:\n" + "# Game Characters\n" + "\u30b7\u30ed (Shiro) - Female; protagonist; player-controlled (Actors.json ID 1); " + "speaks in a flustered, cute register with feminine speech markers; nickname \u30d0\u30ab\u732b\u3002\n" + "\u30af\u30ed\u30cd (Kurone) - Female; antagonist; cold and terse; speaks in short cutting sentences\n" "\n" "Rules:\n" " - Named characters only — no generic enemy types or unnamed NPCs.\n" " - If a character has a player-chosen name (e.g. Actors.json id 1), note it explicitly.\n" "\n" "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" - "SECTION 2 — WORLDBUILDING TERMS\n" + "# Worldbuilding Terms\n" "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" "Identify lore-specific terms that appear in dialogue, descriptions, or narration " "but do NOT have a dedicated database file — i.e. terms the translation tool will NOT " @@ -573,24 +599,53 @@ class WorkflowTab(QWidget): "\n" "Target specifically:\n" " - Faction / organisation names (kingdoms, guilds, cults, nations)\n" - " - Location names that are mentioned in dialogue but are not map titles\n" + " - Location names mentioned in dialogue but not map titles\n" " - Unique magic systems, schools of magic, or power classifications\n" " - Lore titles and honorifics unique to this setting\n" - " - Recurring in-universe concepts, events, or proper nouns with no English equivalent\n" + " - Recurring in-universe concepts or proper nouns with no English equivalent\n" "\n" - "For each term provide:\n" - " - Japanese original (as it appears in-game)\n" - " - Recommended English translation\n" - " - Brief note explaining what it is\n" - "\n" - "Format — one entry per line:\n" - "魔世界 (Demon World) - The demonic realm referenced in NPC dialogue; not a named map\n" - "聖剣教団 (Holy Blade Order) - Antagonist faction controlling the eastern territories\n" + "Format — the header line, then one entry per line:\n" + "# Worldbuilding Terms\n" + "\u9b54\u4e16\u754c (Demon World) - The demonic realm referenced in NPC dialogue; not a named map\n" + "\u8056\u5263\u6559\u56e3 (Holy Blade Order) - Antagonist faction controlling the eastern territories\n" "\n" "Rules:\n" " - Do NOT list skill names, item names, weapon names, armour names — the tool handles those.\n" - " - Skip generic RPG words (ポーション, レベル, ステータス, etc.).\n" - " - Do NOT repeat character names here — those belong in Section 1." + " - Skip generic RPG words (\u30dd\u30fc\u30b7\u30e7\u30f3, \u30ec\u30d9\u30eb, \u30b9\u30c6\u30fc\u30bf\u30b9, etc.).\n" + " - Do NOT repeat character names here.\n" + "\n" + "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" + "COMPLETE EXAMPLE OUTPUT\n" + "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" + "Below is what a correct, well-formed response looks like.\n" + "Your output should follow this structure exactly:\n" + "\n" + "```\n" + "# Game Characters\n" + "アリア (Aria) - Female; protagonist; player-chosen name (Actors.json ID 1); " + "speaks cheerfully in casual feminine speech; nicknamed アリアちゃん by her sister\n" + "ゼクス (Zex) - Male; antagonist; cold and commanding; addresses others with contempt; " + "uses archaic formal register\n" + "カナエ (Kanae) - Female; NPC shopkeeper; warm and motherly; ends sentences with わね\n" + "\n" + "# Worldbuilding Terms\n" + "虚無の穴 (Void Rift) - Dimensional tear referenced repeatedly in Act 2 NPC dialogue; " + "not a named map location\n" + "鋼の誓約 (Iron Vow) - Sacred oath-binding ritual unique to the knightly order; " + "appears in story cutscenes\n" + "裁定者 (Arbiter) - Title held by the ruling council; lore-specific rank with no " + "real-world equivalent\n" + "```\n" + "\n" + "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" + "GLOBAL RULES (apply to both sections)\n" + "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" + " - NEVER give two options for any term (e.g. 'Sylfia / Sylphia' is wrong). " + "Always commit to a single best translation. If multiple transliterations exist, " + "pick the most etymologically accurate or natural-sounding one and use only that.\n" + " - Use a plain hyphen-minus (-) as the separator between the Japanese entry and " + "its description. Never use an em dash (\u2014) or en dash (\u2013) \u2014 the " + "translation tool only recognises the plain hyphen." ) # ── Step 1 (Optional): Pre-process ──────────────────────────────── @@ -693,8 +748,8 @@ class WorkflowTab(QWidget): tc_inner = QVBoxLayout(tc) tc_inner.setSpacing(4) tc_desc = QLabel( - "Copies everything from the gameupdate/ folder in the game root " - "into the game\'s data folder, overwriting existing files." + "Copies everything from the gameupdate/ folder " + "into the game\'s root folder, overwriting existing files." ) tc_desc.setTextFormat(Qt.RichText) tc_desc.setWordWrap(True) @@ -719,7 +774,7 @@ class WorkflowTab(QWidget): tc_dst_row = QHBoxLayout() tc_dst_row.addWidget(QLabel("Destination:")) - self.pp_gameupdate_dst_label = QLabel("(data folder auto-filled from project)") + self.pp_gameupdate_dst_label = QLabel("(game root folder auto-filled from project)") self.pp_gameupdate_dst_label.setStyleSheet("color:#888;font-size:10px;") tc_dst_row.addWidget(self.pp_gameupdate_dst_label, 1) tc_inner.addLayout(tc_dst_row) @@ -752,7 +807,7 @@ class WorkflowTab(QWidget): ) def _build_step1(self, layout: QVBoxLayout): - layout.addWidget(_make_section_label("Step 2 — Glossary (vocab.txt)")) + layout.addWidget(_make_section_label("Step 3 — Glossary (vocab.txt)")) hint = QLabel( "Edit your glossary below. Character names, genders, and worldbuilding terms " "here are injected into every AI prompt for consistent terminology." @@ -761,8 +816,43 @@ class WorkflowTab(QWidget): hint.setStyleSheet("color:#888;font-size:10px;padding-bottom:4px;") layout.addWidget(hint) + # ---- Parse Speakers ------------------------------------------------- + spk_box = QGroupBox("2a — Parse Speakers (Auto-detect Names)") + spk_box.setStyleSheet( + "QGroupBox{color:#ccc;border:1px solid #444;border-radius:3px;" + "margin-top:8px;font-size:11px;}" + "QGroupBox::title{padding:0 6px;}" + ) + spk_inner = QVBoxLayout(spk_box) + spk_inner.setSpacing(6) + + spk_hint = QLabel( + "Scans every file in files/ and extracts all detected speaker names " + "into a # Speakers section of vocab.txt. " + "Run this first — the AI prompt below will then work from the pre-populated " + "list and can enrich entries with gender, role, and speech register." + ) + spk_hint.setTextFormat(Qt.RichText) + spk_hint.setWordWrap(True) + spk_hint.setStyleSheet("color:#888;font-size:10px;") + spk_inner.addWidget(spk_hint) + + spk_row = QHBoxLayout() + self._parse_speakers_status = QLabel("") + self._parse_speakers_status.setStyleSheet("color:#aaa;font-size:10px;") + spk_row.addWidget(self._parse_speakers_status, 1) + self._parse_speakers_btn = _make_btn("🔍 Parse Speakers", "#007acc") + self._parse_speakers_btn.setToolTip( + "Scan all game files for speaker names and write them to vocab.txt" + ) + self._parse_speakers_btn.clicked.connect(self._run_parse_speakers) + spk_row.addWidget(self._parse_speakers_btn) + spk_inner.addLayout(spk_row) + + layout.addWidget(spk_box) + # ---- Copilot / Cursor prompt helpers -------------------------------- - prompt_box = QGroupBox("AI Prompt Helpers (Copilot / Cursor)") + prompt_box = QGroupBox("2b — AI Prompt Helpers (Copilot / Cursor)") prompt_box.setStyleSheet( "QGroupBox{color:#ccc;border:1px solid #444;border-radius:3px;" "margin-top:8px;font-size:11px;}" @@ -772,8 +862,10 @@ class WorkflowTab(QWidget): pb_inner.setSpacing(6) prompt_hint = QLabel( - "Copy a prompt below, paste it into GitHub Copilot Chat or Cursor with your " - "game files open, then paste the AI’s output into vocab.txt." + "After parsing speakers above, copy the prompt below and paste it into " + "GitHub Copilot Chat or Cursor with your game files open. " + "The AI will enrich the speaker list with gender, role, and speech register " + "and add worldbuilding terms. Paste the AI's output back into vocab.txt." ) prompt_hint.setWordWrap(True) prompt_hint.setStyleSheet("color:#888;font-size:10px;") @@ -800,8 +892,10 @@ class WorkflowTab(QWidget): # ---- vocab.txt editor ----------------------------------------------- layout.addWidget(_make_section_label("vocab.txt editor")) format_hint = QLabel( - "Expected line format: Japanese (English) - Gender; role; notes\n" - "Example: 山田太郎 (Yamada Taro) - Male; protagonist; village blacksmith" + "Put character entries under # Game Characters — they are always sent to the AI.\n" + "Format: Japanese (English) - Gender; role; speech register / personality notes\n" + "Example: シロ (Shiro) - Female; protagonist; speaks in a flustered, cute register with feminine speech markers\n" + "Universal terms (honorifics, elements, etc.) live in vocab_base.txt and are auto-appended on every save." ) format_hint.setFont(QFont("Consolas", 9)) format_hint.setStyleSheet( @@ -834,7 +928,7 @@ class WorkflowTab(QWidget): # ── Step 2: Actor Variables ───────────────────────────────────────────── def _build_step2(self, layout: QVBoxLayout): - layout.addWidget(_make_section_label("Step 3 — Actor Variable Substitution")) + layout.addWidget(_make_section_label("Step 4 — Actor Variable Substitution")) hint = QLabel( "Replaces \\n[X] RPGMaker name variables with actual actor names " "so the AI has proper context. After translation the variables are " @@ -872,7 +966,7 @@ class WorkflowTab(QWidget): # ── Step 3: Speaker Detection ─────────────────────────────────────────── def _build_step3(self, layout: QVBoxLayout): - layout.addWidget(_make_section_label("Step 4 — Speaker Format Detection")) + layout.addWidget(_make_section_label("Step 2 — Speaker Format Detection")) hint = QLabel( "Scans map files to determine how speaker names are embedded in dialogue " "and automatically sets the correct INLINE401SPEAKERS / FIRSTLINESPEAKERS " @@ -1099,11 +1193,69 @@ class WorkflowTab(QWidget): # Step 1 – Vocab # ───────────────────────────────────────────────────────────────────────── + def _run_parse_speakers(self): + """Launch a speaker-parse pass over files/ and write results to vocab.txt.""" + try: + from gui.translation_tab import TranslationWorker + except Exception as exc: + self._log(f"Could not import TranslationWorker: {exc}") + return + + project_root = Path(__file__).parent.parent + files_dir = project_root / "files" + all_json = sorted( + p.name for p in files_dir.glob("*.json") if p.name != ".gitkeep" + ) if files_dir.exists() else [] + + if not all_json: + self._log("No JSON files found in files/. Run Step 1 (import) first.") + self._parse_speakers_status.setText("No files found in files/.") + return + + self._log(f"Parsing speakers from {len(all_json)} file(s)...") + self._parse_speakers_btn.setEnabled(False) + self._parse_speakers_status.setText("Running...") + + module_info = ["RPG Maker MV/MZ", [".json"], None] + worker = TranslationWorker( + project_root, module_info, estimate_only=False, + selected_files=all_json, + parse_speakers=True, + ) + worker.log_signal.connect(self._log) + worker.progress_signal.connect( + lambda cur, tot, fn: self._parse_speakers_status.setText( + f"[{cur}/{tot}] {Path(fn).name}" + ) + ) + worker.finished_signal.connect(self._on_parse_speakers_done) + self._worker = worker + worker.start() + + def _on_parse_speakers_done(self, ok: bool, msg: str): + self._parse_speakers_btn.setEnabled(True) + if ok: + self._parse_speakers_status.setText( + "Done — # Speakers section written to vocab.txt" + ) + self._reload_vocab() + self._log("Speaker parse complete. vocab.txt updated with # Speakers.") + else: + self._parse_speakers_status.setText(f"Failed: {msg}") + self._log(f"Speaker parse failed: {msg}") + + _BASE_SEPARATOR = "# ── Base Vocabulary (auto-appended from vocab_base.txt — do not edit below) ──\n" + def _reload_vocab(self): vocab_path = Path("vocab.txt") try: if vocab_path.exists(): - self.vocab_editor.setPlainText(vocab_path.read_text(encoding="utf-8")) + text = vocab_path.read_text(encoding="utf-8") + # Strip the auto-appended base section so editor shows only game-specific content + sep_idx = text.find(self._BASE_SEPARATOR) + if sep_idx != -1: + text = text[:sep_idx].rstrip("\n") + self.vocab_editor.setPlainText(text) else: self.vocab_editor.setPlainText("# Add character glossary entries here\n") except Exception as exc: @@ -1111,10 +1263,12 @@ class WorkflowTab(QWidget): def _save_vocab(self): try: - Path("vocab.txt").write_text( - self.vocab_editor.toPlainText(), encoding="utf-8" - ) - self._log("✅ vocab.txt saved.") + game_text = self.vocab_editor.toPlainText().rstrip("\n") + base_path = Path("vocab_base.txt") + base_text = base_path.read_text(encoding="utf-8") if base_path.exists() else "" + combined = game_text + "\n\n" + self._BASE_SEPARATOR + base_text + Path("vocab.txt").write_text(combined, encoding="utf-8") + self._log("✅ vocab.txt saved (base terms from vocab_base.txt appended).") except Exception as exc: self._log(f"❌ Could not save vocab.txt: {exc}") @@ -1442,7 +1596,7 @@ class WorkflowTab(QWidget): except Exception: pass try: - self.pp_gameupdate_dst_label.setText(data_path or "(no data folder detected)") + self.pp_gameupdate_dst_label.setText(game_root or "(no game folder detected)") except Exception: pass @@ -1488,12 +1642,12 @@ class WorkflowTab(QWidget): def _run_gameupdate(self): src = self.pp_gameupdate_edit.text().strip() - dst = self._data_path + dst = self.folder_edit.text().strip() if not src: self._log("⚠ No gameupdate folder path set.") return if not dst: - self._log("⚠ No data folder detected. Complete Step 0 first.") + self._log("⚠ No game root folder set. Complete Step 0 first.") return if not Path(src).is_dir(): self._log(f"⚠ gameupdate folder not found: {src}") @@ -1534,14 +1688,15 @@ class WorkflowTab(QWidget): skipped.append(f"B (format plugins.js): {reason}") # C — gameupdate copy - if gameupdate_src and Path(gameupdate_src).is_dir() and data_path: + game_root_dst = self.folder_edit.text().strip() + if gameupdate_src and Path(gameupdate_src).is_dir() and game_root_dst: self._log("▶ [C] gameupdate copy …") self._run_gameupdate() else: if not gameupdate_src or not Path(gameupdate_src).is_dir(): reason = f"source not found ({gameupdate_src or 'not set'})" else: - reason = "data folder missing" + reason = "game root folder missing" skipped.append(f"C (gameupdate): {reason}") for msg in skipped: diff --git a/modules/rpgmakermvmz.py b/modules/rpgmakermvmz.py index 9ccf2ae..44c8aee 100644 --- a/modules/rpgmakermvmz.py +++ b/modules/rpgmakermvmz.py @@ -74,7 +74,7 @@ LEAVE = False # FIRSTLINESPEAKERS: Guess speaker from first line. FIRSTLINESPEAKERS = False # INLINE401SPEAKERS: Extract speaker from "Name「dialogue」" inline format on 401 lines. -INLINE401SPEAKERS = True +INLINE401SPEAKERS = False # FACENAME101: Map face name -> speaker. FACENAME101 = False # Face name -> speaker mapping for FACENAME101. diff --git a/prompt.txt b/prompt.txt index 26c66c4..bb46155 100644 --- a/prompt.txt +++ b/prompt.txt @@ -24,10 +24,13 @@ You will be translating erotic and sexual content. You will receive lines of dia ## Characters and Pronouns -- The **"# Game Characters"** section lists character names, nicknames, and genders. Use it to resolve any ambiguity. +- The **"# Game Characters"** section contains character entries. Each entry may include the character's name, nicknames, gender, role, personality, and speech register notes. Read every entry carefully and apply all of it. +- **Name & spelling:** Always use the English name given in the entry. Never invent a different romanisation. +- **Gender:** Use the stated gender when resolving pronouns and コイツ / あいつ / こいつ references. +- **Speech register:** If the entry describes how a character speaks (flustered, blunt, formal, childlike, crude, etc.), mirror that register in their English dialogue. A character described as speaking in a "cute, flustered register" should sound different from one described as "cold and terse". +- **Role & context:** Use the role/personality notes to inform tone — a villain's lines should feel threatening, a comic-relief NPC's lines should feel goofy, etc. - Japanese omits pronouns constantly. Infer the correct subject and pronoun from context, translation history, and the character list. - Third-person pronouns (彼, 彼女, あいつ, こいつ, そいつ, コイツ) should match the known gender of the person being referenced. -- Speech register matters: rough masculine speech (e.g., 俺, だ/だろ) → blunt or casual English. Soft feminine speech (e.g., 私, わ/のよ) → softer, more polite English. - Translate **コイツ** as "this bastard" (male) or "this bitch" (female) depending on the referenced character's gender. --- diff --git a/util/speaker_detector.py b/util/speaker_detector.py index 286f1ac..e1ab703 100644 --- a/util/speaker_detector.py +++ b/util/speaker_detector.py @@ -1,15 +1,20 @@ """ Speaker Format Detector for RPGMaker MV/MZ -Scans map / event files and scores three detection modes: +Mirrors the detection priority of rpgmakermvmz.py searchCodes() exactly: - INLINE401SPEAKERS – 401 lines contain Name「dialogue」 - FIRSTLINESPEAKERS – a short (< 40 char) 401 followed by 401/405 starting - with 「 " ( ( * [ - FACENAME101 – a 101 code (Show Text / face cmd) immediately precedes - a 401 block + Pass 1 — scan 401/405 codes in order: + 1. \\n / \\k inline nametag codes (always active, no flag needed) + 2. 【Name】 alone on a 401 line (always active, no flag needed) + 3. 【Name】dialogue on same 401 line (always active, no flag needed) + 4. Name「dialogue」 inline quote -> INLINE401SPEAKERS + 5. Short 401 (<40 chars) followed by 401 whose + text starts with 「 " ( ( * [ -> FIRSTLINESPEAKERS -Returns the best mode and confidence scores for each. + Pass 2 — only if Pass 1 produced no reliable hits: + 6. 101 code param[0] is a non-empty name string -> FACENAME101 + +Returns the best mode and confidence scores. """ from __future__ import annotations @@ -18,13 +23,28 @@ import json import re from pathlib import Path -# Regex used to detect inline quote speaker: "Name「..." at start of 401 text -_INLINE_RE = re.compile(r"^([^\s「」。、!?…\\\n]{1,20})「") +# ── Regexes matching rpgmakermvmz.py exactly ──────────────────────────────── -# Characters that signal the next 401 is dialogue (first-line speaker heuristic) +# \\n / \\k nametag codes (always active in module) +_NAMETAG_RE = re.compile( + r"[\\]+[kKnN][wWcCrRrEe]?[\[<](?:[\\]*\w\[\d+\])?(.*?)(?:[\\]*\w\[\d+\])?[>]" +) + +# 【Name】 alone on the line (with optional trailing control codes) +_BRACKET_ALONE_RE = re.compile( + r"^\s*【[^】]+】(?:\s*|(?:[\\]+[A-Za-z]+(?:\[(?:[^\[\]]|\[[^\]]*\])*\])+\s*)*)$" +) + +# 【Name】dialogue on the same line +_BRACKET_INLINE_RE = re.compile(r"^\s*【([^】]+)】(.+)", re.DOTALL) + +# Inline quote: Name「dialogue… +_INLINE_QUOTE_RE = re.compile(r"^([^\s「」。、!?…\\\n]{1,20})「") + +# Dialogue starters that follow a FIRSTLINESPEAKERS name line _DIALOGUE_STARTERS = ("「", '"', "(", "(", "*", "[") -# Minimum number of qualifying events to consider a result reliable +# Minimum hits to trust a result _MIN_HITS = 3 @@ -32,41 +52,46 @@ def detect_speaker_format( files_dir: str | Path = "files", sample_size: int = 20, ) -> dict: - """Scan up to *sample_size* map files and score speaker detection modes. + """Scan up to *sample_size* map files and determine the speaker format. Returns: { - "best_mode": "INLINE401SPEAKERS" | "FIRSTLINESPEAKERS" | "FACENAME101" | "NONE", + "best_mode": "INLINE401SPEAKERS" | "FIRSTLINESPEAKERS" | "FACENAME101" + | "ALWAYS_ON" | "NONE", "scores": { - "INLINE401SPEAKERS": int, # raw hit count - "FIRSTLINESPEAKERS": int, - "FACENAME101": int, + "nametag_codes": int, # \\n hits — always handled, no flag + "bracket_401": int, # 【Name】 hits — always handled, no flag + "INLINE401SPEAKERS": int, + "FIRSTLINESPEAKERS": int, + "FACENAME101": int, }, - "total_401_groups": int, # total dialogue groups examined + "total_401_groups": int, "files_scanned": int, - "recommended_config": { # ready to apply to rpgmakermvmz.py - "FIRSTLINESPEAKERS": bool, + "recommended_config": { "INLINE401SPEAKERS": bool, + "FIRSTLINESPEAKERS": bool, "FACENAME101": bool, }, "confidence": "high" | "medium" | "low", - "note": str, # human-readable explanation + "note": str, } """ files_dir = Path(files_dir) scores = { + "nametag_codes": 0, + "bracket_401": 0, "INLINE401SPEAKERS": 0, "FIRSTLINESPEAKERS": 0, - "FACENAME101": 0, + "FACENAME101": 0, } total_groups = 0 files_scanned = 0 - # Collect map files (Maps only, not MapInfos) + # Collect map files (Maps only, not MapInfos), largest first for better signal map_files = sorted( [p for p in files_dir.glob("Map[0-9]*.json") if p.is_file()], key=lambda p: p.stat().st_size, - reverse=True, # scan larger maps first for better signal + reverse=True, )[:sample_size] for fp in map_files: @@ -104,121 +129,155 @@ def detect_speaker_format( for entry in ce_data or []: if not entry: continue - cmd_list = entry.get("list") or [] - _score_command_list(cmd_list, scores) + _score_command_list(entry.get("list") or [], scores) except Exception: pass - # Determine best mode - best = max(scores, key=lambda k: scores[k]) - best_score = scores[best] - second_best = sorted(scores.values(), reverse=True)[1] - - if best_score < _MIN_HITS: - best_mode = "NONE" - confidence = "low" - note = ( - f"No strong speaker pattern detected " - f"(scanned {files_scanned} file(s), {total_groups} dialogue group(s)). " - "Consider checking speaker settings manually." - ) - elif best_score > second_best * 2: - best_mode = best - confidence = "high" - note = ( - f"Strong signal for {best} " - f"({best_score} hits vs {second_best} for next best, " - f"{total_groups} dialogue group(s) in {files_scanned} file(s))." - ) - else: - best_mode = best - confidence = "medium" - note = ( - f"Moderate signal for {best} " - f"({best_score} hits, {second_best} for next-best mode). " - "Review a sample of dialogue manually to confirm." - ) + # ── Decision: 401-based patterns first, FACENAME101 only as fallback ──── + always_on_hits = scores["nametag_codes"] + scores["bracket_401"] + inline_hits = scores["INLINE401SPEAKERS"] + first_hits = scores["FIRSTLINESPEAKERS"] + face_hits = scores["FACENAME101"] + total_401_hits = always_on_hits + inline_hits + first_hits recommended = { - "FIRSTLINESPEAKERS": best_mode == "FIRSTLINESPEAKERS", - "INLINE401SPEAKERS": best_mode == "INLINE401SPEAKERS", - "FACENAME101": best_mode == "FACENAME101", + "INLINE401SPEAKERS": False, + "FIRSTLINESPEAKERS": False, + "FACENAME101": False, } + if total_401_hits >= _MIN_HITS: + if always_on_hits >= _MIN_HITS and always_on_hits >= inline_hits and always_on_hits >= first_hits: + best_mode = "ALWAYS_ON" + confidence = "high" if always_on_hits > (inline_hits + first_hits) else "medium" + note = ( + f"Speakers detected via \\\\n codes or 【Name】 brackets " + f"({always_on_hits} hits) — no extra flags needed. " + f"INLINE={inline_hits}, FIRSTLINE={first_hits}. " + f"Scanned {files_scanned} file(s), {total_groups} dialogue group(s)." + ) + elif inline_hits >= first_hits: + best_mode = "INLINE401SPEAKERS" + recommended["INLINE401SPEAKERS"] = True + confidence = "high" if inline_hits > first_hits * 2 else "medium" + note = ( + f"INLINE401SPEAKERS: {inline_hits} hits " + f"(FIRSTLINESPEAKERS: {first_hits}, always-on: {always_on_hits}). " + f"Scanned {files_scanned} file(s), {total_groups} dialogue group(s)." + ) + else: + best_mode = "FIRSTLINESPEAKERS" + recommended["FIRSTLINESPEAKERS"] = True + confidence = "high" if first_hits > inline_hits * 2 else "medium" + note = ( + f"FIRSTLINESPEAKERS: {first_hits} hits " + f"(INLINE401SPEAKERS: {inline_hits}, always-on: {always_on_hits}). " + f"Scanned {files_scanned} file(s), {total_groups} dialogue group(s)." + ) + elif face_hits >= _MIN_HITS: + best_mode = "FACENAME101" + recommended["FACENAME101"] = True + confidence = "high" if face_hits > _MIN_HITS * 2 else "medium" + note = ( + f"No 401-based speaker pattern found (401 hits: {total_401_hits}). " + f"FACENAME101 recommended based on {face_hits} 101-code name hits. " + f"Scanned {files_scanned} file(s), {total_groups} dialogue group(s)." + ) + else: + best_mode = "NONE" + confidence = "low" + note = ( + f"No reliable speaker pattern detected " + f"(401 hits: {total_401_hits}, FACENAME101 hits: {face_hits}). " + f"Scanned {files_scanned} file(s), {total_groups} dialogue group(s). " + "Check speaker settings manually." + ) + return { - "best_mode": best_mode, - "scores": scores, - "total_401_groups": total_groups, - "files_scanned": files_scanned, + "best_mode": best_mode, + "scores": scores, + "total_401_groups": total_groups, + "files_scanned": files_scanned, "recommended_config": recommended, - "confidence": confidence, - "note": note, + "confidence": confidence, + "note": note, } -# --------------------------------------------------------------------------- -# Internal helpers -# --------------------------------------------------------------------------- +# ── Internal helpers ───────────────────────────────────────────────────────── def _score_command_list(cmd_list: list, scores: dict) -> None: - """Walk a command list and increment scores for detected patterns.""" + """Walk a command list and score speaker patterns in module priority order.""" if not cmd_list: return - last_101_face: str | None = None # face name from most recent 101 cmd - i = 0 while i < len(cmd_list): - cmd = cmd_list[i] - code = cmd.get("code") + cmd = cmd_list[i] + code = cmd.get("code") params = cmd.get("parameters") or [] - # ---- 101: Show Text (face) ---- + # ── 101: Show Text / face name ──────────────────────────────────────── if code == 101: - # params[0] = face name, params[1] = face index - last_101_face = (params[0] if params else None) or "" + # FACENAME101: param[0] is a non-empty face-name string + face = params[0] if params else "" + if isinstance(face, str) and face.strip(): + scores["FACENAME101"] += 1 i += 1 continue - # ---- 401 / 405: dialogue line ---- + # ── 401 / 405: dialogue line ────────────────────────────────────────── if code in (401, 405): text = (params[0] if params else "") or "" - # Score INLINE401SPEAKERS - if _INLINE_RE.match(text): + # 1. \\n nametag codes (always active — no flag needed) + if _NAMETAG_RE.search(text): + scores["nametag_codes"] += 1 + i += 1 + continue # module strips the nametag before further processing + + # 2+3. 【Name】 alone on line, or 【Name】dialogue inline (always active) + if _BRACKET_ALONE_RE.match(text) or _BRACKET_INLINE_RE.match(text): + scores["bracket_401"] += 1 + i += 1 + continue + + # 4. INLINE401SPEAKERS: Name「dialogue + if _INLINE_QUOTE_RE.match(text): scores["INLINE401SPEAKERS"] += 1 + i += 1 + continue - # Score FACENAME101 — if face was set right before this block - if last_101_face is not None: - scores["FACENAME101"] += 1 - last_101_face = None # only count once per block - - # Score FIRSTLINESPEAKERS — short line with no dialogue starters, - # followed by a 401/405 that starts with a dialogue starter + # 5. FIRSTLINESPEAKERS: short line followed by 401 starting with + # a dialogue-starter character if ( len(text) < 40 - and not any(text.lstrip().startswith(s) for s in _DIALOGUE_STARTERS) and _has_japanese(text) + and not any(text.lstrip().startswith(s) for s in _DIALOGUE_STARTERS) ): j = i + 1 while j < len(cmd_list) and cmd_list[j].get("code") == -1: j += 1 if j < len(cmd_list) and cmd_list[j].get("code") in (401, 405): - next_text = ((cmd_list[j].get("parameters") or [""])[0]) or "" - if next_text.lstrip().startswith(_DIALOGUE_STARTERS): + next_params = cmd_list[j].get("parameters") or [] + next_text = (next_params[0] if next_params else "") or "" + # Strip leading RPGMaker control codes before checking starter + next_stripped = re.sub( + r"^(?:[\\]+[^cCnNiIkKvVSs{}]+?\[[\d\w\W]+?\]?\])+", + "", next_text, + ).lstrip() + if next_stripped and next_stripped[0] in _DIALOGUE_STARTERS: scores["FIRSTLINESPEAKERS"] += 1 i += 1 continue - # Any non-dialogue code resets the face context - if code not in (-1,): - last_101_face = None - i += 1 -_JP_RE = re.compile(r"[\u3000-\u303F\u3040-\u309F\u30A0-\u30FF\u4E00-\u9FFF\uF900-\uFAFF\uFF61-\uFF9F]") +_JP_RE = re.compile( + r"[\u3000-\u303F\u3040-\u309F\u30A0-\u30FF\u4E00-\u9FFF\uF900-\uFAFF\uFF61-\uFF9F]" +) def _has_japanese(text: str) -> bool: diff --git a/util/translation.py b/util/translation.py index 47c8dc8..fe862e8 100644 --- a/util/translation.py +++ b/util/translation.py @@ -674,6 +674,9 @@ def buildMatchedVocabText(vocabPairs, subbedText, history=None): return matchedVocabText +_BASE_VOCAB_SEPARATOR = "# ── Base Vocabulary" # Prefix of the separator line written by _save_vocab + + def createContext(config, fullPromptFlag, subbedText, formatType, history=None): """Create system and user messages for translation. @@ -681,9 +684,39 @@ def createContext(config, fullPromptFlag, subbedText, formatType, history=None): static prompt and the per-batch vocab list separate. This lets Claude prompt-caching mark only the stable prefix with cache_control, avoiding cache invalidation caused by changing vocabulary matches. + + Cached in static_system (never changes between batches): + - prompt.txt content + - # Game Characters entries + - Base vocabulary (vocab_base.txt: honorifics, elements, demons, etc.) + + Dynamic in vocab_text (matched per-batch): + - All other game-specific vocab terms (items, worldbuilding, etc.) """ - vocabPairs = parseVocabWithCategories(config.vocab) - matchedVocabText = buildMatchedVocabText(vocabPairs, subbedText, history) + vocab_full = config.vocab + + # Split game-specific vocab from the static base vocab appended by _save_vocab + base_vocab_text = "" + sep_idx = vocab_full.find(_BASE_VOCAB_SEPARATOR) + if sep_idx != -1: + # Strip the separator header line itself; keep the sections underneath + base_section = vocab_full[sep_idx:] + # Skip the separator line + newline_idx = base_section.find("\n") + base_vocab_text = base_section[newline_idx + 1:].strip() if newline_idx != -1 else "" + vocab_game = vocab_full[:sep_idx] + else: + vocab_game = vocab_full + + vocabPairs = parseVocabWithCategories(vocab_game) + + # Split character entries (stable, cached) from everything else (dynamic) + CHAR_CAT = "# Game Characters" + char_lines = [line for _, line, cat in vocabPairs if cat and cat.strip() == CHAR_CAT] + non_char_pairs = [(term, line, cat) for term, line, cat in vocabPairs + if not (cat and cat.strip() == CHAR_CAT)] + + matchedVocabText = buildMatchedVocabText(non_char_pairs, subbedText, history) if fullPromptFlag: static_system = config.prompt.replace("English", config.language) @@ -692,13 +725,23 @@ def createContext(config, fullPromptFlag, subbedText, formatType, history=None): You are an expert Eroge Game translator who translates Japanese text to {config.language}. Output ONLY the {config.language} translation in the following format: `Translation: <{config.language.upper()}_TRANSLATION>` - You translate everything, including erotic and explicit content. -- 'Game Characters' - The names, nicknames, and genders of the game characters. Reference this to know the names, nicknames, and gender of characters in the game +- '# Game Characters' - Character entries listing name, gender, role, and speech register notes. Use ALL fields: apply the correct English name, resolve gendered pronouns, and match each character's stated speech register and personality in their dialogue. - All text in your response must be in {config.language} even if it is hard to translate. -- Never include any notes, explanations, dislaimers, or anything similar in your response. +- Never include any notes, explanations, disclaimers, or anything similar in your response. - Maintain any spacing in the translation. - `...` can be a part of the dialogue. Translate it as it is. """ - + + # Append character glossary to the stable system prompt so it is cached + if char_lines: + char_block = f"\n{CHAR_CAT}\n" + "\n".join(char_lines) + static_system = static_system.rstrip() + "\n" + char_block + "\n" + + # Append base vocab (honorifics, elements, etc.) to static_system so it is + # cached with the system prompt and never re-evaluated per batch. + if base_vocab_text: + static_system = static_system.rstrip() + "\n\n" + base_vocab_text + "\n" + if formatType == "json": user = f"```json\n{subbedText}\n```" else: