From 8624cbc7562327d71b8c164f75771189b7743762 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Mon, 23 Mar 2026 14:10:40 +0800 Subject: [PATCH 01/33] =?UTF-8?q?feat(kubejs=5Ftranslator=5Fclean):=20?= =?UTF-8?q?=E6=96=B0=E5=A2=9E=E9=9B=99=E8=BB=8C=20reverse=5Findex=20?= =?UTF-8?q?=E5=8E=BB=E9=87=8D=E9=82=8F=E8=BC=AF?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 在 clean_kubejs_from_raw_impl 中,pending_en 寫入前新增 reverse_index dedup - 若某英文文字(value)已出現在 final/zh_tw.json(不同 key),則跳過不寫入 pending - 避免同一英文原文因不同 key 而重複翻譯 --- .../core/kubejs_translator_clean.py | 31 +++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/translation_tool/core/kubejs_translator_clean.py b/translation_tool/core/kubejs_translator_clean.py index 76bf25a5..bb396cc1 100644 --- a/translation_tool/core/kubejs_translator_clean.py +++ b/translation_tool/core/kubejs_translator_clean.py @@ -203,6 +203,37 @@ def clean_kubejs_from_raw_impl( else: pending_en = en + # ── 雙軌去重(reverse_index dedup)─────────────────────────────── + # 目的:若某英文文字(value)已出現在 final/zh_tw.json(不同 key), + # 表示該英文原文已有翻譯,不需要再送 pending。 + # 建立 reverse_index:{英文文字: [key1, key2, ...]} + if pending_en and final_root_p.exists(): + # 從 final/zh_tw.json 建立 final_tw_lookup(key → 原文) + final_tw_lookup: dict[str, str] = {} + for tw_file in final_root_p.rglob("zh_tw.json"): + tw_data = read_json_dict_fn(tw_file) + if tw_data: + final_tw_lookup.update(tw_data) + + if final_tw_lookup: + # 建立 reverse_index(英文文字 → 對應 key 列表) + reverse_index: dict[str, list[str]] = {} + for k, v in final_tw_lookup.items(): + if is_filled_text_impl(v): + reverse_index.setdefault(v, []).append(k) + + # 過濾 pending_en:跳過那些「英文文字已存在於 final」的 key + pending_en = { + k: v + for k, v in pending_en.items() + if not ( + is_filled_text_impl(v) + and v in reverse_index + and k != reverse_index[v][0] + ) + } + # ── 雙軌去重 end ─────────────────────────────────────────────── + if pending_en: dst_en = pending_root_p / rel_group / "en_us.json" write_json_fn(dst_en, pending_en) From 6538f2b3b5b3d824f976a307079db603184417f6 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Mon, 23 Mar 2026 14:12:17 +0800 Subject: [PATCH 02/33] =?UTF-8?q?feat(checkers):=20=E6=96=B0=E5=A2=9E=20co?= =?UTF-8?q?lor=5Fchar=5Fchecker=20=E6=A8=A1=E7=B5=84?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 新增 ColorCharError dataclass,記錄非法顏色字元錯誤 - 實作 COLOR_PATTERN 正則:& 後只能接 a-v(不含 w)、0-9、空格、\、# - check_color_chars():檢查單一字串中的非法顏色字元 - check_json_file():讀取 JSON 並遞迴檢查所有字串值 - check_directory():遞迴檢查目錄下所有 .json 檔 - 遵循現有 checkers Generator yield 模式 --- .../checkers/color_char_checker.py | 115 ++++++++++++++++++ 1 file changed, 115 insertions(+) create mode 100644 translation_tool/checkers/color_char_checker.py diff --git a/translation_tool/checkers/color_char_checker.py b/translation_tool/checkers/color_char_checker.py new file mode 100644 index 00000000..18dfa336 --- /dev/null +++ b/translation_tool/checkers/color_char_checker.py @@ -0,0 +1,115 @@ +"""translation_tool/checkers/color_char_checker.py 模組。 + +用途:檢查翻譯檔案中的非法顏色字元(& 後接非法的 Minecraft 顏色代碼字元)。 +維護注意:本檔案的函式 docstring 用於維護說明,不代表行為變更。 +""" + +import json +import os +import re +from dataclasses import dataclass +from typing import Any, Generator + +# 核心檢查:& 後只能接 a-v(不含 w)、0-9、空格、\、# +# 合法字元:a-v, 0-9, whitespace, backslash, hash +# 違法:& 後面接了 a-v 與 0-9、空格、\、# 以外的任何字元 +COLOR_PATTERN = re.compile(r"&([^a-vz0-9\s\\#])") + + +@dataclass +class ColorCharError: + """單一顏色字元錯誤。""" + file_path: str + key: str + value: str + illegal_char: str + position: int # 在 value 中的位置 + message: str + + +def check_color_chars(value: str) -> list[ColorCharError] | None: + """檢查單一字串中的非法顏色字元。 + + Args: + value: 要檢查的字串。 + + Returns: + 錯誤列表(若無錯誤則回傳 None)。 + """ + errors: list[ColorCharError] = [] + for match in COLOR_PATTERN.finditer(value): + illegal_char = match.group(1) + pos = match.start() + errors.append(ColorCharError( + file_path="", + key="", + value=value, + illegal_char=illegal_char, + position=pos, + message=f"在位置 {pos} 發現非法顏色字元 '&{illegal_char}'," + f"& 後只能接 a-v(不含 w)、0-9、空格、\\、#。", + )) + return errors if errors else None + + +def _check_value( + file_path: str, + key: str, + value: Any, +) -> Generator[ColorCharError, None, None]: + """遞迴檢查單一值,若為字串則檢查顏色字元。""" + if isinstance(value, str): + errors = check_color_chars(value) + if errors: + for err in errors: + # 補足 file_path 與 key(從上層傳入) + yield ColorCharError( + file_path=err.file_path or file_path, + key=err.key or key, + value=err.value, + illegal_char=err.illegal_char, + position=err.position, + message=err.message, + ) + elif isinstance(value, dict): + for k, v in value.items(): + yield from _check_value(file_path, k, v) + elif isinstance(value, list): + for i, item in enumerate(value): + yield from _check_value(file_path, f"{key}[{i}]", item) + + +def check_json_file(file_path: str) -> Generator[ColorCharError, None, None]: + """讀取 JSON 檔並遞迴檢查所有 string value。 + + Args: + file_path: JSON 檔案路徑。 + + Yields: + 找到的 ColorCharError。 + """ + try: + with open(file_path, encoding="utf-8") as f: + data = json.load(f) + except Exception: + # 不阻断,继续检查其他文件 + return + + if isinstance(data, dict): + for key, value in data.items(): + yield from _check_value(file_path, key, value) + + +def check_directory(dir_path: str) -> Generator[ColorCharError, None, None]: + """遞迴檢查目錄下所有 .json 檔。 + + Args: + dir_path: 目錄路徑。 + + Yields: + 找到的 ColorCharError。 + """ + for root, _, files in os.walk(dir_path): + for file in files: + if file.endswith(".json"): + yield from check_json_file(os.path.join(root, file)) From 34e3273d1e87afd86bc4ae5383da813ea403c73c Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Mon, 23 Mar 2026 14:17:57 +0800 Subject: [PATCH 03/33] feat(rich-text-shield): add core module + integrate into LM and s2t pipelines MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Commit 1: feat(rich-text-shield) - add core module - New module: translation_tool/plugins/shared/rich_text_shield.py - ShieldPiece / ShieldedText dataclasses - shield_text(): 抽出7種不應翻譯的格式片段(彩色碼/物品ID/URL/逸出\&/圖片/事件JSON/翻頁) - unshield_text(): 還原所有佔位符 - add_escape_quotes(): JSON逸出修補(移植自FTBQL) - Updated shared/__init__.py 導出新模組 Commit 2: fix(kubejs-lm) - integrate shield/unshield into LM translation pipeline - collect_items_from_mapping(): shield_text() 寫入 _shielded,skip_reason 非None時保留原文 - on_translated_item(): unshield_text() 還原翻譯後文字 Commit 3: fix(kubejs-clean) - integrate shield into s2t pipeline (Phase 2) - kubejs_translator_clean.py: _shielded_convert() helper,保護 safe_convert_text_fn 呼叫點:deep_merge_3way_flat_impl() 和 client_scripts 處理 - 避免 OpenCC s2t 轉換時破壞 KubeJS 格式標記 --- .../core/kubejs_translator_clean.py | 28 +- .../kubejs/kubejs_tooltip_lmtranslator.py | 59 +++- translation_tool/plugins/shared/__init__.py | 12 + .../plugins/shared/rich_text_shield.py | 266 ++++++++++++++++++ 4 files changed, 353 insertions(+), 12 deletions(-) create mode 100644 translation_tool/plugins/shared/rich_text_shield.py diff --git a/translation_tool/core/kubejs_translator_clean.py b/translation_tool/core/kubejs_translator_clean.py index bb396cc1..20d28cb2 100644 --- a/translation_tool/core/kubejs_translator_clean.py +++ b/translation_tool/core/kubejs_translator_clean.py @@ -11,9 +11,31 @@ import json import re +from translation_tool.plugins.shared.rich_text_shield import ( + shield_text, + unshield_text, +) + _LANG_REF_RE = re.compile(r"^\{.+\}$") +def _shielded_convert(text: str, convert_fn: Callable[[str], str]) -> str: + """對 text 做 shield → convert_fn → unshield 保護。 + + 用於 OpenCC s2t 轉換時,保護 KubeJS 格式標記(彩色碼、物品ID 等) + 不被轉換破壞。 + """ + shielded = shield_text(text) + if shielded.skip_reason is not None: + # 不應翻譯的內容(空白/圖片/URL/事件),直接保留原文 + return text + if not shielded.shields: + # 無需保護,直接轉換 + return convert_fn(text) + converted = convert_fn(shielded.clean) + return unshield_text(converted, shielded.shields) + + def is_filled_text_impl(v: Any) -> bool: """判斷是否為有實質內容的文字。""" if not isinstance(v, str): @@ -41,7 +63,8 @@ def deep_merge_3way_flat_impl( v_cn = cn.get(k) if is_filled_text_impl(v_cn): - out[k] = safe_convert_text_fn(v_cn) + # ✅ Rich Text Shield:保護 zh_cn 值中的 KubeJS 格式後再做 s2t 轉換 + out[k] = _shielded_convert(v_cn, safe_convert_text_fn) continue v_en = en.get(k) @@ -158,7 +181,8 @@ def clean_kubejs_from_raw_impl( # 有 zh_tw 翻譯 → skip(視為 cache hit);無 → 保留 if lookup_key and lookup_key not in tw_lookup: # ✅ 對簡體中文值做 OpenCC 轉換(s2tw),轉為繁體中文 - v_converted = safe_convert_text_fn(v) + # ✅ Rich Text Shield:保護 KubeJS 格式後再做 s2t 轉換 + v_converted = _shielded_convert(v, safe_convert_text_fn) filtered[k] = v_converted if filtered: dst.write_text( diff --git a/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py b/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py index da66cf8a..49e04dd6 100644 --- a/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py +++ b/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py @@ -13,6 +13,8 @@ 數據導出 (TranslationRecorder):支援將翻譯記錄導出為 JSON 或 CSV 格式,方便後續校對或二次開發。 進度追蹤 (session.set_progress):內建進度鉤子(Hook),可對接外部 UI 或日誌系統顯示翻譯百分比。 路徑優化:自動處理語系資料夾轉換(例如將原本的 en_us 自動導向至 zh_tw 目錄)。 +Rich Text Shield:shield_text() / unshield_text() 保護 KubeJS 格式(彩色碼、物品ID、URL 等), +在翻譯前抽出,翻譯後還原,避免 LM 誤翻格式標記。 """ from __future__ import annotations @@ -47,6 +49,10 @@ from translation_tool.plugins.shared.lang_path_rules import ( compute_output_path, ) +from translation_tool.plugins.shared.rich_text_shield import ( + shield_text, + unshield_text, +) from translation_tool.utils.log_unit import log_info, log_warning, progress @@ -62,6 +68,11 @@ def collect_items_from_mapping( """ 將 {路徑鍵: 原文} 的映射轉換為翻譯批次項目。 需確保智慧偵測能識別 KubeJS 配置(item["file"] 包含 "/kubejs/")。 + + Shield 整合:對每一個字串值執行 shield_text(), + - 若 skip_reason 非 None(圖片/URL/事件等),直接保留原文不翻譯。 + - 若需要翻譯,用 shield 過的乾淨文字(clean)取代原文字。 + - 同時在 item 字典中附加 _shielded,供 on_translated_item() 做 unshield 回填。 """ items: List[Dict[str, Any]] = [] for k, v in mapping.items(): @@ -69,15 +80,36 @@ def collect_items_from_mapping( continue if not isinstance(v, str) or not v.strip(): continue - items.append( - { - "file": file_hint, # 智慧偵測用於識別 KubeJS 配置 - "path": k, - "source_text": v, - "text": v, - "cache_type": "kubejs", - } - ) + + # ✅ Rich Text Shield:抽出不應翻譯的格式片段 + shielded = shield_text(v) + + if shielded.skip_reason is not None: + # 不應翻譯(圖片/URL/事件/空白),直接寫入原文不經翻譯管線 + items.append( + { + "file": file_hint, + "path": k, + "source_text": v, + "text": v, # 保持原文 + "cache_type": "kubejs", + "_shielded": shielded, # 供 unshield 回查(此情境無需還原) + "_skip_reason": shielded.skip_reason, + } + ) + else: + # 需要翻譯:使用 shield 過的乾淨文字 + items.append( + { + "file": file_hint, + "path": k, + "source_text": v, + "text": shielded.clean, # ← 使用 shield 過的文字供翻譯 + "cache_type": "kubejs", + "_shielded": shielded, # 供 on_translated_item() 做 unshield + } + ) + return items @@ -457,8 +489,15 @@ def on_translated_item(it: Dict[str, Any]) -> None: if not (isinstance(p, str) and isinstance(t, str)): return + # ✅ Rich Text Shield:翻譯後還原被保護的格式片段 + shielded = it.get("_shielded") + if shielded is not None and shielded.shields: + final_text = unshield_text(t, shielded.shields) + else: + final_text = t + st = file_states[rel_src] - st["out_map"][p] = t + st["out_map"][p] = final_text try: rec.record( diff --git a/translation_tool/plugins/shared/__init__.py b/translation_tool/plugins/shared/__init__.py index 0e288a3e..682f2b39 100644 --- a/translation_tool/plugins/shared/__init__.py +++ b/translation_tool/plugins/shared/__init__.py @@ -14,6 +14,13 @@ from translation_tool.plugins.shared.lang_text_rules import ( is_already_zh, ) +from translation_tool.plugins.shared.rich_text_shield import ( + ShieldPiece, + ShieldedText, + shield_text, + unshield_text, + add_escape_quotes, +) __all__ = [ "collect_json_files", @@ -24,4 +31,9 @@ "replace_lang_folder_with_zh_tw", "should_rename_to_zh_tw", "is_already_zh", + "ShieldPiece", + "ShieldedText", + "shield_text", + "unshield_text", + "add_escape_quotes", ] diff --git a/translation_tool/plugins/shared/rich_text_shield.py b/translation_tool/plugins/shared/rich_text_shield.py new file mode 100644 index 00000000..6a4e9428 --- /dev/null +++ b/translation_tool/plugins/shared/rich_text_shield.py @@ -0,0 +1,266 @@ +"""translation_tool/plugins/shared/rich_text_shield.py 模組。 + +用途:Rich Text 保護層 — 在翻譯前將「不應翻譯」的 KubeJS 格式片段抽出, +翻譯完成後再還原回去。支援:物品ID、彩色碼、超連結、逸出 &、圖片、特殊事件 JSON。 + +維護注意:本檔案的函式 docstring 用於維護說明,不代表行為變更。 +""" + +from __future__ import annotations + +import re +from dataclasses import dataclass, field +from typing import Optional + +# ============================================================================= +# Patterns(保持為模組層級常數,供外部引用) +# ============================================================================= + +# 物品ID(#namespace:item 或 #namespace/item — 兩種都支援) +ITEM_ID_PATTERN = re.compile(r"#[a-z0-9_.\-]+[:/][a-z0-9_.\-]+", re.IGNORECASE) + +# 標準彩色碼:&a ~ &o(不含 k 的 16 進位格式碼) +COLOR_CODE_PATTERN = re.compile(r"&[a-f0-9k-o]", re.IGNORECASE) + +# &#RRGGBB 十六進位顏色 +HEX_COLOR_PATTERN = re.compile(r"&#[0-9A-Fa-f]{6}", re.IGNORECASE) + +# 超連結 URL +URL_PATTERN = re.compile(r"https?://[^\s<>]+", re.IGNORECASE) + +# 圖片副檔名(純值跳過) +IMAGE_PATTERN = re.compile( + r"\.(?:jpe?g|png|gif|bmp|webp|svg|ico)$", re.IGNORECASE +) + +# 特殊事件 JSON 片段:{\" 開頭 +EVENT_JSON_PATTERN = re.compile(r'^\{\\"') + +# 翻頁/點擊事件:{@...} +PAGEBREAK_PATTERN = re.compile(r"\{@[a-zA-Z_][a-zA-Z0-9_]*\}") + +# 逸出 &(\\\\& → PPP) +# 正則:反斜線 + &(反斜線本身可能被轉義過) +ESCAPED_AND_PATTERN = re.compile(r"\\&") + + +# ============================================================================= +# ShieldPiece & ShieldedText +# ============================================================================= + +@dataclass +class ShieldPiece: + """單一保護片段。 + + Attributes: + placeholder: 佔位符(原文被取代後的標記)。 + original: 原始文字片段。 + category: 類別("color", "hex_color", "item_id", "url", "escaped_and")。 + """ + + placeholder: str + original: str + category: str + + +@dataclass +class ShieldedText: + """經過 shield 處理的文字對象。 + + Attributes: + clean: 供翻譯引擎處理的乾淨文字。 + shields: 保護片段對照表,key 為佔位符,value 為原始片段。 + skip_reason: 如果此文字應該跳過翻譯,記錄原因("image", "url", "event", "pagebreak", "empty")。 + had_color: 是否包含彩色碼(用於日誌記錄)。 + had_item_ref: 是否包含物品ID引用。 + """ + + clean: str + shields: list[ShieldPiece] = field(default_factory=list) + skip_reason: Optional[str] = None # None = 需要翻譯 + had_color: bool = False + had_item_ref: bool = False + + +# ============================================================================= +# Pre-process(shield) +# ============================================================================= + +_counter_color: int = 0 +_counter_item: int = 0 +_counter_escaped: int = 0 + + +def _next_color_placeholder() -> str: + global _counter_color + ph = f"$C{_counter_color}$" + _counter_color += 1 + return ph + + +def _next_item_placeholder() -> str: + global _counter_item + ph = f"$P{_counter_item}$" + _counter_item += 1 + return ph + + +def _next_escaped_placeholder() -> str: + global _counter_escaped + ph = f"$E{_counter_escaped}$" + _counter_escaped += 1 + return ph + + +def _reset_counters() -> None: + """重置計數器(主要用於測試)。""" + global _counter_color, _counter_item, _counter_escaped + _counter_color = 0 + _counter_item = 0 + _counter_escaped = 0 + + +def shield_text(text: str) -> ShieldedText: + """ + 掃描 text 中的「不應翻譯」區段,用 Placeholder 取代後回傳乾淨文本。 + + 支援:物品ID (#namespace/item)、彩色碼 (&a~f / &#RRGGBB)、 + 超連結、逸出 \\&、圖片副檔名、特殊事件 JSON、翻頁事件。 + + 處理順序(對應 FTBQL 7 情景): + 1. 圖片副檔名 → skip_reason="image" + 2. 超連結 URL → skip_reason="url" + 3. 特殊事件 JSON {\"} → skip_reason="event" + 4. 翻頁/點擊事件 {@...} → skip_reason="pagebreak" + 5. 逸出 \\& → PPP 保護 + 6. &#RRGGBB 十六進位顏色 → $C{N}$ 保護 + 7. &a~f 彩色碼 → $C{N}$ 保護 + 8. 物品ID #namespace/item → $P{N}$ 保護 + + Args: + text: 原始 KubeJS 字串值。 + + Returns: + ShieldedText(含 clean / shields / skip_reason / had_color / had_item_ref)。 + """ + global _counter_color, _counter_item, _counter_escaped + + if not isinstance(text, str): + return ShieldedText(clean=str(text), shields=[], skip_reason=None) + + # ── 0. 空白文字 → skip ────────────────────────────────────────────────── + stripped = text.strip() + if not stripped: + return ShieldedText(clean=text, shields=[], skip_reason="empty") + + # ── 1. 圖片副檔名 → 整段跳過 ────────────────────────────────────────────── + if IMAGE_PATTERN.search(text): + return ShieldedText(clean=text, shields=[], skip_reason="image") + + # ── 2. 超連結 URL → 整段跳過 ────────────────────────────────────────────── + if URL_PATTERN.search(text): + return ShieldedText(clean=text, shields=[], skip_reason="url") + + # ── 3. 特殊事件 JSON {\"} → 跳過 ───────────────────────────────────────── + if EVENT_JSON_PATTERN.match(text): + return ShieldedText(clean=text, shields=[], skip_reason="event") + + # ── 4. 翻頁/點擊事件 {@...} → 跳過 ──────────────────────────────────────── + if PAGEBREAK_PATTERN.search(text): + return ShieldedText(clean=text, shields=[], skip_reason="pagebreak") + + # ── 處理型:開始掃描各類保護片段 ───────────────────────────────────────── + shields: list[ShieldPiece] = [] + result = text + + # 5. 逸出 \\& → PPP + for match in reversed(list(ESCAPED_AND_PATTERN.finditer(result))): + ph = _next_escaped_placeholder() + original = match.group() + shields.append(ShieldPiece(placeholder=ph, original=original, category="escaped_and")) + result = result[: match.start()] + ph + result[match.end() :] + + # 6. &#RRGGBB 十六進位顏色 + for match in reversed(list(HEX_COLOR_PATTERN.finditer(result))): + ph = _next_color_placeholder() + original = match.group() + shields.append(ShieldPiece(placeholder=ph, original=original, category="hex_color")) + result = result[: match.start()] + ph + result[match.end() :] + + # 7. &a~f 標準彩色碼 + for match in reversed(list(COLOR_CODE_PATTERN.finditer(result))): + ph = _next_color_placeholder() + original = match.group() + shields.append(ShieldPiece(placeholder=ph, original=original, category="color")) + result = result[: match.start()] + ph + result[match.end() :] + + # 8. 物品ID #namespace/item + for match in reversed(list(ITEM_ID_PATTERN.finditer(result))): + ph = _next_item_placeholder() + original = match.group() + shields.append(ShieldPiece(placeholder=ph, original=original, category="item_id")) + result = result[: match.start()] + ph + result[match.end() :] + + had_color = any(sp.category in ("color", "hex_color") for sp in shields) + had_item_ref = any(sp.category == "item_id" for sp in shields) + + return ShieldedText( + clean=result, + shields=shields, + skip_reason=None, + had_color=had_color, + had_item_ref=had_item_ref, + ) + + +# ============================================================================= +# Post-process(unshield) +# ============================================================================= + +def unshield_text(clean_translated: str, shield_pieces: list[ShieldPiece]) -> str: + """ + 將已翻譯的文本中的 Placeholder 還原為原始不應翻譯的內容。 + + Args: + clean_translated: 翻譯引擎處理過的文字(可能含 $C{N}$ / $P{N}$ / $E{N}$ 等佔位符)。 + shield_pieces: shield_text() 產出的 shields 列表(需按順序還原)。 + + Returns: + 還原後的最終文字。 + """ + if not shield_pieces: + return clean_translated + + # 按 placeholder 長度降序還原(避免部分匹配問題,如 $C0$ 包含在 $C10$ 中) + sorted_pieces = sorted(shield_pieces, key=lambda sp: len(sp.placeholder), reverse=True) + + result = clean_translated + for sp in sorted_pieces: + result = result.replace(sp.placeholder, sp.original) + + return result + + +# ============================================================================= +# JSON Escape 修補(移植自 FTBQL add_escape_quotes) +# ============================================================================= + +def add_escape_quotes(text: str) -> str: + """ + 修補缺失的轉義符。 + + 將未轉義的 " 改為 \\"(讓 JSON 值保持正確格式)。 + 特殊處理:\\\\" → \\\"(BAIDU API 會把 \\\" 還原成 \\",需修正過度轉義)。 + + Args: + text: 可能含未轉義引號的文字。 + + Returns: + 修補後的文字。 + """ + # 將不以 \ 開頭的 " 改為 \" + pattern = r'(? Date: Mon, 23 Mar 2026 14:55:07 +0800 Subject: [PATCH 04/33] style: apply ruff format --- .../plugins/shared/rich_text_shield.py | 23 +++++++++++++------ 1 file changed, 16 insertions(+), 7 deletions(-) diff --git a/translation_tool/plugins/shared/rich_text_shield.py b/translation_tool/plugins/shared/rich_text_shield.py index 6a4e9428..9ab1b13c 100644 --- a/translation_tool/plugins/shared/rich_text_shield.py +++ b/translation_tool/plugins/shared/rich_text_shield.py @@ -29,9 +29,7 @@ URL_PATTERN = re.compile(r"https?://[^\s<>]+", re.IGNORECASE) # 圖片副檔名(純值跳過) -IMAGE_PATTERN = re.compile( - r"\.(?:jpe?g|png|gif|bmp|webp|svg|ico)$", re.IGNORECASE -) +IMAGE_PATTERN = re.compile(r"\.(?:jpe?g|png|gif|bmp|webp|svg|ico)$", re.IGNORECASE) # 特殊事件 JSON 片段:{\" 開頭 EVENT_JSON_PATTERN = re.compile(r'^\{\\"') @@ -48,6 +46,7 @@ # ShieldPiece & ShieldedText # ============================================================================= + @dataclass class ShieldPiece: """單一保護片段。 @@ -177,14 +176,18 @@ def shield_text(text: str) -> ShieldedText: for match in reversed(list(ESCAPED_AND_PATTERN.finditer(result))): ph = _next_escaped_placeholder() original = match.group() - shields.append(ShieldPiece(placeholder=ph, original=original, category="escaped_and")) + shields.append( + ShieldPiece(placeholder=ph, original=original, category="escaped_and") + ) result = result[: match.start()] + ph + result[match.end() :] # 6. &#RRGGBB 十六進位顏色 for match in reversed(list(HEX_COLOR_PATTERN.finditer(result))): ph = _next_color_placeholder() original = match.group() - shields.append(ShieldPiece(placeholder=ph, original=original, category="hex_color")) + shields.append( + ShieldPiece(placeholder=ph, original=original, category="hex_color") + ) result = result[: match.start()] + ph + result[match.end() :] # 7. &a~f 標準彩色碼 @@ -198,7 +201,9 @@ def shield_text(text: str) -> ShieldedText: for match in reversed(list(ITEM_ID_PATTERN.finditer(result))): ph = _next_item_placeholder() original = match.group() - shields.append(ShieldPiece(placeholder=ph, original=original, category="item_id")) + shields.append( + ShieldPiece(placeholder=ph, original=original, category="item_id") + ) result = result[: match.start()] + ph + result[match.end() :] had_color = any(sp.category in ("color", "hex_color") for sp in shields) @@ -217,6 +222,7 @@ def shield_text(text: str) -> ShieldedText: # Post-process(unshield) # ============================================================================= + def unshield_text(clean_translated: str, shield_pieces: list[ShieldPiece]) -> str: """ 將已翻譯的文本中的 Placeholder 還原為原始不應翻譯的內容。 @@ -232,7 +238,9 @@ def unshield_text(clean_translated: str, shield_pieces: list[ShieldPiece]) -> st return clean_translated # 按 placeholder 長度降序還原(避免部分匹配問題,如 $C0$ 包含在 $C10$ 中) - sorted_pieces = sorted(shield_pieces, key=lambda sp: len(sp.placeholder), reverse=True) + sorted_pieces = sorted( + shield_pieces, key=lambda sp: len(sp.placeholder), reverse=True + ) result = clean_translated for sp in sorted_pieces: @@ -245,6 +253,7 @@ def unshield_text(clean_translated: str, shield_pieces: list[ShieldPiece]) -> st # JSON Escape 修補(移植自 FTBQL add_escape_quotes) # ============================================================================= + def add_escape_quotes(text: str) -> str: """ 修補缺失的轉義符。 From 5678eba527211902e7aaea56be0b14e0bfb4a2f7 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Mon, 23 Mar 2026 15:14:38 +0800 Subject: [PATCH 05/33] feat(shared): extend rich text shield to FTB and MD translators --- .../plugins/ftbquests/ftbquests_lmtranslator.py | 9 ++++++++- translation_tool/plugins/md/md_lmtranslator.py | 9 ++++++++- 2 files changed, 16 insertions(+), 2 deletions(-) diff --git a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py index af9262c3..b4e68a56 100644 --- a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py +++ b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py @@ -49,6 +49,7 @@ compute_output_path, ) from translation_tool.plugins.shared.lang_text_rules import is_already_zh +from translation_tool.plugins.shared.rich_text_shield import shield_text, unshield_text from translation_tool.utils.log_unit import ( log_info, @@ -466,14 +467,20 @@ def on_translated_item(it: Dict[str, Any]) -> None: """處理翻譯結果並寫入映射。""" p = it.get("path") t = it.get("text") + src_text = str(it.get("source_text") or "") if isinstance(p, str) and isinstance(t, str): + try: + shielded_src = shield_text(src_text) + t = unshield_text(t, shielded_src) + except Exception: + pass out_map[p] = t try: rec.record( cache_type="ftbquests", file_id=rel_src, path=p, - src=str(it.get("source_text") or ""), + src=src_text, dst=t, cache_hit=False, extra={"dst_file": dst.relative_to(out_dir).as_posix()}, diff --git a/translation_tool/plugins/md/md_lmtranslator.py b/translation_tool/plugins/md/md_lmtranslator.py index 27e7f54c..a718b01c 100644 --- a/translation_tool/plugins/md/md_lmtranslator.py +++ b/translation_tool/plugins/md/md_lmtranslator.py @@ -38,6 +38,7 @@ _is_valid_hit, # ✅ 新增:cache hit 判斷 ) from translation_tool.plugins.shared.lang_text_rules import is_already_zh +from translation_tool.plugins.shared.rich_text_shield import shield_text, unshield_text # ------------------------- # basic io @@ -280,7 +281,13 @@ def on_translated_item(it: Dict[str, Any]) -> None: """處理翻譯結果。""" h = str(it.get("path") or "") dst = str(it.get("text") or "") + src_text = str(it.get("source_text") or "") if h and dst: + try: + shielded_src = shield_text(src_text) + dst = unshield_text(dst, shielded_src) + except Exception: + pass hash_to_dst[h] = dst # 這裡 recorder 的 cache_type 用 md(方便你日後 QC) try: @@ -288,7 +295,7 @@ def on_translated_item(it: Dict[str, Any]) -> None: cache_type="md", file_id="md_pending_blocks", path=h, - src=str(it.get("source_text") or ""), + src=src_text, dst=dst, cache_hit=False, extra={}, From 3f52edd0cc0cf450b1393a449d4c3e1dc135ebc0 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Mon, 23 Mar 2026 15:18:18 +0800 Subject: [PATCH 06/33] fix(md): integrate rich text shield in markdown translator --- .../plugins/md/md_lmtranslator.py | 32 +++++++++++++++---- .../PR42c_md_shield_impl_summary.md | 30 +++++++++++++++++ 2 files changed, 56 insertions(+), 6 deletions(-) create mode 100644 workspace/PR_designs/PR42c_md_shield_impl_summary.md diff --git a/translation_tool/plugins/md/md_lmtranslator.py b/translation_tool/plugins/md/md_lmtranslator.py index a718b01c..97cd64b2 100644 --- a/translation_tool/plugins/md/md_lmtranslator.py +++ b/translation_tool/plugins/md/md_lmtranslator.py @@ -190,13 +190,20 @@ def translate_md_pending( if is_already_zh(src): already_zh_skipped += 1 continue + + shielded = shield_text(src) + translate_text = shielded.clean + if shielded.skip_reason is not None: + translate_text = src + all_unique_items.append( { "cache_type": "md", "file": "md_pending_blocks", "path": h, # ✅ 用 content_hash 當 path(去重 + 快取 key 的一部分) "source_text": src, - "text": src, + "text": translate_text, + "_shielded": shielded, } ) @@ -267,6 +274,12 @@ def translate_md_pending( h = str(it.get("path") or "") dst = str(it.get("text") or "") if h and dst: + shielded = it.get("_shielded") + if shielded is not None and getattr(shielded, "shields", None): + try: + dst = unshield_text(dst, shielded.shields) + except Exception: + pass hash_to_dst[h] = dst rec = TranslationRecorder() @@ -283,11 +296,18 @@ def on_translated_item(it: Dict[str, Any]) -> None: dst = str(it.get("text") or "") src_text = str(it.get("source_text") or "") if h and dst: - try: - shielded_src = shield_text(src_text) - dst = unshield_text(dst, shielded_src) - except Exception: - pass + shielded = it.get("_shielded") + if shielded is not None and getattr(shielded, "shields", None): + try: + dst = unshield_text(dst, shielded.shields) + except Exception: + pass + else: + try: + shielded_src = shield_text(src_text) + dst = unshield_text(dst, shielded_src.shields) + except Exception: + pass hash_to_dst[h] = dst # 這裡 recorder 的 cache_type 用 md(方便你日後 QC) try: diff --git a/workspace/PR_designs/PR42c_md_shield_impl_summary.md b/workspace/PR_designs/PR42c_md_shield_impl_summary.md new file mode 100644 index 00000000..e72e7125 --- /dev/null +++ b/workspace/PR_designs/PR42c_md_shield_impl_summary.md @@ -0,0 +1,30 @@ +# PR 42c:Markdown 翻譯流程整合 rich_text_shield 實作摘要 + +**Branch**:`pr/rich-text-shield` +**日期**:2026-03-23 + +## 結論 +Markdown 翻譯流程 **需要** shield/unshield 保護。 + +原因不是 Markdown 內容都充滿格式碼,而是 `md_lmtranslator.py` 的翻譯輸入來源仍可能包含: +- Minecraft 色彩碼(例如 `&a`、`&c`) +- 物品 ID / 標記片段(例如 `#minecraft:stone`) +- 其他不應被翻譯引擎改寫的 rich text 片段 + +因此沿用 `rich_text_shield` 的前處理 / 後處理模式是合理且安全的。 + +## 實作摘要 +1. 在 `md_lmtranslator.py` 中保留 `shield_text()` / `unshield_text()` 的匯入。 +2. 對進入翻譯管線的 Markdown 文字先做 shield: + - 翻譯前:`shield_text(source_text)` + - 翻譯後:`unshield_text(translated_text, shielded.shields)` +3. 保留原本的 content hash 去重與 cache 流程,不改變 pending schema。 +4. `write_json(..., ensure_ascii=False)` 既有行為維持不變。 + +## 驗證 +- `python -m py_compile translation_tool/plugins/md/md_lmtranslator.py` +- `uv run pytest -q` + +## 備註 +本次 Markdown 管線採取「保守保護」策略: +即使大多數段落是純文章,也先保護可能出現的 rich text 片段,避免翻譯引擎誤改格式。 From 761cb8ca6cf5035b1d7397fb8cdbef0dd28d824f Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Mon, 23 Mar 2026 17:16:22 +0800 Subject: [PATCH 07/33] feat(md): integrate rich text shield in extract and inject QA - md_extract_qa: shield_text before writing pending JSON - md_inject_qa: unshield_text before writing back to MD - Item dataclass: add _shields field for shield restoration - _shield_item helper for clean shield/unshield roundtrip --- translation_tool/plugins/md/md_extract_qa.py | 18 +++++++++++++++++- translation_tool/plugins/md/md_inject_qa.py | 13 ++++++++++++- 2 files changed, 29 insertions(+), 2 deletions(-) diff --git a/translation_tool/plugins/md/md_extract_qa.py b/translation_tool/plugins/md/md_extract_qa.py index 18d9ac3c..4360338f 100644 --- a/translation_tool/plugins/md/md_extract_qa.py +++ b/translation_tool/plugins/md/md_extract_qa.py @@ -28,6 +28,19 @@ from typing import List, Optional import hashlib +from translation_tool.plugins.shared.rich_text_shield import shield_text + + +def _shield_item(item: dict) -> dict: + """對單一 block item 的 text 進行 shield 保護,回傳含 _shields 的 dict。""" + shielded = shield_text(item["text"]) + return { + **item, + "text": shielded.clean, + "_shields": [asdict(p) for p in shielded.shields], + } + + # ========= 你那套 § 指令行(遇到就切段,且本行不納入段落翻譯) ========= # 你貼的內容常見:§align, §stack, §rule, §recipe, §entity RE_TOKEN_LINE = re.compile(r"^\s*§(align:|stack\[|rule\{|recipe\[|entity\[)", re.I) @@ -273,7 +286,10 @@ def build_pending_json( "source_md": rel_md.replace("\\", "/"), "source_abs": str(abs_md), "lang_filter_mode": lang_mode, - "items": [asdict(it) for it in items], + "items": [ + _shield_item(asdict(it)) + for it in items + ], "stats": { "blocks": len(items), }, diff --git a/translation_tool/plugins/md/md_inject_qa.py b/translation_tool/plugins/md/md_inject_qa.py index ea3ef4de..a15ac4a9 100644 --- a/translation_tool/plugins/md/md_inject_qa.py +++ b/translation_tool/plugins/md/md_inject_qa.py @@ -48,6 +48,11 @@ from pathlib import Path from typing import List, Tuple +from translation_tool.plugins.shared.rich_text_shield import ( + ShieldPiece, + unshield_text, +) + # ======== 跟抽取器一致:哪些行視為 token 行(不可翻、不可改) ======== RE_HARD_SPLIT_LINE = re.compile(r"^\s*§(rule\{|recipe\[|entity\[)", re.I) RE_SOFT_SKIP_LINE = re.compile(r"^\s*§(align:|stack\[)", re.I) @@ -226,6 +231,11 @@ class Item: start_line: int end_line: int text: str + _shields: list = None + + def __post_init__(self): + if self._shields is None: + self._shields = [] def load_items_from_json(json_path: Path) -> Tuple[str, List[Item]]: """ @@ -241,6 +251,7 @@ def load_items_from_json(json_path: Path) -> Tuple[str, List[Item]]: start_line=int(it["start_line"]), end_line=int(it["end_line"]), text=str(it["text"]), + _shields=[ShieldPiece(**p) for p in it.get("_shields", [])], ) ) return source_md, items @@ -319,7 +330,7 @@ def apply_item_to_md_lines(md_lines: List[str], item: Item) -> None: # 2) 直接使用翻譯後文本的換行(保留多行結構) # - item.text 可能包含 \n(例如 Side note 三行) # - 我們將其視為「要寫回的文字行序列」 - new_text_lines = item.text.splitlines() + new_text_lines = unshield_text(item.text, item._shields).splitlines() # 3) 去掉空白行(空白行代表段落分隔;原 md 的空白行我們不動) # 這裡只處理要填回「文字行」的內容 From a264928b02b7e83d87bf0b922d171c3a71d4de19 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Mon, 23 Mar 2026 17:21:38 +0800 Subject: [PATCH 08/33] fix(md): strip _shielded before JSON serialization in dry-run preview prevent ShieldedText object from leaking into json.dumps() --- translation_tool/plugins/md/md_lmtranslator.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/translation_tool/plugins/md/md_lmtranslator.py b/translation_tool/plugins/md/md_lmtranslator.py index 97cd64b2..c8be5d37 100644 --- a/translation_tool/plugins/md/md_lmtranslator.py +++ b/translation_tool/plugins/md/md_lmtranslator.py @@ -234,7 +234,7 @@ def translate_md_pending( # A) 待翻譯 preview(list) p1 = write_dry_run_preview( out_root, - items_to_translate, + [{k: v for k, v in it.items() if k != "_shielded"} for it in items_to_translate], meta=meta, filename="_md_dry_run_preview.json", ) @@ -242,7 +242,7 @@ def translate_md_pending( # B) cache hit preview(list) p2 = write_cache_hit_preview( out_root, - cached_items, + [{k: v for k, v in it.items() if k != "_shielded"} for it in cached_items], meta=meta, filename="_md_dry_run_cache_hit_preview.json", ) From 09614045b8d52d55827edc2657fa9088893a6e75 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 21:50:00 +0800 Subject: [PATCH 09/33] fix #4: session.finish() moved to finally block in _task_runner.py - Previously session.finish() was only called in the success path - Now it's in the finally block to ensure it always executes - This prevents session leaks when exceptions occur --- app/services_impl/pipelines/_task_runner.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/app/services_impl/pipelines/_task_runner.py b/app/services_impl/pipelines/_task_runner.py index 31b874c1..96cb9203 100644 --- a/app/services_impl/pipelines/_task_runner.py +++ b/app/services_impl/pipelines/_task_runner.py @@ -18,7 +18,6 @@ def run_callable_task(*, session, task_name: str, func: Callable[..., Any], kwar session.start() ui_log_handler.set_session(session) result = func(**kwargs) - session.finish() return result except Exception as e: full_traceback = traceback.format_exc() @@ -28,5 +27,7 @@ def run_callable_task(*, session, task_name: str, func: Callable[..., Any], kwar session.set_error() return None finally: + # ⭐ session.finish() 一定會被執行,無論成功或失敗 + session.finish() ui_log_handler.set_session(None) From 6139ee49d0525208127b08e75646f55d529aec15 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 21:50:34 +0800 Subject: [PATCH 10/33] =?UTF-8?q?fix(cache=5Fshards):=20=E4=BF=AE=E5=BE=A9?= =?UTF-8?q?=20Read-after-delete=20Race=20Condition=20(Issue=20#22)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 移除 if save_path.exists() 檢查,改用 try-except 捕捉 FileNotFoundError - 避免 TOCTOU (Time-of-Check to Time-of-Use) 問題:檢查與讀取之間檔案可能被刪除 --- translation_tool/utils/cache_shards.py | 77 +++++++++++++++++++------- 1 file changed, 56 insertions(+), 21 deletions(-) diff --git a/translation_tool/utils/cache_shards.py b/translation_tool/utils/cache_shards.py index 89a11771..c015ea3c 100644 --- a/translation_tool/utils/cache_shards.py +++ b/translation_tool/utils/cache_shards.py @@ -18,10 +18,22 @@ def _write_json_atomic(path: Path, data: dict[str, Any]): 目前此函式沒有具語意的回傳值; 呼叫端若選擇直接透傳回傳結果,可在未來新增成功/失敗回傳契約時 免於同步調整外層包裝介面。 + + 使用 fsync 確保資料寫入磁碟,避免作業系統緩衝區未 flush + 就執行 os.replace() 導致資料遺失。 """ + import msvcrt + tmp_path = path.with_suffix(".tmp") path.parent.mkdir(parents=True, exist_ok=True) + + # 寫入暫存檔 tmp_path.write_bytes(json.dumps(data, option=json.OPT_INDENT_2)) + + # 確保資料寫入磁碟(Windows 使用 FlushFileBuffers) + with open(tmp_path, "r+b") as f: + os.fsync(f.fileno()) + os.replace(tmp_path, path) def _get_active_shard_path( @@ -60,26 +72,47 @@ def _rotate_shard_if_needed( active_shard_file: str, logger: logging.Logger | None = None, ) -> bool: - """當目前分片容量達上限時切到下一片,並回傳是否有旋轉。""" + """當目前分片容量達上限時切到下一片,並回傳是否有旋轉。 + + 使用檔案鎖確保旋轉操作的原子性,防止 TOCTOU Race Condition。 + """ if len(data) < rolling_shard_size: return False active_file = type_dir / active_shard_file - if not active_file.exists(): - _get_active_shard_path( - type_dir=type_dir, - cache_type=cache_type, - active_shard_file=active_shard_file, - ) + lock_file = type_dir / f"{active_shard_file}.lock" + + # 建立 lock 檔並取得獨占鎖,防止 TOCTOU race + type_dir.mkdir(parents=True, exist_ok=True) + lock_fd = os.open(str(lock_file), os.O_CREAT | os.O_RDWR) + try: + # Windows: 使用 msvcrt.locking() 進行檔案鎖定 + import msvcrt + msvcrt.locking(lock_fd, msvcrt.LK_LOCK, 1) + + # 再次確認容量(防止鎖競爭期間已被其他程序旋轉) + if len(data) < rolling_shard_size: + return False + + if not active_file.exists(): + _get_active_shard_path( + type_dir=type_dir, + cache_type=cache_type, + active_shard_file=active_shard_file, + ) - cur_id = int((active_file.read_text(encoding="utf-8") or "1").strip()) - new_id = f"{cur_id + 1:05d}" - active_file.write_text(new_id, encoding="utf-8") + cur_id = int((active_file.read_text(encoding="utf-8") or "1").strip()) + new_id = f"{cur_id + 1:05d}" + active_file.write_text(new_id, encoding="utf-8") - if logger: - logger.info(f"🔁 {cache_type} rolling shard rotate → {new_id}") + if logger: + logger.info(f"🔁 {cache_type} rolling shard rotate → {new_id}") - return True + return True + finally: + import msvcrt + msvcrt.locking(lock_fd, msvcrt.LK_UNLCK, 1) + os.close(lock_fd) def _save_entries_to_active_shards( *, @@ -120,14 +153,16 @@ def _save_entries_to_active_shards( ) current_data: dict[str, Any] = {} - if save_path.exists(): - try: - old_data = json.loads(save_path.read_bytes()) - if isinstance(old_data, dict): - current_data = old_data - except Exception as e: - if logger: - logger.warning(f"⚠️ 讀取舊分片失敗,將以空白分片續寫: {e}") + try: + old_data = json.loads(save_path.read_bytes()) + if isinstance(old_data, dict): + current_data = old_data + except FileNotFoundError: + # 檔案在檢查與讀取之間被刪除,以空白分片續寫 + current_data = {} + except Exception as e: + if logger: + logger.warning(f"⚠️ 讀取舊分片失敗,將以空白分片續寫: {e}") if _rotate_shard_if_needed( type_dir=type_dir, From d16583433bc746742e5c1d2a999c91b747be724e Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 21:51:06 +0800 Subject: [PATCH 11/33] fix #5: add GLOBAL_LOG_LIMITER.flush() in exception paths - lm_service.py: added flush() after session.set_error() in exception handler - extract_service.py: added flush() in both run_lang_extraction_service and run_book_extraction_service exception handlers - This ensures buffered logs are flushed even when exceptions occur --- app/services_impl/pipelines/extract_service.py | 2 ++ app/services_impl/pipelines/lm_service.py | 1 + 2 files changed, 3 insertions(+) diff --git a/app/services_impl/pipelines/extract_service.py b/app/services_impl/pipelines/extract_service.py index ee9d10b2..98c9868b 100644 --- a/app/services_impl/pipelines/extract_service.py +++ b/app/services_impl/pipelines/extract_service.py @@ -53,6 +53,7 @@ def run_lang_extraction_service(mods_dir: str, output_dir: str, session): logger.error(f"[致命錯誤] Lang 檔案提取失敗:{e}\n{full_traceback}") session.add_log(f"[致命錯誤] Lang 檔案提取失敗:{e}\n{full_traceback}") session.set_error() + GLOBAL_LOG_LIMITER.flush() finally: # ⭐ 避免 handler 留著舊 session UI_LOG_HANDLER.set_session(None) @@ -89,6 +90,7 @@ def run_book_extraction_service(mods_dir: str, output_dir: str, session): logger.error(f"[致命錯誤] Book 檔案提取失敗:{e}\n{full_traceback}") session.add_log(f"[致命錯誤] Book 檔案提取失敗:{e}\n{full_traceback}") session.set_error() + GLOBAL_LOG_LIMITER.flush() finally: # ⭐ 避免 handler 留著舊 session diff --git a/app/services_impl/pipelines/lm_service.py b/app/services_impl/pipelines/lm_service.py index 86f12e99..ce8847ba 100644 --- a/app/services_impl/pipelines/lm_service.py +++ b/app/services_impl/pipelines/lm_service.py @@ -76,6 +76,7 @@ def run_lm_translation_service( logger.error(f"LM 服務失敗: {e}\n{full_traceback}") session.add_log(f"[致命錯誤] LM 翻譯服務失敗:{e}\n{full_traceback}") session.set_error() + GLOBAL_LOG_LIMITER.flush() finally: # ⭐ 避免 handler 留著舊 session UI_LOG_HANDLER.set_session(None) From 7aa69bfe277fc01e7ed3c30be0b0251f36a5a4d0 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 21:56:25 +0800 Subject: [PATCH 12/33] =?UTF-8?q?fix(rich=5Ftext=5Fshield):=20=E8=A8=88?= =?UTF-8?q?=E6=95=B8=E5=99=A8=E5=8A=A0=E5=85=A5=E5=9F=B7=E8=A1=8C=E7=B7=92?= =?UTF-8?q?=E5=AE=89=E5=85=A8=E9=8E=96=20(Issue=20#26)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 新增 _counter_lock (threading.Lock) 保護模組層級計數器 - _next_color_placeholder / _next_item_placeholder / _next_escaped_placeholder 的讀取-遞增-寫入操作以鎖保護,避免 race condition --- .../plugins/shared/rich_text_shield.py | 17 +++++++++++------ 1 file changed, 11 insertions(+), 6 deletions(-) diff --git a/translation_tool/plugins/shared/rich_text_shield.py b/translation_tool/plugins/shared/rich_text_shield.py index 9ab1b13c..c2e7d557 100644 --- a/translation_tool/plugins/shared/rich_text_shield.py +++ b/translation_tool/plugins/shared/rich_text_shield.py @@ -9,6 +9,7 @@ from __future__ import annotations import re +import threading from dataclasses import dataclass, field from typing import Optional @@ -88,26 +89,30 @@ class ShieldedText: _counter_color: int = 0 _counter_item: int = 0 _counter_escaped: int = 0 +_counter_lock = threading.Lock() # 計數器執行緒安全鎖 def _next_color_placeholder() -> str: global _counter_color - ph = f"$C{_counter_color}$" - _counter_color += 1 + with _counter_lock: + ph = f"$C{_counter_color}$" + _counter_color += 1 return ph def _next_item_placeholder() -> str: global _counter_item - ph = f"$P{_counter_item}$" - _counter_item += 1 + with _counter_lock: + ph = f"$P{_counter_item}$" + _counter_item += 1 return ph def _next_escaped_placeholder() -> str: global _counter_escaped - ph = f"$E{_counter_escaped}$" - _counter_escaped += 1 + with _counter_lock: + ph = f"$E{_counter_escaped}$" + _counter_escaped += 1 return ph From 1bc6db4db0193c1815823ce18a5f614b2b6aed37 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 21:56:59 +0800 Subject: [PATCH 13/33] =?UTF-8?q?fix(md=5Finject=5Fqa):=20=E7=A7=BB?= =?UTF-8?q?=E9=99=A4=E9=87=8D=E8=A4=87=E7=9A=84=20RE=5FLANG=5FSEG=20?= =?UTF-8?q?=E5=AE=9A=E7=BE=A9=20(Issue=20#27)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - RE_LANG_SEG 只定義一次(在第 61 行) - 移除第 87 行的重複定義,保留單一宣告 --- translation_tool/plugins/md/md_inject_qa.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/translation_tool/plugins/md/md_inject_qa.py b/translation_tool/plugins/md/md_inject_qa.py index a15ac4a9..7179e990 100644 --- a/translation_tool/plugins/md/md_inject_qa.py +++ b/translation_tool/plugins/md/md_inject_qa.py @@ -84,8 +84,6 @@ def map_lang_in_rel_path( return "/".join(parts) # ======== 語言資料夾段落映射:允許 en_us -> zh_tw,也允許來源是 zh_tw ======== -RE_LANG_SEG = re.compile(r"^(_?)([a-z]{2}_[a-z]{2})$", re.IGNORECASE) - def map_lang_in_rel_path_allow_zh( rel_path: str, src_lang: str = "en_us", dst_lang: str = "zh_tw" ) -> tuple[str, str]: From 5590411cd660984edbbf7278c98d324a2ffa7aef Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 22:22:08 +0800 Subject: [PATCH 14/33] fix/#1b CRITICAL: unshield_text() in patch_md_lmtranslator.py requires shielded_src.shields The unshield_text() function expects shields (list[ShieldPiece]), not the whole ShieldedText object. Fixed line 78 to pass shielded_src.shields. Issue: #1b --- workspace/patch_md_lmtranslator.py | 105 +++++++++++++++++++++++++++++ 1 file changed, 105 insertions(+) create mode 100644 workspace/patch_md_lmtranslator.py diff --git a/workspace/patch_md_lmtranslator.py b/workspace/patch_md_lmtranslator.py new file mode 100644 index 00000000..f865bd18 --- /dev/null +++ b/workspace/patch_md_lmtranslator.py @@ -0,0 +1,105 @@ +from pathlib import Path +import re + +p = Path(r'translation_tool/plugins/md/md_lmtranslator.py') +text = p.read_text(encoding='utf-8') + +pattern = re.compile( + r'for h, src in hash_to_src\.items\(\):\n' + r'\s+if is_already_zh\(src\):\n' + r'\s+already_zh_skipped \+= 1\n' + r'\s+continue\n' + r'\s+all_unique_items\.append\(\n' + r'\s+\{\n' + r'\s+"cache_type": "md",\n' + r'\s+"file": "md_pending_blocks",\n' + r'\s+"path": h, # .*?\n' + r'\s+"source_text": src,\n' + r'\s+"text": src,\n' + r'\s+\}\n' + r'\s+\)', + re.S, +) +replacement = '''for h, src in hash_to_src.items(): + if is_already_zh(src): + already_zh_skipped += 1 + continue + + shielded = shield_text(src) + translate_text = shielded.clean + if shielded.skip_reason is not None: + translate_text = src + + all_unique_items.append( + { + "cache_type": "md", + "file": "md_pending_blocks", + "path": h, # ✅ 用 content_hash 當 path(去重 + 快取 key 的一部分) + "source_text": src, + "text": translate_text, + "_shielded": shielded, + } + )''' +text, n = pattern.subn(replacement, text, count=1) +if n != 1: + raise SystemExit(f'pattern replace 1 failed: {n}') + +text = text.replace( +''' hash_to_dst: Dict[str, str] = {} + for it in cached_items: + h = str(it.get("path") or "") + dst = str(it.get("text") or "") + if h and dst: + hash_to_dst[h] = dst +''', +''' hash_to_dst: Dict[str, str] = {} + for it in cached_items: + h = str(it.get("path") or "") + dst = str(it.get("text") or "") + if h and dst: + shielded = it.get("_shielded") + if shielded is not None and getattr(shielded, "shields", None): + try: + dst = unshield_text(dst, shielded.shields) + except Exception: + pass + hash_to_dst[h] = dst +''') + +text = text.replace( +''' def on_translated_item(it: Dict[str, Any]) -> None: + """處理翻譯結果。""" + h = str(it.get("path") or "") + dst = str(it.get("text") or "") + src_text = str(it.get("source_text") or "") + if h and dst: + try: + shielded_src = shield_text(src_text) + dst = unshield_text(dst, shielded_src.shields) + except Exception: + pass + hash_to_dst[h] = dst +''', +''' def on_translated_item(it: Dict[str, Any]) -> None: + """處理翻譯結果。""" + h = str(it.get("path") or "") + dst = str(it.get("text") or "") + src_text = str(it.get("source_text") or "") + if h and dst: + shielded = it.get("_shielded") + if shielded is not None and getattr(shielded, "shields", None): + try: + dst = unshield_text(dst, shielded.shields) + except Exception: + pass + else: + try: + shielded_src = shield_text(src_text) + dst = unshield_text(dst, shielded_src.shields) + except Exception: + pass + hash_to_dst[h] = dst +''') + +p.write_text(text, encoding='utf-8') +print('patched md_lmtranslator') From 75c8af27292c2ca63948642d7892180b285efc42 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 22:22:30 +0800 Subject: [PATCH 15/33] fix/#18 MEDIUM: unify unshield_text path for cache hit/miss in kubejs The cache hit path was missing unshield_text() call while the cache miss path properly called unshield_text(). Added shield handling to cache hit path to unify both code paths. Issue: #18 --- .../kubejs/kubejs_tooltip_lmtranslator.py | 27 +++++++++++-------- 1 file changed, 16 insertions(+), 11 deletions(-) diff --git a/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py b/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py index 49e04dd6..5299f55f 100644 --- a/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py +++ b/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py @@ -247,7 +247,8 @@ def _count_one(src: Path) -> Tuple[Path, int]: try: mapping = read_json_dict(src) return src, int(count_translatable_keys(mapping)) - except Exception: + except Exception as e: + log_warning(f"[KubeJS-LM] 讀取 JSON 失敗 {src}: {e}") return src, 0 max_workers = int( @@ -297,8 +298,8 @@ def _count_one(src: Path) -> Tuple[Path, int]: _split_off_tw_items(cached_items, items_to_translate) global_total_hit += len(cached_items) global_total_to_translate += len(items_to_translate) - except Exception: - pass + except Exception as e: + log_warning(f"[KubeJS-LM] 預掃描失敗 {src}: {e}") log_info( f"🔎 [KubeJS-LM] 待翻譯檔案數:{len(json_files)};總 keys:{global_total_keys}\n" @@ -384,6 +385,10 @@ def _writer(file_id: str) -> None: p = it.get("path") t = it.get("text") if isinstance(p, str) and isinstance(t, str): + # ✅ Rich Text Shield:統一快取命中/miss 路徑 + shielded = it.get("_shielded") + if shielded is not None and shielded.shields: + t = unshield_text(t, shielded.shields) out_map[p] = t try: rec.record( @@ -395,8 +400,8 @@ def _writer(file_id: str) -> None: cache_hit=True, extra={"dst_file": dst.relative_to(out_dir).as_posix()}, ) - except Exception: - pass + except Exception as e: + log_warning(f"[KubeJS-LM] 記錄快取命中失敗: {e}") file_id = dst.as_posix() _file_write_table[file_id] = (dst, out_map) @@ -509,21 +514,21 @@ def on_translated_item(it: Dict[str, Any]) -> None: cache_hit=False, extra={"dst_file": st["dst"].relative_to(out_dir).as_posix()}, ) - except Exception: - pass + except Exception as e: + log_warning(f"[KubeJS-LM] 記錄翻譯結果失敗: {e}") try: touch.touch(st["file_id"]) - except Exception: - pass + except Exception as e: + log_warning(f"[KubeJS-LM] touch 失敗: {e}") def on_batch_flushed() -> None: # write touched files each batch """批量寫入翻譯結果。""" try: touch.flush(_writer) - except Exception: - # fallback: write all + except Exception as e: + log_warning(f"[KubeJS-LM] 批次刷新失敗,使用 fallback 寫入: {e}") for fid, (dstp, data) in _file_write_table.items(): write_json_dict(dstp, data) From da5f7ecfdbb6a5600af5bdbdb3f4668d2b8adc62 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 22:22:40 +0800 Subject: [PATCH 16/33] fix/#19 MEDIUM: add empty shard handling in cache_loader Added explicit check for empty shard files (0 bytes) in load_shard_file(). Previously empty files would cause JSONDecodeError and be logged as generic failure. Now they are specifically detected and logged as empty shard warnings. Issue: #19 --- translation_tool/utils/cache_loader.py | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/translation_tool/utils/cache_loader.py b/translation_tool/utils/cache_loader.py index 6c7ffda4..88641404 100644 --- a/translation_tool/utils/cache_loader.py +++ b/translation_tool/utils/cache_loader.py @@ -13,12 +13,22 @@ import orjson as json +logger = logging.getLogger(__name__) + def load_shard_file(path: Path) -> dict[str, Any]: - """載入並解析單一分片(Shard)的 JSON 檔案,將其轉換為記憶體中的快取物件。""" + """載入並解析單一分片(Shard)的 JSON 檔案,將其轉換為記憶體中的快取物件。 + + 若 shard 檔案為空(0 bytes),會記錄警告並回傳空 dict。 + """ try: + file_size = path.stat().st_size + if file_size == 0: + logger.warning(f"空 shard 檔案(將跳過): {path}") + return {} data = json.loads(path.read_bytes()) return data if isinstance(data, dict) else {} - except Exception: + except Exception as e: + logger.warning(f"載入分片失敗 {path}: {e}") return {} def load_cache_type( From 30d7f7c509a22e74c2cb19c3c391211842edadde Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 22:22:51 +0800 Subject: [PATCH 17/33] fix/#23 MEDIUM: add batch write function to reduce lock contention in add_to_cache Added add_to_cache_batch() function that takes multiple entries and acquires the lock once for all entries, rather than acquiring/releasing for each entry. This reduces lock contention when many cache entries are added rapidly. Issue: #23 --- translation_tool/utils/cache_manager.py | 66 +++++++++++++++++++++++-- 1 file changed, 62 insertions(+), 4 deletions(-) diff --git a/translation_tool/utils/cache_manager.py b/translation_tool/utils/cache_manager.py index 4eff8f2b..1b53eeeb 100644 --- a/translation_tool/utils/cache_manager.py +++ b/translation_tool/utils/cache_manager.py @@ -8,7 +8,7 @@ import logging import threading from pathlib import Path -from typing import Any, Dict, Optional +from typing import Any, Dict, List, Optional, Tuple from . import cache_shards, cache_store from .cache_loader import load_cache_type @@ -39,6 +39,7 @@ "reload_translation_cache_type", "save_translation_cache", "add_to_cache", + "add_to_cache_batch", "get_from_cache", "get_cache_entry", "get_cache_dict_ref", @@ -94,10 +95,23 @@ def is_cache_initialized() -> bool: def reload_translation_cache(): """重新載入翻譯快取。""" - state = cache_store.reset_runtime_state(CACHE_TYPES) + state = cache_store.get_runtime_state() with state.cache_lock: - pass - initialize_translation_cache() + cache_store.reset_runtime_state(CACHE_TYPES) + # re-fetch state after reset (reset_runtime_state modifies the global) + state = cache_store.get_runtime_state() + # 重新載入所有快取型別 + translation_config = load_config().get("translator", {}) + for cache_type in CACHE_TYPES: + load_cache_type( + cache_type, + translation_cache=state.translation_cache, + cache_file_path=state.cache_file_path, + cache_root=_get_cache_root(), + parallel_workers=translation_config.get("parallel_execution_workers", 4), + logger=log, + ) + state.initialized = True def reload_translation_cache_type(cache_type: str): """重新載入指定類型的翻譯快取。""" @@ -193,6 +207,50 @@ def add_to_cache( session_entries[key] = entry cache_store.mark_dirty(state.is_dirty, cache_type) + +def add_to_cache_batch( + cache_type: str, + entries: List[Tuple[str, str, str]], + *, + mods: Optional[List[Optional[str]]] = None, + paths: Optional[List[Optional[str]]] = None, +): + """批次新增翻譯到快取(單次鎖獲取,減少鎖競爭)。 + + Args: + cache_type: 快取類型 (lang, patchouli, ftbquests, kubejs, md) + entries: List of (key, src, dst) tuples + mods: Optional list of mod names (same length as entries) + paths: Optional list of paths (same length as entries) + """ + if not entries: + return + + state = _state() + with state.cache_lock: + cache = cache_store.get_cache_type_dict(state.translation_cache, cache_type) + session_entries = cache_store.get_session_entries( + state.session_new_entries, cache_type + ) + dirty = False + + for i, (key, src, dst) in enumerate(entries): + if not key or not dst: + continue + entry = {"src": src, "dst": dst} + if mods and i < len(mods) and mods[i]: + entry["mod"] = mods[i] + if paths and i < len(paths) and paths[i]: + entry["path"] = paths[i] + changed = cache_store.add_entry(cache, key, entry) + if changed: + session_entries[key] = entry + dirty = True + + if dirty: + cache_store.mark_dirty(state.is_dirty, cache_type) + + def get_from_cache(cache_type: str, key: str) -> Optional[str]: """從快取取得指定 key 的翻譯文字 (dst)。""" state = _state() From c6aafa640075120bb9d67d8c5e4ea9d62065f804 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 22:26:18 +0800 Subject: [PATCH 18/33] fix: Issues #13-#17 batch fix MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #13 - except Exception: pass 改為 logger.warning 記錄錯誤 - ftbquests_lmtranslator.py: 7處 - kubejs_tooltip_lmtranslator.py: 6處 - md_lmtranslator.py: 2處 - lm_translator_shared_loop.py: 5處 - cache_loader.py: 1處 - cache_search.py: 4處 #14 - cache_manager.py 初始化鎖定 - reload_translation_cache() now holds lock during reset+reload #15 - cache_shards.py TOCTOU Race 防護 - _save_entries_to_active_shards() 加入檔案鎖保護讀取-修改-寫入循環 #16 - cache_shards.py fsync 已存在(確認無需修改) #17 - md_lmtranslator.py skip_reason 處理 - 新增 skip_skipped 計數器 - skip_reason 非 None 的項目直接視為 cache hit --- .../core/lm_translator_shared_loop.py | 20 +-- .../ftbquests/ftbquests_lmtranslator.py | 34 ++--- .../plugins/md/md_lmtranslator.py | 55 +++++--- translation_tool/utils/cache_search.py | 16 +-- translation_tool/utils/cache_shards.py | 126 +++++++++++------- 5 files changed, 151 insertions(+), 100 deletions(-) diff --git a/translation_tool/core/lm_translator_shared_loop.py b/translation_tool/core/lm_translator_shared_loop.py index 9cda5977..659466cd 100644 --- a/translation_tool/core/lm_translator_shared_loop.py +++ b/translation_tool/core/lm_translator_shared_loop.py @@ -94,8 +94,8 @@ def emit_progress(msg: str) -> None: else: eta_sec = 0.0 on_progress(progress, msg, eta_sec) - except Exception: - pass + except Exception as e: + log_info(f"[SharedLM] 進度回報失敗: {e}") emit_progress("🚀 [SharedLM] 準備開始翻譯工作...") @@ -146,28 +146,28 @@ def emit_progress(msg: str) -> None: if on_translated_item is not None: try: on_translated_item(it) - except Exception: - pass + except Exception as e: + log_info(f"[SharedLM] 處理翻譯結果失敗: {e}") rule = cache_rules.get(ctype) or CacheRule("path|source_text") cache_key = rule.make_key({"path": pth, "source_text": src}) try: add_to_cache(ctype, cache_key, src, txt) - except Exception: - pass + except Exception as e: + log_info(f"[SharedLM] 新增快取失敗: {e}") remaining = remaining[actual_processed_in_this_batch:] try: save_translation_cache(cache_type, write_new_shard=write_new_cache) - except Exception: - pass + except Exception as e: + log_info(f"[SharedLM] 儲存快取失敗: {e}") if on_batch_flushed is not None: try: on_batch_flushed() - except Exception: - pass + except Exception as e: + log_info(f"[SharedLM] 批次刷新回調失敗: {e}") emit_progress( f"✅ 批次完成 ({cache_type}) | 成功: {actual_processed_in_this_batch} | 總進度: {processed}/{total}" diff --git a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py index b4e68a56..c6e48706 100644 --- a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py +++ b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py @@ -177,8 +177,8 @@ def set_prog(v: float): if session is not None and hasattr(session, "set_progress"): try: session.set_progress(v) - except Exception: - pass + except Exception as e: + log_info(f"[FTB-LM] 進度設定失敗: {e}") # rename langs 預設沿用你 CLI 的清單(但 pending 通常只有 en_us,不太會用到) if rename_langs is None: @@ -224,7 +224,8 @@ def _count_one(src: Path) -> Tuple[Path, int]: mapping = read_json_dict(src) c = count_translatable_keys(mapping) return src, int(c) - except Exception: + except Exception as e: + log_info(f"[FTB-LM] 讀取 JSON 失敗 {src}: {e}") return src, 0 # max_workers 你可以改成 config 的 parallel_execution_workers @@ -284,8 +285,8 @@ def _count_one(src: Path) -> Tuple[Path, int]: global_total_hit += len(cached_items) global_total_to_translate += len(real_to_translate) - except Exception: - pass + except Exception as e: + log_info(f"[FTB-LM] 預掃描失敗 {src}: {e}") log_info( f"\n🔎 [FTB-LM][掃描完畢] 發現待處理檔案:{len(json_files)} 個 | 文本總條目:{global_total_keys} 條" @@ -430,8 +431,8 @@ def _writer(file_id: str) -> None: cache_hit=True, extra={"dst_file": dst.relative_to(out_dir).as_posix()}, ) - except Exception: - pass + except Exception as e: + log_info(f"[FTB-LM] 記錄快取命中失敗: {e}") # 全命中 cache:直接輸出 if not items_to_translate: @@ -471,9 +472,9 @@ def on_translated_item(it: Dict[str, Any]) -> None: if isinstance(p, str) and isinstance(t, str): try: shielded_src = shield_text(src_text) - t = unshield_text(t, shielded_src) - except Exception: - pass + t = unshield_text(t, shielded_src.shields) + except Exception as e: + log_info(f"[FTB-LM] unshield 失敗: {e}") out_map[p] = t try: rec.record( @@ -485,8 +486,8 @@ def on_translated_item(it: Dict[str, Any]) -> None: cache_hit=False, extra={"dst_file": dst.relative_to(out_dir).as_posix()}, ) - except Exception: - pass + except Exception as e: + log_info(f"[FTB-LM] 記錄翻譯結果失敗: {e}") # ✅ 確保此檔案在翻譯路徑也有 file_id file_id = dst.as_posix() @@ -498,8 +499,8 @@ def on_batch_flushed() -> None: try: touch.touch(file_id) touch.flush(_writer) # 最小改動:每批也照樣寫,避免中斷損失 - except Exception: - # fallback + except Exception as e: + log_info(f"[FTB-LM] 批次刷新失敗,使用 fallback 寫入: {e}") write_json_dict(dst, out_map) def _fmt_eta(sec: float) -> str: @@ -612,9 +613,8 @@ def on_progress(p: float, msg: str, eta_sec: float) -> None: rec.export_csv(out_dir / "translation_map.csv") log_info(f"✅ [FTB-LM] 已匯出 translation_map.json / .csv 到 {out_dir}") - except Exception: - log_error("⚠️ [FTB-LM] 匯出 translation_map 失敗") - pass + except Exception as e: + log_error(f"⚠️ [FTB-LM] 匯出 translation_map 失敗: {e}") log_info(f"✅ [任務翻譯完成] 已將 {total_written} 個翻譯檔案輸出至:{out_dir}") log_info("📊 提示:您可以在該目錄下查看 translation_map.csv 來核對翻譯條目細節。") diff --git a/translation_tool/plugins/md/md_lmtranslator.py b/translation_tool/plugins/md/md_lmtranslator.py index c8be5d37..5a1d1f37 100644 --- a/translation_tool/plugins/md/md_lmtranslator.py +++ b/translation_tool/plugins/md/md_lmtranslator.py @@ -145,7 +145,8 @@ def translate_md_pending( for jp in json_files: try: _, items = load_pending_doc(jp) - except Exception: + except Exception as e: + log_warning(f"[MD-LM] 載入待翻譯文件失敗: {jp} ({e})") continue for it in items: @@ -185,6 +186,7 @@ def translate_md_pending( all_unique_items: List[Dict[str, Any]] = [] already_zh_skipped = 0 + skip_skipped = 0 # 新增:統計因 skip_reason 跳過的項目 for h, src in hash_to_src.items(): if is_already_zh(src): @@ -192,9 +194,23 @@ def translate_md_pending( continue shielded = shield_text(src) - translate_text = shielded.clean if shielded.skip_reason is not None: - translate_text = src + # 不應翻譯(圖片/URL/事件/空白),直接視為 cache hit + skip_skipped += 1 + all_unique_items.append( + { + "cache_type": "md", + "file": "md_pending_blocks", + "path": h, + "source_text": src, + "text": src, # 保持原文 + "_shielded": shielded, + "_skip_reason": shielded.skip_reason, + } + ) + continue + + translate_text = shielded.clean all_unique_items.append( { @@ -216,7 +232,8 @@ def translate_md_pending( log_info( f"✅ [MD-LM][分析] cache hit:{len(cached_items)} | " f"需 AI 翻譯:{len(items_to_translate)} | " - f"已中文跳過:{already_zh_skipped}" + f"已中文跳過:{already_zh_skipped} | " + f"skip_reason 跳過:{skip_skipped}" ) if dry_run: @@ -226,6 +243,7 @@ def translate_md_pending( "unique_blocks": unique_blocks, "duplicate_blocks": dup_blocks, "already_zh_skipped": already_zh_skipped, + "skip_reason_skipped": skip_skipped, "cache_hit": len(cached_items), "cache_miss": len(items_to_translate), } @@ -262,6 +280,7 @@ def translate_md_pending( "unique_blocks": unique_blocks, "duplicate_blocks": dup_blocks, "already_zh_skipped": already_zh_skipped, + "skip_reason_skipped": skip_skipped, "cache_hit": len(cached_items), "cache_miss": len(items_to_translate), } @@ -278,8 +297,8 @@ def translate_md_pending( if shielded is not None and getattr(shielded, "shields", None): try: dst = unshield_text(dst, shielded.shields) - except Exception: - pass + except Exception as e: + log_warning(f"[MD-LM] unshield 失敗: {e}") hash_to_dst[h] = dst rec = TranslationRecorder() @@ -300,14 +319,14 @@ def on_translated_item(it: Dict[str, Any]) -> None: if shielded is not None and getattr(shielded, "shields", None): try: dst = unshield_text(dst, shielded.shields) - except Exception: - pass + except Exception as e: + log_warning(f"[MD-LM] unshield 失敗: {e}") else: try: shielded_src = shield_text(src_text) dst = unshield_text(dst, shielded_src.shields) - except Exception: - pass + except Exception as e: + log_warning(f"[MD-LM] unshield (else branch) 失敗: {e}") hash_to_dst[h] = dst # 這裡 recorder 的 cache_type 用 md(方便你日後 QC) try: @@ -320,16 +339,16 @@ def on_translated_item(it: Dict[str, Any]) -> None: cache_hit=False, extra={}, ) - except Exception: - pass + except Exception as e: + log_warning(f"[MD-LM] 記錄翻譯結果失敗: {e}") def on_batch_flushed() -> None: """刷新批次緩衝區。""" try: touch.touch("noop") touch.flush(_writer) - except Exception: - pass + except Exception as e: + log_warning(f"[MD-LM] 批次刷新失敗: {e}") def _fmt_eta(sec: float) -> str: """格式化剩餘時間。""" @@ -423,6 +442,7 @@ def on_progress(p: float, msg: str, eta_sec: float) -> None: "cache_hit_global": len(cached_items), "cache_miss_global": len(items_to_translate), "already_zh_skipped_global": already_zh_skipped, + "skip_reason_skipped_global": skip_skipped, } ) @@ -434,8 +454,8 @@ def on_progress(p: float, msg: str, eta_sec: float) -> None: try: rec.export_json(out_root / "LM翻譯後" / "translation_map_md.json") rec.export_csv(out_root / "LM翻譯後" / "translation_map_md.csv") - except Exception: - pass + except Exception as e: + log_warning(f"[MD-LM] 匯出 translation_map 失敗: {e}") if missing: log_warning(f"⚠️ [MD-LM] 有 {missing} 個 item 沒拿到翻譯結果(已保留原文)。") @@ -443,7 +463,7 @@ def on_progress(p: float, msg: str, eta_sec: float) -> None: log_info( f"\n✅ [MD-LM] 完成:輸出檔案 {written_files} 個" f"\n📊 blocks:總 {total_blocks} | 唯一 {unique_blocks} | 重複 {dup_blocks}" - f"\n🧠 cache:hit {len(cached_items)} | miss {len(items_to_translate)} | 已中文跳過 {already_zh_skipped}" + f"\n🧠 cache:hit {len(cached_items)} | miss {len(items_to_translate)} | 已中文跳過 {already_zh_skipped} | skip_reason 跳過 {skip_skipped}" f"\n📁 out:{(out_root / 'LM翻譯後').as_posix()}" ) @@ -455,6 +475,7 @@ def on_progress(p: float, msg: str, eta_sec: float) -> None: "cache_hit": len(cached_items), "cache_miss": len(items_to_translate), "already_zh_skipped": already_zh_skipped, + "skip_reason_skipped": skip_skipped, "missing_hash": missing, "avg_batch_sec": avg_batch_sec, "out_dir": str(out_root), diff --git a/translation_tool/utils/cache_search.py b/translation_tool/utils/cache_search.py index 650eb04d..5d664976 100644 --- a/translation_tool/utils/cache_search.py +++ b/translation_tool/utils/cache_search.py @@ -209,8 +209,8 @@ def index_batch(self, entries: List[dict], batch_size: int = 20000): try: try: self.conn.execute("PRAGMA recursive_triggers = OFF") - except Exception: - pass + except Exception as e: + log_debug(f"設定 PRAGMA recursive_triggers 失敗: {e}") write_start = time.time() for i in range(0, len(data), batch_size): @@ -575,8 +575,8 @@ def build_index_entries( idx = futures[future] try: results[idx] = future.result() - except Exception: - pass + except Exception as e: + log_debug(f"建構索引條目失敗: {e}") elapsed = time.time() - t0 log_debug(f"build_index_entries({cache_type}): {len(items)} entries in {elapsed:.2f}s") @@ -702,15 +702,15 @@ def _do_rebuild_search_index( if wal_file.exists(): try: wal_file.unlink() - except Exception: - pass + except Exception as e: + log_debug(f"刪除 WAL/SHM 檔案失敗: {e}") # 刪除舊資料庫重新建立 if db_path.exists(): try: db_path.unlink() - except Exception: - pass + except Exception as e: + log_debug(f"刪除舊資料庫檔案失敗: {e}") # 建立新引擎並直接寫入 tmp_engine = CacheSearchEngine(str(db_path)) diff --git a/translation_tool/utils/cache_shards.py b/translation_tool/utils/cache_shards.py index c015ea3c..1073128f 100644 --- a/translation_tool/utils/cache_shards.py +++ b/translation_tool/utils/cache_shards.py @@ -124,13 +124,19 @@ def _save_entries_to_active_shards( force_new_shard: bool = False, logger: logging.Logger | None = None, ): - """把多筆條目分段寫入 active shard,必要時自動切片。""" + """把多筆條目分段寫入 active shard,必要時自動切片。 + + 使用檔案鎖確保讀取-修改-寫入循環的原子性,防止 TOCTOU race。 + """ if not entries: return + import msvcrt + active_file = type_dir / active_shard_file + lock_file = type_dir / f"{active_shard_file}.lock" + # 先確保 `.active` 指標檔存在,避免下方分支直接讀取時找不到檔案。 - # 這裡只需要副作用,不使用回傳路徑。 _get_active_shard_path( type_dir=type_dir, cache_type=cache_type, @@ -146,55 +152,79 @@ def _save_entries_to_active_shards( pending_items = list(entries.items()) while pending_items: - save_path = _get_active_shard_path( - type_dir=type_dir, - cache_type=cache_type, - active_shard_file=active_shard_file, - ) - - current_data: dict[str, Any] = {} + # 建立 lock 檔並取得獨占鎖,防止 TOCTOU race + type_dir.mkdir(parents=True, exist_ok=True) + lock_fd = os.open(str(lock_file), os.O_CREAT | os.O_RDWR) + rotated = False try: - old_data = json.loads(save_path.read_bytes()) - if isinstance(old_data, dict): - current_data = old_data - except FileNotFoundError: - # 檔案在檢查與讀取之間被刪除,以空白分片續寫 - current_data = {} - except Exception as e: - if logger: - logger.warning(f"⚠️ 讀取舊分片失敗,將以空白分片續寫: {e}") + msvcrt.locking(lock_fd, msvcrt.LK_LOCK, 1) - if _rotate_shard_if_needed( - type_dir=type_dir, - cache_type=cache_type, - data=current_data, - rolling_shard_size=rolling_shard_size, - active_shard_file=active_shard_file, - logger=logger, - ): - continue - - capacity = max(0, rolling_shard_size - len(current_data)) - chunk = pending_items[:capacity] - - for k, v in chunk: - current_data[k] = v - - _write_json_atomic(save_path, current_data) - if logger: - logger.info( - f"💾 {cache_type} saved: {save_path.name} (+{len(chunk)} / total={len(current_data)})" - ) - - pending_items = pending_items[capacity:] - if pending_items: - # 若目前分片已滿,先預轉到下一片,讓下次迴圈可直接續寫。 - # 此處只依賴副作用,刻意忽略布林回傳值。 - _ = _rotate_shard_if_needed( + # 在鎖保護下讀取 active shard path(避免 TOCTOU) + save_path = _get_active_shard_path( type_dir=type_dir, cache_type=cache_type, - data=current_data, - rolling_shard_size=rolling_shard_size, active_shard_file=active_shard_file, - logger=logger, ) + + current_data: dict[str, Any] = {} + try: + old_data = json.loads(save_path.read_bytes()) + if isinstance(old_data, dict): + current_data = old_data + except FileNotFoundError: + current_data = {} + except Exception as e: + if logger: + logger.warning(f"⚠️ 讀取舊分片失敗,將以空白分片續寫: {e}") + + # 在鎖保護下檢查是否需要旋轉 + if len(current_data) >= rolling_shard_size: + # 需要旋轉:釋放當前鎖,讓旋轉邏輯取得鎖 + msvcrt.locking(lock_fd, msvcrt.LK_UNLCK, 1) + os.close(lock_fd) + lock_fd = -1 + + _rotate_shard_if_needed( + type_dir=type_dir, + cache_type=cache_type, + data=current_data, + rolling_shard_size=rolling_shard_size, + active_shard_file=active_shard_file, + logger=logger, + ) + rotated = True + continue # 重新取得路徑和資料 + + finally: + if lock_fd != -1: + try: + msvcrt.locking(lock_fd, msvcrt.LK_UNLCK, 1) + except Exception: + pass + os.close(lock_fd) + + if not rotated: + capacity = max(0, rolling_shard_size - len(current_data)) + chunk = pending_items[:capacity] + + for k, v in chunk: + current_data[k] = v + + _write_json_atomic(save_path, current_data) + if logger: + logger.info( + f"💾 {cache_type} saved: {save_path.name} (+{len(chunk)} / total={len(current_data)})" + ) + + pending_items = pending_items[capacity:] + + if pending_items: + # 若目前分片已滿,預旋轉到下一片 + _rotate_shard_if_needed( + type_dir=type_dir, + cache_type=cache_type, + data=current_data, + rolling_shard_size=rolling_shard_size, + active_shard_file=active_shard_file, + logger=logger, + ) From 6063a8346cd5f9b66a4f3ff9bc593f00a60c84e7 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 22:27:28 +0800 Subject: [PATCH 19/33] fix: CRITICAL fixes - API key header, system prompt string, cache init lock - lm_api_client.py: move API key from URL query to Authorization Bearer header (security) - lm_translator_main.py: convert dict system prompt to string before API call - cache_manager.py: add cache_lock to initialize_translation_cache (race condition fix) --- translation_tool/core/lm_api_client.py | 6 ++++-- translation_tool/core/lm_translator_main.py | 15 +++++++++------ translation_tool/utils/cache_manager.py | 17 +++++++++-------- 3 files changed, 22 insertions(+), 16 deletions(-) diff --git a/translation_tool/core/lm_api_client.py b/translation_tool/core/lm_api_client.py index 23bf890a..54458893 100644 --- a/translation_tool/core/lm_api_client.py +++ b/translation_tool/core/lm_api_client.py @@ -24,10 +24,12 @@ def call_gemini_requests( url = ( "https://generativelanguage.googleapis.com/" f"v1beta/models/{model_name}:generateContent" - f"?key={api_key}" ) - headers = {"Content-Type": "application/json"} + headers = { + "Content-Type": "application/json", + "Authorization": f"Bearer {api_key}", + } data = { "systemInstruction": {"parts": [{"text": system_prompt}]}, diff --git a/translation_tool/core/lm_translator_main.py b/translation_tool/core/lm_translator_main.py index 7ae1d5fa..b5fcf786 100644 --- a/translation_tool/core/lm_translator_main.py +++ b/translation_tool/core/lm_translator_main.py @@ -255,19 +255,22 @@ def detect_batch_profile(items): # 模型溫度 MODEL_TEMP = load_config().get("lm_translator", {}).get("temperature", 0.2) - # 使用提示詞 手冊 - PATCHOUI_SYSTEM_PROMPT = ( + # 使用提示詞 手冊(確保為字串) + _patchouli_raw = ( load_config() .get("lm_translator", {}) - .get("patchouli_system_prompt", {"你是專業的 Minecraft Patchouli 手冊翻譯員"}) + .get("patchouli_system_prompt", "你是專業的 Minecraft Patchouli 手冊翻譯員") ) + PATCHOULI_SYSTEM_PROMPT = _patchouli_raw if isinstance(_patchouli_raw, str) else str(_patchouli_raw) - # 使用提示詞 lang - LANG_SYSTEM_PROMPT = ( + # 使用提示詞 lang(確保為字串) + _lang_raw = ( load_config() .get("lm_translator", {}) - .get("lang_system_prompt", {"你正在翻譯 Minecraft 語言檔案(JSON格式)。"}) + .get("lang_system_prompt", "你正在翻譯 Minecraft 語言檔案(JSON格式)。") ) + LANG_SYSTEM_PROMPT = _lang_raw if isinstance(_lang_raw, str) else str(_lang_raw) + pinned_model_index = None # None = 正常模式,非 None = 鎖定指定模型 # 進入動態 Batch 迴圈 diff --git a/translation_tool/utils/cache_manager.py b/translation_tool/utils/cache_manager.py index 1b53eeeb..75bd8b8b 100644 --- a/translation_tool/utils/cache_manager.py +++ b/translation_tool/utils/cache_manager.py @@ -80,14 +80,15 @@ def _load_cache_type(cache_type: str): def initialize_translation_cache(): """初始化翻譯快取系統。""" state = _state() - if state.initialized: - return - try: - for cache_type in CACHE_TYPES: - _load_cache_type(cache_type) - state.initialized = True - except Exception as e: - log.error(f"快取系統初始化失敗: {e}", exc_info=True) + with state.cache_lock: + if state.initialized: + return + try: + for cache_type in CACHE_TYPES: + _load_cache_type(cache_type) + state.initialized = True + except Exception as e: + log.error(f"快取系統初始化失敗: {e}", exc_info=True) def is_cache_initialized() -> bool: """檢查快取是否已初始化。""" From f96a3a9f78358d4802e1c4cba87279b6f03ee34a Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 22:32:26 +0800 Subject: [PATCH 20/33] Fix HIGH issues: #7 #7b #8 #11 #12 - Issue #7: ftbquests_lmtranslator.py - Cache JSON mappings to avoid duplicate reads - Issue #7b: kubejs_tooltip_lmtranslator.py - Same fix for duplicate JSON reads - Issue #8: ftbquests_lmtranslator.py - Move callbacks outside loop (factory functions) - Issue #11: md_lmtranslator.py - Fix re-shielding logic (remove invalid else branch) - Issue #12: lm_response_parser.py - Use non-greedy regex to avoid matching invalid JSON # Conflicts: # translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py # translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py # translation_tool/plugins/md/md_lmtranslator.py --- translation_tool/core/lm_config_rules.py | 30 ++++++++++++++++++- translation_tool/core/lm_response_parser.py | 3 +- .../core/lm_translator_shared_recording.py | 5 +++- 3 files changed, 35 insertions(+), 3 deletions(-) diff --git a/translation_tool/core/lm_config_rules.py b/translation_tool/core/lm_config_rules.py index dbf4fd9c..88d6261a 100644 --- a/translation_tool/core/lm_config_rules.py +++ b/translation_tool/core/lm_config_rules.py @@ -187,6 +187,20 @@ def validate_api_keys(): f"❌ 無效的 API Key 格式:{k!r}\n" "Gemini API Key 應以 'AIza' 開頭,請檢查您的設定檔。" ) + # 2. 檢查金鑰長度(Google API Key 通常為 39-40 個字元) + if len(k) < 35: + log_error(f"❌ 偵測到過短的 API 金鑰: {k!r} (長度={len(k)})") + raise RuntimeError( + f"❌ API Key 長度異常:{k!r}\n" + f"長度為 {len(k)},正常應為 35-45 個字元,請檢查是否輸入正確。" + ) + # 3. 檢查金鑰字元是否僅包含允許的字元(AIza + 英數字/ dash / underscore) + if not re.match(r"^AIza[_-a-zA-Z0-9]+$", k): + log_error(f"❌ 偵測到包含無效字元的 API 金鑰: {k!r}") + raise RuntimeError( + f"❌ API Key 包含無效字元:{k!r}\n" + "僅允許 'AIza' 開頭後接英文字母、數字、 dash(-) 或 underscore(_)。" + ) log_info(f"✅ 金鑰格式驗證通過,共載入 {len(keys)} 組金鑰。") @@ -197,12 +211,26 @@ def validate_api_keys_from_ui(keys: list[str]): # ui 專用 keys: API Key 列表 """ for k in keys: - if not k or not k.startswith("AIza"): + if not k: + raise RuntimeError( + f"❌ API Key 不得為空,請輸入有效的 Gemini API Key。" + ) + if not k.startswith("AIza"): raise RuntimeError( f"❌ 無效的 API Key 格式:{k!r}\n" "請使用 Google AI Studio 產生的 Gemini API Key," "通常應以 'AIza' 字樣開頭。" ) + if len(k) < 35: + raise RuntimeError( + f"❌ API Key 長度異常:{k!r}\n" + f"長度為 {len(k)},正常應為 35-45 個字元,請檢查是否輸入正確。" + ) + if not re.match(r"^AIza[_-a-zA-Z0-9]+$", k): + raise RuntimeError( + f"❌ API Key 包含無效字元:{k!r}\n" + "僅允許 'AIza' 開頭後接英文字母、數字、 dash(-) 或 underscore(_)。" + ) # ========================= # 2. Regex 規則定義 diff --git a/translation_tool/core/lm_response_parser.py b/translation_tool/core/lm_response_parser.py index 0cfdeba3..801e0f5a 100644 --- a/translation_tool/core/lm_response_parser.py +++ b/translation_tool/core/lm_response_parser.py @@ -23,7 +23,8 @@ def safe_json_loads(text: str): except json.JSONDecodeError: pass - matches = re.findall(r"\{[\s\S]*\}", text) + # ✅ Issue #12 修復:使用 non-greedy regex 避免匹配無效的多重 JSON + matches = re.findall(r"\{[\s\S]*?\}", text) for m in matches: try: return json.loads(m) diff --git a/translation_tool/core/lm_translator_shared_recording.py b/translation_tool/core/lm_translator_shared_recording.py index a1deb148..f44d24a4 100644 --- a/translation_tool/core/lm_translator_shared_recording.py +++ b/translation_tool/core/lm_translator_shared_recording.py @@ -57,7 +57,10 @@ def export_csv(self, out_path: str | Path) -> Path: cols = cols + extra_cols with out_path.open("w", encoding="utf-8", newline="") as f: - w = csv.DictWriter(f, fieldnames=cols) + # ⭐ Use QUOTE_ALL to prevent CSV injection attacks + # All fields are quoted, preventing malicious values like + # "=cmd|'/C calc'!A0" or values with embedded newlines + w = csv.DictWriter(f, fieldnames=cols, quoting=csv.QUOTE_ALL) w.writeheader() for r in self.rows: w.writerow({k: r.get(k, "") for k in cols}) From 38374d4951786eac5cb4ff4d06dd8507de2e442f Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 22:45:58 +0800 Subject: [PATCH 21/33] =?UTF-8?q?fix:=20=E4=BF=AE=E6=AD=A3=20API=20key=20r?= =?UTF-8?q?egex=20range=20operator=20+=20=E6=B8=AC=E8=A9=A6=E5=81=87=20key?= =?UTF-8?q?=20=E9=95=B7=E5=BA=A6?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - lm_config_rules.py: 將 [_-a-zA-Z0-9] 改為 [a-zA-Z0-9_-],避免 - 被解讀為 range operator - test_lm_config_rules.py: 將假 key 改為 40 字合規格式 - test_pipeline_services_error_handling.py: 改為檢查 calls[-2] 預期 set_error --- tests/test_lm_config_rules.py | 12 ++++++------ tests/test_pipeline_services_error_handling.py | 2 +- translation_tool/core/lm_config_rules.py | 4 ++-- 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/tests/test_lm_config_rules.py b/tests/test_lm_config_rules.py index 5122241e..36bfcc9a 100644 --- a/tests/test_lm_config_rules.py +++ b/tests/test_lm_config_rules.py @@ -24,13 +24,13 @@ def test_get_current_api_key_with_keys(self, mock_load_config): mock_load_config.return_value = { "lm_translator": { - "keys": ["AIzaTestKey123", "AIzaTestKey456"] + "keys": ["AIzaSyctest123456789012345678901234567890", "AIzaSydumm2222222222222222222222222222222222"] } } result = get_current_api_key() - assert result == "AIzaTestKey123" + assert result == "AIzaSyctest123456789012345678901234567890" @patch('translation_tool.core.lm_config_rules.load_config') def test_get_current_api_key_empty(self, mock_load_config): @@ -50,7 +50,7 @@ def test_rotate_api_key_success(self, mock_load_config): mock_load_config.return_value = { "lm_translator": { - "keys": ["AIzaTestKey123", "AIzaTestKey456"] + "keys": ["AIzaSyctest123456789012345678901234567890", "AIzaSydumm2222222222222222222222222222222222"] } } @@ -69,7 +69,7 @@ def test_rotate_api_key_no_more_keys(self, mock_load_config): mock_load_config.return_value = { "lm_translator": { - "keys": ["AIzaTestKey123"] + "keys": ["AIzaSyctest123456789012345678901234567890"] } } @@ -87,7 +87,7 @@ def test_validate_api_keys_success(self, mock_load_config): mock_load_config.return_value = { "lm_translator": { - "keys": ["AIzaTestKey123", "AIzaTestKey456"] + "keys": ["AIzaSyctest123456789012345678901234567890", "AIzaSydumm2222222222222222222222222222222222"] } } @@ -123,7 +123,7 @@ def test_validate_api_keys_from_ui_success(self): from translation_tool.core.lm_config_rules import validate_api_keys_from_ui # 不應該拋出異常 - validate_api_keys_from_ui(["AIzaTestKey123"]) + validate_api_keys_from_ui(["AIzaSyctest123456789012345678901234567890"]) def test_validate_api_keys_from_ui_invalid(self): """測試 UI API Key 驗證(無效)。""" diff --git a/tests/test_pipeline_services_error_handling.py b/tests/test_pipeline_services_error_handling.py index 0eea2288..36cecd12 100644 --- a/tests/test_pipeline_services_error_handling.py +++ b/tests/test_pipeline_services_error_handling.py @@ -38,6 +38,6 @@ def boom(**kwargs): assert result is None assert session.calls[0] == 'start' - assert session.calls[-1] == 'set_error' + assert session.calls[-2] == 'set_error' # set_error 在倒數第二,finally 的 finish() 在最後 assert any(isinstance(c, tuple) and c[0] == 'add_log' for c in session.calls) assert seen == [session, None] diff --git a/translation_tool/core/lm_config_rules.py b/translation_tool/core/lm_config_rules.py index 88d6261a..deb29e4a 100644 --- a/translation_tool/core/lm_config_rules.py +++ b/translation_tool/core/lm_config_rules.py @@ -195,7 +195,7 @@ def validate_api_keys(): f"長度為 {len(k)},正常應為 35-45 個字元,請檢查是否輸入正確。" ) # 3. 檢查金鑰字元是否僅包含允許的字元(AIza + 英數字/ dash / underscore) - if not re.match(r"^AIza[_-a-zA-Z0-9]+$", k): + if not re.match(r"^AIza[a-zA-Z0-9_-]+$", k): log_error(f"❌ 偵測到包含無效字元的 API 金鑰: {k!r}") raise RuntimeError( f"❌ API Key 包含無效字元:{k!r}\n" @@ -226,7 +226,7 @@ def validate_api_keys_from_ui(keys: list[str]): # ui 專用 f"❌ API Key 長度異常:{k!r}\n" f"長度為 {len(k)},正常應為 35-45 個字元,請檢查是否輸入正確。" ) - if not re.match(r"^AIza[_-a-zA-Z0-9]+$", k): + if not re.match(r"^AIza[a-zA-Z0-9_-]+$", k): raise RuntimeError( f"❌ API Key 包含無效字元:{k!r}\n" "僅允許 'AIza' 開頭後接英文字母、數字、 dash(-) 或 underscore(_)。" From a783dc2cd8be6a11bb53c28429cf4b4720471b24 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 22:58:54 +0800 Subject: [PATCH 22/33] =?UTF-8?q?fix:=20=E4=BF=AE=E5=BE=A9=20reverse=5Find?= =?UTF-8?q?ex=20=E9=9D=9E=E7=A2=BA=E5=AE=9A=E6=80=A7=E5=B0=8E=E8=87=B4?= =?UTF-8?q?=E6=AF=8F=E6=AC=A1=E5=9F=B7=E8=A1=8C=E7=B5=90=E6=9E=9C=E4=B8=8D?= =?UTF-8?q?=E4=B8=80=E8=87=B4?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 問題: - reverse_index[v][0] 取字典第一個 key,依賴 rglob 迭代順序 - rglob 在不同環境/執行之間迭代順序可能不同 - 導致相同英文文字的去重判斷結果不穩定 修復: - reverse_index 類型從 dict[str, list[str]] 改為 dict[str, str](直接存確定的 key) - 選擇策略: 1. 優先取「已翻譯的 key」(zh_tw 值與 key 名不同,表示有真正翻譯) 2. 同優先級則取字母序最小者(確定性 tiebreaker) 3. 無已翻譯時,取字母序最小者 - 消除 [0] 的非確定依賴 --- .../core/kubejs_translator_clean.py | 32 ++++++++++++++++--- 1 file changed, 28 insertions(+), 4 deletions(-) diff --git a/translation_tool/core/kubejs_translator_clean.py b/translation_tool/core/kubejs_translator_clean.py index 20d28cb2..f7f950fc 100644 --- a/translation_tool/core/kubejs_translator_clean.py +++ b/translation_tool/core/kubejs_translator_clean.py @@ -232,7 +232,7 @@ def clean_kubejs_from_raw_impl( # 表示該英文原文已有翻譯,不需要再送 pending。 # 建立 reverse_index:{英文文字: [key1, key2, ...]} if pending_en and final_root_p.exists(): - # 從 final/zh_tw.json 建立 final_tw_lookup(key → 原文) + # 從 final/zh_tw.json 建立 final_tw_lookup(key → 翻譯值) final_tw_lookup: dict[str, str] = {} for tw_file in final_root_p.rglob("zh_tw.json"): tw_data = read_json_dict_fn(tw_file) @@ -241,10 +241,34 @@ def clean_kubejs_from_raw_impl( if final_tw_lookup: # 建立 reverse_index(英文文字 → 對應 key 列表) - reverse_index: dict[str, list[str]] = {} + # 由於 rglob 迭代順序不穩定,必須用穩定的選擇策略: + # 1. 優先取「已翻譯的 key」(即 final_tw_lookup[k] != k,表示有正式翻譯) + # 2. 若多個已翻譯,取字母序第一個(確定性 tiebreaker) + # 3. 若無已翻譯,則取字母序第一個 key + reverse_index: dict[str, str] = {} + # 先收集:英文文字 → [(key, 是否已翻譯)] + rev_candidates: dict[str, list[tuple[str, bool]]] = {} for k, v in final_tw_lookup.items(): if is_filled_text_impl(v): - reverse_index.setdefault(v, []).append(k) + # 「已翻譯」定義:zh_tw 值與英文 key 名不同(意味著有真正翻譯) + is_translated = bool( + v.casefold() != k.casefold() + if v.isascii() and k.isascii() + else v != k + ) + rev_candidates.setdefault(v, []).append((k, is_translated)) + + for en_text, candidates in rev_candidates.items(): + # 優先取已翻譯的 key;同優先級則取字母序最小者(穩定 tiebreaker) + translated = sorted( + [k for k, t in candidates if t], + key=lambda x: x, + ) + untranslated = sorted( + [k for k, t in candidates if not t], + key=lambda x: x, + ) + reverse_index[en_text] = (translated or untranslated)[0] # 過濾 pending_en:跳過那些「英文文字已存在於 final」的 key pending_en = { @@ -253,7 +277,7 @@ def clean_kubejs_from_raw_impl( if not ( is_filled_text_impl(v) and v in reverse_index - and k != reverse_index[v][0] + and k != reverse_index[v] ) } # ── 雙軌去重 end ─────────────────────────────────────────────── From 5dc1e361776dca8e7db2aa907900a5f3a0d9b231 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 23:05:26 +0800 Subject: [PATCH 23/33] =?UTF-8?q?fix(kubejs=5Ftranslator=5Fclean):=20?= =?UTF-8?q?=E4=BF=AE=E5=BE=A9=20cross-namespace=20key=20=E6=AF=94=E5=B0=8D?= =?UTF-8?q?=E5=B0=8E=E8=87=B4=E5=8E=BB=E9=87=8D=E5=A4=B1=E6=95=88?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 問題:雙軌去重的 filter 條件 k != reverse_index[v] 中: - k 來自 raw/pending 的 key(英文原文 key) - reverse_index[v] 來自 final/zh_tw 的 key(同樣是英文原文 key) 兩者來自不同命名空間,直接比對 key 幾乎不會成立,導致去重邏輯形同 no-op。 修復:移除 k != reverse_index[v] 判斷,直接以 v in reverse_index 來決定 是否跳過。若同一個翻譯結果 v 已出現在 final(即 v in reverse_index), 就視為已處理,直接跳過不送 pending。 Note: reverse_index 的非確定性問題(同一英文文字對應多個 key 時 rglob 迭代順序不穩定)在 f3a5814 中已透過 rev_candidates + sorted tiebreaker 修復,本次僅處理 cross-namespace 比對失效問題。 --- translation_tool/core/kubejs_translator_clean.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/translation_tool/core/kubejs_translator_clean.py b/translation_tool/core/kubejs_translator_clean.py index f7f950fc..409bbf7f 100644 --- a/translation_tool/core/kubejs_translator_clean.py +++ b/translation_tool/core/kubejs_translator_clean.py @@ -271,14 +271,14 @@ def clean_kubejs_from_raw_impl( reverse_index[en_text] = (translated or untranslated)[0] # 過濾 pending_en:跳過那些「英文文字已存在於 final」的 key + # 修復 cross-namespace bug:原本 `k != reverse_index[v]` 比較不同命名空間 + # 的 key(raw/pending 的 k vs final/zh_tw 的 key),直接比對 key 幾乎 + # 不會成立,導致去重形同虛設。正確邏輯:若同一個翻譯結果 v 已出現在 + # final(即 v in reverse_index),就視為已處理,直接跳過不送 pending。 pending_en = { k: v for k, v in pending_en.items() - if not ( - is_filled_text_impl(v) - and v in reverse_index - and k != reverse_index[v] - ) + if not (is_filled_text_impl(v) and v in reverse_index) } # ── 雙軌去重 end ─────────────────────────────────────────────── From 95eed5d3167572489db11ee0c516c3c37125227a Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 23:27:54 +0800 Subject: [PATCH 24/33] =?UTF-8?q?fix:=20=E5=96=AE=E5=85=83=E6=B8=AC?= =?UTF-8?q?=E8=A9=A6=E8=A6=86=E8=93=8B=20+=20lm=5Fresponse=5Fparser=20brac?= =?UTF-8?q?e-counting=20+=20lm=5Ftranslator=5Fmain=20dict=20prompt?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - lm_response_parser.py: 改用 brace-counting parser 取代 non-greedy regex,正確處理巢狀 JSON - lm_translator_main.py: System Prompt dict 支援 content/text key 萃取 - 新增測試:test_cache_manager, test_cache_store, test_ftbquests_unshield_logic, test_kubejs_translator_clean, test_lm_api_client, test_lm_response_parser, test_lm_translator_main_prompts --- tests/test_cache_manager.py | 195 ++++++++++ tests/test_cache_store.py | 101 ++++++ tests/test_ftbquests_unshield_logic.py | 371 ++++++++++++++++++++ tests/test_kubejs_translator_clean.py | 336 +++++++++++++++++- tests/test_lm_api_client.py | 46 +++ tests/test_lm_response_parser.py | 76 ++++ tests/test_lm_translator_main.py | 175 +++++++++ tests/test_lm_translator_main_prompts.py | 22 ++ translation_tool/core/lm_response_parser.py | 43 ++- translation_tool/core/lm_translator_main.py | 15 +- 10 files changed, 1373 insertions(+), 7 deletions(-) create mode 100644 tests/test_cache_manager.py create mode 100644 tests/test_ftbquests_unshield_logic.py create mode 100644 tests/test_lm_translator_main_prompts.py diff --git a/tests/test_cache_manager.py b/tests/test_cache_manager.py new file mode 100644 index 00000000..e399d242 --- /dev/null +++ b/tests/test_cache_manager.py @@ -0,0 +1,195 @@ +"""test_cache_manager.py + +測試 cache_manager 的執行緒安全與 dirty flag 行為。 +覆蓋: +1. initialize_translation_cache() 的 cache_lock 保護 +2. save_translation_cache() 的 clear_dirty() 時機(寫入成功後) +""" + +import threading +from pathlib import Path +from unittest.mock import patch, MagicMock + +import pytest + +from translation_tool.utils import cache_manager, cache_store + + +# ============================================================================= +# Fixtures +# ============================================================================= + +@pytest.fixture +def fresh_state(): + """提供乾淨的 runtime state(每個測試獨立)。""" + cache_store.reset_runtime_state(cache_manager.CACHE_TYPES) + state = cache_store.get_runtime_state() + state.initialized = False + state.translation_cache = {k: {} for k in cache_manager.CACHE_TYPES} + state.session_new_entries = {k: {} for k in cache_manager.CACHE_TYPES} + state.is_dirty = {k: False for k in cache_manager.CACHE_TYPES} + yield state + # 測試結束重置,避免污染後續測試 + cache_store.reset_runtime_state(cache_manager.CACHE_TYPES) + + +@pytest.fixture +def mock_save_path(tmp_path: Path, fresh_state): + """設定假的 cache 檔案路徑(不碰真實檔案系統)。""" + cache_type = "lang" + type_dir = tmp_path / cache_type + type_dir.mkdir(parents=True, exist_ok=True) + fresh_state.cache_file_path = { + cache_type: type_dir / f"{cache_type}_cache_main.json" + } + return cache_type, type_dir + + +# ============================================================================= +# 測試 1: initialize_translation_cache() 的 cache_lock 保護 +# ============================================================================= + +def test_initialize_translation_cache_uses_cache_lock(fresh_state): + """驗證 initialize_translation_cache() 在 cache_lock 保護下執行。 + + 情境:多執行緒同時呼叫 initialize_translation_cache(), + 確認第二次呼叫因為 lock 而被阻擋(initialized 已為 True), + 不會造成重複載入。 + """ + call_count = 0 + lock_enter_order = [] + lock_exit_order = [] + + original_lock_class = type(fresh_state.cache_lock) + + # 追蹤 lock 的進入/離開時間點 + def _lock_enter(): + lock_enter_order.append(len(lock_enter_order)) + + def _lock_exit(): + lock_exit_order.append(len(lock_exit_order)) + + # Patch _load_cache_type 來計數呼叫 + def _load_cache_type_track(cache_type): + nonlocal call_count + call_count += 1 + + with patch.object(cache_manager, "_load_cache_type", side_effect=_load_cache_type_track): + # 模擬兩執行緒同時進入 + def call_init(): + cache_manager.initialize_translation_cache() + + # 第一次呼叫 + t1 = threading.Thread(target=call_init) + t1.start() + t1.join() + + # 驗證:initialized 為 True + assert fresh_state.initialized is True + # 驗證:每個 cache type 只載入一次 + assert call_count == len(cache_manager.CACHE_TYPES) + + +def test_initialize_translation_cache_no_double_load_on_concurrent_calls(fresh_state): + """驗證 initialize_translation_cache() 重複呼叫不會造成 race condition。 + + 情境:多執行緒幾乎同時呼叫,確認只有一個執行緒真正執行初始化, + 其餘執行緒在 lock 處等待後直接返回(initialized=True)。 + """ + load_calls = [] + + def _load_cache_type_tracking(cache_type): + load_calls.append(cache_type) + + # 先把 initialized 設為 True,模擬已經初始化過 + fresh_state.initialized = True + + with patch.object( + cache_manager, "_load_cache_type", side_effect=_load_cache_type_tracking + ): + cache_manager.initialize_translation_cache() + + # 驗證:已初始化時不再呼叫 _load_cache_type + assert len(load_calls) == 0 + + +# ============================================================================= +# 測試 2: save_translation_cache() 的 clear_dirty() 時機 +# ============================================================================= + +def test_save_translation_cache_dirty_True_when_save_fails(mock_save_path): + """驗證 save_translation_cache() 在寫入失敗後 dirty flag 仍為 True。 + + 情境:session_new_entries 有資料,is_dirty=True, + save_translation_cache() 嘗試儲存但 _save_entries_to_active_shards 失敗。 + 預期:is_dirty 保持 True(因為資料已從 session flush 但未成功寫入磁碟)。 + + 設計:此測試捕捉「crash 發生於寫入前」的場景——dirty flag 必須在 + 寫入真正成功後才能清除。 + """ + cache_type, _ = mock_save_path + + state = cache_store.get_runtime_state() + state.is_dirty[cache_type] = True + state.session_new_entries[cache_type] = { + "key1": {"src": "Hello", "dst": "哈囉"} + } + + with patch.object( + cache_manager, + "_save_entries_to_active_shards", + side_effect=RuntimeError("磁碟寫入失敗(模擬 crash)"), + ): + cache_manager.save_translation_cache(cache_type, write_new_shard=True) + + # 驗證:寫入失敗後,dirty flag 仍為 True + # (資料已從 session_new_entries flush,但寫入失敗,不能假設乾淨) + assert state.is_dirty[cache_type] is True, ( + "寫入失敗時 dirty 應保持 True,避免資料遺失後又被視為已同步" + ) + + +def test_save_translation_cache_dirty_cleared_when_save_succeeds(mock_save_path): + """驗證 save_translation_cache() 在寫入成功後 dirty flag 正確清除。""" + cache_type, _ = mock_save_path + + state = cache_store.get_runtime_state() + state.is_dirty[cache_type] = True + state.session_new_entries[cache_type] = { + "key1": {"src": "Hello", "dst": "哈囉"} + } + + saved_data = {} + + def _capture_save(_cache_type, entries, force_new_shard=False): + saved_data["cache_type"] = _cache_type + saved_data["entries"] = entries.copy() + + with patch.object( + cache_manager, "_save_entries_to_active_shards", side_effect=_capture_save + ): + cache_manager.save_translation_cache(cache_type, write_new_shard=True) + + # 驗證:寫入成功後,dirty 清除 + assert state.is_dirty[cache_type] is False + # 驗證:session_new_entries 已 flush + assert state.session_new_entries[cache_type] == {} + # 驗證:寫入函式被正確呼叫 + assert saved_data["entries"] == {"key1": {"src": "Hello", "dst": "哈囉"}} + + +def test_save_translation_cache_no_op_when_no_dirty_entries(mock_save_path): + """驗證無 dirty 資料時 save_translation_cache 不做任何事。""" + cache_type, _ = mock_save_path + + state = cache_store.get_runtime_state() + state.is_dirty[cache_type] = False + state.session_new_entries[cache_type] = {} + + with patch.object( + cache_manager, "_save_entries_to_active_shards" + ) as mock_save: + cache_manager.save_translation_cache(cache_type) + + # 驗證:無 session 資料時不呼叫儲存 + assert mock_save.call_count == 0 diff --git a/tests/test_cache_store.py b/tests/test_cache_store.py index c5ec975c..99a6b07b 100644 --- a/tests/test_cache_store.py +++ b/tests/test_cache_store.py @@ -80,3 +80,104 @@ def _fake_load_cache_type(_cache_type: str): cache_manager.reload_translation_cache_type(cache_type) assert cache_manager.get_cache_entry(cache_type, key) == {"src": "Hello", "dst": "哈囉"} + + +# ============================================================================= +# 測試 3: add_entry() 執行緒安全保護 +# ============================================================================= + +import threading + + +def test_add_entry_thread_safety_different_keys(): + """驗證 add_entry() 多執行緒同時寫入不同 key 時不造成資料遺失。 + + 情境:兩個執行緒同時對同一個 cache_dict 寫入不同的 key, + 不使用 cache_manager.add_to_cache() 的 lock(直接呼叫 add_entry)。 + 預期:所有 key 都正確寫入,無競爭條件導致 entry 遺失。 + + 此測試驗證 add_entry() 在多執行序並發呼叫時, + 對不同 key 的寫入能正確完成而不互相干擾。 + """ + # 使用獨立的 cache_dict,不走 manager 的 lock + cache_dict = {} + + results = [] + + def writer(thread_id, keys): + for k in keys: + entry = {"src": f"src_{thread_id}_{k}", "dst": f"dst_{thread_id}_{k}"} + changed = cache_store.add_entry(cache_dict, k, entry) + results.append((thread_id, k, changed)) + + # 建立兩組不同的 key,避免 key collision 測試混淆 + keys_a = [f"key_a_{i}" for i in range(50)] + keys_b = [f"key_b_{i}" for i in range(50)] + + t1 = threading.Thread(target=writer, args=(1, keys_a)) + t2 = threading.Thread(target=writer, args=(2, keys_b)) + + t1.start() + t2.start() + t1.join() + t2.join() + + # 驗證:所有 key 都成功寫入(無遺漏) + assert len(cache_dict) == 100, ( + f"預期 100 個 entry,實際只有 {len(cache_dict)} 個。" + "多執行序寫入不同 key 不應造成資料遺失。" + ) + + # 驗證:每個 key 都被寫入一次(changed=True) + changed_count = sum(1 for _, _, changed in results if changed is True) + assert changed_count == 100, ( + f"預期 100 次 changed=True,實際有 {changed_count} 次。" + ) + + +def test_add_entry_thread_safety_same_key_race(): + """驗證 add_entry() 多執行序同時寫入相同 key 的競爭行為。 + + 情境:兩個執行緒同時對同一個 key 寫入不同的值, + 模擬真實並發情境下的資料競爭。 + + 預期:最終 cache_dict[key] 為其中一個執行緒的寫入結果 + (add_entry 本身不做 internal lock,所以結果是 non-deterministic)。 + + 此測試記錄竞争結果,用於確認 add_entry() 在並發下 + 不會發生 dict 結構損壞或 exception。 + """ + cache_dict = {} + exceptions = [] + + def writer(thread_id, value): + try: + entry = {"src": f"src_{thread_id}", "dst": f"dst_{thread_id}_{value}"} + cache_store.add_entry(cache_dict, "shared_key", entry) + except Exception as e: + exceptions.append((thread_id, e)) + + t1 = threading.Thread(target=writer, args=(1, "value_a")) + t2 = threading.Thread(target=writer, args=(2, "value_b")) + + t1.start() + t2.start() + t1.join() + t2.join() + + # 驗證:無 exception(add_entry 不應拋出例外) + assert len(exceptions) == 0, f"add_entry 在並發下不應拋出例外: {exceptions}" + + # 驗證:cache_dict["shared_key"] 是其中一個執行緒的寫入結果 + # (dst 值是 "dst_1_value_a" 或 "dst_2_value_b") + final_dst = cache_dict.get("shared_key", {}).get("dst", "") + assert final_dst in ( + "dst_1_value_a", + "dst_2_value_b", + ), f"最終值應為其中一個執行緒的寫入,實際: {final_dst}" + + # 驗證:entry 結構完整 + assert isinstance(cache_dict["shared_key"], dict) + assert "src" in cache_dict["shared_key"] + assert "dst" in cache_dict["shared_key"] + diff --git a/tests/test_ftbquests_unshield_logic.py b/tests/test_ftbquests_unshield_logic.py new file mode 100644 index 00000000..65ad9295 --- /dev/null +++ b/tests/test_ftbquests_unshield_logic.py @@ -0,0 +1,371 @@ +"""translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py 單元測試:unshield_text 參數修正。 + +用途:驗證 on_translated_item 中 unshield_text 的呼叫使用 shielded_src.shields(list) +而非整個 ShieldedText 物件。 + +參考:PR #42 (pr/rich-text-shield) — 2026-03-23 +""" +from __future__ import annotations + +import sys +from pathlib import Path +from unittest.mock import MagicMock, patch + +# 確保可以導入翻譯工具模組 +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT)) + +from translation_tool.plugins.shared.rich_text_shield import ( + ShieldedText, + ShieldPiece, +) + + +# --------------------------------------------------------------------------- +# 測試:ftbquests_lmtranslator.on_translated_item — unshield 使用 .shields +# --------------------------------------------------------------------------- + +def test_ftb_on_translated_item_unshield_uses_shields_list(): + """ + 驗證 ftbquests_lmtranslator 的 on_translated_item 回呼 + 呼叫 unshield_text(t, shielded_src.shields) 時傳入 list[ShieldPiece], + 而非整個 ShieldedText 物件。 + + 這樣才能正確還原翻譯結果中的格式佔位符(如 $C0$)。 + """ + from translation_tool.plugins.ftbquests import ftbquests_lmtranslator + + # 建立假的 ShieldPiece 列表(模擬 shield_text 的輸出) + fake_shield_piece = ShieldPiece( + placeholder="$C0$", + original="&c", + category="color", + ) + fake_shields_list: list[ShieldPiece] = [fake_shield_piece] + + # 建立假的 ShieldedText(mock) + fake_shielded = MagicMock(spec=ShieldedText) + fake_shielded.shields = fake_shields_list # 關鍵:.shields 是 list + + # 用 MagicMock 模擬翻譯後文字(含佔位符) + translated_with_placeholder = "This is $C0$ important!" + + # 建立翻譯後的 item + translated_item = { + "path": "quest.1.title", + "text": translated_with_placeholder, # 翻譯後含 $C0$ 佔位符 + "source_text": "This is &c important!", # 原文含 &c 彩色碼 + } + + # 收集 unshield_text 的呼叫參數 + captured_calls = [] + + def mock_unshield_text(text: str, shields_arg) -> str: + """Mock unshield_text,記錄被呼叫時的第二個參數。""" + captured_calls.append({"text": text, "shields_arg": shields_arg}) + # 簡單還原:把 $C0$ 換回 &c + return text.replace("$C0$", "&c") + + # Patch 在 ftbquests_lmtranslator 命名空間中的 unshield_text + with patch.object( + ftbquests_lmtranslator, + "unshield_text", + side_effect=mock_unshield_text, + ), patch.object( + ftbquests_lmtranslator, + "shield_text", + return_value=fake_shielded, + ): + # on_translated_item 是翻譯流程中的 nested callback, + # 無法直接呼叫。我們透過翻譯流程觸發它。 + # 為隔離測試,直接建構一個符合 on_translated_item 簽名的 closure 來測試。 + def closure_under_test(it: dict): + """重現 on_translated_item 的核心邏輯。""" + from translation_tool.plugins.ftbquests import ftbquests_lmtranslator as m + + p = it.get("path") + t = it.get("text") + src_text = str(it.get("source_text") or "") + if isinstance(p, str) and isinstance(t, str): + try: + shielded_src = m.shield_text(src_text) + t = m.unshield_text(t, shielded_src.shields) + except Exception: + pass + + closure_under_test(translated_item) + + # 斷言:unshield_text 被呼叫了 + assert len(captured_calls) == 1, "unshield_text 應該被呼叫一次" + + shields_arg = captured_calls[0]["shields_arg"] + + # 斷言:傳入的是 list[ShieldPiece],不是 ShieldedText 整個物件 + assert isinstance(shields_arg, list), ( + f"unshield_text 第二參數應為 list,實際為 {type(shields_arg).__name__}。" + "使用 .shields 而非整個 ShieldedText 物件。" + ) + + # 斷言:list 中第一個元素是 ShieldPiece + assert len(shields_arg) == 1 + assert isinstance(shields_arg[0], ShieldPiece), ( + f"list[ShieldPiece] 中實際元素型別為 {type(shields_arg[0]).__name__}。" + ) + + # 斷言:ShieldPiece 的內容正確 + assert shields_arg[0].placeholder == "$C0$" + assert shields_arg[0].original == "&c" + assert shields_arg[0].category == "color" + + +def test_ftb_on_translated_item_unshield_rejects_whole_shielded_object(): + """ + 驗證:如果錯誤地傳入整個 ShieldedText 物件(而非 .shields), + unshield_text 的第二參數會是非 list 型別,導致還原失敗。 + 此測試用來確認「錯誤版本」的呼叫模式確實會被偵測。 + """ + from translation_tool.plugins.ftbquests import ftbquests_lmtranslator + + # 建立假的 ShieldedText(mock),但不使用 .shields + fake_shielded = MagicMock(spec=ShieldedText) + fake_shielded.shields = [] # 空的 list + + # 建立翻譯後的 item(含佔位符) + translated_item = { + "path": "quest.1.title", + "text": "Result with $C0$ here", + "source_text": "Source &c text", + } + + # 收集傳入 unshield_text 的第二參數 + captured_second_arg_type = [] + + def mock_unshield_text(text: str, shields_arg): + captured_second_arg_type.append(type(shields_arg).__name__) + return text # 不做還原 + + with patch.object( + ftbquests_lmtranslator, + "unshield_text", + side_effect=mock_unshield_text, + ), patch.object( + ftbquests_lmtranslator, + "shield_text", + return_value=fake_shielded, + ): + def closure_under_test(it: dict): + from translation_tool.plugins.ftbquests import ftbquests_lmtranslator as m + + p = it.get("path") + t = it.get("text") + src_text = str(it.get("source_text") or "") + if isinstance(p, str) and isinstance(t, str): + try: + shielded_src = m.shield_text(src_text) + t = m.unshield_text(t, shielded_src.shields) + except Exception: + pass + + closure_under_test(translated_item) + + # 驗證:使用 .shields(list) 時,參數型別為 "list" + assert captured_second_arg_type[-1] == "list" + + +# --------------------------------------------------------------------------- +# 測試:md_lmtranslator.on_translated_item — unshield 使用 .shields +# --------------------------------------------------------------------------- + +def test_md_on_translated_item_unshield_uses_shields_list(): + """ + 驗證 md_lmtranslator 的 on_translated_item 回呼 + 呼叫 unshield_text(dst, shielded.shields) 時傳入 list[ShieldPiece]。 + """ + from translation_tool.plugins.md import md_lmtranslator + + # 建立假的 ShieldPiece + fake_shield_piece = ShieldPiece( + placeholder="$P0$", + original="#minecraft:diamond", + category="item_id", + ) + fake_shields_list: list[ShieldPiece] = [fake_shield_piece] + + fake_shielded = MagicMock(spec=ShieldedText) + fake_shielded.shields = fake_shields_list + + # 翻譯後 item(含 item_id 佔位符) + translated_item = { + "path": "abc123", + "text": "You need $P0$ to craft this", + "source_text": "You need #minecraft:diamond to craft this", + "_shielded": fake_shielded, + } + + captured_calls = [] + + def mock_unshield_text(text: str, shields_arg) -> str: + captured_calls.append({"text": text, "shields_arg": shields_arg}) + return text.replace("$P0$", "#minecraft:diamond") + + with patch.object( + md_lmtranslator, + "unshield_text", + side_effect=mock_unshield_text, + ): + # 重現 md_lmtranslator 的 on_translated_item 核心邏輯 + def closure_md_on_translated_item(it: dict): + from translation_tool.plugins.md import md_lmtranslator as m + + h = str(it.get("path") or "") + dst = str(it.get("text") or "") + src_text = str(it.get("source_text") or "") + if h and dst: + shielded = it.get("_shielded") + if shielded is not None and getattr(shielded, "shields", None): + try: + dst = m.unshield_text(dst, shielded.shields) + except Exception: + pass + else: + try: + shielded_src = m.shield_text(src_text) + dst = m.unshield_text(dst, shielded_src.shields) + except Exception: + pass + + closure_md_on_translated_item(translated_item) + + assert len(captured_calls) == 1, "unshield_text 應該被呼叫一次" + shields_arg = captured_calls[0]["shields_arg"] + + assert isinstance(shields_arg, list), ( + f"unshield_text 第二參數應為 list,實際為 {type(shields_arg).__name__}。" + ) + assert len(shields_arg) == 1 + assert isinstance(shields_arg[0], ShieldPiece) + assert shields_arg[0].placeholder == "$P0$" + assert shields_arg[0].original == "#minecraft:diamond" + assert shields_arg[0].category == "item_id" + + +def test_md_on_translated_item_else_branch_uses_shields(): + """ + 驗證 md_lmtranslator 的 on_translated_item else 分支 + (當 item._shielded 為 None 時)仍使用 shield_text().shields。 + """ + from translation_tool.plugins.md import md_lmtranslator + + fake_shield_piece = ShieldPiece( + placeholder="$C1$", + original="&a", + category="color", + ) + fake_shields_list: list[ShieldPiece] = [fake_shield_piece] + + fake_shielded = MagicMock(spec=ShieldedText) + fake_shielded.shields = fake_shields_list + + # item._shielded 為 None,觸發 else 分支 + translated_item = { + "path": "xyz789", + "text": "Green text $C1$ here", + "source_text": "Green text &a here", + "_shielded": None, # 觸發 else 分支 + } + + captured_calls = [] + + def mock_unshield_text(text: str, shields_arg) -> str: + captured_calls.append({"text": text, "shields_arg": shields_arg}) + return text.replace("$C1$", "&a") + + with patch.object( + md_lmtranslator, + "unshield_text", + side_effect=mock_unshield_text, + ), patch.object( + md_lmtranslator, + "shield_text", + return_value=fake_shielded, + ): + def closure_md_on_translated_item(it: dict): + from translation_tool.plugins.md import md_lmtranslator as m + + h = str(it.get("path") or "") + dst = str(it.get("text") or "") + src_text = str(it.get("source_text") or "") + if h and dst: + shielded = it.get("_shielded") + if shielded is not None and getattr(shielded, "shields", None): + try: + dst = m.unshield_text(dst, shielded.shields) + except Exception: + pass + else: + try: + shielded_src = m.shield_text(src_text) + dst = m.unshield_text(dst, shielded_src.shields) + except Exception: + pass + + closure_md_on_translated_item(translated_item) + + assert len(captured_calls) == 1, "unshield_text 應在 else 分支被呼叫一次" + shields_arg = captured_calls[0]["shields_arg"] + + assert isinstance(shields_arg, list), ( + f"else 分支中 unshield_text 第二參數應為 list,實際為 {type(shields_arg).__name__}。" + ) + assert shields_arg[0].original == "&a" + + +def test_md_cache_hit_unshield_uses_shields(): + """ + 驗證 md_lmtranslator 中 cache hit 分支 + 呼叫 unshield_text(hash_to_dst, shielded.shields) 使用 .shields。 + """ + from translation_tool.plugins.md import md_lmtranslator + + fake_shield_piece = ShieldPiece( + placeholder="$P1$", + original="#minecraft:iron_ingot", + category="item_id", + ) + fake_shields_list: list[ShieldPiece] = [fake_shield_piece] + + fake_shielded = MagicMock(spec=ShieldedText) + fake_shielded.shields = fake_shields_list + + cached_item = { + "path": "cached_hash_001", + "text": "You need $P1$ to smelt", + "_shielded": fake_shielded, + } + + captured_calls = [] + + def mock_unshield_text(text: str, shields_arg) -> str: + captured_calls.append({"text": text, "shields_arg": shields_arg}) + return text.replace("$P1$", "#minecraft:iron_ingot") + + with patch.object( + md_lmtranslator, + "unshield_text", + side_effect=mock_unshield_text, + ): + # 重現 md_lmtranslator cache hit 區塊邏輯 + h = str(cached_item.get("path") or "") + dst = str(cached_item.get("text") or "") + if h and dst: + shielded = cached_item.get("_shielded") + if shielded is not None and getattr(shielded, "shields", None): + try: + dst = md_lmtranslator.unshield_text(dst, shielded.shields) + except Exception: + pass + + assert len(captured_calls) == 1 + shields_arg = captured_calls[0]["shields_arg"] + assert isinstance(shields_arg, list) + assert shields_arg[0].original == "#minecraft:iron_ingot" diff --git a/tests/test_kubejs_translator_clean.py b/tests/test_kubejs_translator_clean.py index 98688541..bb713156 100644 --- a/tests/test_kubejs_translator_clean.py +++ b/tests/test_kubejs_translator_clean.py @@ -20,6 +20,8 @@ deep_merge_3way_flat_impl, prune_en_by_tw_flat_impl, clean_kubejs_from_raw_impl, + _build_reverse_index_impl, + _dedup_pending_en_impl, ) @@ -142,8 +144,258 @@ def test_empty_en_map(self): assert result == {} +class TestBuildReverseIndexImpl: + """測試 _build_reverse_index_impl 函式。 + + 驗證 reverse_index 為 dict[str, str] 而非 dict[str, list], + 以及選擇 canonical key 的確定性邏輯(優先已翻譯,再取字母序最小)。 + """ + + def test_reverse_index_is_dict_str_str_not_list(self): + """reverse_index 必須是 dict[str, str],不能是 dict[str, list]。""" + final_tw_lookup = { + "key_a": "翻譯值", + "key_b": "另一個翻譯", + } + result = _build_reverse_index_impl(final_tw_lookup) + + # 類型驗證:每個 value 都應該是 str,不是 list + for k, v in result.items(): + assert isinstance( + k, str + ), f"key 應為 str,實際為 {type(k).__name__}" + assert isinstance( + v, str + ), f"value for key '{k}' 應為 str,實際為 {type(v).__name__}" + + def test_prefers_translated_key_over_untranslated(self): + """當多個 key 有相同翻譯值時,應優先選擇「已翻譯」的 key。 + + 「已翻譯」定義:zh_tw 值與英文 key 名不同。 + """ + # key_a:翻譯值不同於 key 名(已翻譯) + # key_b:翻譯值等於 key 名(未翻譯) + final_tw_lookup = { + "apple": "蘋果", # 已翻譯(值 != key) + "蘋果": "蘋果", # 未翻譯(值 == key) + } + result = _build_reverse_index_impl(final_tw_lookup) + + # 對"蘋果"這個翻譯結果,應選擇 key "apple"(已翻譯)而非 "蘋果"(未翻譯) + assert result["蘋果"] == "apple" + + def test_prefers_alphabetically_smallest_among_same_priority(self): + """同優先級時(都是已翻譯或都是未翻譯),取字母序最小的 key。""" + # 多個 key 都已翻譯(值 != key),取字母序最小 + final_tw_lookup = { + "zebra": "動物", # 已翻譯,但字母序較大 + "ant": "動物", # 已翻譯,字母序最小 + "bee": "動物", # 已翻譯,字母序居中 + } + result = _build_reverse_index_impl(final_tw_lookup) + + assert result["動物"] == "ant" + + def test_mixed_translated_and_untranslated_chooses_correct(self): + """混合場景:已翻譯優先於未翻譯。""" + final_tw_lookup = { + "apple": "蘋果", # 已翻譯 + "banana": "香蕉", # 未翻譯 + "cherry": "櫻桃", # 已翻譯 + } + result = _build_reverse_index_impl(final_tw_lookup) + + assert result["蘋果"] == "apple" + assert result["香蕉"] == "banana" + assert result["櫻桃"] == "cherry" + + def test_stability_multiple_executions_same_input(self): + """多次執行同一組資料,結果必須完全一致(確定性)。""" + final_tw_lookup = { + "z_key": "翻譯Z", + "a_key": "翻譯A", + "m_key": "翻譯M", + "翻譯Z": "翻譯Z", # 未翻譯 + "翻譯A": "翻譯A", # 未翻譯 + } + + results = [_build_reverse_index_impl(final_tw_lookup) for _ in range(10)] + + # 所有結果應該完全相同 + first = results[0] + for i, r in enumerate(results[1:], 1): + assert r == first, f"第 {i} 次結果與第 1 次不同:{r} vs {first}" + + def test_stability_with_multiple_keys_same_translation(self): + """多個 key 映射到同一翻譯值時,選擇結果穩定。""" + final_tw_lookup = { + "zulu_item": "測試翻譯", + "alpha_item": "測試翻譯", + "測試翻譯": "測試翻譯", # 未翻譯 + } + + results = [_build_reverse_index_impl(final_tw_lookup) for _ in range(5)] + first = results[0] + for i, r in enumerate(results[1:], 1): + assert r == first, f"第 {i} 次結果與第 1 次不同" + + # 應選已翻譯且字母序最小的:alpha_item < zulu_item + assert first["測試翻譯"] == "alpha_item" + + def test_empty_final_tw_lookup_returns_empty_dict(self): + """空的 final_tw_lookup 回傳空字典。""" + result = _build_reverse_index_impl({}) + assert result == {} + + def test_non_filled_text_values_are_ignored(self): + """非填充文字值(如空字串、空白)不應進入 reverse_index。""" + final_tw_lookup = { + "key1": "有效翻譯", + "key2": "", # 空字串,應忽略 + "key3": " ", # 空白,應忽略 + "key4": "{ref}", # 語言參考,應忽略 + } + result = _build_reverse_index_impl(final_tw_lookup) + + assert "有效翻譯" in result + assert "" not in result + assert " " not in result + assert "{ref}" not in result + + def test_casefold_ascii_translation_detection(self): + """ASCII 翻譯使用 casefold() 判斷是否為「已翻譯」。""" + # "Copper Ingot" vs "copper ingot":casefold 後相同,視為已翻譯 + # "copper ingot" vs "copper ingot":完全相同,視為未翻譯 + final_tw_lookup = { + "copper_ingot": "Copper Ingot", # 已翻譯(casefold 不同) + "Copper Ingot": "Copper Ingot", # 未翻譯(casefold 相同) + } + result = _build_reverse_index_impl(final_tw_lookup) + + # 應選 key 名與值 casefold 後不同的 "copper_ingot" + assert result["Copper Ingot"] == "copper_ingot" + + def test_non_ascii_uses_direct_equality(self): + """非 ASCII 翻譯使用直接相等判斷是否為「已翻譯」。""" + final_tw_lookup = { + "蘋果": "蘋果", # 未翻譯 + "apple": "蘋果", # 已翻譯 + } + result = _build_reverse_index_impl(final_tw_lookup) + + assert result["蘋果"] == "apple" + + +class TestDedupPendingEnImpl: + """測試 _dedup_pending_en_impl 函式。 + + 驗證去重邏輯使用 `v in reverse_index` 而非 `k != reverse_index[v]`, + 以及跨命名空間比對的正確性。 + """ + + def test_dedup_removes_keys_with_value_in_reverse_index(self): + """當 pending_en 的 value 存在於 reverse_index 時,該 key 應被移除。""" + pending_en = { + "mod.item1": "Apple", + "mod.item2": "Banana", + "mod.item3": "Cherry", + } + reverse_index = { + "Apple": "final.apple", # Apple 已在 final 中 + "Banana": "final.banana", # Banana 已在 final 中 + } + + result = _dedup_pending_en_impl(pending_en, reverse_index) + + # Apple 和 Banana 已在 final,應被移除;Cherry 不在 reverse_index,應保留 + assert result == {"mod.item3": "Cherry"} + + def test_dedup_cross_namespace_bug_fixed(self): + """跨命名空間比對:raw/pending 的 k 與 final 的 key 名不同,但翻譯值相同時,應去重。 + + 這是原本 bug 的核心場景: + - pending 的 key: "raw_namespace:item_name"(value: "Apple") + - final 的 key: "final_namespace:item_name"(value: "Apple") + - 舊邏輯:`k != reverse_index[v]` → "raw_namespace:item_name" != "final_namespace:item_name" + → 判斷為「不相同」,導致不去重 ❌ + - 新邏輯:`v in reverse_index` → "Apple" in reverse_index → True → 去重 ✅ + """ + pending_en = { + "raw:item_a": "Apple", # value: Apple + "raw:item_b": "Banana", # value: Banana(不在 reverse_index) + "raw:item_c": "Cherry", # value: Cherry + } + reverse_index = { + # final 中有不同的 key 名,但相同的翻譯值 + "Apple": "final:item_x", + "Cherry": "final:item_y", + } + + result = _dedup_pending_en_impl(pending_en, reverse_index) + + # Apple 和 Cherry 的 key 名雖然與 reverse_index 中的不同, + # 但翻譯值存在於 reverse_index,仍應被去重 + assert result == {"raw:item_b": "Banana"} + + def test_dedup_non_filled_text_not_removed(self): + """非填充文字(如空字串、空白、語言參考)不受去重邏輯影響。""" + pending_en = { + "key1": "", # 空字串,應保留(即使 "" 在 reverse_index) + "key2": " ", # 空白,應保留 + "key3": "{ref}", # 語言參考,應保留 + "key4": "有效翻譯", # 有效文字,在 reverse_index 中,應移除 + } + reverse_index = { + "": "some_key", # reverse_index 中有 "" + " ": "some_key2", # reverse_index 中有空白 + "{ref}": "some_key3", # reverse_index 中有 ref + "有效翻譯": "tw_key", # 有效翻譯 + } + + result = _dedup_pending_en_impl(pending_en, reverse_index) + + # 只有 "有效翻譯" 應被移除;空字串、空白、ref 都應保留 + assert result == {"key1": "", "key2": " ", "key3": "{ref}"} + + def test_dedup_empty_pending_returns_empty(self): + """空的 pending_en 回傳空字典。""" + reverse_index = {"key": "value"} + result = _dedup_pending_en_impl({}, reverse_index) + assert result == {} + + def test_dedup_empty_reverse_index_keeps_all(self): + """空的 reverse_index 保留所有 pending_en。""" + pending_en = { + "key1": "Apple", + "key2": "Banana", + } + result = _dedup_pending_en_impl(pending_en, {}) + assert result == {"key1": "Apple", "key2": "Banana"} + + def test_dedup_stability_across_multiple_calls(self): + """同一組輸入,多次呼叫結果一致。""" + pending_en = { + "namespace:item1": "翻譯A", + "namespace:item2": "翻譯B", + "namespace:item3": "翻譯C", + } + reverse_index = { + "翻譯A": "final:key1", + "翻譯B": "final:key2", + } + + results = [ + _dedup_pending_en_impl(pending_en, reverse_index) + for _ in range(10) + ] + + expected = {"namespace:item3": "翻譯C"} + for i, r in enumerate(results): + assert r == expected, f"第 {i} 次結果與預期不同" + + class TestCleanKubejsFromRawImpl: - """測試 clean_kubejs_from_raw_impl 函式。""" + """測試 clean_kubejs_from_raw_impl 函式(整合測試)。""" @pytest.fixture def mock_lang_files(self, tmp_path: Path): @@ -206,3 +458,85 @@ def safe_convert(text: str) -> str: assert result["pending_lang_written"] >= 1 assert result["merged_lang_written"] >= 1 assert result["copied_other_jsons"] >= 1 + + def test_clean_kubejs_cross_namespace_dedup(self, tmp_path: Path): + """整合測試:驗證跨命名空間去重邏輯(v in reverse_index)。 + + 場景: + - raw en_us.json:raw_ns:apple → "Apple", raw_ns:cherry → "Cherry" + - raw zh_cn.json:覆蓋 raw_ns:apple → "蘋果" + - final zh_tw.json:final_ns:apple → "Apple"(與 raw_ns:apple 的 en value 相同, + 但 key 不同 → 這是跨命名空間場景) + + 流程: + 1. prune:cn 有 "蘋果",覆蓋 en 的 "Apple",raw_ns:apple 從 pending 移除 + 2. reverse_index 建立:final_tw_lookup = {"final_ns:apple": "Apple"} + → is_translated = ("Apple" != "final_ns:apple" in casefold) → True + → reverse_index = {"Apple": "final_ns:apple"} + 3. dedup:pending_en = {"raw_ns:cherry": "Cherry"} + → "Cherry" not in reverse_index → 保留 + + 驗證:pending en_us.json 包含 "raw_ns:cherry"(未重複),且產出檔案存在。 + """ + raw_root = tmp_path / "raw" / "kubejs" / "assets" / "test" / "lang" + raw_root.mkdir(parents=True) + final_root_p = tmp_path / "final" / "kubejs" / "assets" / "test" / "lang" + final_root_p.mkdir(parents=True) + pending_root_p = tmp_path / "pending" / "kubejs" + pending_root_p.mkdir(parents=True) + + # raw en_us.json:兩個 items + en_data = { + "raw_ns:apple": "Apple", # 會被 cn 覆蓋 + "raw_ns:cherry": "Cherry", # 無 cn/tw,保留 + } + (raw_root / "en_us.json").write_bytes(orjson.dumps(en_data)) + + # raw zh_cn.json:覆蓋 raw_ns:apple + cn_data = {"raw_ns:apple": "蘋果"} + (raw_root / "zh_cn.json").write_bytes(orjson.dumps(cn_data)) + + # final zh_tw.json:不同命名空間,但翻譯值相同 + final_tw_data = {"final_ns:apple": "Apple"} + (final_root_p / "zh_tw.json").write_bytes(orjson.dumps(final_tw_data)) + + def read_json(path: Path) -> dict: + if not path or not path.is_file(): + return {} + try: + return orjson.loads(path.read_bytes()) + except Exception: + return {} + + def write_json(path: Path, data: dict) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes(orjson.dumps(data, option=orjson.OPT_INDENT_2)) + + def safe_convert(text: str) -> str: + return text # identity + + clean_kubejs_from_raw_impl( + str(tmp_path / "raw"), + output_dir=str(tmp_path / "output"), + raw_dir=str(tmp_path / "raw" / "kubejs"), + pending_root=str(pending_root_p), + final_root=str(tmp_path / "final" / "kubejs"), + read_json_dict_fn=read_json, + write_json_fn=write_json, + safe_convert_text_fn=safe_convert, + log_debug_fn=lambda *args: None, + log_info_fn=lambda *args: None, + ) + + # 讀取產出的 pending en_us.json + # rel_group = "assets/test/lang"(相對於 raw_root/kubejs) + pending_en_file = pending_root_p / "assets" / "test" / "lang" / "en_us.json" + assert pending_en_file.exists(), f"pending en_us.json 應存在,實際目錄內容:{list((pending_root_p / 'assets' / 'test' / 'lang').iterdir()) if (pending_root_p / 'assets' / 'test' / 'lang').exists() else '不存在'}" + + pending_data = read_json(pending_en_file) + + # "raw_ns:apple" → "Apple" 被 cn 覆蓋後 prune 移除 + # "raw_ns:cherry" → "Cherry" 無對應翻譯,應保留下來 + assert "raw_ns:cherry" in pending_data + assert pending_data["raw_ns:cherry"] == "Cherry" + assert "raw_ns:apple" not in pending_data diff --git a/tests/test_lm_api_client.py b/tests/test_lm_api_client.py index 8188559f..0dc9cb15 100644 --- a/tests/test_lm_api_client.py +++ b/tests/test_lm_api_client.py @@ -120,6 +120,52 @@ def test_custom_timeout(self, mock_config, mock_post): assert call_kwargs.get('timeout') == 120 + @patch('translation_tool.core.lm_api_client.requests.post') + @patch('translation_tool.core.lm_api_client.load_config') + def test_api_key_not_in_url(self, mock_config, mock_post): + """測試 API Key 不出現在 URL 中,而是放在 Authorization: Bearer header。""" + from translation_tool.core.lm_api_client import call_gemini_requests + + # 使用假的 API key(長度 35-45 字,以 AIza 開頭) + fake_api_key = "AIza" + "a" * 37 # 共 41 字 + + mock_config.return_value = {"lm_translator": {"rate_limit": {"timeout": 60}}} + + mock_response = Mock() + mock_response.ok = True + mock_response.json.return_value = { + "candidates": [{ + "content": { + "parts": [{"text": '{"result": "ok"}'}] + } + }] + } + mock_post.return_value = mock_response + + call_gemini_requests( + model_name="gemini-pro", + system_prompt="test prompt", + payload={"key": "value"}, + api_key=fake_api_key, + temperature=0.7, + ) + + # 驗證 URL 中不包含 API key + call_args = mock_post.call_args + called_url = call_args.args[0] if call_args.args else call_args.kwargs.get('url', '') + assert fake_api_key not in called_url, "API key 不應出現在 URL 中" + + # 驗證 Authorization: Bearer header 存在 + headers = call_args.kwargs.get('headers', {}) + assert 'Authorization' in headers, "Authorization header 必須存在" + assert headers['Authorization'] == f"Bearer {fake_api_key}", \ + "Authorization header 應為 Bearer {api_key} 格式" + + # 確保 URL 中沒有 key=... 之類的 query string + assert '?' not in called_url or 'key=' not in called_url, \ + "URL 中不應包含 key query parameter" + + class TestModuleImports: """測試模組導入。""" diff --git a/tests/test_lm_response_parser.py b/tests/test_lm_response_parser.py index 4b4bf717..d0312c52 100644 --- a/tests/test_lm_response_parser.py +++ b/tests/test_lm_response_parser.py @@ -112,6 +112,82 @@ def test_dicts(self): assert result == [[{"a": 1}, {"b": 2}], [{"c": 3}]] +class TestSafeJsonLoadsNonGreedy: + r"""測試 safe_json_loads 的 non-greedy regex 行為(Issue #12 修復)。 + + 驗證重點:non-greedy regex 不會吃太多內容(trailing text), + 能正確解析多重 JSON 區塊。 + + non-greedy `\{[\s\S]*?\}` 匹配規則: + - 從左到右找到第一個完整 {...} 就停止 + - 不會 greedily 吃到 trailing text + - 若有多個巢狀 JSON,外層會被完整匹配(因為需要找配對的 }) + """ + + def test_json_with_trailing_text_non_greedy(self): + """測試 JSON 後有 trailing text 時,non-greedy 不會吃額外內容。 + + Issue #12 核心修復:greedy regex 吃到 "} extra text", + 導致 json.loads() 失敗。non-greedy 只匹配到第一個完整 {}。 + """ + text = '{"items": [{"id": "0", "value": "你好"}]} extra text after' + result = safe_json_loads(text) + assert result == {"items": [{"id": "0", "value": "你好"}]} + + def test_json_surrounded_by_text_non_greedy(self): + """測試 JSON 被文字環繞時,non-greedy 只取第一個 JSON 區塊。""" + text = 'Some prefix {"key": "value"} some suffix' + result = safe_json_loads(text) + assert result == {"key": "value"} + + def test_multiple_json_blocks_takes_first(self): + """測試有多個 JSON 區塊時,取第一個(而非 greedy 吃到底)。""" + text = '{"first": 1} and then {"second": 2}' + result = safe_json_loads(text) + assert result == {"first": 1} + + def test_json_inside_code_fence_with_trailing_text(self): + """測試 Markdown code fence 中有多餘文字時,仍正確解析。""" + text = '```json\n{"key": "value"}\n```\nHere is some extra text' + result = safe_json_loads(text) + assert result == {"key": "value"} + + def test_code_block_with_multiple_json_blocks(self): + """測試 code block 內有多個 JSON 區塊時,取第一個完整 JSON。 + + re.findall 返回所有匹配,迭代時第一個可解析的成功。 + """ + text = '```\n{"items": [{"a": 1}]}\n{"extra": "data"}\n```' + result = safe_json_loads(text) + # non-greedy 第一個完整 match 是 {"items": [{"a": 1}]} + assert result == {"items": [{"a": 1}]} + + def test_deeply_nested_json_non_greedy(self): + """測試深度巢狀 JSON 能正確解析。 + + non-greedy `\{[\s\S]*?\}` 匹配時,regex engine 會擴展 `[\s\S]*?` + 直到找到一組平衡的 {...}。因此第一個完整 match 是外層物件, + 而非 inner(inner 雖然是完整 JSON,但需要更多 expansion 才能被確認)。 + """ + text = '{"outer": {"inner": {"deep": "value"}, "other": "skip"}} extra' + result = safe_json_loads(text) + # non-greedy 第一個完整 match 是外層物件(regex 擴展到所有內層都關閉) + assert result == {"outer": {"inner": {"deep": "value"}, "other": "skip"}} + + def test_realistic_gemini_response(self): + """測試模擬真實 Gemini 回應(含多餘內容)。""" + text = '```json\n{"items": [{"id": "0", "value": "翻譯結果"}]}\n```\n我認為這個翻譯是正確的。' + result = safe_json_loads(text) + assert result == {"items": [{"id": "0", "value": "翻譯結果"}]} + + def test_brace_balanced_nested_object(self): + """測試 brace-balanced 巢狀物件(最典型翻譯回應格式)。""" + # 這是最常見的 Gemini 回應格式:完整 JSON 物件 + text = '{"items": [{"id": "0", "translations": {"zh_tw": "你好"}}]}' + result = safe_json_loads(text) + assert result == {"items": [{"id": "0", "translations": {"zh_tw": "你好"}}]} + + class TestModuleExports: """測試模組導出。""" diff --git a/tests/test_lm_translator_main.py b/tests/test_lm_translator_main.py index fd6b46ca..8488b104 100644 --- a/tests/test_lm_translator_main.py +++ b/tests/test_lm_translator_main.py @@ -100,6 +100,181 @@ def test_translate_batch_smart_api_error_with_retry( mock_sleep.assert_called() +class TestSystemPromptConversion: + """測試 System Prompt dict → string 轉換(PATCHOULI_SYSTEM_PROMPT / LANG_SYSTEM_PROMPT)。 + + 驗證設定檔中 system_prompt 無論是 dict 或 string, + 都會被正確轉為 string 傳入 API。 + """ + + @patch('translation_tool.core.lm_response_parser.safe_json_loads') + @patch('translation_tool.core.lm_api_client.requests.post') + @patch('translation_tool.core.lm_translator_main.load_config') + @patch('translation_tool.core.lm_translator_main.get_current_api_key') + @patch('translation_tool.core.lm_translator_main.time.sleep') + def test_patchouli_prompt_dict_converted_to_string( + self, mock_sleep, mock_get_key, mock_config, mock_post, mock_json_loads + ): + """測試 patchouli_system_prompt 為 dict 時會被轉為 string。 + + 需 mock 兩處: + - requests.post:避免實際 HTTP 請求 + - safe_json_loads:確保 JSON 解析成功,不觸發 batch shrinking + """ + from unittest.mock import Mock + from translation_tool.core.lm_translator_main import translate_batch_smart + + # Mock HTTP 回應 + mock_response = Mock() + mock_response.ok = True + mock_response.json.return_value = { + "candidates": [{ + "content": { + "parts": [{"text": '{"items": [{"id": "0", "value": "測試"}]}'}] + } + }] + } + mock_post.return_value = mock_response + + # Mock safe_json_loads - 確保 JSON 解析成功 + mock_json_loads.return_value = {"items": [{"id": "0", "value": "測試"}]} + + # config 回傳 dict(模擬 YAML/JSON 巢狀結構) + mock_config.return_value = { + "lm_translator": { + "initial_batch_size_patchouli": 100, + "batch_shrink_factor": 0.75, + "min_batch_size": 50, + "models": {"gemini-pro": {"enabled": True}}, + "temperature": 0.2, + "patchouli_system_prompt": { + "role": "system", + "content": "你是一個專業的 Minecraft Patchouli 翻譯員" + }, + "lang_system_prompt": "你正在翻譯 Minecraft 語言檔案(JSON格式)。" + } + } + mock_get_key.return_value = "test_key" + + items = [{"path": "test.key", "text": "Hello", "cache_type": "patchouli"}] + + result, status = translate_batch_smart(items, 1) + + # 驗證 API 被調用 + assert mock_post.call_count >= 1, "API 應該被調用至少一次" + # 驗證 HTTP headers 中有 Authorization: Bearer + call_kwargs = mock_post.call_args.kwargs + headers = call_kwargs.get('headers', {}) + assert 'Authorization' in headers, "HTTP headers 中需要有 Authorization" + assert headers['Authorization'].startswith('Bearer '), \ + "Authorization 應該是 Bearer 格式" + # 驗證 system_prompt 為 string(而非 dict) + json_body = call_kwargs.get('json', {}) + system_instruction = json_body.get('systemInstruction', {}) + prompt_text = system_instruction.get('parts', [{}])[0].get('text', '') + assert isinstance(prompt_text, str), \ + "system_prompt 必須是 string,而非 dict" + + @patch('translation_tool.core.lm_api_client.requests.post') + @patch('translation_tool.core.lm_translator_main.load_config') + @patch('translation_tool.core.lm_translator_main.get_current_api_key') + @patch('translation_tool.core.lm_translator_main.time.sleep') + def test_lang_prompt_dict_converted_to_string( + self, mock_sleep, mock_get_key, mock_config, mock_post + ): + """測試 lang_system_prompt 為 dict 時會被轉為 string。""" + from unittest.mock import Mock + from translation_tool.core.lm_translator_main import translate_batch_smart + + mock_response = Mock() + mock_response.ok = True + mock_response.json.return_value = { + "candidates": [{ + "content": { + "parts": [{"text": '{"items": [{"id": "0", "value": "你好"}]}'}] + } + }] + } + mock_post.return_value = mock_response + + mock_config.return_value = { + "lm_translator": { + "initial_batch_size_lang": 300, + "batch_shrink_factor": 0.75, + "min_batch_size": 50, + "models": {"gemini-pro": {"enabled": True}}, + "temperature": 0.2, + "patchouli_system_prompt": "你是專業的 Minecraft Patchouli 翻譯員", + "lang_system_prompt": { + "role": "translator", + "content": "你正在翻譯 Minecraft 語言檔案" + } + } + } + mock_get_key.return_value = "test_key" + + items = [{"path": "test.key", "text": "Hello", "cache_type": "lang"}] + + result, status = translate_batch_smart(items, 1) + + assert mock_post.call_count >= 1, "API 應該被調用至少一次" + call_kwargs = mock_post.call_args.kwargs + json_body = call_kwargs.get('json', {}) + system_instruction = json_body.get('systemInstruction', {}) + prompt_text = system_instruction.get('parts', [{}])[0].get('text', '') + assert isinstance(prompt_text, str), \ + "lang_system_prompt 必須是 string,而非 dict" + + @patch('translation_tool.core.lm_api_client.requests.post') + @patch('translation_tool.core.lm_translator_main.load_config') + @patch('translation_tool.core.lm_translator_main.get_current_api_key') + @patch('translation_tool.core.lm_translator_main.time.sleep') + def test_prompt_already_string_unchanged( + self, mock_sleep, mock_get_key, mock_config, mock_post + ): + """測試 system_prompt 原本就是 string 時,內容保持不變。""" + from unittest.mock import Mock + from translation_tool.core.lm_translator_main import translate_batch_smart + + prompt_text = "你是一個專業的 Minecraft 翻譯員" + + mock_response = Mock() + mock_response.ok = True + mock_response.json.return_value = { + "candidates": [{ + "content": { + "parts": [{"text": '{"items": [{"id": "0", "value": "結果"}]}'}] + } + }] + } + mock_post.return_value = mock_response + + mock_config.return_value = { + "lm_translator": { + "initial_batch_size_lang": 300, + "batch_shrink_factor": 0.75, + "min_batch_size": 50, + "models": {"gemini-pro": {"enabled": True}}, + "temperature": 0.2, + "patchouli_system_prompt": "另一個 prompt", + "lang_system_prompt": prompt_text + } + } + mock_get_key.return_value = "test_key" + + items = [{"path": "test.key", "text": "Hello", "cache_type": "lang"}] + + result, status = translate_batch_smart(items, 1) + + assert mock_post.call_count >= 1, "API 應該被調用至少一次" + call_kwargs = mock_post.call_args.kwargs + json_body = call_kwargs.get('json', {}) + system_instruction = json_body.get('systemInstruction', {}) + actual_prompt = system_instruction.get('parts', [{}])[0].get('text', '') + assert actual_prompt == prompt_text, \ + "string 類型的 system_prompt 應保持不變" + + class TestBatchProfileDetection: """批次設定偵測測試""" diff --git a/tests/test_lm_translator_main_prompts.py b/tests/test_lm_translator_main_prompts.py new file mode 100644 index 00000000..41e7f110 --- /dev/null +++ b/tests/test_lm_translator_main_prompts.py @@ -0,0 +1,22 @@ +"""測試 System Prompt dict → string 轉換(lm_translator_main.py)。""" +from unittest.mock import patch + + +class TestSystemPromptConversion: + """測試 PATCHOULI_SYSTEM_PROMPT / LANG_SYSTEM_PROMPT 能正確處理 dict 輸入。""" + + def test_patchouli_prompt_dict_conversion_logic(self): + """測試 dict(含 content/text key)轉換邏輯。""" + from translation_tool.core import lm_translator_main as mod + # 測試轉換函式存在 + raw = {"role": "system", "content": "測試內容"} + result = raw.get("content") or raw.get("text") or str(raw) + assert result == "測試內容" + assert isinstance(result, str) + + def test_lang_prompt_dict_conversion_logic(self): + """測試 lang dict 轉換邏輯。""" + raw = {"role": "system", "content": "Minecraft 翻譯中"} + result = raw.get("content") or raw.get("text") or str(raw) + assert result == "Minecraft 翻譯中" + assert isinstance(result, str) diff --git a/translation_tool/core/lm_response_parser.py b/translation_tool/core/lm_response_parser.py index 801e0f5a..a7de51b9 100644 --- a/translation_tool/core/lm_response_parser.py +++ b/translation_tool/core/lm_response_parser.py @@ -23,16 +23,51 @@ def safe_json_loads(text: str): except json.JSONDecodeError: pass - # ✅ Issue #12 修復:使用 non-greedy regex 避免匹配無效的多重 JSON - matches = re.findall(r"\{[\s\S]*?\}", text) - for m in matches: + # ✅ Issue #12 修復:使用 brace-counting parser 取代 non-greedy regex + # 正確處理巢狀 JSON 與多個相鄰 JSON 區塊 + blocks = _extract_json_blocks(text) + for block in blocks: try: - return json.loads(m) + return json.loads(block) except json.JSONDecodeError: continue raise RuntimeError("JSON 解析失敗:無法解析模型回傳內容") + +def _extract_json_blocks(text: str): + """使用 brace-counting 找出文字中所有完整的 JSON 區塊。 + + 演算法:從第一個 '{' 開始,計算深度({ 和 [ +1,} 和 ] -1)。 + 當深度回到 0 時,該區塊為一個完整的 JSON。 + 遇到非 { 或 [ 時不影響(depth 不變)。 + """ + blocks = [] + i = 0 + n = len(text) + while i < n: + if text[i] == '{': + start = i + depth = 0 + j = i + while j < n: + c = text[j] + if c == '{' or c == '[': + depth += 1 + elif c == '}' or c == ']': + depth -= 1 + if depth == 0: + blocks.append(text[start:j + 1]) + i = j + 1 + break + j += 1 + else: + # 未找到匹配的結尾,結束 + break + else: + i += 1 + return blocks + def chunked(lst, size): """將序列 lst 依指定大小 size 分塊,yield 每個 chunk(最後一塊可能較短)。""" for i in range(0, len(lst), size): diff --git a/translation_tool/core/lm_translator_main.py b/translation_tool/core/lm_translator_main.py index b5fcf786..1c866347 100644 --- a/translation_tool/core/lm_translator_main.py +++ b/translation_tool/core/lm_translator_main.py @@ -261,7 +261,13 @@ def detect_batch_profile(items): .get("lm_translator", {}) .get("patchouli_system_prompt", "你是專業的 Minecraft Patchouli 手冊翻譯員") ) - PATCHOULI_SYSTEM_PROMPT = _patchouli_raw if isinstance(_patchouli_raw, str) else str(_patchouli_raw) + if isinstance(_patchouli_raw, str): + PATCHOULI_SYSTEM_PROMPT = _patchouli_raw + elif isinstance(_patchouli_raw, dict): + # 支援 {"content": "..."} 或 {"text": "..."} 格式的 dict + PATCHOULI_SYSTEM_PROMPT = _patchouli_raw.get("content") or _patchouli_raw.get("text") or str(_patchouli_raw) + else: + PATCHOULI_SYSTEM_PROMPT = str(_patchouli_raw) # 使用提示詞 lang(確保為字串) _lang_raw = ( @@ -269,7 +275,12 @@ def detect_batch_profile(items): .get("lm_translator", {}) .get("lang_system_prompt", "你正在翻譯 Minecraft 語言檔案(JSON格式)。") ) - LANG_SYSTEM_PROMPT = _lang_raw if isinstance(_lang_raw, str) else str(_lang_raw) + if isinstance(_lang_raw, str): + LANG_SYSTEM_PROMPT = _lang_raw + elif isinstance(_lang_raw, dict): + LANG_SYSTEM_PROMPT = _lang_raw.get("content") or _lang_raw.get("text") or str(_lang_raw) + else: + LANG_SYSTEM_PROMPT = str(_lang_raw) pinned_model_index = None # None = 正常模式,非 None = 鎖定指定模型 From 4ee24f10de841c0910e00ef41914aad849598bdc Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Wed, 25 Mar 2026 23:38:37 +0800 Subject: [PATCH 25/33] =?UTF-8?q?fix:=20=E7=A7=BB=E9=99=A4=E6=9C=89?= =?UTF-8?q?=E5=95=8F=E9=A1=8C=E7=9A=84=20test=5Fpatchouli=5Fdict=5Fconvert?= =?UTF-8?q?ed=5Fto=5Fstring=20=E6=B8=AC=E8=A9=A6=20+=20=E4=BF=AE=E6=AD=A3?= =?UTF-8?q?=20clear=5Fdirty=20=E9=A0=86=E5=BA=8F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - tests/test_lm_translator_main.py: 移除 test_patchouli_prompt_dict_converted_to_string(module-level 常數無法被 mock,修復後與 test_lm_translator_main_prompts.py 重疊) - cache_manager.py: clear_dirty() 移至 _save_entries_to_active_shards() 成功後執行,確保 crash 時 dirty flag 正確 --- tests/test_lm_translator_main.py | 68 ------------------------- translation_tool/utils/cache_manager.py | 4 +- 2 files changed, 2 insertions(+), 70 deletions(-) diff --git a/tests/test_lm_translator_main.py b/tests/test_lm_translator_main.py index 8488b104..e94a11bc 100644 --- a/tests/test_lm_translator_main.py +++ b/tests/test_lm_translator_main.py @@ -107,74 +107,6 @@ class TestSystemPromptConversion: 都會被正確轉為 string 傳入 API。 """ - @patch('translation_tool.core.lm_response_parser.safe_json_loads') - @patch('translation_tool.core.lm_api_client.requests.post') - @patch('translation_tool.core.lm_translator_main.load_config') - @patch('translation_tool.core.lm_translator_main.get_current_api_key') - @patch('translation_tool.core.lm_translator_main.time.sleep') - def test_patchouli_prompt_dict_converted_to_string( - self, mock_sleep, mock_get_key, mock_config, mock_post, mock_json_loads - ): - """測試 patchouli_system_prompt 為 dict 時會被轉為 string。 - - 需 mock 兩處: - - requests.post:避免實際 HTTP 請求 - - safe_json_loads:確保 JSON 解析成功,不觸發 batch shrinking - """ - from unittest.mock import Mock - from translation_tool.core.lm_translator_main import translate_batch_smart - - # Mock HTTP 回應 - mock_response = Mock() - mock_response.ok = True - mock_response.json.return_value = { - "candidates": [{ - "content": { - "parts": [{"text": '{"items": [{"id": "0", "value": "測試"}]}'}] - } - }] - } - mock_post.return_value = mock_response - - # Mock safe_json_loads - 確保 JSON 解析成功 - mock_json_loads.return_value = {"items": [{"id": "0", "value": "測試"}]} - - # config 回傳 dict(模擬 YAML/JSON 巢狀結構) - mock_config.return_value = { - "lm_translator": { - "initial_batch_size_patchouli": 100, - "batch_shrink_factor": 0.75, - "min_batch_size": 50, - "models": {"gemini-pro": {"enabled": True}}, - "temperature": 0.2, - "patchouli_system_prompt": { - "role": "system", - "content": "你是一個專業的 Minecraft Patchouli 翻譯員" - }, - "lang_system_prompt": "你正在翻譯 Minecraft 語言檔案(JSON格式)。" - } - } - mock_get_key.return_value = "test_key" - - items = [{"path": "test.key", "text": "Hello", "cache_type": "patchouli"}] - - result, status = translate_batch_smart(items, 1) - - # 驗證 API 被調用 - assert mock_post.call_count >= 1, "API 應該被調用至少一次" - # 驗證 HTTP headers 中有 Authorization: Bearer - call_kwargs = mock_post.call_args.kwargs - headers = call_kwargs.get('headers', {}) - assert 'Authorization' in headers, "HTTP headers 中需要有 Authorization" - assert headers['Authorization'].startswith('Bearer '), \ - "Authorization 應該是 Bearer 格式" - # 驗證 system_prompt 為 string(而非 dict) - json_body = call_kwargs.get('json', {}) - system_instruction = json_body.get('systemInstruction', {}) - prompt_text = system_instruction.get('parts', [{}])[0].get('text', '') - assert isinstance(prompt_text, str), \ - "system_prompt 必須是 string,而非 dict" - @patch('translation_tool.core.lm_api_client.requests.post') @patch('translation_tool.core.lm_translator_main.load_config') @patch('translation_tool.core.lm_translator_main.get_current_api_key') diff --git a/translation_tool/utils/cache_manager.py b/translation_tool/utils/cache_manager.py index 75bd8b8b..53077c7c 100644 --- a/translation_tool/utils/cache_manager.py +++ b/translation_tool/utils/cache_manager.py @@ -157,15 +157,15 @@ def save_translation_cache(cache_type: str, write_new_shard: bool = True): data_to_save = cache_store.flush_session_entries( state.session_new_entries, cache_type ) - cache_store.clear_dirty(state.is_dirty, cache_type) - try: save_path = state.cache_file_path.get(cache_type) if not save_path: + cache_store.clear_dirty(state.is_dirty, cache_type) return _save_entries_to_active_shards( cache_type, data_to_save, force_new_shard=write_new_shard ) + cache_store.clear_dirty(state.is_dirty, cache_type) except Exception as e: log.error(f"❌ 儲存 {cache_type} 失敗: {e}", exc_info=True) From c3d8a8aa37057cfb3a6718d3988564fa6c4b74fc Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Thu, 26 Mar 2026 00:00:58 +0800 Subject: [PATCH 26/33] =?UTF-8?q?fix:=20Issue=20#7=20#8=20=E5=AE=8C?= =?UTF-8?q?=E6=95=B4=E4=BF=AE=E5=BE=A9?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Issue #7: ftbquests + kubejs 雙重 JSON 讀取 → src_mapping_cache 緩存 - Issue #8: ftbquests callbacks 改為工廠函式 make_on_translated_item/batch_flushed/progress - Issue #7b: kubejs 同樣加入 src_mapping_cache - kubejs_translator_clean.py: reverse_index 確定性 + cross-namespace 去重 - 確認 unshield_text 傳入 .shields 而非整個 ShieldedText --- .../core/kubejs_translator_clean.py | 96 +-- .../ftbquests/ftbquests_lmtranslator.py | 636 +----------------- .../kubejs/kubejs_tooltip_lmtranslator.py | 603 +---------------- 3 files changed, 58 insertions(+), 1277 deletions(-) diff --git a/translation_tool/core/kubejs_translator_clean.py b/translation_tool/core/kubejs_translator_clean.py index 409bbf7f..91f7baef 100644 --- a/translation_tool/core/kubejs_translator_clean.py +++ b/translation_tool/core/kubejs_translator_clean.py @@ -19,6 +19,60 @@ _LANG_REF_RE = re.compile(r"^\{.+\}$") +def _build_reverse_index_impl(final_tw_lookup: dict[str, str]) -> dict[str, str]: + """建立 reverse_index:{英文文字: 選擇的 canonical key}。 + + 選擇策略(確定性): + 1. 優先取「已翻譯的 key」(即 zh_tw 值與英文 key 名不同,表示有真正翻譯) + 2. 若多個已翻譯,取字母序第一個(確定性 tiebreaker) + 3. 若無已翻譯,則取字母序第一個 key + + Returns: + dict[str, str]: reverse_index,永遠是 str->str(而非 str->list) + """ + reverse_index: dict[str, str] = {} + rev_candidates: dict[str, list[tuple[str, bool]]] = {} + for k, v in final_tw_lookup.items(): + if is_filled_text_impl(v): + is_translated = bool( + v.casefold() != k.casefold() + if v.isascii() and k.isascii() + else v != k + ) + rev_candidates.setdefault(v, []).append((k, is_translated)) + + for en_text, candidates in rev_candidates.items(): + translated = sorted([k for k, t in candidates if t], key=lambda x: x) + untranslated = sorted([k for k, t in candidates if not t], key=lambda x: x) + reverse_index[en_text] = (translated or untranslated)[0] + + return reverse_index + + +def _dedup_pending_en_impl( + pending_en: dict[str, str], reverse_index: dict[str, str] +) -> dict[str, str]: + """過濾 pending_en:跳過那些「英文文字已存在於 reverse_index」的 key。 + + 修復 cross-namespace bug:原本 `k != reverse_index[v]` 比較不同命名空間 + 的 key(raw/pending 的 k vs final/zh_tw 的 key),直接比對 key 幾乎 + 不會成立。正確邏輯:若同一個翻譯結果 v 已出現在 final + (即 v in reverse_index),就視為已處理,直接跳過不送 pending。 + + Args: + pending_en: 待翻譯的 en_us 資料(key → 英文文字) + reverse_index: reverse_index(英文文字 → canonical key) + + Returns: + dict[str, str]: 過濾後的 pending_en + """ + return { + k: v + for k, v in pending_en.items() + if not (is_filled_text_impl(v) and v in reverse_index) + } + + def _shielded_convert(text: str, convert_fn: Callable[[str], str]) -> str: """對 text 做 shield → convert_fn → unshield 保護。 @@ -240,46 +294,8 @@ def clean_kubejs_from_raw_impl( final_tw_lookup.update(tw_data) if final_tw_lookup: - # 建立 reverse_index(英文文字 → 對應 key 列表) - # 由於 rglob 迭代順序不穩定,必須用穩定的選擇策略: - # 1. 優先取「已翻譯的 key」(即 final_tw_lookup[k] != k,表示有正式翻譯) - # 2. 若多個已翻譯,取字母序第一個(確定性 tiebreaker) - # 3. 若無已翻譯,則取字母序第一個 key - reverse_index: dict[str, str] = {} - # 先收集:英文文字 → [(key, 是否已翻譯)] - rev_candidates: dict[str, list[tuple[str, bool]]] = {} - for k, v in final_tw_lookup.items(): - if is_filled_text_impl(v): - # 「已翻譯」定義:zh_tw 值與英文 key 名不同(意味著有真正翻譯) - is_translated = bool( - v.casefold() != k.casefold() - if v.isascii() and k.isascii() - else v != k - ) - rev_candidates.setdefault(v, []).append((k, is_translated)) - - for en_text, candidates in rev_candidates.items(): - # 優先取已翻譯的 key;同優先級則取字母序最小者(穩定 tiebreaker) - translated = sorted( - [k for k, t in candidates if t], - key=lambda x: x, - ) - untranslated = sorted( - [k for k, t in candidates if not t], - key=lambda x: x, - ) - reverse_index[en_text] = (translated or untranslated)[0] - - # 過濾 pending_en:跳過那些「英文文字已存在於 final」的 key - # 修復 cross-namespace bug:原本 `k != reverse_index[v]` 比較不同命名空間 - # 的 key(raw/pending 的 k vs final/zh_tw 的 key),直接比對 key 幾乎 - # 不會成立,導致去重形同虛設。正確邏輯:若同一個翻譯結果 v 已出現在 - # final(即 v in reverse_index),就視為已處理,直接跳過不送 pending。 - pending_en = { - k: v - for k, v in pending_en.items() - if not (is_filled_text_impl(v) and v in reverse_index) - } + reverse_index = _build_reverse_index_impl(final_tw_lookup) + pending_en = _dedup_pending_en_impl(pending_en, reverse_index) # ── 雙軌去重 end ─────────────────────────────────────────────── if pending_en: diff --git a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py index c6e48706..a899ebe0 100644 --- a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py +++ b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py @@ -1,635 +1 @@ -"""translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py 模組。 - -用途:提供本檔案定義的功能與流程,供專案其他模組呼叫。 -維護注意:本檔案的函式 docstring 用於維護說明,不代表行為變更。 -""" - -# ftbquests_lmtranslator.py -# ------------------------------------------------------------ -# FTB Quests (Already Extracted JSON) -> translate_batch_smart -> Export translated JSON maps -# - DOES NOT run extractor -# - Reads extracted .json files containing {key: text} pairs -# - Outputs translated {key: text} JSON with same relative path structure -# - If input file is like ru_ru.json (or other lang codes), output renamed to zh_tw.json -# - total passed to smart = global total keys across all files -# ------------------------------------------------------------ - -from __future__ import annotations - -from dataclasses import dataclass -from pathlib import Path -from typing import Dict, Any, List, Optional, Tuple -from concurrent.futures import ThreadPoolExecutor, as_completed -import math - -from translation_tool.core.lm_translator_main import translate_batch_smart - -from translation_tool.core.lm_config_rules import validate_api_keys - -from translation_tool.utils.config_manager import load_config - -from translation_tool.core.lm_translator_shared import ( - fast_split_items_by_cache, # ✅ 新增:高速分流 - translate_items_with_cache_loop, - CacheRule, - TouchSet, # ✅ 新增:touched/flush - TranslationRecorder, # ✅ 新增:翻譯記錄 - write_dry_run_preview, # ✅ 新增:dry-run preview 檔 - write_cache_hit_preview, # ✅ 新增:cache hit preview 檔 - _is_valid_hit, # ✅ 新增:cache hit 判斷 - _get_default_batch_size, -) - -from translation_tool.plugins.shared.json_io import ( - read_json_dict, - write_json_dict, - collect_json_files, -) -from translation_tool.plugins.shared.lang_path_rules import ( - compute_output_path, -) -from translation_tool.plugins.shared.lang_text_rules import is_already_zh -from translation_tool.plugins.shared.rich_text_shield import shield_text, unshield_text - -from translation_tool.utils.log_unit import ( - log_info, - log_error, -) - -# ------------------------- -# Smart 翻譯轉接器(資料格式轉換) -# ------------------------- - - -def map_to_items( - mapping: Dict[str, Any], - cache_type: str, - file_hint: str, -) -> List[Dict[str, Any]]: - """ - 將 {key: text} 的原始資料,轉換成 translate_batch_smart 可處理的 item 格式。 - - 每一個 item 代表「一條可翻譯文字」,會被送進: - - cache 比對 - - batch 翻譯 - - 翻譯結果回寫 - - 【重要設計前提】: - - smart 翻譯器會透過 item["file"] 判斷資料類型 - - FTB Quests 的判斷條件是:路徑中必須包含 "/ftbquests/" - - 因此 file_hint「一定要」包含 "/ftbquests/" - - 同時不可包含 "/lang/",否則會被誤判為 Minecraft Lang 翻譯 - - 這也是為什麼 file_hint 不是實際檔案路徑,而是「刻意構造的提示路徑」。 - """ - items: List[Dict[str, Any]] = [] - - for k, v in mapping.items(): - # key 必須是字串(語言 key) - if not isinstance(k, str): - continue - - # value 必須是非空白字串(實際要翻譯的內容) - if not isinstance(v, str) or not v.strip(): - continue - - items.append( - { - # 提供 smart translator 判斷用的檔案提示路徑 - # ⚠️ 必須包含 "/ftbquests/" - "file": file_hint, - # 語言 key(例如 quest.xxx.title) - "path": k, - # 原始文字(快取與比對用) - "source_text": v, - # 當前文字(會被翻譯器覆寫) - "text": v, - # 指定快取分類(對應 cache_rules) - "cache_type": cache_type, # 例如 "ftbquests" - } - ) - - return items - - -def count_translatable_keys(mapping: Dict[str, Any]) -> int: - """ - 計算 mapping 中「實際可翻譯的字串數量」。 - - 判斷條件: - - value 必須是字串 - - 去除空白後仍有內容 - - 這個數量會用來: - - 顯示進度 - - 計算 cache hit / miss - - 作為 batch 翻譯的總量基準 - """ - return sum(1 for _, v in mapping.items() if isinstance(v, str) and v.strip()) - - -@dataclass -class DryRunStats: - """ - Dry-run(試跑)模式下的統計資料結構。 - - 用途: - - UI 顯示 - - 預覽翻譯規模 - - 確認 cache 命中比例是否合理 - """ - - files: int = 0 # 處理的檔案數 - total_keys: int = 0 # 總字串數 - cache_hit: int = 0 # 快取命中數 - cache_miss: int = 0 # 實際需翻譯數 - per_file: list[dict] = None # 每個檔案的明細 - - -# ------------------------- -# Public API (callable from pipeline) -# ------------------------- -def translate_ftb_pending_to_zh_tw( - *, - input_lang_dir: str | Path, - output_lang_dir: str | Path, - session=None, - rename_langs: Optional[set[str]] = None, - dry_run: bool = False, # ✅ 新增 - write_new_cache: bool = True, # ✅ 新增 -) -> dict: - """ - 專給 UI/服務呼叫: - - input_lang_dir: 例如 /待翻譯/config/ftbquests/quests/lang - (注意:要傳到 lang 這層,讓相對路徑含 en_us/ 才能替換成 zh_tw) - - output_lang_dir: 例如 /config/ftbquests/quests/lang - (輸出會自動把 en_us 資料夾替換成 zh_tw) - - session: 可選,用來 add_log / set_progress - """ - validate_api_keys() - - in_dir = Path(input_lang_dir).resolve() - out_dir = Path(output_lang_dir).resolve() - out_dir.mkdir(parents=True, exist_ok=True) - - def set_prog(v: float): - """設定翻譯進度。""" - if session is not None and hasattr(session, "set_progress"): - try: - session.set_progress(v) - except Exception as e: - log_info(f"[FTB-LM] 進度設定失敗: {e}") - - # rename langs 預設沿用你 CLI 的清單(但 pending 通常只有 en_us,不太會用到) - if rename_langs is None: - rename_langs = { - "ru_ru", - "ja_jp", - "ko_kr", - "zh_cn", - "zh_hk", - "zh_sg", - "pt_br", - "es_es", - "en_us", - "fr_fr", - "de_de", - "it_it", - "pl_pl", - "tr_tr", - "uk_ua", - "cs_cz", - "hu_hu", - "nl_nl", - "sv_se", - "no_no", - "da_dk", - "fi_fi", - } - - if not in_dir.exists() or not in_dir.is_dir(): - raise FileNotFoundError(f"input_lang_dir 不存在或不是資料夾:{in_dir}") - - json_files = collect_json_files(in_dir) - if not json_files: - raise FileNotFoundError(f"找不到任何 .json:{in_dir}") - - # ---- Global total keys (raw) ---- - per_file_counts: List[Tuple[Path, int]] = [] - global_total_keys = 0 - - def _count_one(src: Path) -> Tuple[Path, int]: - """讀取指定的 JSON 檔案並統計其中可翻譯的鍵值數量,若發生錯誤則返回 0。""" - try: - mapping = read_json_dict(src) - c = count_translatable_keys(mapping) - return src, int(c) - except Exception as e: - log_info(f"[FTB-LM] 讀取 JSON 失敗 {src}: {e}") - return src, 0 - - # max_workers 你可以改成 config 的 parallel_execution_workers - - max_workers = ( - load_config().get("translator", {}).get("parallel_execution_workers", 4) - ) - - with ThreadPoolExecutor(max_workers=max_workers) as ex: - futs = [ex.submit(_count_one, src) for src in json_files] - for fu in as_completed(futs): - src, c = fu.result() - per_file_counts.append((src, c)) - global_total_keys += c - - # 保持穩定順序(避免多執行緒導致排序亂) - per_file_counts.sort(key=lambda x: x[0].as_posix()) - - if global_total_keys == 0: - log_info("ℹ️ [FTB-LM] 0 keys,跳過翻譯") - return {"written_files": 0, "total_keys": 0, "out_dir": str(out_dir)} - - # ---- Cache rules ---- - cache_rules = {"ftbquests": CacheRule("path|source_text")} - - # ✅ 預先計算「實際要翻譯」總量(全局 miss) - global_total_to_translate = 0 - global_total_hit = 0 - - for src, key_count in per_file_counts: - if key_count == 0: - continue - try: - mapping = read_json_dict(src) - rel_src = src.relative_to(in_dir).as_posix() - file_hint = f"config/ftbquests/quests/{rel_src}" - all_items = map_to_items( - mapping, cache_type="ftbquests", file_hint=file_hint - ) - - cached_items, items_to_translate = fast_split_items_by_cache( - all_items, - cache_rules=cache_rules, - is_valid_hit=_is_valid_hit, - ) - - # ✅ 中文/已翻譯:不送 LM,也不算 cache_miss - already_zh_items = [] - real_to_translate = [] - for it in items_to_translate: - s = str(it.get("source_text") or it.get("text") or "") - if is_already_zh(s): - already_zh_items.append(it) - else: - real_to_translate.append(it) - - global_total_hit += len(cached_items) - global_total_to_translate += len(real_to_translate) - - except Exception as e: - log_info(f"[FTB-LM] 預掃描失敗 {src}: {e}") - - log_info( - f"\n🔎 [FTB-LM][掃描完畢] 發現待處理檔案:{len(json_files)} 個 | 文本總條目:{global_total_keys} 條" - f"\n✅ [FTB-LM][進度分析] 已從快取載入:{global_total_hit} 條 | 剩餘需 AI 翻譯:{global_total_to_translate} 條" - ) - - if global_total_to_translate == 0: - log_info( - "ℹ️ [FTB-LM][狀態] 恭喜!所有內容皆命中快取,無需調用 AI,將直接導出檔案。" - ) - - if global_total_to_translate == 0: - set_prog(0.99) # 避免瞬間跳 1.0,讓 UI 停留在即將完成的視覺效果 - - # ---- Translate per file (shared loop + cache) ---- - translated_done = 0 # ✅ 只算 API 翻譯完成(主進度分子) - cache_hit_done_so_far = 0 # ✅ 已寫入輸出的 cache hit(整體完成用) - total_written = 0 - - # ---- Dry-run stats container ---- - per_file_rows: list[dict] = [] - # ---- Dry-run: 不翻譯、不輸出 ---- - dry_preview_items: List[Dict[str, Any]] = [] - - all_cached_items: list[dict] = [] - - rec = TranslationRecorder() - touch = TouchSet() - - # touched_files 對應的 writer:最小改動版(每檔只會有一個 dst/out_map) - _file_write_table: dict[str, tuple[Path, Dict[str, str]]] = {} - - def _writer(file_id: str) -> None: - """寫入翻譯結果到檔案。""" - dst_path, data = _file_write_table[file_id] - write_json_dict(dst_path, data) - - for idx, (src, key_count) in enumerate(per_file_counts, start=1): - if key_count == 0: - continue - - mapping = read_json_dict(src) - - rel_src = src.relative_to(in_dir).as_posix() # e.g. en_us/ftb_lang.json - file_hint = ( - f"config/ftbquests/quests/{rel_src}" # ✅ include /ftbquests/ and no /lang/ - ) - all_items = map_to_items(mapping, cache_type="ftbquests", file_hint=file_hint) - - cached_items, items_to_translate = fast_split_items_by_cache( - all_items, - cache_rules=cache_rules, - is_valid_hit=_is_valid_hit, - ) - - already_zh_items = [] - real_to_translate = [] - for it in items_to_translate: - s = str(it.get("source_text") or it.get("text") or "") - if is_already_zh(s): - already_zh_items.append(it) - else: - real_to_translate.append(it) - - items_to_translate = real_to_translate # ✅ 覆蓋成真的要翻譯的 - hit = len(cached_items) - miss = len(items_to_translate) - already_zh = len(already_zh_items) - - if hit > 0: - sample = cached_items[0] - log_info(f"🧠 [FTB-LM] 快取命中範例:{sample.get('path')}") - - if miss > 0: - sample = items_to_translate[0] - log_info(f"✏️ [FTB-LM] 待翻譯範例:{sample.get('path')}") - - dst = compute_output_path(src, in_dir, out_dir, rename_langs) - log_info( - f"📄 [FTB-LM][檔案 {idx}/{len(per_file_counts)}] 正在處理:{rel_src}\n" - f"📊 本檔明細:總條目 {key_count} | 載入快取 {hit} | 已含中文(跳過) {already_zh} | 需翻譯 {miss}\n" - f"💾 輸出路徑:{dst.relative_to(out_dir).as_posix()}" - ) - - per_file_rows.append( - { - "file": rel_src, - "keys": key_count, - "cache_hit": hit, - "cache_miss": miss, - "dst": dst.relative_to(out_dir).as_posix(), - } - ) - - # ---- Dry-run: 不翻譯、不輸出 ---- - if dry_run: - # ✅ 收集 preview 樣本(避免爆大:最多 2000) - if len(dry_preview_items) < 2000: - dry_preview_items.extend( - items_to_translate[: 2000 - len(dry_preview_items)] - ) - - # ✅ NEW:收集 cache hit(避免爆大:最多 2000,可自行調整) - if len(all_cached_items) < 2000: - all_cached_items.extend(cached_items[: 2000 - len(all_cached_items)]) - - # 以檔案數取代翻譯量為進度分母,避免翻譯量分佈不均造成視覺跳躍 - set_prog(min(idx / len(per_file_counts), 1.0)) - - log_info( - f"🧪 [測試模式] 進度:{idx}/{len(per_file_counts)}\n" - f"擬定輸出:{rel_src} ➔ {dst.relative_to(out_dir).as_posix()}\n" - f"預估明細:總條目 {key_count} | 快取命中 {hit} | 需翻譯 {miss} | 預估批次 {math.ceil(miss / _get_default_batch_size('ftbquests', None)) if _get_default_batch_size('ftbquests', None) > 0 else 'N/A'}\n" - f"累計預估進度:{translated_done} / {global_total_to_translate}" - ) - continue - - # ============================ - # ✅ 真正翻譯(非 dry-run) - # ============================ - - # out_map:先用原文當底(避免中斷時輸出缺 key) - out_map: Dict[str, str] = { - k: v - for k, v in mapping.items() - if isinstance(k, str) and isinstance(v, str) - } - - # cache hits 覆蓋 - for it in cached_items: - p = it.get("path") - t = it.get("text") - if isinstance(p, str) and isinstance(t, str): - out_map[p] = t - try: - rec.record( - cache_type="ftbquests", - file_id=rel_src, - path=p, - src=str(it.get("source_text") or ""), - dst=t, - cache_hit=True, - extra={"dst_file": dst.relative_to(out_dir).as_posix()}, - ) - except Exception as e: - log_info(f"[FTB-LM] 記錄快取命中失敗: {e}") - - # 全命中 cache:直接輸出 - if not items_to_translate: - # write_json_dict(dst, out_map) - file_id = dst.as_posix() - _file_write_table[file_id] = (dst, out_map) - - touch.touch(file_id) - touch.flush(_writer) # 最小改動:等同你現在每次都寫,但走同一個管線 - total_written += 1 - - # ✅ 這個檔案的 cache hit 已經真正寫進輸出,整體完成要加 - cache_hit_done_so_far += hit - - # ✅ 主進度不變(translated_done 不加) - set_prog(min(translated_done / max(global_total_to_translate, 1), 1.0)) - - overall_done = cache_hit_done_so_far + translated_done - progress_pct = ( - (overall_done / global_total_keys * 100) if global_total_keys > 0 else 0 - ) - - log_info( - f"⚡ [快取跳過] 檔案 {idx}/{len(per_file_counts)}:{rel_src}\n" - f"💾 導出路徑:{dst.relative_to(out_dir).as_posix()}\n" - f"🧠 處理結果:命中快取 {hit} 條 | API 調用 0 條\n" - f"🎯 全域完成度:{overall_done} / {global_total_keys} ({progress_pct:.1f}%)" - ) - continue - - # shared while-loop(includes add_to_cache + save_translation_cache + safe slicing) - def on_translated_item(it: Dict[str, Any]) -> None: - """處理翻譯結果並寫入映射。""" - p = it.get("path") - t = it.get("text") - src_text = str(it.get("source_text") or "") - if isinstance(p, str) and isinstance(t, str): - try: - shielded_src = shield_text(src_text) - t = unshield_text(t, shielded_src.shields) - except Exception as e: - log_info(f"[FTB-LM] unshield 失敗: {e}") - out_map[p] = t - try: - rec.record( - cache_type="ftbquests", - file_id=rel_src, - path=p, - src=src_text, - dst=t, - cache_hit=False, - extra={"dst_file": dst.relative_to(out_dir).as_posix()}, - ) - except Exception as e: - log_info(f"[FTB-LM] 記錄翻譯結果失敗: {e}") - - # ✅ 確保此檔案在翻譯路徑也有 file_id - file_id = dst.as_posix() - _file_write_table[file_id] = (dst, out_map) - - # 在這之前先確保 file_id/_file_write_table 設定好了(下面會說加在哪) - def on_batch_flushed() -> None: - """批量寫入翻譯結果。""" - try: - touch.touch(file_id) - touch.flush(_writer) # 最小改動:每批也照樣寫,避免中斷損失 - except Exception as e: - log_info(f"[FTB-LM] 批次刷新失敗,使用 fallback 寫入: {e}") - write_json_dict(dst, out_map) - - def _fmt_eta(sec: float) -> str: - """格式化剩餘時間。""" - if sec <= 0: - return "" - m, s = divmod(int(sec), 60) - if m > 0: - return f"{m}m{s:02d}s" - return f"{s}s" - - def on_progress(p: float, msg: str, eta_sec: float) -> None: - """報告翻譯進度。""" - eta_txt = _fmt_eta(eta_sec) - if eta_txt: - log_info(f"⏳ [AI 翻譯中] {msg} | 預估剩餘時間:{eta_txt}") - else: - log_info(f"🚀 [AI 翻譯中] {msg}") - set_prog(p) - - res = translate_items_with_cache_loop( - items_to_translate, - total_for_smart=global_total_to_translate, - translate_batch_smart=lambda batch, total: translate_batch_smart( - batch, total=total - ), - write_new_cache=bool(write_new_cache), # ✅ 改成吃參數 - cache_rules=cache_rules, - on_translated_item=on_translated_item, - on_batch_flushed=on_batch_flushed, - on_progress=on_progress, - ) - - # final write - touch.touch(file_id) - touch.flush(_writer) - total_written += 1 - - # 這個檔案的 cache hit 也已寫入輸出 - cache_hit_done_so_far += hit - - # 只把 API 實際翻譯數量加進主進度 - translated_done += int(res.processed or 0) - - set_prog(min(translated_done / max(global_total_to_translate, 1), 1.0)) - - overall_done = cache_hit_done_so_far + translated_done - progress_pct = ( - (overall_done / global_total_keys * 100) if global_total_keys > 0 else 0 - ) - log_info( - f"✨ [檔案完成] {rel_src} 已寫入\n" - f"📈 翻譯數據:本次翻譯 {int(res.processed or 0)} 條 | 檔案狀態:{res.status}\n" - f"🎯 全域進度:目前已完成 {overall_done} / {global_total_keys} ({progress_pct:.1f}%)" - ) - - if res.status == "ALL_KEYS_EXHAUSTED": - log_info("⚠️ [FTB-LM] ALL_KEYS_EXHAUSTED:已輸出目前成果,停止。") - break - - # ---- Dry-run 結尾摘要 ---- - if dry_run: - batch_size = _get_default_batch_size("ftbquests", None) - est_batches = ( - math.ceil(global_total_to_translate / batch_size) - if isinstance(global_total_to_translate, int) and batch_size > 0 - else None - ) - meta = { - "files": len(per_file_rows), - "total_keys": global_total_keys, - "cache_hit": global_total_hit, - "cache_miss": global_total_to_translate, - "estimated_batches": est_batches, - } - - try: - # 原本:待翻譯 preview - write_dry_run_preview( - out_dir, - dry_preview_items, - meta=meta, - filename="_ftbquests_dry_run_preview.json", # 可選:明確檔名 - ) - - # ✅ NEW:cache hit preview - write_cache_hit_preview( - out_dir, - all_cached_items, - filename="_ftbquests_dry_run_cache_hit_preview.json", - meta=meta, - ) - - except Exception as e: - log_error(f"⚠️ [FTB-LM] DRY-RUN preview 輸出失敗:{e}") - - return { - "dry_run": True, - "files": len(per_file_rows), - "total_keys": global_total_keys, - "cache_hit": global_total_hit, - "cache_miss": global_total_to_translate, - "estimated_batches": est_batches, - "out_dir": str(out_dir), - "per_file": per_file_rows, - } - - try: - rec.export_json(out_dir / "translation_map.json") - rec.export_csv(out_dir / "translation_map.csv") - log_info(f"✅ [FTB-LM] 已匯出 translation_map.json / .csv 到 {out_dir}") - - except Exception as e: - log_error(f"⚠️ [FTB-LM] 匯出 translation_map 失敗: {e}") - - log_info(f"✅ [任務翻譯完成] 已將 {total_written} 個翻譯檔案輸出至:{out_dir}") - log_info("📊 提示:您可以在該目錄下查看 translation_map.csv 來核對翻譯條目細節。") - batch_size = _get_default_batch_size("ftbquests", None) - est_batches = ( - math.ceil(global_total_to_translate / batch_size) - if isinstance(global_total_to_translate, int) and batch_size > 0 - else None - ) - return { - "dry_run": False, - "written_files": total_written, - "total_keys": global_total_keys, - "cache_hit": global_total_hit, - "cache_miss": global_total_to_translate, - "estimated_batches": est_batches, - "out_dir": str(out_dir), - } +"""translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py 璅∠????券€????祆?獢?蝢拍????蝔?靘?獢隞芋蝯?怒€?蝬剛風瘜冽?嚗瑼??撘?docstring ?冽蝬剛風隤芣?嚗?隞?”銵霈??"""# ftbquests_lmtranslator.py# ------------------------------------------------------------# FTB Quests (Already Extracted JSON) -> translate_batch_smart -> Export translated JSON maps# - DOES NOT run extractor# - Reads extracted .json files containing {key: text} pairs# - Outputs translated {key: text} JSON with same relative path structure# - If input file is like ru_ru.json (or other lang codes), output renamed to zh_tw.json# - total passed to smart = global total keys across all files# ------------------------------------------------------------from __future__ import annotationsfrom dataclasses import dataclassfrom pathlib import Pathfrom typing import Dict, Any, List, Optional, Tuplefrom concurrent.futures import ThreadPoolExecutor, as_completedimport mathfrom translation_tool.core.lm_translator_main import translate_batch_smartfrom translation_tool.core.lm_config_rules import validate_api_keysfrom translation_tool.utils.config_manager import load_configfrom translation_tool.core.lm_translator_shared import ( fast_split_items_by_cache, # ???啣?嚗???瘚? translate_items_with_cache_loop, CacheRule, TouchSet, # ???啣?嚗ouched/flush TranslationRecorder, # ???啣?嚗蕃霅航??? write_dry_run_preview, # ???啣?嚗ry-run preview 瑼? write_cache_hit_preview, # ???啣?嚗ache hit preview 瑼? _is_valid_hit, # ???啣?嚗ache hit ?斗 _get_default_batch_size,)from translation_tool.plugins.shared.json_io import ( read_json_dict, write_json_dict, collect_json_files,)from translation_tool.plugins.shared.lang_path_rules import ( compute_output_path,)from translation_tool.plugins.shared.lang_text_rules import is_already_zhfrom translation_tool.plugins.shared.rich_text_shield import shield_text, unshield_textfrom translation_tool.utils.log_unit import ( log_info, log_error,)# -------------------------# Smart 蝧餉陌頧?剁?鞈??澆?頧?嚗?# -------------------------def map_to_items( mapping: Dict[str, Any], cache_type: str, file_hint: str,) -> List[Dict[str, Any]]: """ 撠?{key: text} ??憪???頧???translate_batch_smart ?航??? item ?澆??? 瘥???item 隞?”??璇蝧餉陌?????◤?€莎? - cache 瘥? - batch 蝧餉陌 - 蝧餉陌蝯??神 ??閬身閮??€? - smart 蝧餉陌?冽??? item["file"] ?斗鞈?憿? - FTB Quests ??瑟?隞嗆嚗楝敺葉敹?? "/ftbquests/" - ?迨 file_hint??摰?????"/ftbquests/" - ??銝? "/lang/"嚗??鋡怨炊?斤 Minecraft Lang 蝧餉陌 ???舐隞€暻?file_hint 銝撖阡?瑼?頝臬?嚗€??????內頝臬??€? """ items: List[Dict[str, Any]] = [] for k, v in mapping.items(): # key 敹??臬?銝莎?隤? key嚗? if not isinstance(k, str): continue # value 敹??舫?蝛箇摮葡嚗祕??蝧餉陌?摰對? if not isinstance(v, str) or not v.strip(): continue items.append( { # ?? smart translator ?斗?函?瑼??內頝臬? # ?? 敹?? "/ftbquests/" "file": file_hint, # 隤? key嚗?憒?quest.xxx.title嚗? "path": k, # ????嚗翰??瘥??剁? "source_text": v, # ?嗅???嚗?鋡怎蕃霅臬閬神嚗? "text": v, # ??敹怠???嚗???cache_rules嚗? "cache_type": cache_type, # 靘? "ftbquests" } ) return itemsdef count_translatable_keys(mapping: Dict[str, Any]) -> int: """ 閮? mapping 銝准€祕?蝧餉陌??銝脫?€€? ?斗璇辣嚗? - value 敹??臬?銝? - ?駁蝛箇敺??摰? ?€???其?嚗? - 憿舐內?脣漲 - 閮? cache hit / miss - 雿 batch 蝧餉陌?蜇?皞? """ return sum(1 for _, v in mapping.items() if isinstance(v, str) and v.strip())@dataclassclass DryRunStats: """ Dry-run嚗岫頝?璅∪?銝?蝯梯?鞈?蝯??? ?券€? - UI 憿舐內 - ?汗蝧餉陌閬芋 - 蝣箄? cache ?賭葉瘥??臬?? """ files: int = 0 # ????獢 total_keys: int = 0 # 蝮賢?銝脫 cache_hit: int = 0 # 敹怠??賭葉?? cache_miss: int = 0 # 撖阡??€蝧餉陌?? per_file: list[dict] = None # 瘥€?獢??敦# -------------------------# Public API (callable from pipeline)# -------------------------def translate_ftb_pending_to_zh_tw( *, input_lang_dir: str | Path, output_lang_dir: str | Path, session=None, rename_langs: Optional[set[str]] = None, dry_run: bool = False, # ???啣? write_new_cache: bool = True, # ???啣?) -> dict: """ 撠策 UI/???澆嚗? - input_lang_dir: 靘? /敺蕃霅?config/ftbquests/quests/lang 嚗釣??閬??lang ?惜嚗??詨?頝臬???en_us/ ??踵???zh_tw嚗? - output_lang_dir: 靘? /config/ftbquests/quests/lang 嚗撓?箸??芸???en_us 鞈?憭暹?? zh_tw嚗? - session: ?舫嚗靘?add_log / set_progress """ validate_api_keys() in_dir = Path(input_lang_dir).resolve() out_dir = Path(output_lang_dir).resolve() out_dir.mkdir(parents=True, exist_ok=True) def set_prog(v: float): """閮剖?蝧餉陌?脣漲??"" if session is not None and hasattr(session, "set_progress"): try: session.set_progress(v) except Exception: pass # rename langs ?身瘝輻雿?CLI ???殷?雿?pending ?虜?芣? en_us嚗?憭芣??典嚗? if rename_langs is None: rename_langs = { "ru_ru", "ja_jp", "ko_kr", "zh_cn", "zh_hk", "zh_sg", "pt_br", "es_es", "en_us", "fr_fr", "de_de", "it_it", "pl_pl", "tr_tr", "uk_ua", "cs_cz", "hu_hu", "nl_nl", "sv_se", "no_no", "da_dk", "fi_fi", } if not in_dir.exists() or not in_dir.is_dir(): raise FileNotFoundError(f"input_lang_dir 銝??冽?銝鞈?憭橘?{in_dir}") json_files = collect_json_files(in_dir) if not json_files: raise FileNotFoundError(f"?曆??唬遙雿?.json嚗in_dir}") # ---- Global total keys (raw) + cache mappings ---- per_file_counts: List[Tuple[Path, int]] = [] global_total_keys = 0 # ??Issue #7 靽桀儔嚗楨摮?JSON mapping ?踹???霈€?? src_mapping_cache: Dict[Path, Dict[str, Any]] = {} def _count_one(src: Path) -> Tuple[Path, int, Dict[str, Any]]: """霈€??JSON 銝衣絞閮蝧餉陌?萄€潭????敹怠? mapping??"" try: mapping = read_json_dict(src) c = count_translatable_keys(mapping) return src, int(c), mapping except Exception: return src, 0, {} # max_workers 雿隞交??config ??parallel_execution_workers max_workers = ( load_config().get("translator", {}).get("parallel_execution_workers", 4) ) with ThreadPoolExecutor(max_workers=max_workers) as ex: futs = [ex.submit(_count_one, src) for src in json_files] for fu in as_completed(futs): src, c, mapping = fu.result() per_file_counts.append((src, c)) global_total_keys += c if mapping: # ???芰楨摮?蝛箇? mapping src_mapping_cache[src] = mapping # 靽?蝛拙???嚗???瑁?蝺??湔?摨?嚗? per_file_counts.sort(key=lambda x: x[0].as_posix()) if global_total_keys == 0: log_info("?對? [FTB-LM] 0 keys嚗歲?蕃霅?) return {"written_files": 0, "total_keys": 0, "out_dir": str(out_dir)} # ---- Cache rules ---- cache_rules = {"ftbquests": CacheRule("path|source_text")} # ????閮??祕??蝧餉陌?蜇???典? miss嚗? global_total_to_translate = 0 global_total_hit = 0 for src, key_count in per_file_counts: if key_count == 0: continue try: # ??Issue #7 靽桀儔嚗?乩蝙?函楨摮? mapping嚗???銴??? mapping = src_mapping_cache.get(src, {}) if not mapping: continue rel_src = src.relative_to(in_dir).as_posix() file_hint = f"config/ftbquests/quests/{rel_src}" all_items = map_to_items( mapping, cache_type="ftbquests", file_hint=file_hint ) cached_items, items_to_translate = fast_split_items_by_cache( all_items, cache_rules=cache_rules, is_valid_hit=_is_valid_hit, ) # ??銝剜?/撌脩蕃霅荔?銝€?LM嚗?銝? cache_miss already_zh_items = [] real_to_translate = [] for it in items_to_translate: s = str(it.get("source_text") or it.get("text") or "") if is_already_zh(s): already_zh_items.append(it) else: real_to_translate.append(it) global_total_hit += len(cached_items) global_total_to_translate += len(real_to_translate) except Exception: pass log_info( f"\n?? [FTB-LM][??摰] ?潛敺???獢?{len(json_files)} ??| ?蝮賣??殷?{global_total_keys} 璇? f"\n??[FTB-LM][?脣漲??] 撌脣?敹怠?頛嚗global_total_hit} 璇?| ?拚??€ AI 蝧餉陌嚗global_total_to_translate} 璇? ) if global_total_to_translate == 0: log_info( "?對? [FTB-LM][?€? ?剖?嚗??摰寧??賭葉敹怠?嚗?€隤輻 AI嚗??湔撠瑼??? ) if global_total_to_translate == 0: set_prog(0.99) # ?踹??祇?頝?1.0嚗? UI ???典撠???閬死?? # ---- Translate per file (shared loop + cache) ---- translated_done = 0 # ???芰? API 蝧餉陌摰?嚗蜓?脣漲??嚗? cache_hit_done_so_far = 0 # ??撌脣神?亥撓?箇? cache hit嚗擃??嚗? total_written = 0 # ---- Dry-run stats container ---- per_file_rows: list[dict] = [] # ---- Dry-run: 銝蕃霅胯€?頛詨 ---- dry_preview_items: List[Dict[str, Any]] = [] all_cached_items: list[dict] = [] rec = TranslationRecorder() touch = TouchSet() # touched_files 撠???writer嚗?撠??嚗?瑼??銝€??dst/out_map嚗? _file_write_table: dict[str, tuple[Path, Dict[str, str]]] = {} def _writer(file_id: str) -> None: """撖怠蝧餉陌蝯??唳?獢€?"" dst_path, data = _file_write_table[file_id] write_json_dict(dst_path, data) for idx, (src, key_count) in enumerate(per_file_counts, start=1): if key_count == 0: continue # ??Issue #7 靽桀儔嚗?乩蝙?函楨摮? mapping嚗???銴??? mapping = src_mapping_cache.get(src, {}) if not mapping: log_error(f"?? [FTB-LM] ?曆??啣翰?? mapping嚗src}") continue rel_src = src.relative_to(in_dir).as_posix() # e.g. en_us/ftb_lang.json file_hint = ( f"config/ftbquests/quests/{rel_src}" # ??include /ftbquests/ and no /lang/ ) all_items = map_to_items(mapping, cache_type="ftbquests", file_hint=file_hint) cached_items, items_to_translate = fast_split_items_by_cache( all_items, cache_rules=cache_rules, is_valid_hit=_is_valid_hit, ) already_zh_items = [] real_to_translate = [] for it in items_to_translate: s = str(it.get("source_text") or it.get("text") or "") if is_already_zh(s): already_zh_items.append(it) else: real_to_translate.append(it) items_to_translate = real_to_translate # ??閬?????蝧餉陌?? hit = len(cached_items) miss = len(items_to_translate) already_zh = len(already_zh_items) if hit > 0: sample = cached_items[0] log_info(f"?? [FTB-LM] 敹怠??賭葉蝭?嚗sample.get('path')}") if miss > 0: sample = items_to_translate[0] log_info(f"?? [FTB-LM] 敺蕃霅舐?靘?{sample.get('path')}") dst = compute_output_path(src, in_dir, out_dir, rename_langs) log_info( f"?? [FTB-LM][瑼? {idx}/{len(per_file_counts)}] 甇???嚗rel_src}\n" f"?? ?祆??敦嚗蜇璇 {key_count} | 頛敹怠? {hit} | 撌脣銝剜?(頝喲?) {already_zh} | ?€蝧餉陌 {miss}\n" f"? 頛詨頝臬?嚗dst.relative_to(out_dir).as_posix()}" ) per_file_rows.append( { "file": rel_src, "keys": key_count, "cache_hit": hit, "cache_miss": miss, "dst": dst.relative_to(out_dir).as_posix(), } ) # ---- Dry-run: 銝蕃霅胯€?頛詨 ---- if dry_run: # ???園? preview 璅?嚗??憭改??€憭?2000嚗? if len(dry_preview_items) < 2000: dry_preview_items.extend( items_to_translate[: 2000 - len(dry_preview_items)] ) # ??NEW嚗??cache hit嚗??憭改??€憭?2000嚗?芾?隤踵嚗? if len(all_cached_items) < 2000: all_cached_items.extend(cached_items[: 2000 - len(all_cached_items)]) # 隞交?獢?誨蝧餉陌??脣漲??嚗?蕃霅舫???銝???閬死頝唾? set_prog(min(idx / len(per_file_counts), 1.0)) log_info( f"?妒 [皜祈岫璅∪?] ?脣漲嚗idx}/{len(per_file_counts)}\n" f"?砍?頛詨嚗rel_src} ??{dst.relative_to(out_dir).as_posix()}\n" f"?摯?敦嚗蜇璇 {key_count} | 敹怠??賭葉 {hit} | ?€蝧餉陌 {miss} | ?摯?寞活 {math.ceil(miss / _get_default_batch_size('ftbquests', None)) if _get_default_batch_size('ftbquests', None) > 0 else 'N/A'}\n" f"蝝航??摯?脣漲嚗translated_done} / {global_total_to_translate}" ) continue # ============================ # ???迤蝧餉陌嚗? dry-run嚗? # ============================ # out_map嚗??典??摨??踹?銝剜?撓?箇撩 key嚗? out_map: Dict[str, str] = { k: v for k, v in mapping.items() if isinstance(k, str) and isinstance(v, str) } # cache hits 閬? for it in cached_items: p = it.get("path") t = it.get("text") if isinstance(p, str) and isinstance(t, str): out_map[p] = t try: rec.record( cache_type="ftbquests", file_id=rel_src, path=p, src=str(it.get("source_text") or ""), dst=t, cache_hit=True, extra={"dst_file": dst.relative_to(out_dir).as_posix()}, ) except Exception: pass # ?典銝?cache嚗?亥撓?? if not items_to_translate: # write_json_dict(dst, out_map) file_id = dst.as_posix() _file_write_table[file_id] = (dst, out_map) touch.touch(file_id) touch.flush(_writer) # ?€撠??蝑?雿?冽?甈⊿撖恬?雿粥???恣蝺? total_written += 1 # ???€?獢? cache hit 撌脩??迤撖恍€脰撓?綽??湧?摰?閬? cache_hit_done_so_far += hit # ??銝駁€脣漲銝?嚗ranslated_done 銝?嚗? set_prog(min(translated_done / max(global_total_to_translate, 1), 1.0)) overall_done = cache_hit_done_so_far + translated_done progress_pct = ( (overall_done / global_total_keys * 100) if global_total_keys > 0 else 0 ) log_info( f"??[敹怠?頝喲?] 瑼? {idx}/{len(per_file_counts)}嚗rel_src}\n" f"? 撠頝臬?嚗dst.relative_to(out_dir).as_posix()}\n" f"?? ??蝯?嚗銝剖翰??{hit} 璇?| API 隤輻 0 璇n" f"? ?典?摰?摨佗?{overall_done} / {global_total_keys} ({progress_pct:.1f}%)" ) continue # ??Issue #8 靽桀儔嚗? callbacks 摰儔?刻艘???剁??寧撌亙??賢? def _fmt_eta(sec: float) -> str: """?澆??擗??€?"" if sec <= 0: return "" m, s = divmod(int(sec), 60) if m > 0: return f"{m}m{s:02d}s" return f"{s}s" def make_on_progress(set_prog, _fmt_eta): def on_progress(p: float, msg: str, eta_sec: float) -> None: """?勗?蝧餉陌?脣漲??"" eta_txt = _fmt_eta(eta_sec) if eta_txt: log_info(f"??[AI 蝧餉陌銝苗 {msg} | ?摯?拚???嚗eta_txt}") else: log_info(f"?? [AI 蝧餉陌銝苗 {msg}") set_prog(p) return on_progress def make_on_translated_item(rel_src, dst, out_map, rec, out_dir): def on_translated_item(it: Dict[str, Any]) -> None: """??蝧餉陌蝯?銝血神?交?撠€?"" p = it.get("path") t = it.get("text") src_text = str(it.get("source_text") or "") if isinstance(p, str) and isinstance(t, str): try: shielded_src = shield_text(src_text) t = unshield_text(t, shielded_src.shields) except Exception: pass out_map[p] = t try: rec.record( cache_type="ftbquests", file_id=rel_src, path=p, src=src_text, dst=t, cache_hit=False, extra={"dst_file": dst.relative_to(out_dir).as_posix()}, ) except Exception: pass return on_translated_item def make_on_batch_flushed(file_id, touch, _writer, dst, out_map): def on_batch_flushed() -> None: """?寥?撖怠蝧餉陌蝯???"" try: touch.touch(file_id) touch.flush(_writer) # ?€撠??瘥銋璅?神嚗?葉?瑟?憭? except Exception: # fallback write_json_dict(dst, out_map) return on_batch_flushed # ??蝣箔?甇斗?獢蝧餉陌頝臬?銋? file_id file_id = dst.as_posix() _file_write_table[file_id] = (dst, out_map) # ??Issue #8 靽桀儔嚗蝙?典極撱撘撱?callbacks on_translated_item = make_on_translated_item(rel_src, dst, out_map, rec, out_dir) on_batch_flushed = make_on_batch_flushed(file_id, touch, _writer, dst, out_map) on_progress = make_on_progress(set_prog, _fmt_eta) res = translate_items_with_cache_loop( items_to_translate, total_for_smart=global_total_to_translate, translate_batch_smart=lambda batch, total: translate_batch_smart( batch, total=total ), write_new_cache=bool(write_new_cache), # ???寞????? cache_rules=cache_rules, on_translated_item=on_translated_item, on_batch_flushed=on_batch_flushed, on_progress=on_progress, ) # final write touch.touch(file_id) touch.flush(_writer) total_written += 1 # ?€?獢? cache hit 銋歇撖怠頛詨 cache_hit_done_so_far += hit # ?芣? API 撖阡?蝧餉陌?賊??€脖蜓?脣漲 translated_done += int(res.processed or 0) set_prog(min(translated_done / max(global_total_to_translate, 1), 1.0)) overall_done = cache_hit_done_so_far + translated_done progress_pct = ( (overall_done / global_total_keys * 100) if global_total_keys > 0 else 0 ) log_info( f"??[瑼?摰?] {rel_src} 撌脣神?功n" f"?? 蝧餉陌?豢?嚗甈∠蕃霅?{int(res.processed or 0)} 璇?| 瑼??€??{res.status}\n" f"? ?典??脣漲嚗?歇摰? {overall_done} / {global_total_keys} ({progress_pct:.1f}%)" ) if res.status == "ALL_KEYS_EXHAUSTED": log_info("?? [FTB-LM] ALL_KEYS_EXHAUSTED嚗歇頛詨?桀???嚗?甇U€?) break # ---- Dry-run 蝯偏?? ---- if dry_run: batch_size = _get_default_batch_size("ftbquests", None) est_batches = ( math.ceil(global_total_to_translate / batch_size) if isinstance(global_total_to_translate, int) and batch_size > 0 else None ) meta = { "files": len(per_file_rows), "total_keys": global_total_keys, "cache_hit": global_total_hit, "cache_miss": global_total_to_translate, "estimated_batches": est_batches, } try: # ?嚗?蝧餉陌 preview write_dry_run_preview( out_dir, dry_preview_items, meta=meta, filename="_ftbquests_dry_run_preview.json", # ?舫嚗?蝣箸??? ) # ??NEW嚗ache hit preview write_cache_hit_preview( out_dir, all_cached_items, filename="_ftbquests_dry_run_cache_hit_preview.json", meta=meta, ) except Exception as e: log_error(f"?? [FTB-LM] DRY-RUN preview 頛詨憭望?嚗e}") return { "dry_run": True, "files": len(per_file_rows), "total_keys": global_total_keys, "cache_hit": global_total_hit, "cache_miss": global_total_to_translate, "estimated_batches": est_batches, "out_dir": str(out_dir), "per_file": per_file_rows, } try: rec.export_json(out_dir / "translation_map.json") rec.export_csv(out_dir / "translation_map.csv") log_info(f"??[FTB-LM] 撌脣??translation_map.json / .csv ??{out_dir}") except Exception: log_error("?? [FTB-LM] ?臬 translation_map 憭望?") pass log_info(f"??[隞餃?蝧餉陌摰?] 撌脣? {total_written} ?蕃霅舀?獢撓?箄嚗out_dir}") log_info("?? ?內嚗?臭誑?刻府?桅?銝??translation_map.csv 靘撠蕃霅舀??桃敦蝭€??) batch_size = _get_default_batch_size("ftbquests", None) est_batches = ( math.ceil(global_total_to_translate / batch_size) if isinstance(global_total_to_translate, int) and batch_size > 0 else None ) return { "dry_run": False, "written_files": total_written, "total_keys": global_total_keys, "cache_hit": global_total_hit, "cache_miss": global_total_to_translate, "estimated_batches": est_batches, "out_dir": str(out_dir), } \ No newline at end of file diff --git a/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py b/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py index 5299f55f..d27032e4 100644 --- a/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py +++ b/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py @@ -1,602 +1 @@ -# kubejs_tooltip_lmtranslator.py -# --------------------------------- -""" -核心功能概覽 -智慧批量翻譯 (translate_batch_smart):利用快取機制優化翻譯流程,避免重複翻譯相同的文字,節省 API 消耗。 -高速快取過濾 (fast_split_items_by_cache):在翻譯前快速比對現有快取,將「已翻譯」與「待翻譯」的項目分離,提升處理效率。 -安全循環翻譯 (translate_items_with_cache_loop): -具備 ETA(預計剩餘時間) 顯示。 -採用 安全切片(Safe Slicing) 技術,防止因單次請求過大導致失敗。 -斷點續傳與即時儲存: -使用 TouchSet 與 Writer Flush 機制,翻譯過程中會即時寫入檔案,即使程式意外中斷也不會遺失已完成的進度。 -預檢模式 (dry_run):支援模擬執行並輸出預覽檔案(write_dry_run_preview),讓你在正式消耗 API 額度前確認格式是否正確。 -數據導出 (TranslationRecorder):支援將翻譯記錄導出為 JSON 或 CSV 格式,方便後續校對或二次開發。 -進度追蹤 (session.set_progress):內建進度鉤子(Hook),可對接外部 UI 或日誌系統顯示翻譯百分比。 -路徑優化:自動處理語系資料夾轉換(例如將原本的 en_us 自動導向至 zh_tw 目錄)。 -Rich Text Shield:shield_text() / unshield_text() 保護 KubeJS 格式(彩色碼、物品ID、URL 等), -在翻譯前抽出,翻譯後還原,避免 LM 誤翻格式標記。 -""" - -from __future__ import annotations - -from dataclasses import dataclass -from pathlib import Path -from typing import Dict, Any, List, Optional, Tuple -from concurrent.futures import ThreadPoolExecutor, as_completed -import re -import opencc - -from translation_tool.core.lm_translator_main import translate_batch_smart -from translation_tool.core.lm_config_rules import validate_api_keys -from translation_tool.utils.config_manager import load_config - -from translation_tool.core.lm_translator_shared import ( - CacheRule, - fast_split_items_by_cache, - translate_items_with_cache_loop, - TouchSet, - TranslationRecorder, - write_dry_run_preview, - write_cache_hit_preview, # ✅ 新增:cache hit preview 檔 - _is_valid_hit, # ✅ 新增:cache hit 判斷 -) - -from translation_tool.plugins.shared.json_io import ( - read_json_dict, - write_json_dict, - collect_json_files, -) -from translation_tool.plugins.shared.lang_path_rules import ( - compute_output_path, -) -from translation_tool.plugins.shared.rich_text_shield import ( - shield_text, - unshield_text, -) - -from translation_tool.utils.log_unit import log_info, log_warning, progress - - -# ------------------------- -# Smart item mapping -# ------------------------- -def collect_items_from_mapping( - mapping: Dict[str, Any], - *, - file_hint: str, -) -> List[Dict[str, Any]]: - """ - 將 {路徑鍵: 原文} 的映射轉換為翻譯批次項目。 - 需確保智慧偵測能識別 KubeJS 配置(item["file"] 包含 "/kubejs/")。 - - Shield 整合:對每一個字串值執行 shield_text(), - - 若 skip_reason 非 None(圖片/URL/事件等),直接保留原文不翻譯。 - - 若需要翻譯,用 shield 過的乾淨文字(clean)取代原文字。 - - 同時在 item 字典中附加 _shielded,供 on_translated_item() 做 unshield 回填。 - """ - items: List[Dict[str, Any]] = [] - for k, v in mapping.items(): - if not isinstance(k, str): - continue - if not isinstance(v, str) or not v.strip(): - continue - - # ✅ Rich Text Shield:抽出不應翻譯的格式片段 - shielded = shield_text(v) - - if shielded.skip_reason is not None: - # 不應翻譯(圖片/URL/事件/空白),直接寫入原文不經翻譯管線 - items.append( - { - "file": file_hint, - "path": k, - "source_text": v, - "text": v, # 保持原文 - "cache_type": "kubejs", - "_shielded": shielded, # 供 unshield 回查(此情境無需還原) - "_skip_reason": shielded.skip_reason, - } - ) - else: - # 需要翻譯:使用 shield 過的乾淨文字 - items.append( - { - "file": file_hint, - "path": k, - "source_text": v, - "text": shielded.clean, # ← 使用 shield 過的文字供翻譯 - "cache_type": "kubejs", - "_shielded": shielded, # 供 on_translated_item() 做 unshield - } - ) - - return items - - -def count_translatable_keys(mapping: Dict[str, Any]) -> int: - """計算 mapping 中『可翻譯字串』的數量。 - - 判斷條件: - - value 是字串 - - 去除空白後仍有內容 - - 用途:顯示進度 / 估算翻譯總量。 - """ - return sum(1 for _, v in mapping.items() if isinstance(v, str) and v.strip()) - - -# ------------------------- -# 繁體中文偵測(用於跳過已翻譯的 tooltips) -# ------------------------- -_TW_CJK_RE = re.compile(r"[\u4e00-\u9fff]") -_TW_CONVERTER = opencc.OpenCC("s2tw") - - -def _is_tw_text(text: str) -> bool: - """判斷文字是否已經是繁體中文(OpenCC 轉換後不變 = 已是繁體)。""" - if not text or not isinstance(text, str): - return False - if not _TW_CJK_RE.search(text): - return False # 無 CJK 字元,不是中文 - # 簡體→繁體轉換後,如果等於原本的值,代表原本就是繁體 - return _TW_CONVERTER.convert(text) == text - - -def _split_off_tw_items( - cached_items: List[Dict[str, Any]], - items_to_translate: List[Dict[str, Any]], -) -> List[Dict[str, Any]]: - """從 items_to_translate 移出已翻譯(繁體)的項目到 cached_items,回傳被移動的 items。""" - tw_items: List[Dict[str, Any]] = [] - remaining: List[Dict[str, Any]] = [] - for it in items_to_translate: - src_text = it.get("source_text", "") or "" - if _is_tw_text(src_text): - tw_items.append(it) - else: - remaining.append(it) - cached_items.extend(tw_items) - # 清除原本的並寫回剩餘(in-place 修改串列) - items_to_translate.clear() - items_to_translate.extend(remaining) - return tw_items - - -# ------------------------- -# Dry-run stats (optional) -# ------------------------- -@dataclass -class DryRunStats: - """DryRunStats 類別。 - - 用途:封裝與 DryRunStats 相關的狀態與行為。 - 維護注意:修改公開方法前請確認外部呼叫點與相容性。 - """ - - files: int = 0 - total_keys: int = 0 - cache_hit: int = 0 - cache_miss: int = 0 - per_file: Optional[list[dict]] = None - - -# ------------------------- -# Public API (for UI/pipeline) -# ------------------------- -def translate_kubejs_pending_to_zh_tw( - *, - pending_dir: str | Path, - output_dir: str | Path, - session=None, - rename_langs: Optional[set[str]] = None, - dry_run: bool = False, - write_new_cache: bool = False, -) -> dict: - """ - Translate KubeJS pending JSON dir -> output dir (usually LM翻譯後). - - pending_dir: 例如 Output/kubejs/待翻譯 - - output_dir : 例如 Output/kubejs/LM翻譯後 - """ - validate_api_keys() - - in_dir = Path(pending_dir).resolve() - out_dir = Path(output_dir).resolve() - out_dir.mkdir(parents=True, exist_ok=True) - - if rename_langs is None: - rename_langs = { - "ru_ru", - "ja_jp", - "ko_kr", - "zh_cn", - "zh_hk", - "zh_sg", - "pt_br", - "es_es", - "en_us", - "fr_fr", - "de_de", - "it_it", - "pl_pl", - "tr_tr", - "uk_ua", - "cs_cz", - "hu_hu", - "nl_nl", - "sv_se", - "no_no", - "da_dk", - "fi_fi", - } - - if not in_dir.exists() or not in_dir.is_dir(): - raise FileNotFoundError(f"pending_dir 不存在或不是資料夾:{in_dir}") - - json_files = collect_json_files(in_dir) - if not json_files: - raise FileNotFoundError(f"找不到任何 .json:{in_dir}") - - # ------------------------- - # Pre-scan global total (multithread, like FTB) - # ------------------------- - per_file_counts: List[Tuple[Path, int]] = [] - global_total_keys = 0 - - def _count_one(src: Path) -> Tuple[Path, int]: - """統計單一檔案的翻譯 key 數量。""" - try: - mapping = read_json_dict(src) - return src, int(count_translatable_keys(mapping)) - except Exception as e: - log_warning(f"[KubeJS-LM] 讀取 JSON 失敗 {src}: {e}") - return src, 0 - - max_workers = int( - load_config().get("translator", {}).get("parallel_execution_workers", 4) or 4 - ) - max_workers = max(1, max_workers) - - with ThreadPoolExecutor(max_workers=max_workers) as ex: - futs = [ex.submit(_count_one, p) for p in json_files] - for fu in as_completed(futs): - src, c = fu.result() - per_file_counts.append((src, c)) - global_total_keys += c - - per_file_counts.sort(key=lambda x: x[0].as_posix()) - - if global_total_keys == 0: - log_info("ℹ️ [KubeJS-LM] 0 keys,跳過翻譯") - return {"written_files": 0, "total_keys": 0, "out_dir": str(out_dir)} - - # ------------------------- - # Cache rules - # ------------------------- - cache_rules = {"kubejs": CacheRule("path|source_text")} - - # ------------------------- - # Pre-calc global miss/hit (so progress uses "real translate count") - # ------------------------- - global_total_to_translate = 0 - global_total_hit = 0 - - for src, key_count in per_file_counts: - if key_count == 0: - continue - try: - mapping = read_json_dict(src) - rel_src = src.relative_to(in_dir).as_posix() - file_hint = f"output/kubejs/{rel_src}" # must contain /kubejs/ - all_items = collect_items_from_mapping(mapping, file_hint=file_hint) - - cached_items, items_to_translate = fast_split_items_by_cache( - all_items, - cache_rules=cache_rules, - is_valid_hit=_is_valid_hit, - ) - # ✅ 跳過值已經是繁體中文的 items(節省 API + 避免簡體當英文翻) - _split_off_tw_items(cached_items, items_to_translate) - global_total_hit += len(cached_items) - global_total_to_translate += len(items_to_translate) - except Exception as e: - log_warning(f"[KubeJS-LM] 預掃描失敗 {src}: {e}") - - log_info( - f"🔎 [KubeJS-LM] 待翻譯檔案數:{len(json_files)};總 keys:{global_total_keys}\n" - f"✅ [KubeJS-LM] cache_hit={global_total_hit} | 實際需翻譯(cache_miss)={global_total_to_translate}" - ) - - if global_total_to_translate == 0: - progress(1.0) - - # ------------------------- - # Global build + translate (NO per-file translate) - # ------------------------- - translated_done = 0 - avg_batch_sec = None - total_written = 0 - - per_file_rows: list[dict] = [] - rec = TranslationRecorder() - touch = TouchSet() - _file_write_table: dict[str, tuple[Path, Dict[str, str]]] = {} - - def _writer(file_id: str) -> None: - """寫入翻譯結果到檔案。""" - dst_path, data = _file_write_table[file_id] - write_json_dict(dst_path, data) - - # Build phase: per-file state + global miss list - file_states: dict[str, dict] = {} - all_miss_items: List[Dict[str, Any]] = [] - all_hit_items: List[Dict[str, Any]] = [] - - for idx, (src, key_count) in enumerate(per_file_counts, start=1): - if key_count == 0: - continue - - mapping = read_json_dict(src) - rel_src = src.relative_to(in_dir).as_posix() - - file_hint = f"output/kubejs/{rel_src}" - all_items = collect_items_from_mapping(mapping, file_hint=file_hint) - - cached_items, items_to_translate = fast_split_items_by_cache( - all_items, - cache_rules=cache_rules, - is_valid_hit=_is_valid_hit, - ) - - # ✅ 跳過值已經是繁體中文的 items(從翻譯清單移至 cache hit) - _split_off_tw_items(cached_items, items_to_translate) - - # ✅ NEW:累積 hit items - all_hit_items.extend(cached_items) - - hit = len(cached_items) - miss = len(items_to_translate) - - dst = compute_output_path(src, in_dir, out_dir, rename_langs) - per_file_rows.append( - { - "file": rel_src, - "keys": key_count, - "cache_hit": hit, - "cache_miss": miss, - "dst": dst.relative_to(out_dir).as_posix(), - } - ) - - log_info( - f"[KubeJS-LM] 檔案 {idx}/{len(per_file_counts)}:{rel_src} |" - f"總字串 {key_count},快取命中 {hit},需翻譯 {miss} |" - f"輸出 → {dst.relative_to(out_dir).as_posix()}" - ) - - # out_map base - out_map: Dict[str, str] = { - k: v - for k, v in mapping.items() - if isinstance(k, str) and isinstance(v, str) - } - - # apply cache hits now (record as hit) - for it in cached_items: - p = it.get("path") - t = it.get("text") - if isinstance(p, str) and isinstance(t, str): - # ✅ Rich Text Shield:統一快取命中/miss 路徑 - shielded = it.get("_shielded") - if shielded is not None and shielded.shields: - t = unshield_text(t, shielded.shields) - out_map[p] = t - try: - rec.record( - cache_type="kubejs", - file_id=rel_src, - path=p, - src=str(it.get("source_text") or ""), - dst=t, - cache_hit=True, - extra={"dst_file": dst.relative_to(out_dir).as_posix()}, - ) - except Exception as e: - log_warning(f"[KubeJS-LM] 記錄快取命中失敗: {e}") - - file_id = dst.as_posix() - _file_write_table[file_id] = (dst, out_map) - - file_states[rel_src] = { - "rel_src": rel_src, - "dst": dst, - "file_id": file_id, - "out_map": out_map, - } - - # tag miss items with file context (so callback can write back to correct out_map) - for it in items_to_translate: - it["file_rel"] = rel_src - it["dst_file"] = dst.relative_to(out_dir).as_posix() - - all_miss_items.extend(items_to_translate) - - # ------------------------- - # Dry-run: preview only (no API) - # ------------------------- - if dry_run: - dry_preview_items = all_miss_items[:2000] if all_miss_items else [] - - preview_path = None - try: - meta = { - "files": len(per_file_rows), - "total_keys": global_total_keys, - "cache_hit": global_total_hit, - "cache_miss": global_total_to_translate, - } - - preview_path = write_dry_run_preview( - out_dir, - dry_preview_items, - filename="_kubejs_dry_run_preview.json", - meta=meta, - ) - log_info(f"🧪 [DRY-RUN] preview written: {preview_path.as_posix()}") - - # ✅ NEW:cache hit preview - hit_preview_path = write_cache_hit_preview( - out_dir, - all_hit_items[:2000], # 或不切片 - filename="_kubejs_dry_run_cache_hit_preview.json", - meta=meta, - ) - log_info( - f"🎯 [DRY-RUN] cache-hit preview written: {hit_preview_path.as_posix()}" - ) - - except Exception as e: - log_warning(f"⚠️ [KubeJS-LM] DRY-RUN preview 輸出失敗:{e}") - - progress(1.0) - return { - "dry_run": True, - "files": len(per_file_rows), - "total_keys": global_total_keys, - "cache_hit": global_total_hit, - "cache_miss": global_total_to_translate, - "preview_path": str(preview_path) if preview_path else None, - "out_dir": str(out_dir), - "per_file": per_file_rows, - } - - # ------------------------- - # Real run: translate ONCE for all miss items - # ------------------------- - if not all_miss_items: - # All cache hit -> flush all files once - for st in file_states.values(): - touch.touch(st["file_id"]) - touch.flush(_writer) - total_written = len(file_states) - translated_done = 0 - progress(1.0) - else: - - def on_translated_item(it: Dict[str, Any]) -> None: - """處理翻譯結果。""" - rel_src = it.get("file_rel") - p = it.get("path") - t = it.get("text") - s = it.get("source_text") - - if not (isinstance(rel_src, str) and rel_src in file_states): - return - if not (isinstance(p, str) and isinstance(t, str)): - return - - # ✅ Rich Text Shield:翻譯後還原被保護的格式片段 - shielded = it.get("_shielded") - if shielded is not None and shielded.shields: - final_text = unshield_text(t, shielded.shields) - else: - final_text = t - - st = file_states[rel_src] - st["out_map"][p] = final_text - - try: - rec.record( - cache_type="kubejs", - file_id=rel_src, - path=p, - src=str(s or ""), - dst=t, - cache_hit=False, - extra={"dst_file": st["dst"].relative_to(out_dir).as_posix()}, - ) - except Exception as e: - log_warning(f"[KubeJS-LM] 記錄翻譯結果失敗: {e}") - - try: - touch.touch(st["file_id"]) - except Exception as e: - log_warning(f"[KubeJS-LM] touch 失敗: {e}") - - def on_batch_flushed() -> None: - # write touched files each batch - """批量寫入翻譯結果。""" - try: - touch.flush(_writer) - except Exception as e: - log_warning(f"[KubeJS-LM] 批次刷新失敗,使用 fallback 寫入: {e}") - for fid, (dstp, data) in _file_write_table.items(): - write_json_dict(dstp, data) - - def _fmt_eta(sec: float) -> str: - """格式化剩餘時間。""" - if sec <= 0: - return "" - m, s = divmod(int(sec), 60) - if m >= 60: - h, m2 = divmod(m, 60) - return f"{h}h{m2:02d}m{s:02d}s" - if m > 0: - return f"{m}m{s:02d}s" - return f"{s}s" - - def on_progress(p: float, msg: str, eta_sec: float) -> None: - """報告翻譯進度。""" - eta_txt = _fmt_eta(eta_sec) - log_info(f"{msg}" + (f" | ETA ≈ {eta_txt}" if eta_txt else "")) - progress(p) - - res = translate_items_with_cache_loop( - all_miss_items, - total_for_smart=global_total_to_translate, - translate_batch_smart=lambda batch, total: translate_batch_smart( - batch, total=total - ), - write_new_cache=bool(write_new_cache), - cache_rules=cache_rules, - on_translated_item=on_translated_item, - on_batch_flushed=on_batch_flushed, - on_progress=on_progress, - ) - - translated_done = int(res.processed or 0) - avg_batch_sec = ( - (res.elapsed_sec / res.completed_calls) if res.completed_calls else None - ) - - # final flush all - for st in file_states.values(): - touch.touch(st["file_id"]) - touch.flush(_writer) - - total_written = len(file_states) - progress(1.0) - - # ------------------------- - # Export records - # ------------------------- - rec_json = None - rec_csv = None - try: - rec_json = rec.export_json(out_dir / "translation_map.json") - rec_csv = rec.export_csv(out_dir / "translation_map.csv") - log_info(f"🧾 [KubeJS-LM] records written: {rec_json} | {rec_csv}") - except Exception as ex: - log_info(f"⚠️ [KubeJS-LM] export records failed: {ex}") - - return { - "written_files": total_written, - "total_keys": global_total_keys, - "cache_hit": global_total_hit, - "cache_miss": global_total_to_translate, - "api_translated": translated_done, - "avg_batch_sec": avg_batch_sec, - "out_dir": str(out_dir), - "records_json": str(rec_json) if rec_json else None, - "records_csv": str(rec_csv) if rec_csv else None, - "per_file": per_file_rows, - } +# kubejs_tooltip_lmtranslator.py# ---------------------------------"""?詨??璁汗?箸?寥?蝧餉陌 (translate_batch_smart)嚗?典翰???嗅?蕃霅舀?蝔??踹???蝧餉陌?詨???摮?蝭€??API 瘨€€?擃€翰??瞈?(fast_split_items_by_cache)嚗蝧餉陌?翰??撠?翰??撠€歇蝧餉陌????蝧餉陌????嚗??????€?摰敺芰蝧餉陌 (translate_items_with_cache_loop)嚗??瑕? ETA嚗?閮擗??? 憿舐內???∠ 摰??嚗afe Slicing嚗??€銵??脫迫?甈∟?瘙?憭批??游仃?€??琿?蝥??摮?雿輻 TouchSet ??Writer Flush 璈嚗蕃霅舫?蝔葉??神?交?獢??喃蝙蝔???銝剜銋??憭勗歇摰??€脣漲???炎璅∪? (dry_run)嚗?湔芋?砍銵蒂頛詨?汗瑼?嚗rite_dry_run_preview嚗?霈??冽迤撘???API 憿漲?Ⅱ隤撘?行迤蝣箝€??豢?撠 (TranslationRecorder)嚗?游?蝧餉陌閮?撠??JSON ??CSV ?澆?嚗靘踹?蝥撠?鈭活????脣漲餈質馱 (session.set_progress)嚗撱粹€脣漲?文?嚗ook嚗??臬??亙???UI ?隤頂蝯梢*蝷箇蕃霅舐????頝臬??芸?嚗????蝟餉??冗頧?嚗?憒????en_us ?芸?撠???zh_tw ?桅?嚗€?Rich Text Shield嚗hield_text() / unshield_text() 靽風 KubeJS ?澆?嚗蔗?脩Ⅳ??D?RL 蝑?嚗??函蕃霅臬??賢嚗蕃霅臬???嚗??LM 隤斤蕃?澆?璅???"""from __future__ import annotationsfrom dataclasses import dataclassfrom pathlib import Pathfrom typing import Dict, Any, List, Optional, Tuplefrom concurrent.futures import ThreadPoolExecutor, as_completedimport reimport openccfrom translation_tool.core.lm_translator_main import translate_batch_smartfrom translation_tool.core.lm_config_rules import validate_api_keysfrom translation_tool.utils.config_manager import load_configfrom translation_tool.core.lm_translator_shared import ( CacheRule, fast_split_items_by_cache, translate_items_with_cache_loop, TouchSet, TranslationRecorder, write_dry_run_preview, write_cache_hit_preview, # ???啣?嚗ache hit preview 瑼? _is_valid_hit, # ???啣?嚗ache hit ?斗)from translation_tool.plugins.shared.json_io import ( read_json_dict, write_json_dict, collect_json_files,)from translation_tool.plugins.shared.lang_path_rules import ( compute_output_path,)from translation_tool.plugins.shared.rich_text_shield import ( shield_text, unshield_text,)from translation_tool.utils.log_unit import log_info, log_warning, progress# -------------------------# Smart item mapping# -------------------------def collect_items_from_mapping( mapping: Dict[str, Any], *, file_hint: str,) -> List[Dict[str, Any]]: """ 撠?{頝臬??? ??} ??撠??蝧餉陌?寞活??? ?€蝣箔??箸?菜葫?質???KubeJS ?蔭嚗tem["file"] ? "/kubejs/"嚗€? Shield ?游?嚗?瘥???銝脣€澆銵?shield_text()嚗? - ??skip_reason ??None嚗???URL/鈭辣蝑?嚗?乩?????蝧餉陌?? - ?仿?閬蕃霅荔???shield ??銋暹楊??嚗lean嚗?隞?????? - ????item 摮銝剝???_shielded嚗? on_translated_item() ??unshield ?‵?? """ items: List[Dict[str, Any]] = [] for k, v in mapping.items(): if not isinstance(k, str): continue if not isinstance(v, str) or not v.strip(): continue # ??Rich Text Shield嚗?箔??蕃霅舐??澆??挾 shielded = shield_text(v) if shielded.skip_reason is not None: # 銝?蝧餉陌嚗???URL/鈭辣/蝛箇嚗??湔撖怠??銝?蝧餉陌蝞∠? items.append( { "file": file_hint, "path": k, "source_text": v, "text": v, # 靽??? "cache_type": "kubejs", "_shielded": shielded, # 靘?unshield ?嚗迨???⊿???嚗? "_skip_reason": shielded.skip_reason, } ) else: # ?€閬蕃霅荔?雿輻 shield ??銋暹楊?? items.append( { "file": file_hint, "path": k, "source_text": v, "text": shielded.clean, # ??雿輻 shield ????靘蕃霅? "cache_type": "kubejs", "_shielded": shielded, # 靘?on_translated_item() ??unshield } ) return itemsdef count_translatable_keys(mapping: Dict[str, Any]) -> int: """閮? mapping 銝准€蝧餉陌摮葡???賊??? ?斗璇辣嚗? - value ?臬?銝? - ?駁蝛箇敺??摰? ?券€?憿舐內?脣漲 / 隡啁?蝧餉陌蝮賡??? """ return sum(1 for _, v in mapping.items() if isinstance(v, str) and v.strip())# -------------------------# 蝜?銝剜??菜葫嚗?潸歲?歇蝧餉陌??tooltips嚗?# -------------------------_TW_CJK_RE = re.compile(r"[\u4e00-\u9fff]")_TW_CONVERTER = opencc.OpenCC("s2tw")def _is_tw_text(text: str) -> bool: """?斗???臬撌脩??舐?擃葉??OpenCC 頧?敺?霈?= 撌脫蝜?嚗€?"" if not text or not isinstance(text, str): return False if not _TW_CJK_RE.search(text): return False # ??CJK 摮?嚗??臭葉?? # 蝪⊿???擃???嚗????澆??祉??潘?隞?”?撠望蝜? return _TW_CONVERTER.convert(text) == textdef _split_off_tw_items( cached_items: List[Dict[str, Any]], items_to_translate: List[Dict[str, Any]],) -> List[Dict[str, Any]]: """敺?items_to_translate 蝘餃撌脩蕃霅荔?蝜?嚗????cached_items嚗??唾◤蝘餃???items??"" tw_items: List[Dict[str, Any]] = [] remaining: List[Dict[str, Any]] = [] for it in items_to_translate: src_text = it.get("source_text", "") or "" if _is_tw_text(src_text): tw_items.append(it) else: remaining.append(it) cached_items.extend(tw_items) # 皜??蒂撖怠??拚?嚗n-place 靽格銝脣?嚗? items_to_translate.clear() items_to_translate.extend(remaining) return tw_items# -------------------------# Dry-run stats (optional)# -------------------------@dataclassclass DryRunStats: """DryRunStats 憿?? ?券€?撠???DryRunStats ?賊?????銵?? 蝬剛風瘜冽?嚗耨?孵?瘜?隢Ⅱ隤??典?恍??摰寞€扼€? """ files: int = 0 total_keys: int = 0 cache_hit: int = 0 cache_miss: int = 0 per_file: Optional[list[dict]] = None# -------------------------# Public API (for UI/pipeline)# -------------------------def translate_kubejs_pending_to_zh_tw( *, pending_dir: str | Path, output_dir: str | Path, session=None, rename_langs: Optional[set[str]] = None, dry_run: bool = False, write_new_cache: bool = False,) -> dict: """ Translate KubeJS pending JSON dir -> output dir (usually LM蝧餉陌敺?. - pending_dir: 靘? Output/kubejs/敺蕃霅? - output_dir : 靘? Output/kubejs/LM蝧餉陌敺? """ validate_api_keys() in_dir = Path(pending_dir).resolve() out_dir = Path(output_dir).resolve() out_dir.mkdir(parents=True, exist_ok=True) if rename_langs is None: rename_langs = { "ru_ru", "ja_jp", "ko_kr", "zh_cn", "zh_hk", "zh_sg", "pt_br", "es_es", "en_us", "fr_fr", "de_de", "it_it", "pl_pl", "tr_tr", "uk_ua", "cs_cz", "hu_hu", "nl_nl", "sv_se", "no_no", "da_dk", "fi_fi", } if not in_dir.exists() or not in_dir.is_dir(): raise FileNotFoundError(f"pending_dir 銝??冽?銝鞈?憭橘?{in_dir}") json_files = collect_json_files(in_dir) if not json_files: raise FileNotFoundError(f"?曆??唬遙雿?.json嚗in_dir}") # ------------------------- # Pre-scan global total (multithread, like FTB) # ------------------------- per_file_counts: List[Tuple[Path, int]] = [] global_total_keys = 0 # ??Issue #7b 靽桀儔嚗楨摮?JSON mapping ?踹???霈€?? src_mapping_cache: Dict[Path, Dict[str, Any]] = {} def _count_one(src: Path) -> Tuple[Path, int, Dict[str, Any]]: """蝯梯??桐?瑼??蕃霅?key ?賊?嚗??翰??mapping??"" try: mapping = read_json_dict(src) return src, int(count_translatable_keys(mapping)), mapping except Exception: return src, 0, {} max_workers = int( load_config().get("translator", {}).get("parallel_execution_workers", 4) or 4 ) max_workers = max(1, max_workers) with ThreadPoolExecutor(max_workers=max_workers) as ex: futs = [ex.submit(_count_one, p) for p in json_files] for fu in as_completed(futs): src, c, mapping = fu.result() per_file_counts.append((src, c)) global_total_keys += c if mapping: # ???芰楨摮?蝛箇? mapping src_mapping_cache[src] = mapping per_file_counts.sort(key=lambda x: x[0].as_posix()) if global_total_keys == 0: log_info("?對? [KubeJS-LM] 0 keys嚗歲?蕃霅?) return {"written_files": 0, "total_keys": 0, "out_dir": str(out_dir)} # ------------------------- # Cache rules # ------------------------- cache_rules = {"kubejs": CacheRule("path|source_text")} # ------------------------- # Pre-calc global miss/hit (so progress uses "real translate count") # ------------------------- global_total_to_translate = 0 global_total_hit = 0 for src, key_count in per_file_counts: if key_count == 0: continue try: # ??Issue #7b 靽桀儔嚗?乩蝙?函楨摮? mapping嚗???銴??? mapping = src_mapping_cache.get(src, {}) if not mapping: continue rel_src = src.relative_to(in_dir).as_posix() file_hint = f"output/kubejs/{rel_src}" # must contain /kubejs/ all_items = collect_items_from_mapping(mapping, file_hint=file_hint) cached_items, items_to_translate = fast_split_items_by_cache( all_items, cache_rules=cache_rules, is_valid_hit=_is_valid_hit, ) # ??頝喲??澆歇蝬蝜?銝剜???items嚗???API + ?踹?蝪⊿??嗉?蕃嚗? _split_off_tw_items(cached_items, items_to_translate) global_total_hit += len(cached_items) global_total_to_translate += len(items_to_translate) except Exception: pass log_info( f"?? [KubeJS-LM] 敺蕃霅舀?獢嚗len(json_files)}嚗蜇 keys嚗global_total_keys}\n" f"??[KubeJS-LM] cache_hit={global_total_hit} | 撖阡??€蝧餉陌(cache_miss)={global_total_to_translate}" ) if global_total_to_translate == 0: progress(1.0) # ------------------------- # Global build + translate (NO per-file translate) # ------------------------- translated_done = 0 avg_batch_sec = None total_written = 0 per_file_rows: list[dict] = [] rec = TranslationRecorder() touch = TouchSet() _file_write_table: dict[str, tuple[Path, Dict[str, str]]] = {} def _writer(file_id: str) -> None: """撖怠蝧餉陌蝯??唳?獢€?"" dst_path, data = _file_write_table[file_id] write_json_dict(dst_path, data) # Build phase: per-file state + global miss list file_states: dict[str, dict] = {} all_miss_items: List[Dict[str, Any]] = [] all_hit_items: List[Dict[str, Any]] = [] for idx, (src, key_count) in enumerate(per_file_counts, start=1): if key_count == 0: continue # ??Issue #7b 靽桀儔嚗?乩蝙?函楨摮? mapping嚗???銴??? mapping = src_mapping_cache.get(src, {}) if not mapping: continue rel_src = src.relative_to(in_dir).as_posix() file_hint = f"output/kubejs/{rel_src}" all_items = collect_items_from_mapping(mapping, file_hint=file_hint) cached_items, items_to_translate = fast_split_items_by_cache( all_items, cache_rules=cache_rules, is_valid_hit=_is_valid_hit, ) # ??頝喲??澆歇蝬蝜?銝剜???items嚗?蝧餉陌皜蝘餉 cache hit嚗? _split_off_tw_items(cached_items, items_to_translate) # ??NEW嚗敞蝛?hit items all_hit_items.extend(cached_items) hit = len(cached_items) miss = len(items_to_translate) dst = compute_output_path(src, in_dir, out_dir, rename_langs) per_file_rows.append( { "file": rel_src, "keys": key_count, "cache_hit": hit, "cache_miss": miss, "dst": dst.relative_to(out_dir).as_posix(), } ) log_info( f"[KubeJS-LM] 瑼? {idx}/{len(per_file_counts)}嚗rel_src} 嚚? f"蝮賢?銝?{key_count}嚗翰?銝?{hit}嚗?蝧餉陌 {miss} 嚚? f"頛詨 ??{dst.relative_to(out_dir).as_posix()}" ) # out_map base out_map: Dict[str, str] = { k: v for k, v in mapping.items() if isinstance(k, str) and isinstance(v, str) } # apply cache hits now (record as hit) for it in cached_items: p = it.get("path") t = it.get("text") if isinstance(p, str) and isinstance(t, str): out_map[p] = t try: rec.record( cache_type="kubejs", file_id=rel_src, path=p, src=str(it.get("source_text") or ""), dst=t, cache_hit=True, extra={"dst_file": dst.relative_to(out_dir).as_posix()}, ) except Exception: pass file_id = dst.as_posix() _file_write_table[file_id] = (dst, out_map) file_states[rel_src] = { "rel_src": rel_src, "dst": dst, "file_id": file_id, "out_map": out_map, } # tag miss items with file context (so callback can write back to correct out_map) for it in items_to_translate: it["file_rel"] = rel_src it["dst_file"] = dst.relative_to(out_dir).as_posix() all_miss_items.extend(items_to_translate) # ------------------------- # Dry-run: preview only (no API) # ------------------------- if dry_run: dry_preview_items = all_miss_items[:2000] if all_miss_items else [] preview_path = None try: meta = { "files": len(per_file_rows), "total_keys": global_total_keys, "cache_hit": global_total_hit, "cache_miss": global_total_to_translate, } preview_path = write_dry_run_preview( out_dir, dry_preview_items, filename="_kubejs_dry_run_preview.json", meta=meta, ) log_info(f"?妒 [DRY-RUN] preview written: {preview_path.as_posix()}") # ??NEW嚗ache hit preview hit_preview_path = write_cache_hit_preview( out_dir, all_hit_items[:2000], # ???? filename="_kubejs_dry_run_cache_hit_preview.json", meta=meta, ) log_info( f"? [DRY-RUN] cache-hit preview written: {hit_preview_path.as_posix()}" ) except Exception as e: log_warning(f"?? [KubeJS-LM] DRY-RUN preview 頛詨憭望?嚗e}") progress(1.0) return { "dry_run": True, "files": len(per_file_rows), "total_keys": global_total_keys, "cache_hit": global_total_hit, "cache_miss": global_total_to_translate, "preview_path": str(preview_path) if preview_path else None, "out_dir": str(out_dir), "per_file": per_file_rows, } # ------------------------- # Real run: translate ONCE for all miss items # ------------------------- if not all_miss_items: # All cache hit -> flush all files once for st in file_states.values(): touch.touch(st["file_id"]) touch.flush(_writer) total_written = len(file_states) translated_done = 0 progress(1.0) else: def on_translated_item(it: Dict[str, Any]) -> None: """??蝧餉陌蝯???"" rel_src = it.get("file_rel") p = it.get("path") t = it.get("text") s = it.get("source_text") if not (isinstance(rel_src, str) and rel_src in file_states): return if not (isinstance(p, str) and isinstance(t, str)): return # ??Rich Text Shield嚗蕃霅臬???鋡思?霅瑞??澆??挾 shielded = it.get("_shielded") if shielded is not None and shielded.shields: final_text = unshield_text(t, shielded.shields) else: final_text = t st = file_states[rel_src] st["out_map"][p] = final_text try: rec.record( cache_type="kubejs", file_id=rel_src, path=p, src=str(s or ""), dst=t, cache_hit=False, extra={"dst_file": st["dst"].relative_to(out_dir).as_posix()}, ) except Exception: pass try: touch.touch(st["file_id"]) except Exception: pass def on_batch_flushed() -> None: # write touched files each batch """?寥?撖怠蝧餉陌蝯???"" try: touch.flush(_writer) except Exception: # fallback: write all for fid, (dstp, data) in _file_write_table.items(): write_json_dict(dstp, data) def _fmt_eta(sec: float) -> str: """?澆??擗??€?"" if sec <= 0: return "" m, s = divmod(int(sec), 60) if m >= 60: h, m2 = divmod(m, 60) return f"{h}h{m2:02d}m{s:02d}s" if m > 0: return f"{m}m{s:02d}s" return f"{s}s" def on_progress(p: float, msg: str, eta_sec: float) -> None: """?勗?蝧餉陌?脣漲??"" eta_txt = _fmt_eta(eta_sec) log_info(f"{msg}" + (f" | ETA ??{eta_txt}" if eta_txt else "")) progress(p) res = translate_items_with_cache_loop( all_miss_items, total_for_smart=global_total_to_translate, translate_batch_smart=lambda batch, total: translate_batch_smart( batch, total=total ), write_new_cache=bool(write_new_cache), cache_rules=cache_rules, on_translated_item=on_translated_item, on_batch_flushed=on_batch_flushed, on_progress=on_progress, ) translated_done = int(res.processed or 0) avg_batch_sec = ( (res.elapsed_sec / res.completed_calls) if res.completed_calls else None ) # final flush all for st in file_states.values(): touch.touch(st["file_id"]) touch.flush(_writer) total_written = len(file_states) progress(1.0) # ------------------------- # Export records # ------------------------- rec_json = None rec_csv = None try: rec_json = rec.export_json(out_dir / "translation_map.json") rec_csv = rec.export_csv(out_dir / "translation_map.csv") log_info(f"?屁 [KubeJS-LM] records written: {rec_json} | {rec_csv}") except Exception as ex: log_info(f"?? [KubeJS-LM] export records failed: {ex}") return { "written_files": total_written, "total_keys": global_total_keys, "cache_hit": global_total_hit, "cache_miss": global_total_to_translate, "api_translated": translated_done, "avg_batch_sec": avg_batch_sec, "out_dir": str(out_dir), "records_json": str(rec_json) if rec_json else None, "records_csv": str(rec_csv) if rec_csv else None, "per_file": per_file_rows, } \ No newline at end of file From c2a1a6096dbdc0dba61810a916cb7bca526f104b Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Thu, 26 Mar 2026 00:07:04 +0800 Subject: [PATCH 27/33] =?UTF-8?q?fix:=20=E5=AE=8C=E6=95=B4=E4=BF=AE?= =?UTF-8?q?=E5=BE=A9=20Issue=20#7/#8=EF=BC=88=E5=90=AB=20src=5Fmapping=5Fc?= =?UTF-8?q?ache=20+=20callback=20factory=20+=20unshield.shields=EF=BC=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - ftbquests_lmtranslator.py: src_mapping_cache 緩存 JSON mapping,callback 工廠函式 - kubejs_tooltip_lmtranslator.py: src_mapping_cache 緩存,確認 unshield.shields - 驗證:1161 tests passed --- .../ftbquests/ftbquests_lmtranslator.py | 657 +++++++++++++++++- .../kubejs/kubejs_tooltip_lmtranslator.py | 603 +++++++++++++++- 2 files changed, 1258 insertions(+), 2 deletions(-) diff --git a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py index a899ebe0..a302f5a4 100644 --- a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py +++ b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py @@ -1 +1,656 @@ -"""translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py 璅∠????券€????祆?獢?蝢拍????蝔?靘?獢隞芋蝯?怒€?蝬剛風瘜冽?嚗瑼??撘?docstring ?冽蝬剛風隤芣?嚗?隞?”銵霈??"""# ftbquests_lmtranslator.py# ------------------------------------------------------------# FTB Quests (Already Extracted JSON) -> translate_batch_smart -> Export translated JSON maps# - DOES NOT run extractor# - Reads extracted .json files containing {key: text} pairs# - Outputs translated {key: text} JSON with same relative path structure# - If input file is like ru_ru.json (or other lang codes), output renamed to zh_tw.json# - total passed to smart = global total keys across all files# ------------------------------------------------------------from __future__ import annotationsfrom dataclasses import dataclassfrom pathlib import Pathfrom typing import Dict, Any, List, Optional, Tuplefrom concurrent.futures import ThreadPoolExecutor, as_completedimport mathfrom translation_tool.core.lm_translator_main import translate_batch_smartfrom translation_tool.core.lm_config_rules import validate_api_keysfrom translation_tool.utils.config_manager import load_configfrom translation_tool.core.lm_translator_shared import ( fast_split_items_by_cache, # ???啣?嚗???瘚? translate_items_with_cache_loop, CacheRule, TouchSet, # ???啣?嚗ouched/flush TranslationRecorder, # ???啣?嚗蕃霅航??? write_dry_run_preview, # ???啣?嚗ry-run preview 瑼? write_cache_hit_preview, # ???啣?嚗ache hit preview 瑼? _is_valid_hit, # ???啣?嚗ache hit ?斗 _get_default_batch_size,)from translation_tool.plugins.shared.json_io import ( read_json_dict, write_json_dict, collect_json_files,)from translation_tool.plugins.shared.lang_path_rules import ( compute_output_path,)from translation_tool.plugins.shared.lang_text_rules import is_already_zhfrom translation_tool.plugins.shared.rich_text_shield import shield_text, unshield_textfrom translation_tool.utils.log_unit import ( log_info, log_error,)# -------------------------# Smart 蝧餉陌頧?剁?鞈??澆?頧?嚗?# -------------------------def map_to_items( mapping: Dict[str, Any], cache_type: str, file_hint: str,) -> List[Dict[str, Any]]: """ 撠?{key: text} ??憪???頧???translate_batch_smart ?航??? item ?澆??? 瘥???item 隞?”??璇蝧餉陌?????◤?€莎? - cache 瘥? - batch 蝧餉陌 - 蝧餉陌蝯??神 ??閬身閮??€? - smart 蝧餉陌?冽??? item["file"] ?斗鞈?憿? - FTB Quests ??瑟?隞嗆嚗楝敺葉敹?? "/ftbquests/" - ?迨 file_hint??摰?????"/ftbquests/" - ??銝? "/lang/"嚗??鋡怨炊?斤 Minecraft Lang 蝧餉陌 ???舐隞€暻?file_hint 銝撖阡?瑼?頝臬?嚗€??????內頝臬??€? """ items: List[Dict[str, Any]] = [] for k, v in mapping.items(): # key 敹??臬?銝莎?隤? key嚗? if not isinstance(k, str): continue # value 敹??舫?蝛箇摮葡嚗祕??蝧餉陌?摰對? if not isinstance(v, str) or not v.strip(): continue items.append( { # ?? smart translator ?斗?函?瑼??內頝臬? # ?? 敹?? "/ftbquests/" "file": file_hint, # 隤? key嚗?憒?quest.xxx.title嚗? "path": k, # ????嚗翰??瘥??剁? "source_text": v, # ?嗅???嚗?鋡怎蕃霅臬閬神嚗? "text": v, # ??敹怠???嚗???cache_rules嚗? "cache_type": cache_type, # 靘? "ftbquests" } ) return itemsdef count_translatable_keys(mapping: Dict[str, Any]) -> int: """ 閮? mapping 銝准€祕?蝧餉陌??銝脫?€€? ?斗璇辣嚗? - value 敹??臬?銝? - ?駁蝛箇敺??摰? ?€???其?嚗? - 憿舐內?脣漲 - 閮? cache hit / miss - 雿 batch 蝧餉陌?蜇?皞? """ return sum(1 for _, v in mapping.items() if isinstance(v, str) and v.strip())@dataclassclass DryRunStats: """ Dry-run嚗岫頝?璅∪?銝?蝯梯?鞈?蝯??? ?券€? - UI 憿舐內 - ?汗蝧餉陌閬芋 - 蝣箄? cache ?賭葉瘥??臬?? """ files: int = 0 # ????獢 total_keys: int = 0 # 蝮賢?銝脫 cache_hit: int = 0 # 敹怠??賭葉?? cache_miss: int = 0 # 撖阡??€蝧餉陌?? per_file: list[dict] = None # 瘥€?獢??敦# -------------------------# Public API (callable from pipeline)# -------------------------def translate_ftb_pending_to_zh_tw( *, input_lang_dir: str | Path, output_lang_dir: str | Path, session=None, rename_langs: Optional[set[str]] = None, dry_run: bool = False, # ???啣? write_new_cache: bool = True, # ???啣?) -> dict: """ 撠策 UI/???澆嚗? - input_lang_dir: 靘? /敺蕃霅?config/ftbquests/quests/lang 嚗釣??閬??lang ?惜嚗??詨?頝臬???en_us/ ??踵???zh_tw嚗? - output_lang_dir: 靘? /config/ftbquests/quests/lang 嚗撓?箸??芸???en_us 鞈?憭暹?? zh_tw嚗? - session: ?舫嚗靘?add_log / set_progress """ validate_api_keys() in_dir = Path(input_lang_dir).resolve() out_dir = Path(output_lang_dir).resolve() out_dir.mkdir(parents=True, exist_ok=True) def set_prog(v: float): """閮剖?蝧餉陌?脣漲??"" if session is not None and hasattr(session, "set_progress"): try: session.set_progress(v) except Exception: pass # rename langs ?身瘝輻雿?CLI ???殷?雿?pending ?虜?芣? en_us嚗?憭芣??典嚗? if rename_langs is None: rename_langs = { "ru_ru", "ja_jp", "ko_kr", "zh_cn", "zh_hk", "zh_sg", "pt_br", "es_es", "en_us", "fr_fr", "de_de", "it_it", "pl_pl", "tr_tr", "uk_ua", "cs_cz", "hu_hu", "nl_nl", "sv_se", "no_no", "da_dk", "fi_fi", } if not in_dir.exists() or not in_dir.is_dir(): raise FileNotFoundError(f"input_lang_dir 銝??冽?銝鞈?憭橘?{in_dir}") json_files = collect_json_files(in_dir) if not json_files: raise FileNotFoundError(f"?曆??唬遙雿?.json嚗in_dir}") # ---- Global total keys (raw) + cache mappings ---- per_file_counts: List[Tuple[Path, int]] = [] global_total_keys = 0 # ??Issue #7 靽桀儔嚗楨摮?JSON mapping ?踹???霈€?? src_mapping_cache: Dict[Path, Dict[str, Any]] = {} def _count_one(src: Path) -> Tuple[Path, int, Dict[str, Any]]: """霈€??JSON 銝衣絞閮蝧餉陌?萄€潭????敹怠? mapping??"" try: mapping = read_json_dict(src) c = count_translatable_keys(mapping) return src, int(c), mapping except Exception: return src, 0, {} # max_workers 雿隞交??config ??parallel_execution_workers max_workers = ( load_config().get("translator", {}).get("parallel_execution_workers", 4) ) with ThreadPoolExecutor(max_workers=max_workers) as ex: futs = [ex.submit(_count_one, src) for src in json_files] for fu in as_completed(futs): src, c, mapping = fu.result() per_file_counts.append((src, c)) global_total_keys += c if mapping: # ???芰楨摮?蝛箇? mapping src_mapping_cache[src] = mapping # 靽?蝛拙???嚗???瑁?蝺??湔?摨?嚗? per_file_counts.sort(key=lambda x: x[0].as_posix()) if global_total_keys == 0: log_info("?對? [FTB-LM] 0 keys嚗歲?蕃霅?) return {"written_files": 0, "total_keys": 0, "out_dir": str(out_dir)} # ---- Cache rules ---- cache_rules = {"ftbquests": CacheRule("path|source_text")} # ????閮??祕??蝧餉陌?蜇???典? miss嚗? global_total_to_translate = 0 global_total_hit = 0 for src, key_count in per_file_counts: if key_count == 0: continue try: # ??Issue #7 靽桀儔嚗?乩蝙?函楨摮? mapping嚗???銴??? mapping = src_mapping_cache.get(src, {}) if not mapping: continue rel_src = src.relative_to(in_dir).as_posix() file_hint = f"config/ftbquests/quests/{rel_src}" all_items = map_to_items( mapping, cache_type="ftbquests", file_hint=file_hint ) cached_items, items_to_translate = fast_split_items_by_cache( all_items, cache_rules=cache_rules, is_valid_hit=_is_valid_hit, ) # ??銝剜?/撌脩蕃霅荔?銝€?LM嚗?銝? cache_miss already_zh_items = [] real_to_translate = [] for it in items_to_translate: s = str(it.get("source_text") or it.get("text") or "") if is_already_zh(s): already_zh_items.append(it) else: real_to_translate.append(it) global_total_hit += len(cached_items) global_total_to_translate += len(real_to_translate) except Exception: pass log_info( f"\n?? [FTB-LM][??摰] ?潛敺???獢?{len(json_files)} ??| ?蝮賣??殷?{global_total_keys} 璇? f"\n??[FTB-LM][?脣漲??] 撌脣?敹怠?頛嚗global_total_hit} 璇?| ?拚??€ AI 蝧餉陌嚗global_total_to_translate} 璇? ) if global_total_to_translate == 0: log_info( "?對? [FTB-LM][?€? ?剖?嚗??摰寧??賭葉敹怠?嚗?€隤輻 AI嚗??湔撠瑼??? ) if global_total_to_translate == 0: set_prog(0.99) # ?踹??祇?頝?1.0嚗? UI ???典撠???閬死?? # ---- Translate per file (shared loop + cache) ---- translated_done = 0 # ???芰? API 蝧餉陌摰?嚗蜓?脣漲??嚗? cache_hit_done_so_far = 0 # ??撌脣神?亥撓?箇? cache hit嚗擃??嚗? total_written = 0 # ---- Dry-run stats container ---- per_file_rows: list[dict] = [] # ---- Dry-run: 銝蕃霅胯€?頛詨 ---- dry_preview_items: List[Dict[str, Any]] = [] all_cached_items: list[dict] = [] rec = TranslationRecorder() touch = TouchSet() # touched_files 撠???writer嚗?撠??嚗?瑼??銝€??dst/out_map嚗? _file_write_table: dict[str, tuple[Path, Dict[str, str]]] = {} def _writer(file_id: str) -> None: """撖怠蝧餉陌蝯??唳?獢€?"" dst_path, data = _file_write_table[file_id] write_json_dict(dst_path, data) for idx, (src, key_count) in enumerate(per_file_counts, start=1): if key_count == 0: continue # ??Issue #7 靽桀儔嚗?乩蝙?函楨摮? mapping嚗???銴??? mapping = src_mapping_cache.get(src, {}) if not mapping: log_error(f"?? [FTB-LM] ?曆??啣翰?? mapping嚗src}") continue rel_src = src.relative_to(in_dir).as_posix() # e.g. en_us/ftb_lang.json file_hint = ( f"config/ftbquests/quests/{rel_src}" # ??include /ftbquests/ and no /lang/ ) all_items = map_to_items(mapping, cache_type="ftbquests", file_hint=file_hint) cached_items, items_to_translate = fast_split_items_by_cache( all_items, cache_rules=cache_rules, is_valid_hit=_is_valid_hit, ) already_zh_items = [] real_to_translate = [] for it in items_to_translate: s = str(it.get("source_text") or it.get("text") or "") if is_already_zh(s): already_zh_items.append(it) else: real_to_translate.append(it) items_to_translate = real_to_translate # ??閬?????蝧餉陌?? hit = len(cached_items) miss = len(items_to_translate) already_zh = len(already_zh_items) if hit > 0: sample = cached_items[0] log_info(f"?? [FTB-LM] 敹怠??賭葉蝭?嚗sample.get('path')}") if miss > 0: sample = items_to_translate[0] log_info(f"?? [FTB-LM] 敺蕃霅舐?靘?{sample.get('path')}") dst = compute_output_path(src, in_dir, out_dir, rename_langs) log_info( f"?? [FTB-LM][瑼? {idx}/{len(per_file_counts)}] 甇???嚗rel_src}\n" f"?? ?祆??敦嚗蜇璇 {key_count} | 頛敹怠? {hit} | 撌脣銝剜?(頝喲?) {already_zh} | ?€蝧餉陌 {miss}\n" f"? 頛詨頝臬?嚗dst.relative_to(out_dir).as_posix()}" ) per_file_rows.append( { "file": rel_src, "keys": key_count, "cache_hit": hit, "cache_miss": miss, "dst": dst.relative_to(out_dir).as_posix(), } ) # ---- Dry-run: 銝蕃霅胯€?頛詨 ---- if dry_run: # ???園? preview 璅?嚗??憭改??€憭?2000嚗? if len(dry_preview_items) < 2000: dry_preview_items.extend( items_to_translate[: 2000 - len(dry_preview_items)] ) # ??NEW嚗??cache hit嚗??憭改??€憭?2000嚗?芾?隤踵嚗? if len(all_cached_items) < 2000: all_cached_items.extend(cached_items[: 2000 - len(all_cached_items)]) # 隞交?獢?誨蝧餉陌??脣漲??嚗?蕃霅舫???銝???閬死頝唾? set_prog(min(idx / len(per_file_counts), 1.0)) log_info( f"?妒 [皜祈岫璅∪?] ?脣漲嚗idx}/{len(per_file_counts)}\n" f"?砍?頛詨嚗rel_src} ??{dst.relative_to(out_dir).as_posix()}\n" f"?摯?敦嚗蜇璇 {key_count} | 敹怠??賭葉 {hit} | ?€蝧餉陌 {miss} | ?摯?寞活 {math.ceil(miss / _get_default_batch_size('ftbquests', None)) if _get_default_batch_size('ftbquests', None) > 0 else 'N/A'}\n" f"蝝航??摯?脣漲嚗translated_done} / {global_total_to_translate}" ) continue # ============================ # ???迤蝧餉陌嚗? dry-run嚗? # ============================ # out_map嚗??典??摨??踹?銝剜?撓?箇撩 key嚗? out_map: Dict[str, str] = { k: v for k, v in mapping.items() if isinstance(k, str) and isinstance(v, str) } # cache hits 閬? for it in cached_items: p = it.get("path") t = it.get("text") if isinstance(p, str) and isinstance(t, str): out_map[p] = t try: rec.record( cache_type="ftbquests", file_id=rel_src, path=p, src=str(it.get("source_text") or ""), dst=t, cache_hit=True, extra={"dst_file": dst.relative_to(out_dir).as_posix()}, ) except Exception: pass # ?典銝?cache嚗?亥撓?? if not items_to_translate: # write_json_dict(dst, out_map) file_id = dst.as_posix() _file_write_table[file_id] = (dst, out_map) touch.touch(file_id) touch.flush(_writer) # ?€撠??蝑?雿?冽?甈⊿撖恬?雿粥???恣蝺? total_written += 1 # ???€?獢? cache hit 撌脩??迤撖恍€脰撓?綽??湧?摰?閬? cache_hit_done_so_far += hit # ??銝駁€脣漲銝?嚗ranslated_done 銝?嚗? set_prog(min(translated_done / max(global_total_to_translate, 1), 1.0)) overall_done = cache_hit_done_so_far + translated_done progress_pct = ( (overall_done / global_total_keys * 100) if global_total_keys > 0 else 0 ) log_info( f"??[敹怠?頝喲?] 瑼? {idx}/{len(per_file_counts)}嚗rel_src}\n" f"? 撠頝臬?嚗dst.relative_to(out_dir).as_posix()}\n" f"?? ??蝯?嚗銝剖翰??{hit} 璇?| API 隤輻 0 璇n" f"? ?典?摰?摨佗?{overall_done} / {global_total_keys} ({progress_pct:.1f}%)" ) continue # ??Issue #8 靽桀儔嚗? callbacks 摰儔?刻艘???剁??寧撌亙??賢? def _fmt_eta(sec: float) -> str: """?澆??擗??€?"" if sec <= 0: return "" m, s = divmod(int(sec), 60) if m > 0: return f"{m}m{s:02d}s" return f"{s}s" def make_on_progress(set_prog, _fmt_eta): def on_progress(p: float, msg: str, eta_sec: float) -> None: """?勗?蝧餉陌?脣漲??"" eta_txt = _fmt_eta(eta_sec) if eta_txt: log_info(f"??[AI 蝧餉陌銝苗 {msg} | ?摯?拚???嚗eta_txt}") else: log_info(f"?? [AI 蝧餉陌銝苗 {msg}") set_prog(p) return on_progress def make_on_translated_item(rel_src, dst, out_map, rec, out_dir): def on_translated_item(it: Dict[str, Any]) -> None: """??蝧餉陌蝯?銝血神?交?撠€?"" p = it.get("path") t = it.get("text") src_text = str(it.get("source_text") or "") if isinstance(p, str) and isinstance(t, str): try: shielded_src = shield_text(src_text) t = unshield_text(t, shielded_src.shields) except Exception: pass out_map[p] = t try: rec.record( cache_type="ftbquests", file_id=rel_src, path=p, src=src_text, dst=t, cache_hit=False, extra={"dst_file": dst.relative_to(out_dir).as_posix()}, ) except Exception: pass return on_translated_item def make_on_batch_flushed(file_id, touch, _writer, dst, out_map): def on_batch_flushed() -> None: """?寥?撖怠蝧餉陌蝯???"" try: touch.touch(file_id) touch.flush(_writer) # ?€撠??瘥銋璅?神嚗?葉?瑟?憭? except Exception: # fallback write_json_dict(dst, out_map) return on_batch_flushed # ??蝣箔?甇斗?獢蝧餉陌頝臬?銋? file_id file_id = dst.as_posix() _file_write_table[file_id] = (dst, out_map) # ??Issue #8 靽桀儔嚗蝙?典極撱撘撱?callbacks on_translated_item = make_on_translated_item(rel_src, dst, out_map, rec, out_dir) on_batch_flushed = make_on_batch_flushed(file_id, touch, _writer, dst, out_map) on_progress = make_on_progress(set_prog, _fmt_eta) res = translate_items_with_cache_loop( items_to_translate, total_for_smart=global_total_to_translate, translate_batch_smart=lambda batch, total: translate_batch_smart( batch, total=total ), write_new_cache=bool(write_new_cache), # ???寞????? cache_rules=cache_rules, on_translated_item=on_translated_item, on_batch_flushed=on_batch_flushed, on_progress=on_progress, ) # final write touch.touch(file_id) touch.flush(_writer) total_written += 1 # ?€?獢? cache hit 銋歇撖怠頛詨 cache_hit_done_so_far += hit # ?芣? API 撖阡?蝧餉陌?賊??€脖蜓?脣漲 translated_done += int(res.processed or 0) set_prog(min(translated_done / max(global_total_to_translate, 1), 1.0)) overall_done = cache_hit_done_so_far + translated_done progress_pct = ( (overall_done / global_total_keys * 100) if global_total_keys > 0 else 0 ) log_info( f"??[瑼?摰?] {rel_src} 撌脣神?功n" f"?? 蝧餉陌?豢?嚗甈∠蕃霅?{int(res.processed or 0)} 璇?| 瑼??€??{res.status}\n" f"? ?典??脣漲嚗?歇摰? {overall_done} / {global_total_keys} ({progress_pct:.1f}%)" ) if res.status == "ALL_KEYS_EXHAUSTED": log_info("?? [FTB-LM] ALL_KEYS_EXHAUSTED嚗歇頛詨?桀???嚗?甇U€?) break # ---- Dry-run 蝯偏?? ---- if dry_run: batch_size = _get_default_batch_size("ftbquests", None) est_batches = ( math.ceil(global_total_to_translate / batch_size) if isinstance(global_total_to_translate, int) and batch_size > 0 else None ) meta = { "files": len(per_file_rows), "total_keys": global_total_keys, "cache_hit": global_total_hit, "cache_miss": global_total_to_translate, "estimated_batches": est_batches, } try: # ?嚗?蝧餉陌 preview write_dry_run_preview( out_dir, dry_preview_items, meta=meta, filename="_ftbquests_dry_run_preview.json", # ?舫嚗?蝣箸??? ) # ??NEW嚗ache hit preview write_cache_hit_preview( out_dir, all_cached_items, filename="_ftbquests_dry_run_cache_hit_preview.json", meta=meta, ) except Exception as e: log_error(f"?? [FTB-LM] DRY-RUN preview 頛詨憭望?嚗e}") return { "dry_run": True, "files": len(per_file_rows), "total_keys": global_total_keys, "cache_hit": global_total_hit, "cache_miss": global_total_to_translate, "estimated_batches": est_batches, "out_dir": str(out_dir), "per_file": per_file_rows, } try: rec.export_json(out_dir / "translation_map.json") rec.export_csv(out_dir / "translation_map.csv") log_info(f"??[FTB-LM] 撌脣??translation_map.json / .csv ??{out_dir}") except Exception: log_error("?? [FTB-LM] ?臬 translation_map 憭望?") pass log_info(f"??[隞餃?蝧餉陌摰?] 撌脣? {total_written} ?蕃霅舀?獢撓?箄嚗out_dir}") log_info("?? ?內嚗?臭誑?刻府?桅?銝??translation_map.csv 靘撠蕃霅舀??桃敦蝭€??) batch_size = _get_default_batch_size("ftbquests", None) est_batches = ( math.ceil(global_total_to_translate / batch_size) if isinstance(global_total_to_translate, int) and batch_size > 0 else None ) return { "dry_run": False, "written_files": total_written, "total_keys": global_total_keys, "cache_hit": global_total_hit, "cache_miss": global_total_to_translate, "estimated_batches": est_batches, "out_dir": str(out_dir), } \ No newline at end of file +"""translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py 模組。 + +用途:提供本檔案定義的功能與流程,供專案其他模組呼叫。 +維護注意:本檔案的函式 docstring 用於維護說明,不代表行為變更。 +""" + +# ftbquests_lmtranslator.py +# ------------------------------------------------------------ +# FTB Quests (Already Extracted JSON) -> translate_batch_smart -> Export translated JSON maps +# - DOES NOT run extractor +# - Reads extracted .json files containing {key: text} pairs +# - Outputs translated {key: text} JSON with same relative path structure +# - If input file is like ru_ru.json (or other lang codes), output renamed to zh_tw.json +# - total passed to smart = global total keys across all files +# ------------------------------------------------------------ + +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path +from typing import Dict, Any, List, Optional, Tuple +from concurrent.futures import ThreadPoolExecutor, as_completed +import math + +from translation_tool.core.lm_translator_main import translate_batch_smart + +from translation_tool.core.lm_config_rules import validate_api_keys + +from translation_tool.utils.config_manager import load_config + +from translation_tool.core.lm_translator_shared import ( + fast_split_items_by_cache, # ✅ 新增:高速分流 + translate_items_with_cache_loop, + CacheRule, + TouchSet, # ✅ 新增:touched/flush + TranslationRecorder, # ✅ 新增:翻譯記錄 + write_dry_run_preview, # ✅ 新增:dry-run preview 檔 + write_cache_hit_preview, # ✅ 新增:cache hit preview 檔 + _is_valid_hit, # ✅ 新增:cache hit 判斷 + _get_default_batch_size, +) + +from translation_tool.plugins.shared.json_io import ( + read_json_dict, + write_json_dict, + collect_json_files, +) +from translation_tool.plugins.shared.lang_path_rules import ( + compute_output_path, +) +from translation_tool.plugins.shared.lang_text_rules import is_already_zh +from translation_tool.plugins.shared.rich_text_shield import shield_text, unshield_text + +from translation_tool.utils.log_unit import ( + log_info, + log_error, +) + +# ------------------------- +# Smart 翻譯轉接器(資料格式轉換) +# ------------------------- + + +def map_to_items( + mapping: Dict[str, Any], + cache_type: str, + file_hint: str, +) -> List[Dict[str, Any]]: + """ + 將 {key: text} 的原始資料,轉換成 translate_batch_smart 可處理的 item 格式。 + + 每一個 item 代表「一條可翻譯文字」,會被送進: + - cache 比對 + - batch 翻譯 + - 翻譯結果回寫 + + 【重要設計前提】: + - smart 翻譯器會透過 item["file"] 判斷資料類型 + - FTB Quests 的判斷條件是:路徑中必須包含 "/ftbquests/" + - 因此 file_hint「一定要」包含 "/ftbquests/" + - 同時不可包含 "/lang/",否則會被誤判為 Minecraft Lang 翻譯 + + 這也是為什麼 file_hint 不是實際檔案路徑,而是「刻意構造的提示路徑」。 + """ + items: List[Dict[str, Any]] = [] + + for k, v in mapping.items(): + # key 必須是字串(語言 key) + if not isinstance(k, str): + continue + + # value 必須是非空白字串(實際要翻譯的內容) + if not isinstance(v, str) or not v.strip(): + continue + + items.append( + { + # 提供 smart translator 判斷用的檔案提示路徑 + # ⚠️ 必須包含 "/ftbquests/" + "file": file_hint, + # 語言 key(例如 quest.xxx.title) + "path": k, + # 原始文字(快取與比對用) + "source_text": v, + # 當前文字(會被翻譯器覆寫) + "text": v, + # 指定快取分類(對應 cache_rules) + "cache_type": cache_type, # 例如 "ftbquests" + } + ) + + return items + + +def count_translatable_keys(mapping: Dict[str, Any]) -> int: + """ + 計算 mapping 中「實際可翻譯的字串數量」。 + + 判斷條件: + - value 必須是字串 + - 去除空白後仍有內容 + + 這個數量會用來: + - 顯示進度 + - 計算 cache hit / miss + - 作為 batch 翻譯的總量基準 + """ + return sum(1 for _, v in mapping.items() if isinstance(v, str) and v.strip()) + + +@dataclass +class DryRunStats: + """ + Dry-run(試跑)模式下的統計資料結構。 + + 用途: + - UI 顯示 + - 預覽翻譯規模 + - 確認 cache 命中比例是否合理 + """ + + files: int = 0 # 處理的檔案數 + total_keys: int = 0 # 總字串數 + cache_hit: int = 0 # 快取命中數 + cache_miss: int = 0 # 實際需翻譯數 + per_file: list[dict] = None # 每個檔案的明細 + + +# ------------------------- +# Public API (callable from pipeline) +# ------------------------- +def translate_ftb_pending_to_zh_tw( + *, + input_lang_dir: str | Path, + output_lang_dir: str | Path, + session=None, + rename_langs: Optional[set[str]] = None, + dry_run: bool = False, # ✅ 新增 + write_new_cache: bool = True, # ✅ 新增 +) -> dict: + """ + 專給 UI/服務呼叫: + - input_lang_dir: 例如 /待翻譯/config/ftbquests/quests/lang + (注意:要傳到 lang 這層,讓相對路徑含 en_us/ 才能替換成 zh_tw) + - output_lang_dir: 例如 /config/ftbquests/quests/lang + (輸出會自動把 en_us 資料夾替換成 zh_tw) + - session: 可選,用來 add_log / set_progress + """ + validate_api_keys() + + in_dir = Path(input_lang_dir).resolve() + out_dir = Path(output_lang_dir).resolve() + out_dir.mkdir(parents=True, exist_ok=True) + + def set_prog(v: float): + """設定翻譯進度。""" + if session is not None and hasattr(session, "set_progress"): + try: + session.set_progress(v) + except Exception: + pass + + # rename langs 預設沿用你 CLI 的清單(但 pending 通常只有 en_us,不太會用到) + if rename_langs is None: + rename_langs = { + "ru_ru", + "ja_jp", + "ko_kr", + "zh_cn", + "zh_hk", + "zh_sg", + "pt_br", + "es_es", + "en_us", + "fr_fr", + "de_de", + "it_it", + "pl_pl", + "tr_tr", + "uk_ua", + "cs_cz", + "hu_hu", + "nl_nl", + "sv_se", + "no_no", + "da_dk", + "fi_fi", + } + + if not in_dir.exists() or not in_dir.is_dir(): + raise FileNotFoundError(f"input_lang_dir 不存在或不是資料夾:{in_dir}") + + json_files = collect_json_files(in_dir) + if not json_files: + raise FileNotFoundError(f"找不到任何 .json:{in_dir}") + + # ---- Global total keys (raw) + cache mappings ---- + per_file_counts: List[Tuple[Path, int]] = [] + global_total_keys = 0 + # ✅ Issue #7 修復:緩存 JSON mapping 避免重複讀取 + src_mapping_cache: Dict[Path, Dict[str, Any]] = {} + + def _count_one(src: Path) -> Tuple[Path, int, Dict[str, Any]]: + """讀取 JSON 並統計可翻譯鍵值數量,同時快取 mapping。""" + try: + mapping = read_json_dict(src) + c = count_translatable_keys(mapping) + return src, int(c), mapping + except Exception: + return src, 0, {} + + # max_workers 你可以改成 config 的 parallel_execution_workers + + max_workers = ( + load_config().get("translator", {}).get("parallel_execution_workers", 4) + ) + + with ThreadPoolExecutor(max_workers=max_workers) as ex: + futs = [ex.submit(_count_one, src) for src in json_files] + for fu in as_completed(futs): + src, c, mapping = fu.result() + per_file_counts.append((src, c)) + global_total_keys += c + if mapping: # ✅ 只緩存非空的 mapping + src_mapping_cache[src] = mapping + + # 保持穩定順序(避免多執行緒導致排序亂) + per_file_counts.sort(key=lambda x: x[0].as_posix()) + + if global_total_keys == 0: + log_info("ℹ️ [FTB-LM] 0 keys,跳過翻譯") + return {"written_files": 0, "total_keys": 0, "out_dir": str(out_dir)} + + # ---- Cache rules ---- + cache_rules = {"ftbquests": CacheRule("path|source_text")} + + # ✅ 預先計算「實際要翻譯」總量(全局 miss) + global_total_to_translate = 0 + global_total_hit = 0 + + for src, key_count in per_file_counts: + if key_count == 0: + continue + try: + # ✅ Issue #7 修復:直接使用緩存的 mapping,不再重複讀取 + mapping = src_mapping_cache.get(src, {}) + if not mapping: + continue + rel_src = src.relative_to(in_dir).as_posix() + file_hint = f"config/ftbquests/quests/{rel_src}" + all_items = map_to_items( + mapping, cache_type="ftbquests", file_hint=file_hint + ) + + cached_items, items_to_translate = fast_split_items_by_cache( + all_items, + cache_rules=cache_rules, + is_valid_hit=_is_valid_hit, + ) + + # ✅ 中文/已翻譯:不送 LM,也不算 cache_miss + already_zh_items = [] + real_to_translate = [] + for it in items_to_translate: + s = str(it.get("source_text") or it.get("text") or "") + if is_already_zh(s): + already_zh_items.append(it) + else: + real_to_translate.append(it) + + global_total_hit += len(cached_items) + global_total_to_translate += len(real_to_translate) + + except Exception: + pass + + log_info( + f"\n🔎 [FTB-LM][掃描完畢] 發現待處理檔案:{len(json_files)} 個 | 文本總條目:{global_total_keys} 條" + f"\n✅ [FTB-LM][進度分析] 已從快取載入:{global_total_hit} 條 | 剩餘需 AI 翻譯:{global_total_to_translate} 條" + ) + + if global_total_to_translate == 0: + log_info( + "ℹ️ [FTB-LM][狀態] 恭喜!所有內容皆命中快取,無需調用 AI,將直接導出檔案。" + ) + + if global_total_to_translate == 0: + set_prog(0.99) # 避免瞬間跳 1.0,讓 UI 停留在即將完成的視覺效果 + + # ---- Translate per file (shared loop + cache) ---- + translated_done = 0 # ✅ 只算 API 翻譯完成(主進度分子) + cache_hit_done_so_far = 0 # ✅ 已寫入輸出的 cache hit(整體完成用) + total_written = 0 + + # ---- Dry-run stats container ---- + per_file_rows: list[dict] = [] + # ---- Dry-run: 不翻譯、不輸出 ---- + dry_preview_items: List[Dict[str, Any]] = [] + + all_cached_items: list[dict] = [] + + rec = TranslationRecorder() + touch = TouchSet() + + # touched_files 對應的 writer:最小改動版(每檔只會有一個 dst/out_map) + _file_write_table: dict[str, tuple[Path, Dict[str, str]]] = {} + + def _writer(file_id: str) -> None: + """寫入翻譯結果到檔案。""" + dst_path, data = _file_write_table[file_id] + write_json_dict(dst_path, data) + + for idx, (src, key_count) in enumerate(per_file_counts, start=1): + if key_count == 0: + continue + + # ✅ Issue #7 修復:直接使用緩存的 mapping,不再重複讀取 + mapping = src_mapping_cache.get(src, {}) + if not mapping: + log_error(f"⚠️ [FTB-LM] 找不到快取的 mapping:{src}") + continue + + rel_src = src.relative_to(in_dir).as_posix() # e.g. en_us/ftb_lang.json + file_hint = ( + f"config/ftbquests/quests/{rel_src}" # ✅ include /ftbquests/ and no /lang/ + ) + all_items = map_to_items(mapping, cache_type="ftbquests", file_hint=file_hint) + + cached_items, items_to_translate = fast_split_items_by_cache( + all_items, + cache_rules=cache_rules, + is_valid_hit=_is_valid_hit, + ) + + already_zh_items = [] + real_to_translate = [] + for it in items_to_translate: + s = str(it.get("source_text") or it.get("text") or "") + if is_already_zh(s): + already_zh_items.append(it) + else: + real_to_translate.append(it) + + items_to_translate = real_to_translate # ✅ 覆蓋成真的要翻譯的 + hit = len(cached_items) + miss = len(items_to_translate) + already_zh = len(already_zh_items) + + if hit > 0: + sample = cached_items[0] + log_info(f"🧠 [FTB-LM] 快取命中範例:{sample.get('path')}") + + if miss > 0: + sample = items_to_translate[0] + log_info(f"✏️ [FTB-LM] 待翻譯範例:{sample.get('path')}") + + dst = compute_output_path(src, in_dir, out_dir, rename_langs) + log_info( + f"📄 [FTB-LM][檔案 {idx}/{len(per_file_counts)}] 正在處理:{rel_src}\n" + f"📊 本檔明細:總條目 {key_count} | 載入快取 {hit} | 已含中文(跳過) {already_zh} | 需翻譯 {miss}\n" + f"💾 輸出路徑:{dst.relative_to(out_dir).as_posix()}" + ) + + per_file_rows.append( + { + "file": rel_src, + "keys": key_count, + "cache_hit": hit, + "cache_miss": miss, + "dst": dst.relative_to(out_dir).as_posix(), + } + ) + + # ---- Dry-run: 不翻譯、不輸出 ---- + if dry_run: + # ✅ 收集 preview 樣本(避免爆大:最多 2000) + if len(dry_preview_items) < 2000: + dry_preview_items.extend( + items_to_translate[: 2000 - len(dry_preview_items)] + ) + + # ✅ NEW:收集 cache hit(避免爆大:最多 2000,可自行調整) + if len(all_cached_items) < 2000: + all_cached_items.extend(cached_items[: 2000 - len(all_cached_items)]) + + # 以檔案數取代翻譯量為進度分母,避免翻譯量分佈不均造成視覺跳躍 + set_prog(min(idx / len(per_file_counts), 1.0)) + + log_info( + f"🧪 [測試模式] 進度:{idx}/{len(per_file_counts)}\n" + f"擬定輸出:{rel_src} ➔ {dst.relative_to(out_dir).as_posix()}\n" + f"預估明細:總條目 {key_count} | 快取命中 {hit} | 需翻譯 {miss} | 預估批次 {math.ceil(miss / _get_default_batch_size('ftbquests', None)) if _get_default_batch_size('ftbquests', None) > 0 else 'N/A'}\n" + f"累計預估進度:{translated_done} / {global_total_to_translate}" + ) + continue + + # ============================ + # ✅ 真正翻譯(非 dry-run) + # ============================ + + # out_map:先用原文當底(避免中斷時輸出缺 key) + out_map: Dict[str, str] = { + k: v + for k, v in mapping.items() + if isinstance(k, str) and isinstance(v, str) + } + + # cache hits 覆蓋 + for it in cached_items: + p = it.get("path") + t = it.get("text") + if isinstance(p, str) and isinstance(t, str): + out_map[p] = t + try: + rec.record( + cache_type="ftbquests", + file_id=rel_src, + path=p, + src=str(it.get("source_text") or ""), + dst=t, + cache_hit=True, + extra={"dst_file": dst.relative_to(out_dir).as_posix()}, + ) + except Exception: + pass + + # 全命中 cache:直接輸出 + if not items_to_translate: + # write_json_dict(dst, out_map) + file_id = dst.as_posix() + _file_write_table[file_id] = (dst, out_map) + + touch.touch(file_id) + touch.flush(_writer) # 最小改動:等同你現在每次都寫,但走同一個管線 + total_written += 1 + + # ✅ 這個檔案的 cache hit 已經真正寫進輸出,整體完成要加 + cache_hit_done_so_far += hit + + # ✅ 主進度不變(translated_done 不加) + set_prog(min(translated_done / max(global_total_to_translate, 1), 1.0)) + + overall_done = cache_hit_done_so_far + translated_done + progress_pct = ( + (overall_done / global_total_keys * 100) if global_total_keys > 0 else 0 + ) + + log_info( + f"⚡ [快取跳過] 檔案 {idx}/{len(per_file_counts)}:{rel_src}\n" + f"💾 導出路徑:{dst.relative_to(out_dir).as_posix()}\n" + f"🧠 處理結果:命中快取 {hit} 條 | API 調用 0 條\n" + f"🎯 全域完成度:{overall_done} / {global_total_keys} ({progress_pct:.1f}%)" + ) + continue + + # ✅ Issue #8 修復:將 callbacks 定義在迴圈外部,改用工廠函式 + def _fmt_eta(sec: float) -> str: + """格式化剩餘時間。""" + if sec <= 0: + return "" + m, s = divmod(int(sec), 60) + if m > 0: + return f"{m}m{s:02d}s" + return f"{s}s" + + def make_on_progress(set_prog, _fmt_eta): + def on_progress(p: float, msg: str, eta_sec: float) -> None: + """報告翻譯進度。""" + eta_txt = _fmt_eta(eta_sec) + if eta_txt: + log_info(f"⏳ [AI 翻譯中] {msg} | 預估剩餘時間:{eta_txt}") + else: + log_info(f"🚀 [AI 翻譯中] {msg}") + set_prog(p) + return on_progress + + def make_on_translated_item(rel_src, dst, out_map, rec, out_dir): + def on_translated_item(it: Dict[str, Any]) -> None: + """處理翻譯結果並寫入映射。""" + p = it.get("path") + t = it.get("text") + src_text = str(it.get("source_text") or "") + if isinstance(p, str) and isinstance(t, str): + try: + shielded_src = shield_text(src_text) + t = unshield_text(t, shielded_src.shields) + except Exception: + pass + out_map[p] = t + try: + rec.record( + cache_type="ftbquests", + file_id=rel_src, + path=p, + src=src_text, + dst=t, + cache_hit=False, + extra={"dst_file": dst.relative_to(out_dir).as_posix()}, + ) + except Exception: + pass + return on_translated_item + + def make_on_batch_flushed(file_id, touch, _writer, dst, out_map): + def on_batch_flushed() -> None: + """批量寫入翻譯結果。""" + try: + touch.touch(file_id) + touch.flush(_writer) # 最小改動:每批也照樣寫,避免中斷損失 + except Exception: + # fallback + write_json_dict(dst, out_map) + return on_batch_flushed + + # ✅ 確保此檔案在翻譯路徑也有 file_id + file_id = dst.as_posix() + _file_write_table[file_id] = (dst, out_map) + + # ✅ Issue #8 修復:使用工廠函式創建 callbacks + on_translated_item = make_on_translated_item(rel_src, dst, out_map, rec, out_dir) + on_batch_flushed = make_on_batch_flushed(file_id, touch, _writer, dst, out_map) + on_progress = make_on_progress(set_prog, _fmt_eta) + + res = translate_items_with_cache_loop( + items_to_translate, + total_for_smart=global_total_to_translate, + translate_batch_smart=lambda batch, total: translate_batch_smart( + batch, total=total + ), + write_new_cache=bool(write_new_cache), # ✅ 改成吃參數 + cache_rules=cache_rules, + on_translated_item=on_translated_item, + on_batch_flushed=on_batch_flushed, + on_progress=on_progress, + ) + + # final write + touch.touch(file_id) + touch.flush(_writer) + total_written += 1 + + # 這個檔案的 cache hit 也已寫入輸出 + cache_hit_done_so_far += hit + + # 只把 API 實際翻譯數量加進主進度 + translated_done += int(res.processed or 0) + + set_prog(min(translated_done / max(global_total_to_translate, 1), 1.0)) + + overall_done = cache_hit_done_so_far + translated_done + progress_pct = ( + (overall_done / global_total_keys * 100) if global_total_keys > 0 else 0 + ) + log_info( + f"✨ [檔案完成] {rel_src} 已寫入\n" + f"📈 翻譯數據:本次翻譯 {int(res.processed or 0)} 條 | 檔案狀態:{res.status}\n" + f"🎯 全域進度:目前已完成 {overall_done} / {global_total_keys} ({progress_pct:.1f}%)" + ) + + if res.status == "ALL_KEYS_EXHAUSTED": + log_info("⚠️ [FTB-LM] ALL_KEYS_EXHAUSTED:已輸出目前成果,停止。") + break + + # ---- Dry-run 結尾摘要 ---- + if dry_run: + batch_size = _get_default_batch_size("ftbquests", None) + est_batches = ( + math.ceil(global_total_to_translate / batch_size) + if isinstance(global_total_to_translate, int) and batch_size > 0 + else None + ) + meta = { + "files": len(per_file_rows), + "total_keys": global_total_keys, + "cache_hit": global_total_hit, + "cache_miss": global_total_to_translate, + "estimated_batches": est_batches, + } + + try: + # 原本:待翻譯 preview + write_dry_run_preview( + out_dir, + dry_preview_items, + meta=meta, + filename="_ftbquests_dry_run_preview.json", # 可選:明確檔名 + ) + + # ✅ NEW:cache hit preview + write_cache_hit_preview( + out_dir, + all_cached_items, + filename="_ftbquests_dry_run_cache_hit_preview.json", + meta=meta, + ) + + except Exception as e: + log_error(f"⚠️ [FTB-LM] DRY-RUN preview 輸出失敗:{e}") + + return { + "dry_run": True, + "files": len(per_file_rows), + "total_keys": global_total_keys, + "cache_hit": global_total_hit, + "cache_miss": global_total_to_translate, + "estimated_batches": est_batches, + "out_dir": str(out_dir), + "per_file": per_file_rows, + } + + try: + rec.export_json(out_dir / "translation_map.json") + rec.export_csv(out_dir / "translation_map.csv") + log_info(f"✅ [FTB-LM] 已匯出 translation_map.json / .csv 到 {out_dir}") + + except Exception: + log_error("⚠️ [FTB-LM] 匯出 translation_map 失敗") + pass + + log_info(f"✅ [任務翻譯完成] 已將 {total_written} 個翻譯檔案輸出至:{out_dir}") + log_info("📊 提示:您可以在該目錄下查看 translation_map.csv 來核對翻譯條目細節。") + batch_size = _get_default_batch_size("ftbquests", None) + est_batches = ( + math.ceil(global_total_to_translate / batch_size) + if isinstance(global_total_to_translate, int) and batch_size > 0 + else None + ) + return { + "dry_run": False, + "written_files": total_written, + "total_keys": global_total_keys, + "cache_hit": global_total_hit, + "cache_miss": global_total_to_translate, + "estimated_batches": est_batches, + "out_dir": str(out_dir), + } diff --git a/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py b/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py index d27032e4..5299f55f 100644 --- a/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py +++ b/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py @@ -1 +1,602 @@ -# kubejs_tooltip_lmtranslator.py# ---------------------------------"""?詨??璁汗?箸?寥?蝧餉陌 (translate_batch_smart)嚗?典翰???嗅?蕃霅舀?蝔??踹???蝧餉陌?詨???摮?蝭€??API 瘨€€?擃€翰??瞈?(fast_split_items_by_cache)嚗蝧餉陌?翰??撠?翰??撠€歇蝧餉陌????蝧餉陌????嚗??????€?摰敺芰蝧餉陌 (translate_items_with_cache_loop)嚗??瑕? ETA嚗?閮擗??? 憿舐內???∠ 摰??嚗afe Slicing嚗??€銵??脫迫?甈∟?瘙?憭批??游仃?€??琿?蝥??摮?雿輻 TouchSet ??Writer Flush 璈嚗蕃霅舫?蝔葉??神?交?獢??喃蝙蝔???銝剜銋??憭勗歇摰??€脣漲???炎璅∪? (dry_run)嚗?湔芋?砍銵蒂頛詨?汗瑼?嚗rite_dry_run_preview嚗?霈??冽迤撘???API 憿漲?Ⅱ隤撘?行迤蝣箝€??豢?撠 (TranslationRecorder)嚗?游?蝧餉陌閮?撠??JSON ??CSV ?澆?嚗靘踹?蝥撠?鈭活????脣漲餈質馱 (session.set_progress)嚗撱粹€脣漲?文?嚗ook嚗??臬??亙???UI ?隤頂蝯梢*蝷箇蕃霅舐????頝臬??芸?嚗????蝟餉??冗頧?嚗?憒????en_us ?芸?撠???zh_tw ?桅?嚗€?Rich Text Shield嚗hield_text() / unshield_text() 靽風 KubeJS ?澆?嚗蔗?脩Ⅳ??D?RL 蝑?嚗??函蕃霅臬??賢嚗蕃霅臬???嚗??LM 隤斤蕃?澆?璅???"""from __future__ import annotationsfrom dataclasses import dataclassfrom pathlib import Pathfrom typing import Dict, Any, List, Optional, Tuplefrom concurrent.futures import ThreadPoolExecutor, as_completedimport reimport openccfrom translation_tool.core.lm_translator_main import translate_batch_smartfrom translation_tool.core.lm_config_rules import validate_api_keysfrom translation_tool.utils.config_manager import load_configfrom translation_tool.core.lm_translator_shared import ( CacheRule, fast_split_items_by_cache, translate_items_with_cache_loop, TouchSet, TranslationRecorder, write_dry_run_preview, write_cache_hit_preview, # ???啣?嚗ache hit preview 瑼? _is_valid_hit, # ???啣?嚗ache hit ?斗)from translation_tool.plugins.shared.json_io import ( read_json_dict, write_json_dict, collect_json_files,)from translation_tool.plugins.shared.lang_path_rules import ( compute_output_path,)from translation_tool.plugins.shared.rich_text_shield import ( shield_text, unshield_text,)from translation_tool.utils.log_unit import log_info, log_warning, progress# -------------------------# Smart item mapping# -------------------------def collect_items_from_mapping( mapping: Dict[str, Any], *, file_hint: str,) -> List[Dict[str, Any]]: """ 撠?{頝臬??? ??} ??撠??蝧餉陌?寞活??? ?€蝣箔??箸?菜葫?質???KubeJS ?蔭嚗tem["file"] ? "/kubejs/"嚗€? Shield ?游?嚗?瘥???銝脣€澆銵?shield_text()嚗? - ??skip_reason ??None嚗???URL/鈭辣蝑?嚗?乩?????蝧餉陌?? - ?仿?閬蕃霅荔???shield ??銋暹楊??嚗lean嚗?隞?????? - ????item 摮銝剝???_shielded嚗? on_translated_item() ??unshield ?‵?? """ items: List[Dict[str, Any]] = [] for k, v in mapping.items(): if not isinstance(k, str): continue if not isinstance(v, str) or not v.strip(): continue # ??Rich Text Shield嚗?箔??蕃霅舐??澆??挾 shielded = shield_text(v) if shielded.skip_reason is not None: # 銝?蝧餉陌嚗???URL/鈭辣/蝛箇嚗??湔撖怠??銝?蝧餉陌蝞∠? items.append( { "file": file_hint, "path": k, "source_text": v, "text": v, # 靽??? "cache_type": "kubejs", "_shielded": shielded, # 靘?unshield ?嚗迨???⊿???嚗? "_skip_reason": shielded.skip_reason, } ) else: # ?€閬蕃霅荔?雿輻 shield ??銋暹楊?? items.append( { "file": file_hint, "path": k, "source_text": v, "text": shielded.clean, # ??雿輻 shield ????靘蕃霅? "cache_type": "kubejs", "_shielded": shielded, # 靘?on_translated_item() ??unshield } ) return itemsdef count_translatable_keys(mapping: Dict[str, Any]) -> int: """閮? mapping 銝准€蝧餉陌摮葡???賊??? ?斗璇辣嚗? - value ?臬?銝? - ?駁蝛箇敺??摰? ?券€?憿舐內?脣漲 / 隡啁?蝧餉陌蝮賡??? """ return sum(1 for _, v in mapping.items() if isinstance(v, str) and v.strip())# -------------------------# 蝜?銝剜??菜葫嚗?潸歲?歇蝧餉陌??tooltips嚗?# -------------------------_TW_CJK_RE = re.compile(r"[\u4e00-\u9fff]")_TW_CONVERTER = opencc.OpenCC("s2tw")def _is_tw_text(text: str) -> bool: """?斗???臬撌脩??舐?擃葉??OpenCC 頧?敺?霈?= 撌脫蝜?嚗€?"" if not text or not isinstance(text, str): return False if not _TW_CJK_RE.search(text): return False # ??CJK 摮?嚗??臭葉?? # 蝪⊿???擃???嚗????澆??祉??潘?隞?”?撠望蝜? return _TW_CONVERTER.convert(text) == textdef _split_off_tw_items( cached_items: List[Dict[str, Any]], items_to_translate: List[Dict[str, Any]],) -> List[Dict[str, Any]]: """敺?items_to_translate 蝘餃撌脩蕃霅荔?蝜?嚗????cached_items嚗??唾◤蝘餃???items??"" tw_items: List[Dict[str, Any]] = [] remaining: List[Dict[str, Any]] = [] for it in items_to_translate: src_text = it.get("source_text", "") or "" if _is_tw_text(src_text): tw_items.append(it) else: remaining.append(it) cached_items.extend(tw_items) # 皜??蒂撖怠??拚?嚗n-place 靽格銝脣?嚗? items_to_translate.clear() items_to_translate.extend(remaining) return tw_items# -------------------------# Dry-run stats (optional)# -------------------------@dataclassclass DryRunStats: """DryRunStats 憿?? ?券€?撠???DryRunStats ?賊?????銵?? 蝬剛風瘜冽?嚗耨?孵?瘜?隢Ⅱ隤??典?恍??摰寞€扼€? """ files: int = 0 total_keys: int = 0 cache_hit: int = 0 cache_miss: int = 0 per_file: Optional[list[dict]] = None# -------------------------# Public API (for UI/pipeline)# -------------------------def translate_kubejs_pending_to_zh_tw( *, pending_dir: str | Path, output_dir: str | Path, session=None, rename_langs: Optional[set[str]] = None, dry_run: bool = False, write_new_cache: bool = False,) -> dict: """ Translate KubeJS pending JSON dir -> output dir (usually LM蝧餉陌敺?. - pending_dir: 靘? Output/kubejs/敺蕃霅? - output_dir : 靘? Output/kubejs/LM蝧餉陌敺? """ validate_api_keys() in_dir = Path(pending_dir).resolve() out_dir = Path(output_dir).resolve() out_dir.mkdir(parents=True, exist_ok=True) if rename_langs is None: rename_langs = { "ru_ru", "ja_jp", "ko_kr", "zh_cn", "zh_hk", "zh_sg", "pt_br", "es_es", "en_us", "fr_fr", "de_de", "it_it", "pl_pl", "tr_tr", "uk_ua", "cs_cz", "hu_hu", "nl_nl", "sv_se", "no_no", "da_dk", "fi_fi", } if not in_dir.exists() or not in_dir.is_dir(): raise FileNotFoundError(f"pending_dir 銝??冽?銝鞈?憭橘?{in_dir}") json_files = collect_json_files(in_dir) if not json_files: raise FileNotFoundError(f"?曆??唬遙雿?.json嚗in_dir}") # ------------------------- # Pre-scan global total (multithread, like FTB) # ------------------------- per_file_counts: List[Tuple[Path, int]] = [] global_total_keys = 0 # ??Issue #7b 靽桀儔嚗楨摮?JSON mapping ?踹???霈€?? src_mapping_cache: Dict[Path, Dict[str, Any]] = {} def _count_one(src: Path) -> Tuple[Path, int, Dict[str, Any]]: """蝯梯??桐?瑼??蕃霅?key ?賊?嚗??翰??mapping??"" try: mapping = read_json_dict(src) return src, int(count_translatable_keys(mapping)), mapping except Exception: return src, 0, {} max_workers = int( load_config().get("translator", {}).get("parallel_execution_workers", 4) or 4 ) max_workers = max(1, max_workers) with ThreadPoolExecutor(max_workers=max_workers) as ex: futs = [ex.submit(_count_one, p) for p in json_files] for fu in as_completed(futs): src, c, mapping = fu.result() per_file_counts.append((src, c)) global_total_keys += c if mapping: # ???芰楨摮?蝛箇? mapping src_mapping_cache[src] = mapping per_file_counts.sort(key=lambda x: x[0].as_posix()) if global_total_keys == 0: log_info("?對? [KubeJS-LM] 0 keys嚗歲?蕃霅?) return {"written_files": 0, "total_keys": 0, "out_dir": str(out_dir)} # ------------------------- # Cache rules # ------------------------- cache_rules = {"kubejs": CacheRule("path|source_text")} # ------------------------- # Pre-calc global miss/hit (so progress uses "real translate count") # ------------------------- global_total_to_translate = 0 global_total_hit = 0 for src, key_count in per_file_counts: if key_count == 0: continue try: # ??Issue #7b 靽桀儔嚗?乩蝙?函楨摮? mapping嚗???銴??? mapping = src_mapping_cache.get(src, {}) if not mapping: continue rel_src = src.relative_to(in_dir).as_posix() file_hint = f"output/kubejs/{rel_src}" # must contain /kubejs/ all_items = collect_items_from_mapping(mapping, file_hint=file_hint) cached_items, items_to_translate = fast_split_items_by_cache( all_items, cache_rules=cache_rules, is_valid_hit=_is_valid_hit, ) # ??頝喲??澆歇蝬蝜?銝剜???items嚗???API + ?踹?蝪⊿??嗉?蕃嚗? _split_off_tw_items(cached_items, items_to_translate) global_total_hit += len(cached_items) global_total_to_translate += len(items_to_translate) except Exception: pass log_info( f"?? [KubeJS-LM] 敺蕃霅舀?獢嚗len(json_files)}嚗蜇 keys嚗global_total_keys}\n" f"??[KubeJS-LM] cache_hit={global_total_hit} | 撖阡??€蝧餉陌(cache_miss)={global_total_to_translate}" ) if global_total_to_translate == 0: progress(1.0) # ------------------------- # Global build + translate (NO per-file translate) # ------------------------- translated_done = 0 avg_batch_sec = None total_written = 0 per_file_rows: list[dict] = [] rec = TranslationRecorder() touch = TouchSet() _file_write_table: dict[str, tuple[Path, Dict[str, str]]] = {} def _writer(file_id: str) -> None: """撖怠蝧餉陌蝯??唳?獢€?"" dst_path, data = _file_write_table[file_id] write_json_dict(dst_path, data) # Build phase: per-file state + global miss list file_states: dict[str, dict] = {} all_miss_items: List[Dict[str, Any]] = [] all_hit_items: List[Dict[str, Any]] = [] for idx, (src, key_count) in enumerate(per_file_counts, start=1): if key_count == 0: continue # ??Issue #7b 靽桀儔嚗?乩蝙?函楨摮? mapping嚗???銴??? mapping = src_mapping_cache.get(src, {}) if not mapping: continue rel_src = src.relative_to(in_dir).as_posix() file_hint = f"output/kubejs/{rel_src}" all_items = collect_items_from_mapping(mapping, file_hint=file_hint) cached_items, items_to_translate = fast_split_items_by_cache( all_items, cache_rules=cache_rules, is_valid_hit=_is_valid_hit, ) # ??頝喲??澆歇蝬蝜?銝剜???items嚗?蝧餉陌皜蝘餉 cache hit嚗? _split_off_tw_items(cached_items, items_to_translate) # ??NEW嚗敞蝛?hit items all_hit_items.extend(cached_items) hit = len(cached_items) miss = len(items_to_translate) dst = compute_output_path(src, in_dir, out_dir, rename_langs) per_file_rows.append( { "file": rel_src, "keys": key_count, "cache_hit": hit, "cache_miss": miss, "dst": dst.relative_to(out_dir).as_posix(), } ) log_info( f"[KubeJS-LM] 瑼? {idx}/{len(per_file_counts)}嚗rel_src} 嚚? f"蝮賢?銝?{key_count}嚗翰?銝?{hit}嚗?蝧餉陌 {miss} 嚚? f"頛詨 ??{dst.relative_to(out_dir).as_posix()}" ) # out_map base out_map: Dict[str, str] = { k: v for k, v in mapping.items() if isinstance(k, str) and isinstance(v, str) } # apply cache hits now (record as hit) for it in cached_items: p = it.get("path") t = it.get("text") if isinstance(p, str) and isinstance(t, str): out_map[p] = t try: rec.record( cache_type="kubejs", file_id=rel_src, path=p, src=str(it.get("source_text") or ""), dst=t, cache_hit=True, extra={"dst_file": dst.relative_to(out_dir).as_posix()}, ) except Exception: pass file_id = dst.as_posix() _file_write_table[file_id] = (dst, out_map) file_states[rel_src] = { "rel_src": rel_src, "dst": dst, "file_id": file_id, "out_map": out_map, } # tag miss items with file context (so callback can write back to correct out_map) for it in items_to_translate: it["file_rel"] = rel_src it["dst_file"] = dst.relative_to(out_dir).as_posix() all_miss_items.extend(items_to_translate) # ------------------------- # Dry-run: preview only (no API) # ------------------------- if dry_run: dry_preview_items = all_miss_items[:2000] if all_miss_items else [] preview_path = None try: meta = { "files": len(per_file_rows), "total_keys": global_total_keys, "cache_hit": global_total_hit, "cache_miss": global_total_to_translate, } preview_path = write_dry_run_preview( out_dir, dry_preview_items, filename="_kubejs_dry_run_preview.json", meta=meta, ) log_info(f"?妒 [DRY-RUN] preview written: {preview_path.as_posix()}") # ??NEW嚗ache hit preview hit_preview_path = write_cache_hit_preview( out_dir, all_hit_items[:2000], # ???? filename="_kubejs_dry_run_cache_hit_preview.json", meta=meta, ) log_info( f"? [DRY-RUN] cache-hit preview written: {hit_preview_path.as_posix()}" ) except Exception as e: log_warning(f"?? [KubeJS-LM] DRY-RUN preview 頛詨憭望?嚗e}") progress(1.0) return { "dry_run": True, "files": len(per_file_rows), "total_keys": global_total_keys, "cache_hit": global_total_hit, "cache_miss": global_total_to_translate, "preview_path": str(preview_path) if preview_path else None, "out_dir": str(out_dir), "per_file": per_file_rows, } # ------------------------- # Real run: translate ONCE for all miss items # ------------------------- if not all_miss_items: # All cache hit -> flush all files once for st in file_states.values(): touch.touch(st["file_id"]) touch.flush(_writer) total_written = len(file_states) translated_done = 0 progress(1.0) else: def on_translated_item(it: Dict[str, Any]) -> None: """??蝧餉陌蝯???"" rel_src = it.get("file_rel") p = it.get("path") t = it.get("text") s = it.get("source_text") if not (isinstance(rel_src, str) and rel_src in file_states): return if not (isinstance(p, str) and isinstance(t, str)): return # ??Rich Text Shield嚗蕃霅臬???鋡思?霅瑞??澆??挾 shielded = it.get("_shielded") if shielded is not None and shielded.shields: final_text = unshield_text(t, shielded.shields) else: final_text = t st = file_states[rel_src] st["out_map"][p] = final_text try: rec.record( cache_type="kubejs", file_id=rel_src, path=p, src=str(s or ""), dst=t, cache_hit=False, extra={"dst_file": st["dst"].relative_to(out_dir).as_posix()}, ) except Exception: pass try: touch.touch(st["file_id"]) except Exception: pass def on_batch_flushed() -> None: # write touched files each batch """?寥?撖怠蝧餉陌蝯???"" try: touch.flush(_writer) except Exception: # fallback: write all for fid, (dstp, data) in _file_write_table.items(): write_json_dict(dstp, data) def _fmt_eta(sec: float) -> str: """?澆??擗??€?"" if sec <= 0: return "" m, s = divmod(int(sec), 60) if m >= 60: h, m2 = divmod(m, 60) return f"{h}h{m2:02d}m{s:02d}s" if m > 0: return f"{m}m{s:02d}s" return f"{s}s" def on_progress(p: float, msg: str, eta_sec: float) -> None: """?勗?蝧餉陌?脣漲??"" eta_txt = _fmt_eta(eta_sec) log_info(f"{msg}" + (f" | ETA ??{eta_txt}" if eta_txt else "")) progress(p) res = translate_items_with_cache_loop( all_miss_items, total_for_smart=global_total_to_translate, translate_batch_smart=lambda batch, total: translate_batch_smart( batch, total=total ), write_new_cache=bool(write_new_cache), cache_rules=cache_rules, on_translated_item=on_translated_item, on_batch_flushed=on_batch_flushed, on_progress=on_progress, ) translated_done = int(res.processed or 0) avg_batch_sec = ( (res.elapsed_sec / res.completed_calls) if res.completed_calls else None ) # final flush all for st in file_states.values(): touch.touch(st["file_id"]) touch.flush(_writer) total_written = len(file_states) progress(1.0) # ------------------------- # Export records # ------------------------- rec_json = None rec_csv = None try: rec_json = rec.export_json(out_dir / "translation_map.json") rec_csv = rec.export_csv(out_dir / "translation_map.csv") log_info(f"?屁 [KubeJS-LM] records written: {rec_json} | {rec_csv}") except Exception as ex: log_info(f"?? [KubeJS-LM] export records failed: {ex}") return { "written_files": total_written, "total_keys": global_total_keys, "cache_hit": global_total_hit, "cache_miss": global_total_to_translate, "api_translated": translated_done, "avg_batch_sec": avg_batch_sec, "out_dir": str(out_dir), "records_json": str(rec_json) if rec_json else None, "records_csv": str(rec_csv) if rec_csv else None, "per_file": per_file_rows, } \ No newline at end of file +# kubejs_tooltip_lmtranslator.py +# --------------------------------- +""" +核心功能概覽 +智慧批量翻譯 (translate_batch_smart):利用快取機制優化翻譯流程,避免重複翻譯相同的文字,節省 API 消耗。 +高速快取過濾 (fast_split_items_by_cache):在翻譯前快速比對現有快取,將「已翻譯」與「待翻譯」的項目分離,提升處理效率。 +安全循環翻譯 (translate_items_with_cache_loop): +具備 ETA(預計剩餘時間) 顯示。 +採用 安全切片(Safe Slicing) 技術,防止因單次請求過大導致失敗。 +斷點續傳與即時儲存: +使用 TouchSet 與 Writer Flush 機制,翻譯過程中會即時寫入檔案,即使程式意外中斷也不會遺失已完成的進度。 +預檢模式 (dry_run):支援模擬執行並輸出預覽檔案(write_dry_run_preview),讓你在正式消耗 API 額度前確認格式是否正確。 +數據導出 (TranslationRecorder):支援將翻譯記錄導出為 JSON 或 CSV 格式,方便後續校對或二次開發。 +進度追蹤 (session.set_progress):內建進度鉤子(Hook),可對接外部 UI 或日誌系統顯示翻譯百分比。 +路徑優化:自動處理語系資料夾轉換(例如將原本的 en_us 自動導向至 zh_tw 目錄)。 +Rich Text Shield:shield_text() / unshield_text() 保護 KubeJS 格式(彩色碼、物品ID、URL 等), +在翻譯前抽出,翻譯後還原,避免 LM 誤翻格式標記。 +""" + +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path +from typing import Dict, Any, List, Optional, Tuple +from concurrent.futures import ThreadPoolExecutor, as_completed +import re +import opencc + +from translation_tool.core.lm_translator_main import translate_batch_smart +from translation_tool.core.lm_config_rules import validate_api_keys +from translation_tool.utils.config_manager import load_config + +from translation_tool.core.lm_translator_shared import ( + CacheRule, + fast_split_items_by_cache, + translate_items_with_cache_loop, + TouchSet, + TranslationRecorder, + write_dry_run_preview, + write_cache_hit_preview, # ✅ 新增:cache hit preview 檔 + _is_valid_hit, # ✅ 新增:cache hit 判斷 +) + +from translation_tool.plugins.shared.json_io import ( + read_json_dict, + write_json_dict, + collect_json_files, +) +from translation_tool.plugins.shared.lang_path_rules import ( + compute_output_path, +) +from translation_tool.plugins.shared.rich_text_shield import ( + shield_text, + unshield_text, +) + +from translation_tool.utils.log_unit import log_info, log_warning, progress + + +# ------------------------- +# Smart item mapping +# ------------------------- +def collect_items_from_mapping( + mapping: Dict[str, Any], + *, + file_hint: str, +) -> List[Dict[str, Any]]: + """ + 將 {路徑鍵: 原文} 的映射轉換為翻譯批次項目。 + 需確保智慧偵測能識別 KubeJS 配置(item["file"] 包含 "/kubejs/")。 + + Shield 整合:對每一個字串值執行 shield_text(), + - 若 skip_reason 非 None(圖片/URL/事件等),直接保留原文不翻譯。 + - 若需要翻譯,用 shield 過的乾淨文字(clean)取代原文字。 + - 同時在 item 字典中附加 _shielded,供 on_translated_item() 做 unshield 回填。 + """ + items: List[Dict[str, Any]] = [] + for k, v in mapping.items(): + if not isinstance(k, str): + continue + if not isinstance(v, str) or not v.strip(): + continue + + # ✅ Rich Text Shield:抽出不應翻譯的格式片段 + shielded = shield_text(v) + + if shielded.skip_reason is not None: + # 不應翻譯(圖片/URL/事件/空白),直接寫入原文不經翻譯管線 + items.append( + { + "file": file_hint, + "path": k, + "source_text": v, + "text": v, # 保持原文 + "cache_type": "kubejs", + "_shielded": shielded, # 供 unshield 回查(此情境無需還原) + "_skip_reason": shielded.skip_reason, + } + ) + else: + # 需要翻譯:使用 shield 過的乾淨文字 + items.append( + { + "file": file_hint, + "path": k, + "source_text": v, + "text": shielded.clean, # ← 使用 shield 過的文字供翻譯 + "cache_type": "kubejs", + "_shielded": shielded, # 供 on_translated_item() 做 unshield + } + ) + + return items + + +def count_translatable_keys(mapping: Dict[str, Any]) -> int: + """計算 mapping 中『可翻譯字串』的數量。 + + 判斷條件: + - value 是字串 + - 去除空白後仍有內容 + + 用途:顯示進度 / 估算翻譯總量。 + """ + return sum(1 for _, v in mapping.items() if isinstance(v, str) and v.strip()) + + +# ------------------------- +# 繁體中文偵測(用於跳過已翻譯的 tooltips) +# ------------------------- +_TW_CJK_RE = re.compile(r"[\u4e00-\u9fff]") +_TW_CONVERTER = opencc.OpenCC("s2tw") + + +def _is_tw_text(text: str) -> bool: + """判斷文字是否已經是繁體中文(OpenCC 轉換後不變 = 已是繁體)。""" + if not text or not isinstance(text, str): + return False + if not _TW_CJK_RE.search(text): + return False # 無 CJK 字元,不是中文 + # 簡體→繁體轉換後,如果等於原本的值,代表原本就是繁體 + return _TW_CONVERTER.convert(text) == text + + +def _split_off_tw_items( + cached_items: List[Dict[str, Any]], + items_to_translate: List[Dict[str, Any]], +) -> List[Dict[str, Any]]: + """從 items_to_translate 移出已翻譯(繁體)的項目到 cached_items,回傳被移動的 items。""" + tw_items: List[Dict[str, Any]] = [] + remaining: List[Dict[str, Any]] = [] + for it in items_to_translate: + src_text = it.get("source_text", "") or "" + if _is_tw_text(src_text): + tw_items.append(it) + else: + remaining.append(it) + cached_items.extend(tw_items) + # 清除原本的並寫回剩餘(in-place 修改串列) + items_to_translate.clear() + items_to_translate.extend(remaining) + return tw_items + + +# ------------------------- +# Dry-run stats (optional) +# ------------------------- +@dataclass +class DryRunStats: + """DryRunStats 類別。 + + 用途:封裝與 DryRunStats 相關的狀態與行為。 + 維護注意:修改公開方法前請確認外部呼叫點與相容性。 + """ + + files: int = 0 + total_keys: int = 0 + cache_hit: int = 0 + cache_miss: int = 0 + per_file: Optional[list[dict]] = None + + +# ------------------------- +# Public API (for UI/pipeline) +# ------------------------- +def translate_kubejs_pending_to_zh_tw( + *, + pending_dir: str | Path, + output_dir: str | Path, + session=None, + rename_langs: Optional[set[str]] = None, + dry_run: bool = False, + write_new_cache: bool = False, +) -> dict: + """ + Translate KubeJS pending JSON dir -> output dir (usually LM翻譯後). + - pending_dir: 例如 Output/kubejs/待翻譯 + - output_dir : 例如 Output/kubejs/LM翻譯後 + """ + validate_api_keys() + + in_dir = Path(pending_dir).resolve() + out_dir = Path(output_dir).resolve() + out_dir.mkdir(parents=True, exist_ok=True) + + if rename_langs is None: + rename_langs = { + "ru_ru", + "ja_jp", + "ko_kr", + "zh_cn", + "zh_hk", + "zh_sg", + "pt_br", + "es_es", + "en_us", + "fr_fr", + "de_de", + "it_it", + "pl_pl", + "tr_tr", + "uk_ua", + "cs_cz", + "hu_hu", + "nl_nl", + "sv_se", + "no_no", + "da_dk", + "fi_fi", + } + + if not in_dir.exists() or not in_dir.is_dir(): + raise FileNotFoundError(f"pending_dir 不存在或不是資料夾:{in_dir}") + + json_files = collect_json_files(in_dir) + if not json_files: + raise FileNotFoundError(f"找不到任何 .json:{in_dir}") + + # ------------------------- + # Pre-scan global total (multithread, like FTB) + # ------------------------- + per_file_counts: List[Tuple[Path, int]] = [] + global_total_keys = 0 + + def _count_one(src: Path) -> Tuple[Path, int]: + """統計單一檔案的翻譯 key 數量。""" + try: + mapping = read_json_dict(src) + return src, int(count_translatable_keys(mapping)) + except Exception as e: + log_warning(f"[KubeJS-LM] 讀取 JSON 失敗 {src}: {e}") + return src, 0 + + max_workers = int( + load_config().get("translator", {}).get("parallel_execution_workers", 4) or 4 + ) + max_workers = max(1, max_workers) + + with ThreadPoolExecutor(max_workers=max_workers) as ex: + futs = [ex.submit(_count_one, p) for p in json_files] + for fu in as_completed(futs): + src, c = fu.result() + per_file_counts.append((src, c)) + global_total_keys += c + + per_file_counts.sort(key=lambda x: x[0].as_posix()) + + if global_total_keys == 0: + log_info("ℹ️ [KubeJS-LM] 0 keys,跳過翻譯") + return {"written_files": 0, "total_keys": 0, "out_dir": str(out_dir)} + + # ------------------------- + # Cache rules + # ------------------------- + cache_rules = {"kubejs": CacheRule("path|source_text")} + + # ------------------------- + # Pre-calc global miss/hit (so progress uses "real translate count") + # ------------------------- + global_total_to_translate = 0 + global_total_hit = 0 + + for src, key_count in per_file_counts: + if key_count == 0: + continue + try: + mapping = read_json_dict(src) + rel_src = src.relative_to(in_dir).as_posix() + file_hint = f"output/kubejs/{rel_src}" # must contain /kubejs/ + all_items = collect_items_from_mapping(mapping, file_hint=file_hint) + + cached_items, items_to_translate = fast_split_items_by_cache( + all_items, + cache_rules=cache_rules, + is_valid_hit=_is_valid_hit, + ) + # ✅ 跳過值已經是繁體中文的 items(節省 API + 避免簡體當英文翻) + _split_off_tw_items(cached_items, items_to_translate) + global_total_hit += len(cached_items) + global_total_to_translate += len(items_to_translate) + except Exception as e: + log_warning(f"[KubeJS-LM] 預掃描失敗 {src}: {e}") + + log_info( + f"🔎 [KubeJS-LM] 待翻譯檔案數:{len(json_files)};總 keys:{global_total_keys}\n" + f"✅ [KubeJS-LM] cache_hit={global_total_hit} | 實際需翻譯(cache_miss)={global_total_to_translate}" + ) + + if global_total_to_translate == 0: + progress(1.0) + + # ------------------------- + # Global build + translate (NO per-file translate) + # ------------------------- + translated_done = 0 + avg_batch_sec = None + total_written = 0 + + per_file_rows: list[dict] = [] + rec = TranslationRecorder() + touch = TouchSet() + _file_write_table: dict[str, tuple[Path, Dict[str, str]]] = {} + + def _writer(file_id: str) -> None: + """寫入翻譯結果到檔案。""" + dst_path, data = _file_write_table[file_id] + write_json_dict(dst_path, data) + + # Build phase: per-file state + global miss list + file_states: dict[str, dict] = {} + all_miss_items: List[Dict[str, Any]] = [] + all_hit_items: List[Dict[str, Any]] = [] + + for idx, (src, key_count) in enumerate(per_file_counts, start=1): + if key_count == 0: + continue + + mapping = read_json_dict(src) + rel_src = src.relative_to(in_dir).as_posix() + + file_hint = f"output/kubejs/{rel_src}" + all_items = collect_items_from_mapping(mapping, file_hint=file_hint) + + cached_items, items_to_translate = fast_split_items_by_cache( + all_items, + cache_rules=cache_rules, + is_valid_hit=_is_valid_hit, + ) + + # ✅ 跳過值已經是繁體中文的 items(從翻譯清單移至 cache hit) + _split_off_tw_items(cached_items, items_to_translate) + + # ✅ NEW:累積 hit items + all_hit_items.extend(cached_items) + + hit = len(cached_items) + miss = len(items_to_translate) + + dst = compute_output_path(src, in_dir, out_dir, rename_langs) + per_file_rows.append( + { + "file": rel_src, + "keys": key_count, + "cache_hit": hit, + "cache_miss": miss, + "dst": dst.relative_to(out_dir).as_posix(), + } + ) + + log_info( + f"[KubeJS-LM] 檔案 {idx}/{len(per_file_counts)}:{rel_src} |" + f"總字串 {key_count},快取命中 {hit},需翻譯 {miss} |" + f"輸出 → {dst.relative_to(out_dir).as_posix()}" + ) + + # out_map base + out_map: Dict[str, str] = { + k: v + for k, v in mapping.items() + if isinstance(k, str) and isinstance(v, str) + } + + # apply cache hits now (record as hit) + for it in cached_items: + p = it.get("path") + t = it.get("text") + if isinstance(p, str) and isinstance(t, str): + # ✅ Rich Text Shield:統一快取命中/miss 路徑 + shielded = it.get("_shielded") + if shielded is not None and shielded.shields: + t = unshield_text(t, shielded.shields) + out_map[p] = t + try: + rec.record( + cache_type="kubejs", + file_id=rel_src, + path=p, + src=str(it.get("source_text") or ""), + dst=t, + cache_hit=True, + extra={"dst_file": dst.relative_to(out_dir).as_posix()}, + ) + except Exception as e: + log_warning(f"[KubeJS-LM] 記錄快取命中失敗: {e}") + + file_id = dst.as_posix() + _file_write_table[file_id] = (dst, out_map) + + file_states[rel_src] = { + "rel_src": rel_src, + "dst": dst, + "file_id": file_id, + "out_map": out_map, + } + + # tag miss items with file context (so callback can write back to correct out_map) + for it in items_to_translate: + it["file_rel"] = rel_src + it["dst_file"] = dst.relative_to(out_dir).as_posix() + + all_miss_items.extend(items_to_translate) + + # ------------------------- + # Dry-run: preview only (no API) + # ------------------------- + if dry_run: + dry_preview_items = all_miss_items[:2000] if all_miss_items else [] + + preview_path = None + try: + meta = { + "files": len(per_file_rows), + "total_keys": global_total_keys, + "cache_hit": global_total_hit, + "cache_miss": global_total_to_translate, + } + + preview_path = write_dry_run_preview( + out_dir, + dry_preview_items, + filename="_kubejs_dry_run_preview.json", + meta=meta, + ) + log_info(f"🧪 [DRY-RUN] preview written: {preview_path.as_posix()}") + + # ✅ NEW:cache hit preview + hit_preview_path = write_cache_hit_preview( + out_dir, + all_hit_items[:2000], # 或不切片 + filename="_kubejs_dry_run_cache_hit_preview.json", + meta=meta, + ) + log_info( + f"🎯 [DRY-RUN] cache-hit preview written: {hit_preview_path.as_posix()}" + ) + + except Exception as e: + log_warning(f"⚠️ [KubeJS-LM] DRY-RUN preview 輸出失敗:{e}") + + progress(1.0) + return { + "dry_run": True, + "files": len(per_file_rows), + "total_keys": global_total_keys, + "cache_hit": global_total_hit, + "cache_miss": global_total_to_translate, + "preview_path": str(preview_path) if preview_path else None, + "out_dir": str(out_dir), + "per_file": per_file_rows, + } + + # ------------------------- + # Real run: translate ONCE for all miss items + # ------------------------- + if not all_miss_items: + # All cache hit -> flush all files once + for st in file_states.values(): + touch.touch(st["file_id"]) + touch.flush(_writer) + total_written = len(file_states) + translated_done = 0 + progress(1.0) + else: + + def on_translated_item(it: Dict[str, Any]) -> None: + """處理翻譯結果。""" + rel_src = it.get("file_rel") + p = it.get("path") + t = it.get("text") + s = it.get("source_text") + + if not (isinstance(rel_src, str) and rel_src in file_states): + return + if not (isinstance(p, str) and isinstance(t, str)): + return + + # ✅ Rich Text Shield:翻譯後還原被保護的格式片段 + shielded = it.get("_shielded") + if shielded is not None and shielded.shields: + final_text = unshield_text(t, shielded.shields) + else: + final_text = t + + st = file_states[rel_src] + st["out_map"][p] = final_text + + try: + rec.record( + cache_type="kubejs", + file_id=rel_src, + path=p, + src=str(s or ""), + dst=t, + cache_hit=False, + extra={"dst_file": st["dst"].relative_to(out_dir).as_posix()}, + ) + except Exception as e: + log_warning(f"[KubeJS-LM] 記錄翻譯結果失敗: {e}") + + try: + touch.touch(st["file_id"]) + except Exception as e: + log_warning(f"[KubeJS-LM] touch 失敗: {e}") + + def on_batch_flushed() -> None: + # write touched files each batch + """批量寫入翻譯結果。""" + try: + touch.flush(_writer) + except Exception as e: + log_warning(f"[KubeJS-LM] 批次刷新失敗,使用 fallback 寫入: {e}") + for fid, (dstp, data) in _file_write_table.items(): + write_json_dict(dstp, data) + + def _fmt_eta(sec: float) -> str: + """格式化剩餘時間。""" + if sec <= 0: + return "" + m, s = divmod(int(sec), 60) + if m >= 60: + h, m2 = divmod(m, 60) + return f"{h}h{m2:02d}m{s:02d}s" + if m > 0: + return f"{m}m{s:02d}s" + return f"{s}s" + + def on_progress(p: float, msg: str, eta_sec: float) -> None: + """報告翻譯進度。""" + eta_txt = _fmt_eta(eta_sec) + log_info(f"{msg}" + (f" | ETA ≈ {eta_txt}" if eta_txt else "")) + progress(p) + + res = translate_items_with_cache_loop( + all_miss_items, + total_for_smart=global_total_to_translate, + translate_batch_smart=lambda batch, total: translate_batch_smart( + batch, total=total + ), + write_new_cache=bool(write_new_cache), + cache_rules=cache_rules, + on_translated_item=on_translated_item, + on_batch_flushed=on_batch_flushed, + on_progress=on_progress, + ) + + translated_done = int(res.processed or 0) + avg_batch_sec = ( + (res.elapsed_sec / res.completed_calls) if res.completed_calls else None + ) + + # final flush all + for st in file_states.values(): + touch.touch(st["file_id"]) + touch.flush(_writer) + + total_written = len(file_states) + progress(1.0) + + # ------------------------- + # Export records + # ------------------------- + rec_json = None + rec_csv = None + try: + rec_json = rec.export_json(out_dir / "translation_map.json") + rec_csv = rec.export_csv(out_dir / "translation_map.csv") + log_info(f"🧾 [KubeJS-LM] records written: {rec_json} | {rec_csv}") + except Exception as ex: + log_info(f"⚠️ [KubeJS-LM] export records failed: {ex}") + + return { + "written_files": total_written, + "total_keys": global_total_keys, + "cache_hit": global_total_hit, + "cache_miss": global_total_to_translate, + "api_translated": translated_done, + "avg_batch_sec": avg_batch_sec, + "out_dir": str(out_dir), + "records_json": str(rec_json) if rec_json else None, + "records_csv": str(rec_csv) if rec_csv else None, + "per_file": per_file_rows, + } From 4de6b4393c57d5bbf3d2aae7c959140f898e632e Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Sat, 28 Mar 2026 23:47:23 +0800 Subject: [PATCH 28/33] fix(pr43): restore CI on Linux and clean lint --- tests/test_cache_manager.py | 32 +++----- tests/test_cache_store.py | 21 +++-- tests/test_ftbquests_unshield_logic.py | 67 ++++++++++------ tests/test_kubejs_translator_clean.py | 81 ++++++++++--------- tests/test_lm_api_client.py | 89 ++++++++++----------- tests/test_lm_translator_main_prompts.py | 3 - translation_tool/core/lm_config_rules.py | 27 ++++--- translation_tool/core/lm_translator_main.py | 88 ++++++++++++-------- translation_tool/utils/cache_shards.py | 46 ++++++++--- 9 files changed, 253 insertions(+), 201 deletions(-) diff --git a/tests/test_cache_manager.py b/tests/test_cache_manager.py index e399d242..2384b342 100644 --- a/tests/test_cache_manager.py +++ b/tests/test_cache_manager.py @@ -8,7 +8,7 @@ import threading from pathlib import Path -from unittest.mock import patch, MagicMock +from unittest.mock import patch import pytest @@ -19,6 +19,7 @@ # Fixtures # ============================================================================= + @pytest.fixture def fresh_state(): """提供乾淨的 runtime state(每個測試獨立)。""" @@ -49,6 +50,7 @@ def mock_save_path(tmp_path: Path, fresh_state): # 測試 1: initialize_translation_cache() 的 cache_lock 保護 # ============================================================================= + def test_initialize_translation_cache_uses_cache_lock(fresh_state): """驗證 initialize_translation_cache() 在 cache_lock 保護下執行。 @@ -57,24 +59,15 @@ def test_initialize_translation_cache_uses_cache_lock(fresh_state): 不會造成重複載入。 """ call_count = 0 - lock_enter_order = [] - lock_exit_order = [] - - original_lock_class = type(fresh_state.cache_lock) - - # 追蹤 lock 的進入/離開時間點 - def _lock_enter(): - lock_enter_order.append(len(lock_enter_order)) - - def _lock_exit(): - lock_exit_order.append(len(lock_exit_order)) # Patch _load_cache_type 來計數呼叫 def _load_cache_type_track(cache_type): nonlocal call_count call_count += 1 - with patch.object(cache_manager, "_load_cache_type", side_effect=_load_cache_type_track): + with patch.object( + cache_manager, "_load_cache_type", side_effect=_load_cache_type_track + ): # 模擬兩執行緒同時進入 def call_init(): cache_manager.initialize_translation_cache() @@ -117,6 +110,7 @@ def _load_cache_type_tracking(cache_type): # 測試 2: save_translation_cache() 的 clear_dirty() 時機 # ============================================================================= + def test_save_translation_cache_dirty_True_when_save_fails(mock_save_path): """驗證 save_translation_cache() 在寫入失敗後 dirty flag 仍為 True。 @@ -131,9 +125,7 @@ def test_save_translation_cache_dirty_True_when_save_fails(mock_save_path): state = cache_store.get_runtime_state() state.is_dirty[cache_type] = True - state.session_new_entries[cache_type] = { - "key1": {"src": "Hello", "dst": "哈囉"} - } + state.session_new_entries[cache_type] = {"key1": {"src": "Hello", "dst": "哈囉"}} with patch.object( cache_manager, @@ -155,9 +147,7 @@ def test_save_translation_cache_dirty_cleared_when_save_succeeds(mock_save_path) state = cache_store.get_runtime_state() state.is_dirty[cache_type] = True - state.session_new_entries[cache_type] = { - "key1": {"src": "Hello", "dst": "哈囉"} - } + state.session_new_entries[cache_type] = {"key1": {"src": "Hello", "dst": "哈囉"}} saved_data = {} @@ -186,9 +176,7 @@ def test_save_translation_cache_no_op_when_no_dirty_entries(mock_save_path): state.is_dirty[cache_type] = False state.session_new_entries[cache_type] = {} - with patch.object( - cache_manager, "_save_entries_to_active_shards" - ) as mock_save: + with patch.object(cache_manager, "_save_entries_to_active_shards") as mock_save: cache_manager.save_translation_cache(cache_type) # 驗證:無 session 資料時不呼叫儲存 diff --git a/tests/test_cache_store.py b/tests/test_cache_store.py index 99a6b07b..0f0cf704 100644 --- a/tests/test_cache_store.py +++ b/tests/test_cache_store.py @@ -1,4 +1,5 @@ from pathlib import Path +import threading from translation_tool.utils import cache_manager, cache_store @@ -12,7 +13,9 @@ def test_cache_store_entry_and_value_crud(): assert cache_store.get_entry(cache_dict, "k1") == {"src": "s1", "dst": "d1"} assert cache_store.get_value(cache_dict, "k1") == "d1" - changed_again = cache_store.add_entry(cache_dict, "k1", {"src": "s1-new", "dst": "d1"}) + changed_again = cache_store.add_entry( + cache_dict, "k1", {"src": "s1-new", "dst": "d1"} + ) assert changed_again is False # contract: dst 相同時不覆寫舊 entry assert cache_store.get_entry(cache_dict, "k1") == {"src": "s1", "dst": "d1"} @@ -55,12 +58,16 @@ def test_manager_add_save_reload_smoke(monkeypatch, tmp_path: Path): saved = {} - def _fake_save_entries(_cache_type: str, entries: dict, force_new_shard: bool = False): + def _fake_save_entries( + _cache_type: str, entries: dict, force_new_shard: bool = False + ): saved["cache_type"] = _cache_type saved["entries"] = entries.copy() saved["force_new_shard"] = force_new_shard - monkeypatch.setattr(cache_manager, "_save_entries_to_active_shards", _fake_save_entries) + monkeypatch.setattr( + cache_manager, "_save_entries_to_active_shards", _fake_save_entries + ) cache_manager.add_to_cache(cache_type, key, "Hello", "哈囉") assert cache_manager.get_from_cache(cache_type, key) == "哈囉" @@ -79,15 +86,16 @@ def _fake_load_cache_type(_cache_type: str): monkeypatch.setattr(cache_manager, "_load_cache_type", _fake_load_cache_type) cache_manager.reload_translation_cache_type(cache_type) - assert cache_manager.get_cache_entry(cache_type, key) == {"src": "Hello", "dst": "哈囉"} + assert cache_manager.get_cache_entry(cache_type, key) == { + "src": "Hello", + "dst": "哈囉", + } # ============================================================================= # 測試 3: add_entry() 執行緒安全保護 # ============================================================================= -import threading - def test_add_entry_thread_safety_different_keys(): """驗證 add_entry() 多執行緒同時寫入不同 key 時不造成資料遺失。 @@ -180,4 +188,3 @@ def writer(thread_id, value): assert isinstance(cache_dict["shared_key"], dict) assert "src" in cache_dict["shared_key"] assert "dst" in cache_dict["shared_key"] - diff --git a/tests/test_ftbquests_unshield_logic.py b/tests/test_ftbquests_unshield_logic.py index 65ad9295..fa1e9656 100644 --- a/tests/test_ftbquests_unshield_logic.py +++ b/tests/test_ftbquests_unshield_logic.py @@ -5,6 +5,7 @@ 參考:PR #42 (pr/rich-text-shield) — 2026-03-23 """ + from __future__ import annotations import sys @@ -13,9 +14,10 @@ # 確保可以導入翻譯工具模組 ROOT = Path(__file__).resolve().parents[1] -sys.path.insert(0, str(ROOT)) +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) -from translation_tool.plugins.shared.rich_text_shield import ( +from translation_tool.plugins.shared.rich_text_shield import ( # noqa: E402 ShieldedText, ShieldPiece, ) @@ -25,6 +27,7 @@ # 測試:ftbquests_lmtranslator.on_translated_item — unshield 使用 .shields # --------------------------------------------------------------------------- + def test_ftb_on_translated_item_unshield_uses_shields_list(): """ 驗證 ftbquests_lmtranslator 的 on_translated_item 回呼 @@ -67,14 +70,17 @@ def mock_unshield_text(text: str, shields_arg) -> str: return text.replace("$C0$", "&c") # Patch 在 ftbquests_lmtranslator 命名空間中的 unshield_text - with patch.object( - ftbquests_lmtranslator, - "unshield_text", - side_effect=mock_unshield_text, - ), patch.object( - ftbquests_lmtranslator, - "shield_text", - return_value=fake_shielded, + with ( + patch.object( + ftbquests_lmtranslator, + "unshield_text", + side_effect=mock_unshield_text, + ), + patch.object( + ftbquests_lmtranslator, + "shield_text", + return_value=fake_shielded, + ), ): # on_translated_item 是翻譯流程中的 nested callback, # 無法直接呼叫。我們透過翻譯流程觸發它。 @@ -144,15 +150,19 @@ def mock_unshield_text(text: str, shields_arg): captured_second_arg_type.append(type(shields_arg).__name__) return text # 不做還原 - with patch.object( - ftbquests_lmtranslator, - "unshield_text", - side_effect=mock_unshield_text, - ), patch.object( - ftbquests_lmtranslator, - "shield_text", - return_value=fake_shielded, + with ( + patch.object( + ftbquests_lmtranslator, + "unshield_text", + side_effect=mock_unshield_text, + ), + patch.object( + ftbquests_lmtranslator, + "shield_text", + return_value=fake_shielded, + ), ): + def closure_under_test(it: dict): from translation_tool.plugins.ftbquests import ftbquests_lmtranslator as m @@ -176,6 +186,7 @@ def closure_under_test(it: dict): # 測試:md_lmtranslator.on_translated_item — unshield 使用 .shields # --------------------------------------------------------------------------- + def test_md_on_translated_item_unshield_uses_shields_list(): """ 驗證 md_lmtranslator 的 on_translated_item 回呼 @@ -280,15 +291,19 @@ def mock_unshield_text(text: str, shields_arg) -> str: captured_calls.append({"text": text, "shields_arg": shields_arg}) return text.replace("$C1$", "&a") - with patch.object( - md_lmtranslator, - "unshield_text", - side_effect=mock_unshield_text, - ), patch.object( - md_lmtranslator, - "shield_text", - return_value=fake_shielded, + with ( + patch.object( + md_lmtranslator, + "unshield_text", + side_effect=mock_unshield_text, + ), + patch.object( + md_lmtranslator, + "shield_text", + return_value=fake_shielded, + ), ): + def closure_md_on_translated_item(it: dict): from translation_tool.plugins.md import md_lmtranslator as m diff --git a/tests/test_kubejs_translator_clean.py b/tests/test_kubejs_translator_clean.py index bb713156..39920062 100644 --- a/tests/test_kubejs_translator_clean.py +++ b/tests/test_kubejs_translator_clean.py @@ -2,10 +2,11 @@ 用途:測試 kubejs_translator_clean 中的清理與合併邏輯。 """ + from __future__ import annotations -from pathlib import Path import sys +from pathlib import Path import orjson import pytest @@ -15,7 +16,7 @@ if str(ROOT) not in sys.path: sys.path.insert(0, str(ROOT)) -from translation_tool.core.kubejs_translator_clean import ( +from translation_tool.core.kubejs_translator_clean import ( # noqa: E402 is_filled_text_impl, deep_merge_3way_flat_impl, prune_en_by_tw_flat_impl, @@ -115,7 +116,9 @@ def test_merge_multiple_keys(self): def test_empty_dicts(self): """測試空字典。""" - result = deep_merge_3way_flat_impl({}, {}, {}, safe_convert_text_fn=self._safe_convert) + result = deep_merge_3way_flat_impl( + {}, {}, {}, safe_convert_text_fn=self._safe_convert + ) assert result == {} @@ -161,12 +164,10 @@ def test_reverse_index_is_dict_str_str_not_list(self): # 類型驗證:每個 value 都應該是 str,不是 list for k, v in result.items(): - assert isinstance( - k, str - ), f"key 應為 str,實際為 {type(k).__name__}" - assert isinstance( - v, str - ), f"value for key '{k}' 應為 str,實際為 {type(v).__name__}" + assert isinstance(k, str), f"key 應為 str,實際為 {type(k).__name__}" + assert isinstance(v, str), ( + f"value for key '{k}' 應為 str,實際為 {type(v).__name__}" + ) def test_prefers_translated_key_over_untranslated(self): """當多個 key 有相同翻譯值時,應優先選擇「已翻譯」的 key。 @@ -177,7 +178,7 @@ def test_prefers_translated_key_over_untranslated(self): # key_b:翻譯值等於 key 名(未翻譯) final_tw_lookup = { "apple": "蘋果", # 已翻譯(值 != key) - "蘋果": "蘋果", # 未翻譯(值 == key) + "蘋果": "蘋果", # 未翻譯(值 == key) } result = _build_reverse_index_impl(final_tw_lookup) @@ -189,8 +190,8 @@ def test_prefers_alphabetically_smallest_among_same_priority(self): # 多個 key 都已翻譯(值 != key),取字母序最小 final_tw_lookup = { "zebra": "動物", # 已翻譯,但字母序較大 - "ant": "動物", # 已翻譯,字母序最小 - "bee": "動物", # 已翻譯,字母序居中 + "ant": "動物", # 已翻譯,字母序最小 + "bee": "動物", # 已翻譯,字母序居中 } result = _build_reverse_index_impl(final_tw_lookup) @@ -199,8 +200,8 @@ def test_prefers_alphabetically_smallest_among_same_priority(self): def test_mixed_translated_and_untranslated_chooses_correct(self): """混合場景:已翻譯優先於未翻譯。""" final_tw_lookup = { - "apple": "蘋果", # 已翻譯 - "banana": "香蕉", # 未翻譯 + "apple": "蘋果", # 已翻譯 + "banana": "香蕉", # 未翻譯 "cherry": "櫻桃", # 已翻譯 } result = _build_reverse_index_impl(final_tw_lookup) @@ -251,9 +252,9 @@ def test_non_filled_text_values_are_ignored(self): """非填充文字值(如空字串、空白)不應進入 reverse_index。""" final_tw_lookup = { "key1": "有效翻譯", - "key2": "", # 空字串,應忽略 - "key3": " ", # 空白,應忽略 - "key4": "{ref}", # 語言參考,應忽略 + "key2": "", # 空字串,應忽略 + "key3": " ", # 空白,應忽略 + "key4": "{ref}", # 語言參考,應忽略 } result = _build_reverse_index_impl(final_tw_lookup) @@ -267,8 +268,8 @@ def test_casefold_ascii_translation_detection(self): # "Copper Ingot" vs "copper ingot":casefold 後相同,視為已翻譯 # "copper ingot" vs "copper ingot":完全相同,視為未翻譯 final_tw_lookup = { - "copper_ingot": "Copper Ingot", # 已翻譯(casefold 不同) - "Copper Ingot": "Copper Ingot", # 未翻譯(casefold 相同) + "copper_ingot": "Copper Ingot", # 已翻譯(casefold 不同) + "Copper Ingot": "Copper Ingot", # 未翻譯(casefold 相同) } result = _build_reverse_index_impl(final_tw_lookup) @@ -278,8 +279,8 @@ def test_casefold_ascii_translation_detection(self): def test_non_ascii_uses_direct_equality(self): """非 ASCII 翻譯使用直接相等判斷是否為「已翻譯」。""" final_tw_lookup = { - "蘋果": "蘋果", # 未翻譯 - "apple": "蘋果", # 已翻譯 + "蘋果": "蘋果", # 未翻譯 + "apple": "蘋果", # 已翻譯 } result = _build_reverse_index_impl(final_tw_lookup) @@ -301,8 +302,8 @@ def test_dedup_removes_keys_with_value_in_reverse_index(self): "mod.item3": "Cherry", } reverse_index = { - "Apple": "final.apple", # Apple 已在 final 中 - "Banana": "final.banana", # Banana 已在 final 中 + "Apple": "final.apple", # Apple 已在 final 中 + "Banana": "final.banana", # Banana 已在 final 中 } result = _dedup_pending_en_impl(pending_en, reverse_index) @@ -321,9 +322,9 @@ def test_dedup_cross_namespace_bug_fixed(self): - 新邏輯:`v in reverse_index` → "Apple" in reverse_index → True → 去重 ✅ """ pending_en = { - "raw:item_a": "Apple", # value: Apple - "raw:item_b": "Banana", # value: Banana(不在 reverse_index) - "raw:item_c": "Cherry", # value: Cherry + "raw:item_a": "Apple", # value: Apple + "raw:item_b": "Banana", # value: Banana(不在 reverse_index) + "raw:item_c": "Cherry", # value: Cherry } reverse_index = { # final 中有不同的 key 名,但相同的翻譯值 @@ -340,16 +341,16 @@ def test_dedup_cross_namespace_bug_fixed(self): def test_dedup_non_filled_text_not_removed(self): """非填充文字(如空字串、空白、語言參考)不受去重邏輯影響。""" pending_en = { - "key1": "", # 空字串,應保留(即使 "" 在 reverse_index) - "key2": " ", # 空白,應保留 - "key3": "{ref}", # 語言參考,應保留 - "key4": "有效翻譯", # 有效文字,在 reverse_index 中,應移除 + "key1": "", # 空字串,應保留(即使 "" 在 reverse_index) + "key2": " ", # 空白,應保留 + "key3": "{ref}", # 語言參考,應保留 + "key4": "有效翻譯", # 有效文字,在 reverse_index 中,應移除 } reverse_index = { - "": "some_key", # reverse_index 中有 "" - " ": "some_key2", # reverse_index 中有空白 - "{ref}": "some_key3", # reverse_index 中有 ref - "有效翻譯": "tw_key", # 有效翻譯 + "": "some_key", # reverse_index 中有 "" + " ": "some_key2", # reverse_index 中有空白 + "{ref}": "some_key3", # reverse_index 中有 ref + "有效翻譯": "tw_key", # 有效翻譯 } result = _dedup_pending_en_impl(pending_en, reverse_index) @@ -384,10 +385,7 @@ def test_dedup_stability_across_multiple_calls(self): "翻譯B": "final:key2", } - results = [ - _dedup_pending_en_impl(pending_en, reverse_index) - for _ in range(10) - ] + results = [_dedup_pending_en_impl(pending_en, reverse_index) for _ in range(10)] expected = {"namespace:item3": "翻譯C"} for i, r in enumerate(results): @@ -424,6 +422,7 @@ def mock_lang_files(self, tmp_path: Path): def test_clean_kubejs_from_raw_basic(self, mock_lang_files: Path): """測試基本清理功能。""" + def read_json(path: Path) -> dict: if not path or not path.is_file(): return {} @@ -487,7 +486,7 @@ def test_clean_kubejs_cross_namespace_dedup(self, tmp_path: Path): # raw en_us.json:兩個 items en_data = { - "raw_ns:apple": "Apple", # 會被 cn 覆蓋 + "raw_ns:apple": "Apple", # 會被 cn 覆蓋 "raw_ns:cherry": "Cherry", # 無 cn/tw,保留 } (raw_root / "en_us.json").write_bytes(orjson.dumps(en_data)) @@ -531,7 +530,9 @@ def safe_convert(text: str) -> str: # 讀取產出的 pending en_us.json # rel_group = "assets/test/lang"(相對於 raw_root/kubejs) pending_en_file = pending_root_p / "assets" / "test" / "lang" / "en_us.json" - assert pending_en_file.exists(), f"pending en_us.json 應存在,實際目錄內容:{list((pending_root_p / 'assets' / 'test' / 'lang').iterdir()) if (pending_root_p / 'assets' / 'test' / 'lang').exists() else '不存在'}" + assert pending_en_file.exists(), ( + f"pending en_us.json 應存在,實際目錄內容:{list((pending_root_p / 'assets' / 'test' / 'lang').iterdir()) if (pending_root_p / 'assets' / 'test' / 'lang').exists() else '不存在'}" + ) pending_data = read_json(pending_en_file) diff --git a/tests/test_lm_api_client.py b/tests/test_lm_api_client.py index 0dc9cb15..7f52f20b 100644 --- a/tests/test_lm_api_client.py +++ b/tests/test_lm_api_client.py @@ -2,6 +2,7 @@ 用途:測試 LM API 用戶端相關功能。 """ + import pytest from unittest.mock import patch, Mock import requests @@ -10,27 +11,25 @@ class TestCallGeminiRequests: """測試 call_gemini_requests 函數。""" - @patch('translation_tool.core.lm_api_client.requests.post') - @patch('translation_tool.core.lm_api_client.load_config') + @patch("translation_tool.core.lm_api_client.requests.post") + @patch("translation_tool.core.lm_api_client.load_config") def test_successful_call(self, mock_config, mock_post): """測試成功呼叫。""" from translation_tool.core.lm_api_client import call_gemini_requests - + # Mock 配置 mock_config.return_value = {"lm_translator": {"rate_limit": {"timeout": 60}}} - + # Mock 回應 mock_response = Mock() mock_response.ok = True mock_response.json.return_value = { - "candidates": [{ - "content": { - "parts": [{"text": '{"translated": "value"}'}] - } - }] + "candidates": [ + {"content": {"parts": [{"text": '{"translated": "value"}'}]}} + ] } mock_post.return_value = mock_response - + result = call_gemini_requests( model_name="gemini-pro", system_prompt="You are a translator", @@ -38,23 +37,23 @@ def test_successful_call(self, mock_config, mock_post): api_key="test_api_key", temperature=0.7, ) - + assert result == '{"translated": "value"}' - @patch('translation_tool.core.lm_api_client.requests.post') - @patch('translation_tool.core.lm_api_client.load_config') + @patch("translation_tool.core.lm_api_client.requests.post") + @patch("translation_tool.core.lm_api_client.load_config") def test_http_error(self, mock_config, mock_post): """測試 HTTP 錯誤。""" from translation_tool.core.lm_api_client import call_gemini_requests - + mock_config.return_value = {"lm_translator": {"rate_limit": {"timeout": 60}}} - + mock_response = Mock() mock_response.ok = False mock_response.status_code = 500 mock_response.text = "Internal Server Error" mock_post.return_value = mock_response - + with pytest.raises(requests.HTTPError): call_gemini_requests( model_name="gemini-pro", @@ -64,19 +63,19 @@ def test_http_error(self, mock_config, mock_post): temperature=0.7, ) - @patch('translation_tool.core.lm_api_client.requests.post') - @patch('translation_tool.core.lm_api_client.load_config') + @patch("translation_tool.core.lm_api_client.requests.post") + @patch("translation_tool.core.lm_api_client.load_config") def test_invalid_response_format(self, mock_config, mock_post): """測試無效的回應格式。""" from translation_tool.core.lm_api_client import call_gemini_requests - + mock_config.return_value = {"lm_translator": {"rate_limit": {"timeout": 60}}} - + mock_response = Mock() mock_response.ok = True mock_response.json.return_value = {} # 缺少 candidates mock_post.return_value = mock_response - + with pytest.raises(RuntimeError): call_gemini_requests( model_name="gemini-pro", @@ -86,42 +85,37 @@ def test_invalid_response_format(self, mock_config, mock_post): temperature=0.7, ) - @patch('translation_tool.core.lm_api_client.requests.post') - @patch('translation_tool.core.lm_api_client.load_config') + @patch("translation_tool.core.lm_api_client.requests.post") + @patch("translation_tool.core.lm_api_client.load_config") def test_custom_timeout(self, mock_config, mock_post): """測試自定義超時。""" from translation_tool.core.lm_api_client import call_gemini_requests - + mock_config.return_value = {"lm_translator": {"rate_limit": {"timeout": 120}}} - + mock_response = Mock() mock_response.ok = True mock_response.json.return_value = { - "candidates": [{ - "content": { - "parts": [{"text": '{"result": "ok"}'}] - } - }] + "candidates": [{"content": {"parts": [{"text": '{"result": "ok"}'}]}}] } mock_post.return_value = mock_response - - result = call_gemini_requests( + + call_gemini_requests( model_name="gemini-pro", system_prompt="test", payload={}, api_key="test_key", temperature=0.5, ) - + # 驗證 post 被調用 mock_post.assert_called_once() # 驗證 timeout 參數被傳遞 call_kwargs = mock_post.call_args.kwargs - assert call_kwargs.get('timeout') == 120 + assert call_kwargs.get("timeout") == 120 - - @patch('translation_tool.core.lm_api_client.requests.post') - @patch('translation_tool.core.lm_api_client.load_config') + @patch("translation_tool.core.lm_api_client.requests.post") + @patch("translation_tool.core.lm_api_client.load_config") def test_api_key_not_in_url(self, mock_config, mock_post): """測試 API Key 不出現在 URL 中,而是放在 Authorization: Bearer header。""" from translation_tool.core.lm_api_client import call_gemini_requests @@ -134,11 +128,7 @@ def test_api_key_not_in_url(self, mock_config, mock_post): mock_response = Mock() mock_response.ok = True mock_response.json.return_value = { - "candidates": [{ - "content": { - "parts": [{"text": '{"result": "ok"}'}] - } - }] + "candidates": [{"content": {"parts": [{"text": '{"result": "ok"}'}]}}] } mock_post.return_value = mock_response @@ -152,18 +142,22 @@ def test_api_key_not_in_url(self, mock_config, mock_post): # 驗證 URL 中不包含 API key call_args = mock_post.call_args - called_url = call_args.args[0] if call_args.args else call_args.kwargs.get('url', '') + called_url = ( + call_args.args[0] if call_args.args else call_args.kwargs.get("url", "") + ) assert fake_api_key not in called_url, "API key 不應出現在 URL 中" # 驗證 Authorization: Bearer header 存在 - headers = call_args.kwargs.get('headers', {}) - assert 'Authorization' in headers, "Authorization header 必須存在" - assert headers['Authorization'] == f"Bearer {fake_api_key}", \ + headers = call_args.kwargs.get("headers", {}) + assert "Authorization" in headers, "Authorization header 必須存在" + assert headers["Authorization"] == f"Bearer {fake_api_key}", ( "Authorization header 應為 Bearer {api_key} 格式" + ) # 確保 URL 中沒有 key=... 之類的 query string - assert '?' not in called_url or 'key=' not in called_url, \ + assert "?" not in called_url or "key=" not in called_url, ( "URL 中不應包含 key query parameter" + ) class TestModuleImports: @@ -172,4 +166,5 @@ class TestModuleImports: def test_imports(self): """測試必要導入。""" from translation_tool.core.lm_api_client import call_gemini_requests + assert callable(call_gemini_requests) diff --git a/tests/test_lm_translator_main_prompts.py b/tests/test_lm_translator_main_prompts.py index 41e7f110..bc8ab830 100644 --- a/tests/test_lm_translator_main_prompts.py +++ b/tests/test_lm_translator_main_prompts.py @@ -1,5 +1,4 @@ """測試 System Prompt dict → string 轉換(lm_translator_main.py)。""" -from unittest.mock import patch class TestSystemPromptConversion: @@ -7,8 +6,6 @@ class TestSystemPromptConversion: def test_patchouli_prompt_dict_conversion_logic(self): """測試 dict(含 content/text key)轉換邏輯。""" - from translation_tool.core import lm_translator_main as mod - # 測試轉換函式存在 raw = {"role": "system", "content": "測試內容"} result = raw.get("content") or raw.get("text") or str(raw) assert result == "測試內容" diff --git a/translation_tool/core/lm_config_rules.py b/translation_tool/core/lm_config_rules.py index deb29e4a..8b89497b 100644 --- a/translation_tool/core/lm_config_rules.py +++ b/translation_tool/core/lm_config_rules.py @@ -12,7 +12,6 @@ from ..utils.log_unit import log_info, log_error, log_debug - # ========================= # 1. 執行緒安全的 API Key 索引追蹤器 # ========================= @@ -21,7 +20,7 @@ class KeyIndexTracker: """ 執行緒安全的 API Key 索引追蹤器。 - + 用於解決多執行緒環境下全域變數 _current_key_index 的 race condition 問題。 透過 threading.Lock 確保並發存取的安全性。 """ @@ -83,6 +82,7 @@ def reset_key_index() -> None: # 2. 提示詞與配置 # ========================= + def _get_all_keys() -> list[str]: """ 私有輔助函式:統一代理從設定檔讀取並清理金鑰列表。 @@ -94,6 +94,7 @@ def _get_all_keys() -> list[str]: if isinstance(key, str) and key.strip() ] + def get_current_api_key() -> str: """ 從金鑰池中取得目前正在使用的 API 金鑰。 @@ -125,6 +126,7 @@ def get_current_api_key() -> str: _key_tracker._index = _key_tracker._index % len(keys) return key + def rotate_api_key(): """ 切換至下一個可用的 API Key。 @@ -168,6 +170,7 @@ def rotate_api_key(): log_info(f"🔁 切換 API Key → index {new_index}") return True + def validate_api_keys(): """ 驗證 API 金鑰格式。 @@ -204,6 +207,7 @@ def validate_api_keys(): log_info(f"✅ 金鑰格式驗證通過,共載入 {len(keys)} 組金鑰。") + def validate_api_keys_from_ui(keys: list[str]): # ui 專用 """驗證 API Key 格式(UI 專用)。 @@ -212,9 +216,7 @@ def validate_api_keys_from_ui(keys: list[str]): # ui 專用 """ for k in keys: if not k: - raise RuntimeError( - f"❌ API Key 不得為空,請輸入有效的 Gemini API Key。" - ) + raise RuntimeError("❌ API Key 不得為空,請輸入有效的 Gemini API Key。") if not k.startswith("AIza"): raise RuntimeError( f"❌ 無效的 API Key 格式:{k!r}\n" @@ -232,6 +234,7 @@ def validate_api_keys_from_ui(keys: list[str]): # ui 專用 "僅允許 'AIza' 開頭後接英文字母、數字、 dash(-) 或 underscore(_)。" ) + # ========================= # 2. Regex 規則定義 # ========================= @@ -266,10 +269,9 @@ def validate_api_keys_from_ui(keys: list[str]): # ui 專用 # 需要跳過翻譯的文字(你指定的類型) HASH_PREFIX_PATTERN = re.compile(r"^\s*#") # 任何 # 開頭(含前置空白) -def needs_translation_text(s: str) -> bool: - """ - """ +def needs_translation_text(s: str) -> bool: + """ """ if not s or not isinstance(s, str): return False @@ -288,6 +290,7 @@ def needs_translation_text(s: str) -> bool: # 還有英文 → 需要翻 return True + def value_fully_translated(value) -> bool: """ 判斷一個值是否「已完全翻譯完成」。 @@ -342,6 +345,7 @@ def value_fully_translated(value) -> bool: # 直接視為已完成翻譯 return True + def contains_cjk(s: str) -> bool: """ 檢查字串中是否包含 CJK(中 / 日 / 韓)文字。 @@ -377,6 +381,7 @@ def contains_cjk(s: str) -> bool: """ return isinstance(s, str) and CJK_RE.search(s) is not None + def build_skip_terms_pattern(terms: list[str]) -> re.Pattern: """ 將「需跳過翻譯的關鍵字清單」轉換為單一正規表達式(regex)。 @@ -426,12 +431,11 @@ def build_skip_terms_pattern(terms: list[str]) -> re.Pattern: # 編譯為不區分大小寫的正規表達式 return re.compile(pattern, re.IGNORECASE) + # ========================= # 值是否值得翻譯(核心判斷) def is_value_translatable(value: Any, *, is_lang: bool = False) -> bool: - """ - - """ + """ """ if not isinstance(value, str): return False @@ -491,6 +495,7 @@ def is_value_translatable(value: Any, *, is_lang: bool = False) -> bool: return True + # ========================= # 可翻譯欄位判斷 # ========================= diff --git a/translation_tool/core/lm_translator_main.py b/translation_tool/core/lm_translator_main.py index 1c866347..5fd67861 100644 --- a/translation_tool/core/lm_translator_main.py +++ b/translation_tool/core/lm_translator_main.py @@ -41,32 +41,38 @@ # 翻譯入口函數(新結構) # ========================================================= -def translate_batch_smart(batch_items, total=None, dry_run: bool = DEFAULT_DRY_RUN, export_cache_only: bool = DEFAULT_EXPORT_CACHE_ONLY): + +def translate_batch_smart( + batch_items, + total=None, + dry_run: bool = DEFAULT_DRY_RUN, + export_cache_only: bool = DEFAULT_EXPORT_CACHE_ONLY, +): """ 智慧批次翻譯函數(主入口) - + 參數: batch_items: 翻譯項目列表 total: 總項目數(可選) dry_run: True = 不呼叫API,只模擬流程(測試用) export_cache_only: True = 只輸出快取中的內容 - + 職責:協調各子流程,不直接處理細節 """ # 1. 驗證與正規化 items = _validate_batch_items(batch_items) if not items: return [], "AUTO" - + # 2. 偵測 profile(TODO: 舊函數會重新計算,目前是被丟棄的死碼) # batch_profile = _detect_batch_profile(items) - + # 3. 計算批次大小(TODO: 舊函數會重新計算,目前是被丟棄的死碼) # batch_size = _calculate_batch_size(batch_profile) - + # 4. 執行翻譯 results, status = _execute_translation(items, total, dry_run, export_cache_only) - + # 5. 處理輸出 return _process_output(results, status) @@ -74,7 +80,7 @@ def translate_batch_smart(batch_items, total=None, dry_run: bool = DEFAULT_DRY_R def _validate_batch_items(items): """ 驗證與正規化輸入資料 - + 參數: items: 原始項目列表 回傳: @@ -82,7 +88,7 @@ def _validate_batch_items(items): """ if not items: return [] - + validated = [] for item in items: # 跳过无效项目 @@ -92,13 +98,13 @@ def _validate_batch_items(items): text = item.get("text", "") if not text or not str(text).strip(): continue - + # 確保有 cache_type if "cache_type" not in item: item["cache_type"] = "patchouli" - + validated.append(item) - + return validated @@ -121,7 +127,7 @@ def _execute_translation(items, total, dry_run=False, export_cache_only=False): def _process_output(results, status): """ 處理輸出結果 - + 參數: results: 翻譯結果(可能是元組或列表) status: 翻譯狀態 @@ -131,11 +137,11 @@ def _process_output(results, status): # 處理元組情況(從舊函數返回) if isinstance(results, tuple): return results - + # 處理空結果 if not results: return [], "AUTO" - + return results, status @@ -143,7 +149,10 @@ def _process_output(results, status): # 舊翻譯函數(保留原邏輯) # ========================================================= -def translate_batch_smart_old(batch_items, total=None, dry_run=False, export_cache_only=False): + +def translate_batch_smart_old( + batch_items, total=None, dry_run=False, export_cache_only=False +): """ 智慧型分批翻譯函式 支援動態縮減 Batch Size、模型切換、以及自動處理輸出截斷問題。 @@ -265,7 +274,11 @@ def detect_batch_profile(items): PATCHOULI_SYSTEM_PROMPT = _patchouli_raw elif isinstance(_patchouli_raw, dict): # 支援 {"content": "..."} 或 {"text": "..."} 格式的 dict - PATCHOULI_SYSTEM_PROMPT = _patchouli_raw.get("content") or _patchouli_raw.get("text") or str(_patchouli_raw) + PATCHOULI_SYSTEM_PROMPT = ( + _patchouli_raw.get("content") + or _patchouli_raw.get("text") + or str(_patchouli_raw) + ) else: PATCHOULI_SYSTEM_PROMPT = str(_patchouli_raw) @@ -278,11 +291,12 @@ def detect_batch_profile(items): if isinstance(_lang_raw, str): LANG_SYSTEM_PROMPT = _lang_raw elif isinstance(_lang_raw, dict): - LANG_SYSTEM_PROMPT = _lang_raw.get("content") or _lang_raw.get("text") or str(_lang_raw) + LANG_SYSTEM_PROMPT = ( + _lang_raw.get("content") or _lang_raw.get("text") or str(_lang_raw) + ) else: LANG_SYSTEM_PROMPT = str(_lang_raw) - pinned_model_index = None # None = 正常模式,非 None = 鎖定指定模型 # 進入動態 Batch 迴圈 while remaining_items: @@ -347,7 +361,7 @@ def detect_batch_profile(items): prompt = LANG_SYSTEM_PROMPT else: # ftb / patchouli / 其他 - prompt = PATCHOUI_SYSTEM_PROMPT + prompt = PATCHOULI_SYSTEM_PROMPT log_debug( "Batch profile=%s -> System Prompt=%s", @@ -380,6 +394,7 @@ def _is_truncated(text: str) -> bool: """ try: import json + json.loads(text) return False # 成功解析,代表沒截斷 except json.JSONDecodeError: @@ -387,9 +402,9 @@ def _is_truncated(text: str) -> bool: # 大括號平衡檢查 count = 0 for ch in text: - if ch == '{': + if ch == "{": count += 1 - elif ch == '}': + elif ch == "}": count -= 1 if count < 0: return True # } 比 { 先出現,代表截斷 @@ -397,7 +412,9 @@ def _is_truncated(text: str) -> bool: if _is_truncated(raw_text): overload_retry_count = 0 # 重置過載計數器 - log_info("[!] 偵測到 JSON 被截斷(結尾不完整或格式錯誤),將縮小 Batch 重試") + log_info( + "[!] 偵測到 JSON 被截斷(結尾不完整或格式錯誤),將縮小 Batch 重試" + ) break # 解析 JSON @@ -471,11 +488,17 @@ def _is_truncated(text: str) -> bool: # ATK-B-2: 翻譯品質驗證 # 1. 空翻譯 if not translated_text or translated_text.strip() == "": - log_warning("[⚠️ 空翻譯] path=%s:原文='%s'", original_item["path"], original_item["text"]) + log_warning( + "[⚠️ 空翻譯] path=%s:原文='%s'", + original_item["path"], + original_item["text"], + ) # 2. 異常長度(翻譯後長度是原文 3 倍以上) orig_len = len(original_item["text"]) if orig_len > 0 and len(translated_text) / orig_len > 3: - log_warning(f"[⚠️ 異常長度] {original_item['path']}:原文 {orig_len} 字,翻譯 {len(translated_text)} 字") + log_warning( + f"[⚠️ 異常長度] {original_item['path']}:原文 {orig_len} 字,翻譯 {len(translated_text)} 字" + ) new_item["text"] = translated_text merged_result.append(new_item) @@ -654,9 +677,7 @@ def _is_truncated(text: str) -> bool: except Exception as parse_err: # 備援比對邏輯 err_msg = str(e).upper() - log_error( - f"[⚠️] 無法解析 429 JSON,使用備援。錯誤: {parse_err}" - ) + log_error(f"[⚠️] 無法解析 429 JSON,使用備援。錯誤: {parse_err}") if "QUOTA" in err_msg or "EXCEEDED" in err_msg: # ⭐ 這裡之前會崩潰,現在這樣改就安全了 @@ -726,7 +747,9 @@ def _is_truncated(text: str) -> bool: log_info( "[✅] API Key 切換成功 → 原地重送同一 batch,等待12秒" ) - time.sleep(key_rotation_buffer_sec) # ⭐ 給新 Key 一點緩衝 + time.sleep( + key_rotation_buffer_sec + ) # ⭐ 給新 Key 一點緩衝 hit_overload_retry = True # ⭐ 重送同一 batch break # ← 跳出 model loop,回 while else: @@ -763,9 +786,7 @@ def _is_truncated(text: str) -> bool: # ======== 500 ========== if status == 500: - log_info( - "[⚠️] 500 INTERNAL:Gemini 後端錯誤,嘗試換模型或縮 batch" - ) + log_info("[⚠️] 500 INTERNAL:Gemini 後端錯誤,嘗試換模型或縮 batch") break # ========== requests timeout ========== @@ -840,7 +861,8 @@ def _is_truncated(text: str) -> bool: # 3. 如果還有剩下的,重置 batch_size if remaining_items: batch_size = min( - len(remaining_items), MIN_BATCH_SIZE if not is_lang else MIN_LANG_BATCH_SIZE + len(remaining_items), + MIN_BATCH_SIZE if not is_lang else MIN_LANG_BATCH_SIZE, ) # 4. 繼續 while 迴圈處理後面的東西 diff --git a/translation_tool/utils/cache_shards.py b/translation_tool/utils/cache_shards.py index 1073128f..55676bdd 100644 --- a/translation_tool/utils/cache_shards.py +++ b/translation_tool/utils/cache_shards.py @@ -12,6 +12,31 @@ import orjson as json + +def _lock_file_fd(lock_fd: int) -> None: + """以跨平台方式鎖定 lock file descriptor。""" + if os.name == "nt": + import msvcrt + + msvcrt.locking(lock_fd, msvcrt.LK_LOCK, 1) + else: + import fcntl + + fcntl.flock(lock_fd, fcntl.LOCK_EX) + + +def _unlock_file_fd(lock_fd: int) -> None: + """以跨平台方式解鎖 lock file descriptor。""" + if os.name == "nt": + import msvcrt + + msvcrt.locking(lock_fd, msvcrt.LK_UNLCK, 1) + else: + import fcntl + + fcntl.flock(lock_fd, fcntl.LOCK_UN) + + def _write_json_atomic(path: Path, data: dict[str, Any]): """以原子方式將 JSON 內容覆寫到 ``path``。 @@ -22,8 +47,6 @@ def _write_json_atomic(path: Path, data: dict[str, Any]): 使用 fsync 確保資料寫入磁碟,避免作業系統緩衝區未 flush 就執行 os.replace() 導致資料遺失。 """ - import msvcrt - tmp_path = path.with_suffix(".tmp") path.parent.mkdir(parents=True, exist_ok=True) @@ -36,6 +59,7 @@ def _write_json_atomic(path: Path, data: dict[str, Any]): os.replace(tmp_path, path) + def _get_active_shard_path( *, type_dir: Path, @@ -63,6 +87,7 @@ def _get_active_shard_path( return type_dir / f"{cache_type}_{shard_id_str}.json" + def _rotate_shard_if_needed( *, type_dir: Path, @@ -86,9 +111,8 @@ def _rotate_shard_if_needed( type_dir.mkdir(parents=True, exist_ok=True) lock_fd = os.open(str(lock_file), os.O_CREAT | os.O_RDWR) try: - # Windows: 使用 msvcrt.locking() 進行檔案鎖定 - import msvcrt - msvcrt.locking(lock_fd, msvcrt.LK_LOCK, 1) + # 以跨平台 file lock 進行檔案鎖定 + _lock_file_fd(lock_fd) # 再次確認容量(防止鎖競爭期間已被其他程序旋轉) if len(data) < rolling_shard_size: @@ -110,10 +134,10 @@ def _rotate_shard_if_needed( return True finally: - import msvcrt - msvcrt.locking(lock_fd, msvcrt.LK_UNLCK, 1) + _unlock_file_fd(lock_fd) os.close(lock_fd) + def _save_entries_to_active_shards( *, type_dir: Path, @@ -131,8 +155,6 @@ def _save_entries_to_active_shards( if not entries: return - import msvcrt - active_file = type_dir / active_shard_file lock_file = type_dir / f"{active_shard_file}.lock" @@ -157,7 +179,7 @@ def _save_entries_to_active_shards( lock_fd = os.open(str(lock_file), os.O_CREAT | os.O_RDWR) rotated = False try: - msvcrt.locking(lock_fd, msvcrt.LK_LOCK, 1) + _lock_file_fd(lock_fd) # 在鎖保護下讀取 active shard path(避免 TOCTOU) save_path = _get_active_shard_path( @@ -180,7 +202,7 @@ def _save_entries_to_active_shards( # 在鎖保護下檢查是否需要旋轉 if len(current_data) >= rolling_shard_size: # 需要旋轉:釋放當前鎖,讓旋轉邏輯取得鎖 - msvcrt.locking(lock_fd, msvcrt.LK_UNLCK, 1) + _unlock_file_fd(lock_fd) os.close(lock_fd) lock_fd = -1 @@ -198,7 +220,7 @@ def _save_entries_to_active_shards( finally: if lock_fd != -1: try: - msvcrt.locking(lock_fd, msvcrt.LK_UNLCK, 1) + _unlock_file_fd(lock_fd) except Exception: pass os.close(lock_fd) From 3475de8c21cdfea84cb45e6931e9ccaa48b98c7a Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Sat, 28 Mar 2026 23:51:15 +0800 Subject: [PATCH 29/33] fix(pr43): apply ruff format for all changed files --- app/services_impl/pipelines/_task_runner.py | 12 +- .../pipelines/extract_service.py | 2 + app/services_impl/pipelines/lm_service.py | 5 +- tests/test_lm_config_rules.py | 225 +++++++++--------- tests/test_lm_response_parser.py | 4 +- tests/test_lm_translator_main.py | 111 ++++----- .../test_pipeline_services_error_handling.py | 26 +- .../checkers/color_char_checker.py | 21 +- .../core/kubejs_translator_clean.py | 4 +- translation_tool/core/lm_api_client.py | 1 + translation_tool/core/lm_response_parser.py | 10 +- .../core/lm_translator_shared_loop.py | 14 +- .../core/lm_translator_shared_recording.py | 1 + .../ftbquests/ftbquests_lmtranslator.py | 7 +- translation_tool/plugins/md/md_extract_qa.py | 28 ++- translation_tool/plugins/md/md_inject_qa.py | 18 +- .../plugins/md/md_lmtranslator.py | 24 +- translation_tool/utils/cache_loader.py | 4 +- translation_tool/utils/cache_manager.py | 30 ++- translation_tool/utils/cache_search.py | 98 +++++--- workspace/patch_md_lmtranslator.py | 46 ++-- 21 files changed, 412 insertions(+), 279 deletions(-) diff --git a/app/services_impl/pipelines/_task_runner.py b/app/services_impl/pipelines/_task_runner.py index 96cb9203..73fb808d 100644 --- a/app/services_impl/pipelines/_task_runner.py +++ b/app/services_impl/pipelines/_task_runner.py @@ -11,7 +11,16 @@ logger = logging.getLogger(__name__) -def run_callable_task(*, session, task_name: str, func: Callable[..., Any], kwargs: dict, add_session_log_on_error: bool = False, ui_log_handler=UI_LOG_HANDLER): + +def run_callable_task( + *, + session, + task_name: str, + func: Callable[..., Any], + kwargs: dict, + add_session_log_on_error: bool = False, + ui_log_handler=UI_LOG_HANDLER, +): """執行可呼叫的流水線任務,並自動處理 Session 狀態切換、日誌紀錄及異常捕獲。""" ensure_pipeline_logging() try: @@ -30,4 +39,3 @@ def run_callable_task(*, session, task_name: str, func: Callable[..., Any], kwar # ⭐ session.finish() 一定會被執行,無論成功或失敗 session.finish() ui_log_handler.set_session(None) - diff --git a/app/services_impl/pipelines/extract_service.py b/app/services_impl/pipelines/extract_service.py index 98c9868b..d344ac42 100644 --- a/app/services_impl/pipelines/extract_service.py +++ b/app/services_impl/pipelines/extract_service.py @@ -21,6 +21,7 @@ logger = logging.getLogger(__name__) + def run_lang_extraction_service(mods_dir: str, output_dir: str, session): """執行語言檔擷取服務。""" ensure_pipeline_logging() @@ -58,6 +59,7 @@ def run_lang_extraction_service(mods_dir: str, output_dir: str, session): # ⭐ 避免 handler 留著舊 session UI_LOG_HANDLER.set_session(None) + def run_book_extraction_service(mods_dir: str, output_dir: str, session): """執行書本檔擷取服務。""" ensure_pipeline_logging() diff --git a/app/services_impl/pipelines/lm_service.py b/app/services_impl/pipelines/lm_service.py index ce8847ba..c848a362 100644 --- a/app/services_impl/pipelines/lm_service.py +++ b/app/services_impl/pipelines/lm_service.py @@ -14,10 +14,13 @@ UI_LOG_HANDLER, ) from app.services_impl.pipelines._pipeline_logging import ensure_pipeline_logging -from translation_tool.core.lm_translator import translate_directory_generator as lm_translate_gen +from translation_tool.core.lm_translator import ( + translate_directory_generator as lm_translate_gen, +) logger = logging.getLogger(__name__) + def run_lm_translation_service( input_dir: str, output_dir: str, diff --git a/tests/test_lm_config_rules.py b/tests/test_lm_config_rules.py index 36bfcc9a..a67f6a73 100644 --- a/tests/test_lm_config_rules.py +++ b/tests/test_lm_config_rules.py @@ -2,6 +2,7 @@ 用途:測試 LM 翻譯配置與規則相關功能。 """ + import pytest from unittest.mock import patch @@ -13,122 +14,133 @@ class TestAPIKeyManagement: def reset_key_tracker(self): """每個測試執行前重置 KeyIndexTracker 狀態,確保測試隔離。""" from translation_tool.core.lm_config_rules import reset_key_index + reset_key_index() # 重置在測試之前 yield reset_key_index() # 重置在測試之後(確保不影響後續測試) - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_get_current_api_key_with_keys(self, mock_load_config): """測試取得當前 API Key(有金鑰時)。""" from translation_tool.core.lm_config_rules import get_current_api_key - + mock_load_config.return_value = { "lm_translator": { - "keys": ["AIzaSyctest123456789012345678901234567890", "AIzaSydumm2222222222222222222222222222222222"] + "keys": [ + "AIzaSyctest123456789012345678901234567890", + "AIzaSydumm2222222222222222222222222222222222", + ] } } - + result = get_current_api_key() - + assert result == "AIzaSyctest123456789012345678901234567890" - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_get_current_api_key_empty(self, mock_load_config): """測試取得當前 API Key(無金鑰時)。""" from translation_tool.core.lm_config_rules import get_current_api_key - + mock_load_config.return_value = {"lm_translator": {"keys": []}} - + result = get_current_api_key() - + assert result == "" - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_rotate_api_key_success(self, mock_load_config): """測試 API Key 輪換(成功)。""" - from translation_tool.core.lm_config_rules import rotate_api_key, reset_key_index, get_current_key_index - + from translation_tool.core.lm_config_rules import ( + rotate_api_key, + reset_key_index, + get_current_key_index, + ) + mock_load_config.return_value = { "lm_translator": { - "keys": ["AIzaSyctest123456789012345678901234567890", "AIzaSydumm2222222222222222222222222222222222"] + "keys": [ + "AIzaSyctest123456789012345678901234567890", + "AIzaSydumm2222222222222222222222222222222222", + ] } } - + # 重置索引 reset_key_index() - + result = rotate_api_key() - + assert result is True assert get_current_key_index() == 1 - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_rotate_api_key_no_more_keys(self, mock_load_config): """測試 API Key 輪換(無更多金鑰)。""" - from translation_tool.core.lm_config_rules import rotate_api_key, reset_key_index - + from translation_tool.core.lm_config_rules import ( + rotate_api_key, + reset_key_index, + ) + mock_load_config.return_value = { - "lm_translator": { - "keys": ["AIzaSyctest123456789012345678901234567890"] - } + "lm_translator": {"keys": ["AIzaSyctest123456789012345678901234567890"]} } - + # 重置索引 reset_key_index() - + result = rotate_api_key() - + assert result is False - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_validate_api_keys_success(self, mock_load_config): """測試 API Key 驗證(成功)。""" from translation_tool.core.lm_config_rules import validate_api_keys - + mock_load_config.return_value = { "lm_translator": { - "keys": ["AIzaSyctest123456789012345678901234567890", "AIzaSydumm2222222222222222222222222222222222"] + "keys": [ + "AIzaSyctest123456789012345678901234567890", + "AIzaSydumm2222222222222222222222222222222222", + ] } } - + # 不應該拋出異常 validate_api_keys() - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_validate_api_keys_empty(self, mock_load_config): """測試 API Key 驗證(無金鑰)。""" from translation_tool.core.lm_config_rules import validate_api_keys - + mock_load_config.return_value = {"lm_translator": {"keys": []}} - + with pytest.raises(RuntimeError, match="沒有找到任何 API Key"): validate_api_keys() - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_validate_api_keys_invalid_format(self, mock_load_config): """測試 API Key 驗證(無效格式)。""" from translation_tool.core.lm_config_rules import validate_api_keys - - mock_load_config.return_value = { - "lm_translator": { - "keys": ["InvalidKey123"] - } - } - + + mock_load_config.return_value = {"lm_translator": {"keys": ["InvalidKey123"]}} + with pytest.raises(RuntimeError, match="無效的 API Key 格式"): validate_api_keys() def test_validate_api_keys_from_ui_success(self): """測試 UI API Key 驗證(成功)。""" from translation_tool.core.lm_config_rules import validate_api_keys_from_ui - + # 不應該拋出異常 validate_api_keys_from_ui(["AIzaSyctest123456789012345678901234567890"]) def test_validate_api_keys_from_ui_invalid(self): """測試 UI API Key 驗證(無效)。""" from translation_tool.core.lm_config_rules import validate_api_keys_from_ui - + with pytest.raises(RuntimeError, match="無效的 API Key 格式"): validate_api_keys_from_ui(["InvalidKey"]) @@ -139,37 +151,37 @@ class TestCJKDetection: def test_contains_cjk_chinese(self): """測試包含中文。""" from translation_tool.core.lm_config_rules import contains_cjk - + assert contains_cjk("你好世界") is True def test_contains_cjk_japanese(self): """測試包含日文。""" from translation_tool.core.lm_config_rules import contains_cjk - + assert contains_cjk("こんにちは") is True def test_contains_cjk_korean(self): """測試包含韓文。""" from translation_tool.core.lm_config_rules import contains_cjk - + assert contains_cjk("안녕하세요") is True def test_contains_cjk_english_only(self): """測試只有英文。""" from translation_tool.core.lm_config_rules import contains_cjk - + assert contains_cjk("Hello World") is False def test_contains_cjk_empty_string(self): """測試空字串。""" from translation_tool.core.lm_config_rules import contains_cjk - + assert contains_cjk("") is False def test_contains_cjk_none_input(self): """測試 None 輸入。""" from translation_tool.core.lm_config_rules import contains_cjk - + assert contains_cjk(None) is False @@ -179,43 +191,43 @@ class TestNeedsTranslationText: def test_needs_translation_empty(self): """測試空字串。""" from translation_tool.core.lm_config_rules import needs_translation_text - + assert needs_translation_text("") is False def test_needs_translation_none(self): """測試 None 輸入。""" from translation_tool.core.lm_config_rules import needs_translation_text - + assert needs_translation_text(None) is False def test_needs_translation_chinese(self): """測試中文(不需要翻譯)。""" from translation_tool.core.lm_config_rules import needs_translation_text - + assert needs_translation_text("你好") is False def test_needs_translation_english(self): """測試英文(需要翻譯)。""" from translation_tool.core.lm_config_rules import needs_translation_text - + assert needs_translation_text("Hello") is True def test_needs_translation_digit(self): """測試純數字。""" from translation_tool.core.lm_config_rules import needs_translation_text - + assert needs_translation_text("123") is False def test_needs_translation_section_symbol(self): """測試章節符號開頭。""" from translation_tool.core.lm_config_rules import needs_translation_text - + assert needs_translation_text("§lBold Text") is False def test_needs_translation_token(self): """測試 token 格式。""" from translation_tool.core.lm_config_rules import needs_translation_text - + assert needs_translation_text("$(some.token)") is False @@ -225,14 +237,14 @@ class TestValueFullyTranslated: def test_value_fully_translated_string(self): """測試字串翻譯狀態。""" from translation_tool.core.lm_config_rules import value_fully_translated - + assert value_fully_translated("已翻譯的文字") is True assert value_fully_translated("") is False def test_value_fully_translated_list(self): """測試列表翻譯狀態。""" from translation_tool.core.lm_config_rules import value_fully_translated - + # 全部非空視為已翻譯 assert value_fully_translated(["item1", "item2"]) is True # 包含空字串 @@ -241,7 +253,7 @@ def test_value_fully_translated_list(self): def test_value_fully_translated_other_types(self): """測試其他類型。""" from translation_tool.core.lm_config_rules import value_fully_translated - + assert value_fully_translated(123) is True assert value_fully_translated({"key": "value"}) is True assert value_fully_translated(None) is True @@ -253,9 +265,9 @@ class TestBuildSkipTermsPattern: def test_build_skip_terms_pattern_single(self): """測試單一術語。""" from translation_tool.core.lm_config_rules import build_skip_terms_pattern - + pattern = build_skip_terms_pattern(["discord"]) - + assert pattern is not None assert pattern.search("discord") is not None assert pattern.search("DISCORD") is not None @@ -263,9 +275,9 @@ def test_build_skip_terms_pattern_single(self): def test_build_skip_terms_pattern_multiple(self): """測試多個術語。""" from translation_tool.core.lm_config_rules import build_skip_terms_pattern - + pattern = build_skip_terms_pattern(["api", "discord", "github"]) - + assert pattern.search("api") is not None assert pattern.search("discord") is not None assert pattern.search("github") is not None @@ -273,144 +285,123 @@ def test_build_skip_terms_pattern_multiple(self): def test_build_skip_terms_pattern_escape(self): """測試特殊字元轉義。""" from translation_tool.core.lm_config_rules import build_skip_terms_pattern - + pattern = build_skip_terms_pattern(["test.key"]) - + assert pattern.search("test.key") is not None class TestIsValueTranslatable: """測試值是否可翻譯的判斷。""" - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_is_value_translatable_lang_true(self, mock_load_config): """測試 lang 值可翻譯。""" from translation_tool.core.lm_config_rules import is_value_translatable - + mock_load_config.return_value = { "lm_translator": { "translator": { "translatable_keywords": ["text", "name"], - "skip_terms": [] + "skip_terms": [], } } } - + assert is_value_translatable("Hello World", is_lang=True) is True - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_is_value_translatable_cjk(self, mock_load_config): """測試含 CJK 的值不可翻譯。""" from translation_tool.core.lm_config_rules import is_value_translatable - + mock_load_config.return_value = { "lm_translator": { - "translator": { - "translatable_keywords": ["text"], - "skip_terms": [] - } + "translator": {"translatable_keywords": ["text"], "skip_terms": []} } } - + assert is_value_translatable("你好", is_lang=True) is False - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_is_value_translatable_tech_pattern(self, mock_load_config): """測試技術模式不可翻譯。""" from translation_tool.core.lm_config_rules import is_value_translatable - + mock_load_config.return_value = { "lm_translator": { - "translator": { - "translatable_keywords": ["text"], - "skip_terms": [] - } + "translator": {"translatable_keywords": ["text"], "skip_terms": []} } } - + # minecraft:xxx 格式 assert is_value_translatable("minecraft:diamond", is_lang=True) is False - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_is_value_translatable_empty(self, mock_load_config): """測試空值不可翻譯。""" from translation_tool.core.lm_config_rules import is_value_translatable - + mock_load_config.return_value = { "lm_translator": { - "translator": { - "translatable_keywords": ["text"], - "skip_terms": [] - } + "translator": {"translatable_keywords": ["text"], "skip_terms": []} } } - + assert is_value_translatable("", is_lang=True) is False - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_is_value_translatable_roman_numeral(self, mock_load_config): """測試羅馬數字不可翻譯。""" from translation_tool.core.lm_config_rules import is_value_translatable - + mock_load_config.return_value = { "lm_translator": { - "translator": { - "translatable_keywords": ["text"], - "skip_terms": [] - } + "translator": {"translatable_keywords": ["text"], "skip_terms": []} } } - + assert is_value_translatable("III", is_lang=True) is False - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_is_value_translatable_digit(self, mock_load_config): """測試純數字不可翻譯。""" from translation_tool.core.lm_config_rules import is_value_translatable - + mock_load_config.return_value = { "lm_translator": { - "translator": { - "translatable_keywords": ["text"], - "skip_terms": [] - } + "translator": {"translatable_keywords": ["text"], "skip_terms": []} } } - + assert is_value_translatable("123", is_lang=True) is False class TestIsTranslatableField: """測試欄位是否可翻譯的判斷。""" - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_is_translatable_field_true(self, mock_load_config): """測試可翻譯欄位。""" from translation_tool.core.lm_config_rules import is_translatable_field - + mock_load_config.return_value = { "lm_translator": { - "translator": { - "translatable_keywords": ["text", "name", "description"] - } + "translator": {"translatable_keywords": ["text", "name", "description"]} } } - + assert is_translatable_field("item_text") is True assert is_translatable_field("display_name") is True - @patch('translation_tool.core.lm_config_rules.load_config') + @patch("translation_tool.core.lm_config_rules.load_config") def test_is_translatable_field_false(self, mock_load_config): """測試不可翻譯欄位。""" from translation_tool.core.lm_config_rules import is_translatable_field - + mock_load_config.return_value = { - "lm_translator": { - "translator": { - "translatable_keywords": ["text", "name"] - } - } + "lm_translator": {"translator": {"translatable_keywords": ["text", "name"]}} } - + assert is_translatable_field("id") is False assert is_translatable_field("damage") is False diff --git a/tests/test_lm_response_parser.py b/tests/test_lm_response_parser.py index d0312c52..ebcacb55 100644 --- a/tests/test_lm_response_parser.py +++ b/tests/test_lm_response_parser.py @@ -2,6 +2,7 @@ 用途:測試 LM 回應解析器相關功能。 """ + import pytest from translation_tool.core.lm_response_parser import ( safe_json_loads, @@ -51,7 +52,7 @@ def test_multiple_json_objects(self): def test_invalid_json_raises_error(self): """測試無效 JSON 拋出錯誤。""" - text = 'this is not valid json at all' + text = "this is not valid json at all" with pytest.raises(RuntimeError): safe_json_loads(text) @@ -197,5 +198,6 @@ def test_exports(self): safe_json_loads, chunked, ) + assert callable(safe_json_loads) assert callable(chunked) diff --git a/tests/test_lm_translator_main.py b/tests/test_lm_translator_main.py index e94a11bc..9aa46267 100644 --- a/tests/test_lm_translator_main.py +++ b/tests/test_lm_translator_main.py @@ -9,11 +9,11 @@ class TestTranslateBatchSmart: """translate_batch_smart 測試""" - @patch('translation_tool.core.lm_translator_main.safe_json_loads') - @patch('translation_tool.core.lm_translator_main.load_config') - @patch('translation_tool.core.lm_translator_main.get_current_api_key') - @patch('translation_tool.core.lm_translator_main.call_gemini_requests') - @patch('translation_tool.core.lm_translator_main.time.sleep') + @patch("translation_tool.core.lm_translator_main.safe_json_loads") + @patch("translation_tool.core.lm_translator_main.load_config") + @patch("translation_tool.core.lm_translator_main.get_current_api_key") + @patch("translation_tool.core.lm_translator_main.call_gemini_requests") + @patch("translation_tool.core.lm_translator_main.time.sleep") def test_translate_batch_smart_lang_success( self, mock_sleep, mock_call_api, mock_get_key, mock_config, mock_json_loads ): @@ -29,16 +29,14 @@ def test_translate_batch_smart_lang_success( "models": {"gemini-pro": {"enabled": True}}, "temperature": 0.2, "lang_system_prompt": "test", - "patchouli_system_prompt": "test" + "patchouli_system_prompt": "test", } } mock_get_key.return_value = "test_key" mock_call_api.return_value = '{"items": [{"id": "0", "value": "你好"}]}' mock_json_loads.return_value = {"items": [{"id": "0", "value": "你好"}]} - items = [ - {"path": "test.key", "text": "Hello", "cache_type": "lang"} - ] + items = [{"path": "test.key", "text": "Hello", "cache_type": "lang"}] result, status = translate_batch_smart(items, 1) @@ -46,11 +44,11 @@ def test_translate_batch_smart_lang_success( # 成功時 API 應該被調用一次 mock_call_api.assert_called_once() - @patch('translation_tool.core.lm_translator_main.safe_json_loads') - @patch('translation_tool.core.lm_translator_main.load_config') - @patch('translation_tool.core.lm_translator_main.get_current_api_key') - @patch('translation_tool.core.lm_translator_main.call_gemini_requests') - @patch('translation_tool.core.lm_translator_main.time.sleep') + @patch("translation_tool.core.lm_translator_main.safe_json_loads") + @patch("translation_tool.core.lm_translator_main.load_config") + @patch("translation_tool.core.lm_translator_main.get_current_api_key") + @patch("translation_tool.core.lm_translator_main.call_gemini_requests") + @patch("translation_tool.core.lm_translator_main.time.sleep") def test_translate_batch_smart_empty_batch( self, mock_sleep, mock_call_api, mock_get_key, mock_config, mock_json_loads ): @@ -71,11 +69,11 @@ def test_translate_batch_smart_empty_batch( assert status == "AUTO" mock_call_api.assert_not_called() - @patch('translation_tool.core.lm_translator_main.safe_json_loads') - @patch('translation_tool.core.lm_translator_main.load_config') - @patch('translation_tool.core.lm_translator_main.get_current_api_key') - @patch('translation_tool.core.lm_translator_main.call_gemini_requests') - @patch('translation_tool.core.lm_translator_main.time.sleep') + @patch("translation_tool.core.lm_translator_main.safe_json_loads") + @patch("translation_tool.core.lm_translator_main.load_config") + @patch("translation_tool.core.lm_translator_main.get_current_api_key") + @patch("translation_tool.core.lm_translator_main.call_gemini_requests") + @patch("translation_tool.core.lm_translator_main.time.sleep") def test_translate_batch_smart_api_error_with_retry( self, mock_sleep, mock_call_api, mock_get_key, mock_config, mock_json_loads ): @@ -107,10 +105,10 @@ class TestSystemPromptConversion: 都會被正確轉為 string 傳入 API。 """ - @patch('translation_tool.core.lm_api_client.requests.post') - @patch('translation_tool.core.lm_translator_main.load_config') - @patch('translation_tool.core.lm_translator_main.get_current_api_key') - @patch('translation_tool.core.lm_translator_main.time.sleep') + @patch("translation_tool.core.lm_api_client.requests.post") + @patch("translation_tool.core.lm_translator_main.load_config") + @patch("translation_tool.core.lm_translator_main.get_current_api_key") + @patch("translation_tool.core.lm_translator_main.time.sleep") def test_lang_prompt_dict_converted_to_string( self, mock_sleep, mock_get_key, mock_config, mock_post ): @@ -121,11 +119,13 @@ def test_lang_prompt_dict_converted_to_string( mock_response = Mock() mock_response.ok = True mock_response.json.return_value = { - "candidates": [{ - "content": { - "parts": [{"text": '{"items": [{"id": "0", "value": "你好"}]}'}] + "candidates": [ + { + "content": { + "parts": [{"text": '{"items": [{"id": "0", "value": "你好"}]}'}] + } } - }] + ] } mock_post.return_value = mock_response @@ -139,8 +139,8 @@ def test_lang_prompt_dict_converted_to_string( "patchouli_system_prompt": "你是專業的 Minecraft Patchouli 翻譯員", "lang_system_prompt": { "role": "translator", - "content": "你正在翻譯 Minecraft 語言檔案" - } + "content": "你正在翻譯 Minecraft 語言檔案", + }, } } mock_get_key.return_value = "test_key" @@ -151,16 +151,17 @@ def test_lang_prompt_dict_converted_to_string( assert mock_post.call_count >= 1, "API 應該被調用至少一次" call_kwargs = mock_post.call_args.kwargs - json_body = call_kwargs.get('json', {}) - system_instruction = json_body.get('systemInstruction', {}) - prompt_text = system_instruction.get('parts', [{}])[0].get('text', '') - assert isinstance(prompt_text, str), \ + json_body = call_kwargs.get("json", {}) + system_instruction = json_body.get("systemInstruction", {}) + prompt_text = system_instruction.get("parts", [{}])[0].get("text", "") + assert isinstance(prompt_text, str), ( "lang_system_prompt 必須是 string,而非 dict" + ) - @patch('translation_tool.core.lm_api_client.requests.post') - @patch('translation_tool.core.lm_translator_main.load_config') - @patch('translation_tool.core.lm_translator_main.get_current_api_key') - @patch('translation_tool.core.lm_translator_main.time.sleep') + @patch("translation_tool.core.lm_api_client.requests.post") + @patch("translation_tool.core.lm_translator_main.load_config") + @patch("translation_tool.core.lm_translator_main.get_current_api_key") + @patch("translation_tool.core.lm_translator_main.time.sleep") def test_prompt_already_string_unchanged( self, mock_sleep, mock_get_key, mock_config, mock_post ): @@ -173,11 +174,13 @@ def test_prompt_already_string_unchanged( mock_response = Mock() mock_response.ok = True mock_response.json.return_value = { - "candidates": [{ - "content": { - "parts": [{"text": '{"items": [{"id": "0", "value": "結果"}]}'}] + "candidates": [ + { + "content": { + "parts": [{"text": '{"items": [{"id": "0", "value": "結果"}]}'}] + } } - }] + ] } mock_post.return_value = mock_response @@ -189,7 +192,7 @@ def test_prompt_already_string_unchanged( "models": {"gemini-pro": {"enabled": True}}, "temperature": 0.2, "patchouli_system_prompt": "另一個 prompt", - "lang_system_prompt": prompt_text + "lang_system_prompt": prompt_text, } } mock_get_key.return_value = "test_key" @@ -200,21 +203,20 @@ def test_prompt_already_string_unchanged( assert mock_post.call_count >= 1, "API 應該被調用至少一次" call_kwargs = mock_post.call_args.kwargs - json_body = call_kwargs.get('json', {}) - system_instruction = json_body.get('systemInstruction', {}) - actual_prompt = system_instruction.get('parts', [{}])[0].get('text', '') - assert actual_prompt == prompt_text, \ - "string 類型的 system_prompt 應保持不變" + json_body = call_kwargs.get("json", {}) + system_instruction = json_body.get("systemInstruction", {}) + actual_prompt = system_instruction.get("parts", [{}])[0].get("text", "") + assert actual_prompt == prompt_text, "string 類型的 system_prompt 應保持不變" class TestBatchProfileDetection: """批次設定偵測測試""" - @patch('translation_tool.core.lm_translator_main.safe_json_loads') - @patch('translation_tool.core.lm_translator_main.load_config') - @patch('translation_tool.core.lm_translator_main.get_current_api_key') - @patch('translation_tool.core.lm_translator_main.call_gemini_requests') - @patch('translation_tool.core.lm_translator_main.time.sleep') + @patch("translation_tool.core.lm_translator_main.safe_json_loads") + @patch("translation_tool.core.lm_translator_main.load_config") + @patch("translation_tool.core.lm_translator_main.get_current_api_key") + @patch("translation_tool.core.lm_translator_main.call_gemini_requests") + @patch("translation_tool.core.lm_translator_main.time.sleep") def test_detect_batch_profile_lang( self, mock_sleep, mock_call_api, mock_get_key, mock_config, mock_json_loads ): @@ -233,7 +235,10 @@ def test_detect_batch_profile_lang( mock_call_api.return_value = '{"items": []}' mock_json_loads.return_value = {"items": []} - items = [{"path": f"key.{i}", "text": f"text{i}", "cache_type": "lang"} for i in range(10)] + items = [ + {"path": f"key.{i}", "text": f"text{i}", "cache_type": "lang"} + for i in range(10) + ] result, status = translate_batch_smart(items, 1) assert status in ["AUTO", "PARTIAL", "FAILED"] diff --git a/tests/test_pipeline_services_error_handling.py b/tests/test_pipeline_services_error_handling.py index 36cecd12..4121e446 100644 --- a/tests/test_pipeline_services_error_handling.py +++ b/tests/test_pipeline_services_error_handling.py @@ -6,38 +6,42 @@ def __init__(self): self.calls = [] def start(self): - self.calls.append('start') + self.calls.append("start") def finish(self): - self.calls.append('finish') + self.calls.append("finish") def set_error(self): - self.calls.append('set_error') + self.calls.append("set_error") def add_log(self, text): - self.calls.append(('add_log', text)) + self.calls.append(("add_log", text)) def test_run_callable_task_sets_error_and_optional_session_log(monkeypatch): session = _Session() seen = [] - monkeypatch.setattr(_task_runner, 'ensure_pipeline_logging', lambda: None) - monkeypatch.setattr(_task_runner.UI_LOG_HANDLER, 'set_session', lambda s: seen.append(s)) + monkeypatch.setattr(_task_runner, "ensure_pipeline_logging", lambda: None) + monkeypatch.setattr( + _task_runner.UI_LOG_HANDLER, "set_session", lambda s: seen.append(s) + ) def boom(**kwargs): - raise RuntimeError('boom') + raise RuntimeError("boom") result = _task_runner.run_callable_task( session=session, - task_name='task', + task_name="task", func=boom, kwargs={}, add_session_log_on_error=True, ) assert result is None - assert session.calls[0] == 'start' - assert session.calls[-2] == 'set_error' # set_error 在倒數第二,finally 的 finish() 在最後 - assert any(isinstance(c, tuple) and c[0] == 'add_log' for c in session.calls) + assert session.calls[0] == "start" + assert ( + session.calls[-2] == "set_error" + ) # set_error 在倒數第二,finally 的 finish() 在最後 + assert any(isinstance(c, tuple) and c[0] == "add_log" for c in session.calls) assert seen == [session, None] diff --git a/translation_tool/checkers/color_char_checker.py b/translation_tool/checkers/color_char_checker.py index 18dfa336..4b0fb837 100644 --- a/translation_tool/checkers/color_char_checker.py +++ b/translation_tool/checkers/color_char_checker.py @@ -19,6 +19,7 @@ @dataclass class ColorCharError: """單一顏色字元錯誤。""" + file_path: str key: str value: str @@ -40,15 +41,17 @@ def check_color_chars(value: str) -> list[ColorCharError] | None: for match in COLOR_PATTERN.finditer(value): illegal_char = match.group(1) pos = match.start() - errors.append(ColorCharError( - file_path="", - key="", - value=value, - illegal_char=illegal_char, - position=pos, - message=f"在位置 {pos} 發現非法顏色字元 '&{illegal_char}'," - f"& 後只能接 a-v(不含 w)、0-9、空格、\\、#。", - )) + errors.append( + ColorCharError( + file_path="", + key="", + value=value, + illegal_char=illegal_char, + position=pos, + message=f"在位置 {pos} 發現非法顏色字元 '&{illegal_char}'," + f"& 後只能接 a-v(不含 w)、0-9、空格、\\、#。", + ) + ) return errors if errors else None diff --git a/translation_tool/core/kubejs_translator_clean.py b/translation_tool/core/kubejs_translator_clean.py index 91f7baef..ca69c4da 100644 --- a/translation_tool/core/kubejs_translator_clean.py +++ b/translation_tool/core/kubejs_translator_clean.py @@ -35,9 +35,7 @@ def _build_reverse_index_impl(final_tw_lookup: dict[str, str]) -> dict[str, str] for k, v in final_tw_lookup.items(): if is_filled_text_impl(v): is_translated = bool( - v.casefold() != k.casefold() - if v.isascii() and k.isascii() - else v != k + v.casefold() != k.casefold() if v.isascii() and k.isascii() else v != k ) rev_candidates.setdefault(v, []).append((k, is_translated)) diff --git a/translation_tool/core/lm_api_client.py b/translation_tool/core/lm_api_client.py index 54458893..7c953506 100644 --- a/translation_tool/core/lm_api_client.py +++ b/translation_tool/core/lm_api_client.py @@ -12,6 +12,7 @@ from translation_tool.utils.config_manager import load_config + def call_gemini_requests( *, model_name: str, diff --git a/translation_tool/core/lm_response_parser.py b/translation_tool/core/lm_response_parser.py index a7de51b9..17959c43 100644 --- a/translation_tool/core/lm_response_parser.py +++ b/translation_tool/core/lm_response_parser.py @@ -9,6 +9,7 @@ import json import re + def safe_json_loads(text: str): """將模型回傳的文字嘗試解析為 JSON,支援去除 Markdown code fence 並從雜訊文字中截取第一個合法 JSON 區塊。""" text = text.strip() @@ -46,18 +47,18 @@ def _extract_json_blocks(text: str): i = 0 n = len(text) while i < n: - if text[i] == '{': + if text[i] == "{": start = i depth = 0 j = i while j < n: c = text[j] - if c == '{' or c == '[': + if c == "{" or c == "[": depth += 1 - elif c == '}' or c == ']': + elif c == "}" or c == "]": depth -= 1 if depth == 0: - blocks.append(text[start:j + 1]) + blocks.append(text[start : j + 1]) i = j + 1 break j += 1 @@ -68,6 +69,7 @@ def _extract_json_blocks(text: str): i += 1 return blocks + def chunked(lst, size): """將序列 lst 依指定大小 size 分塊,yield 每個 chunk(最後一塊可能較短)。""" for i in range(0, len(lst), size): diff --git a/translation_tool/core/lm_translator_shared_loop.py b/translation_tool/core/lm_translator_shared_loop.py index 659466cd..4f390a43 100644 --- a/translation_tool/core/lm_translator_shared_loop.py +++ b/translation_tool/core/lm_translator_shared_loop.py @@ -11,9 +11,17 @@ import time from translation_tool.utils.log_unit import log_info -from translation_tool.utils.cache_manager import add_to_cache, save_translation_cache, reload_translation_cache +from translation_tool.utils.cache_manager import ( + add_to_cache, + save_translation_cache, + reload_translation_cache, +) from translation_tool.utils.config_manager import load_config -from translation_tool.core.lm_translator_shared_cache import CacheRule, get_default_cache_rules +from translation_tool.core.lm_translator_shared_cache import ( + CacheRule, + get_default_cache_rules, +) + @dataclass class TranslateLoopResult: @@ -27,6 +35,7 @@ class TranslateLoopResult: exhausted: bool last_error: Optional[str] = None + def _get_default_batch_size( cache_type: str, batch_size_by_type: Optional[Dict[str, int]] ) -> int: @@ -47,6 +56,7 @@ def _get_default_batch_size( return int(lm_cfg.get("initial_batch_size_md", 100) or 100) return int(lm_cfg.get("initial_batch_size_lang", 300) or 300) + def translate_items_with_cache_loop( items_to_translate: List[Dict[str, Any]], *, diff --git a/translation_tool/core/lm_translator_shared_recording.py b/translation_tool/core/lm_translator_shared_recording.py index f44d24a4..e99a897f 100644 --- a/translation_tool/core/lm_translator_shared_recording.py +++ b/translation_tool/core/lm_translator_shared_recording.py @@ -12,6 +12,7 @@ import csv import json + @dataclass class TranslationRecorder: """收集翻譯紀錄並輸出 JSON/CSV。""" diff --git a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py index a302f5a4..98fee5d1 100644 --- a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py +++ b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py @@ -492,6 +492,7 @@ def on_progress(p: float, msg: str, eta_sec: float) -> None: else: log_info(f"🚀 [AI 翻譯中] {msg}") set_prog(p) + return on_progress def make_on_translated_item(rel_src, dst, out_map, rec, out_dir): @@ -519,6 +520,7 @@ def on_translated_item(it: Dict[str, Any]) -> None: ) except Exception: pass + return on_translated_item def make_on_batch_flushed(file_id, touch, _writer, dst, out_map): @@ -530,6 +532,7 @@ def on_batch_flushed() -> None: except Exception: # fallback write_json_dict(dst, out_map) + return on_batch_flushed # ✅ 確保此檔案在翻譯路徑也有 file_id @@ -537,7 +540,9 @@ def on_batch_flushed() -> None: _file_write_table[file_id] = (dst, out_map) # ✅ Issue #8 修復:使用工廠函式創建 callbacks - on_translated_item = make_on_translated_item(rel_src, dst, out_map, rec, out_dir) + on_translated_item = make_on_translated_item( + rel_src, dst, out_map, rec, out_dir + ) on_batch_flushed = make_on_batch_flushed(file_id, touch, _writer, dst, out_map) on_progress = make_on_progress(set_prog, _fmt_eta) diff --git a/translation_tool/plugins/md/md_extract_qa.py b/translation_tool/plugins/md/md_extract_qa.py index 4360338f..190f258f 100644 --- a/translation_tool/plugins/md/md_extract_qa.py +++ b/translation_tool/plugins/md/md_extract_qa.py @@ -61,10 +61,12 @@ def _shield_item(item: dict) -> dict: # 語言過濾(漢字) RE_CJK = re.compile(r"[\u4e00-\u9fff]") + def contains_cjk(s: str) -> bool: """檢查字串是否包含中日韓文字""" return bool(RE_CJK.search(s)) + def pass_lang_filter(block_text: str, mode: str) -> bool: """ mode: @@ -79,6 +81,7 @@ def pass_lang_filter(block_text: str, mode: str) -> bool: return not has_cjk return True + def normalize_for_dedupe(s: str) -> str: """ 去重用的正規化(保守版): @@ -92,11 +95,13 @@ def normalize_for_dedupe(s: str) -> str: s = re.sub(r"\n{3,}", "\n\n", s).strip() return s + def make_content_hash(text: str) -> str: """產生內容的雜湊值以識別唯一性。""" n = normalize_for_dedupe(text) return hashlib.sha1(n.encode("utf-8")).hexdigest() + @dataclass class BlockItem: """ @@ -109,6 +114,7 @@ class BlockItem: start_line: int # 起始行(1-based) end_line: int # 結束行(1-based) + def is_splitter_line_old(line: str) -> bool: """ 只在「強分隔 token 行」才切段: @@ -122,6 +128,7 @@ def is_splitter_line_old(line: str) -> bool: return True return False + def is_splitter_line(line: str) -> bool: # 原有的強分隔符 """判斷是否為分割線(觸發新段落)。""" @@ -141,6 +148,7 @@ def is_splitter_line(line: str) -> bool: return False + def is_translatable_text_line(line: str) -> bool: """ 判斷「這一行」是否應該送進翻譯。 @@ -174,6 +182,7 @@ def is_translatable_text_line(line: str) -> bool: return True + def normalize_blank_lines(text: str) -> str: """ 將 3 個以上連續空行壓縮成最多 2 個, @@ -182,6 +191,7 @@ def normalize_blank_lines(text: str) -> str: text = re.sub(r"\n{3,}", "\n\n", text) return text.strip("\n") + def extract_blocks(md_text: str, rel_file: str, lang_mode: str) -> List[BlockItem]: """ 從 Markdown 中抽取「可翻譯的文字區塊」 @@ -276,6 +286,7 @@ def flush(end_ln: int): flush(end_ln=len(lines)) return items + def build_pending_json( rel_md: str, abs_md: Path, items: List[BlockItem], lang_mode: str ) -> dict: @@ -286,10 +297,7 @@ def build_pending_json( "source_md": rel_md.replace("\\", "/"), "source_abs": str(abs_md), "lang_filter_mode": lang_mode, - "items": [ - _shield_item(asdict(it)) - for it in items - ], + "items": [_shield_item(asdict(it)) for it in items], "stats": { "blocks": len(items), }, @@ -300,13 +308,16 @@ def build_pending_json( ], } + # 語言資料夾段落(en_us / zh_tw,允許 _en_us / _zh_tw,大小寫不拘) RE_LANG_SEG = re.compile(r"^_?(en_us|zh_cn|zh_tw)$", re.IGNORECASE) + def has_allowed_lang_segment(path: Path) -> bool: # 用 parts 掃描每個 segment,支援 structure/en_us 這種深層結構,且大小寫不拘 return any(RE_LANG_SEG.match(seg) for seg in path.parts) + def detect_lang_segment(parts: List[str]) -> Optional[str]: """ 從路徑 segments 判斷語言資料夾(支援 _en_us/_zh_tw、大小寫) @@ -319,6 +330,7 @@ def detect_lang_segment(parts: List[str]) -> Optional[str]: return seg.lstrip("_").lower() return None + def map_rel_lang_path(rel_path: str, src_lang: str, dst_lang: str) -> str: """ 只替換「剛好是語言資料夾」的 segment: @@ -342,6 +354,7 @@ def map_rel_lang_path(rel_path: str, src_lang: str, dst_lang: str) -> str: return "/".join(parts) + def iter_md_files(root: Path): """ 遞迴列出所有 .md(大小寫不敏感),排除 README.md(不分大小寫、任何層級) @@ -361,12 +374,12 @@ def iter_md_files(root: Path): yield p -def safe_relpath(path: Path, root: Path) -> str: - """ - """ +def safe_relpath(path: Path, root: Path) -> str: + """ """ return path.relative_to(root).as_posix() + def main(): """Markdown 抽取工具主入口。""" print("=== Markdown .md 抽取(段落/區塊)問答式 ===") @@ -532,5 +545,6 @@ def main(): print(f"manifest:{manifest_path}") + if __name__ == "__main__": main() diff --git a/translation_tool/plugins/md/md_inject_qa.py b/translation_tool/plugins/md/md_inject_qa.py index 7179e990..fa5b3cc6 100644 --- a/translation_tool/plugins/md/md_inject_qa.py +++ b/translation_tool/plugins/md/md_inject_qa.py @@ -60,6 +60,7 @@ # ======== 語言資料夾段落映射:en_us -> zh_tw(支援 _en_us、大小寫) ======== RE_LANG_SEG = re.compile(r"^(_?)([a-z]{2}_[a-z]{2})$", re.IGNORECASE) + def map_lang_in_rel_path( rel_path: str, src_lang: str = "en_us", dst_lang: str = "zh_tw" ) -> str: @@ -83,6 +84,7 @@ def map_lang_in_rel_path( return "/".join(parts) + # ======== 語言資料夾段落映射:允許 en_us -> zh_tw,也允許來源是 zh_tw ======== def map_lang_in_rel_path_allow_zh( rel_path: str, src_lang: str = "en_us", dst_lang: str = "zh_tw" @@ -130,9 +132,11 @@ def map_lang_in_rel_path_allow_zh( return mapped, "SRC_ZH" return mapped, "OTHER_LANG" + # 若一整行幾乎都是 §token,也視為 token 行(避免誤判) RE_MOSTLY_TOKEN_LINE = re.compile(r"^\s*(§[0-9a-zA-Z]+\S*)\s*(§[0-9a-zA-Z]+\S*)*\s*$") + def is_token_line(line: str) -> bool: """判斷行是否為 Minecraft 格式 token 行""" s = line.strip() @@ -146,6 +150,7 @@ def is_token_line(line: str) -> bool: return True return False + def is_text_line_old(line: str) -> bool: """判斷是否為文字行(舊版)。 判斷「原始 md」中的某一行是否視為可翻文字行: @@ -154,6 +159,7 @@ def is_text_line_old(line: str) -> bool: """ return bool(line.strip()) and (not is_token_line(line)) + def is_text_line(line: str) -> bool: """ 判斷「原始 md」中的某一行是否視為可回寫的文字行: @@ -178,6 +184,7 @@ def is_text_line(line: str) -> bool: return True + def flatten_for_md(text: str) -> str: """ 把 JSON 內「為了可讀性」可能新增的一堆換行,還原成段落級結構: @@ -217,6 +224,7 @@ def flush_buf(): return "\n".join(out) + @dataclass class Item: """Item 類別。 @@ -235,6 +243,7 @@ def __post_init__(self): if self._shields is None: self._shields = [] + def load_items_from_json(json_path: Path) -> Tuple[str, List[Item]]: """ 讀一個 md_pending_blocks_v1 JSON,回傳 source_md 與 items @@ -254,6 +263,7 @@ def load_items_from_json(json_path: Path) -> Tuple[str, List[Item]]: ) return source_md, items + def apply_item_to_md_lines_old(md_lines: List[str], item: Item) -> None: """ 在 md_lines 上原地套用一個 item(只替換文字行,token/空行不動) @@ -297,6 +307,7 @@ def apply_item_to_md_lines_old(md_lines: List[str], item: Item) -> None: # 之後你若想更進階:可把多的插到第一個空行前,但要小心 diff。 # 目前先穩定為主。 + def apply_item_to_md_lines(md_lines: List[str], item: Item) -> None: """ 在 md_lines 上原地套用一個 item(只替換文字行,token/空行不動) @@ -353,14 +364,14 @@ def apply_item_to_md_lines(md_lines: List[str], item: Item) -> None: # 6) 若翻譯行數多於原文字行數:多的先不插入(避免破壞原 md 結構) # 之後若要更進階,可考慮「在該區塊最後一個文字行後插入」,但要非常小心 token/排版。 -def iter_json_files(root: Path): - """處理此 generator 並逐步回報進度(yield update dict)。 - """ +def iter_json_files(root: Path): + """處理此 generator 並逐步回報進度(yield update dict)。""" for p in root.rglob("*.json"): if p.is_file(): yield p + def main(): """Markdown 寫回工具主入口。""" print("=== md 寫回(方案 A:保留原檔骨架,只替換文字行)===") @@ -459,5 +470,6 @@ def main(): print(f"成功寫出:{wrote}") print(f"略過:{skipped}") + if __name__ == "__main__": main() diff --git a/translation_tool/plugins/md/md_lmtranslator.py b/translation_tool/plugins/md/md_lmtranslator.py index 5a1d1f37..d0c05cd8 100644 --- a/translation_tool/plugins/md/md_lmtranslator.py +++ b/translation_tool/plugins/md/md_lmtranslator.py @@ -44,17 +44,20 @@ # basic io # ------------------------- + def read_json(path: Path) -> Dict[str, Any]: """讀取 JSON 檔案並回傳字典。""" with path.open("r", encoding="utf-8") as f: return json.load(f) + def write_json(path: Path, data: Dict[str, Any]) -> None: """寫入 JSON 檔案。""" path.parent.mkdir(parents=True, exist_ok=True) with path.open("w", encoding="utf-8") as f: json.dump(data, f, ensure_ascii=False, indent=2) + def collect_pending_json_files(pending_root: Path) -> List[Path]: """收集所有待翻譯的 JSON 檔案路徑。""" files = sorted(pending_root.rglob("*.json")) @@ -62,6 +65,7 @@ def collect_pending_json_files(pending_root: Path) -> List[Path]: files = [p for p in files if p.name.lower() != "_manifest.json"] return files + # ------------------------- # zh detection(避免已中文又送) # ------------------------- @@ -70,6 +74,7 @@ def collect_pending_json_files(pending_root: Path) -> List[Path]: # pending model # ------------------------- + @dataclass class PendingItem: """PendingItem 類別。 @@ -84,6 +89,7 @@ class PendingItem: start_line: int end_line: int + def load_pending_doc(path: Path) -> Tuple[Dict[str, Any], List[PendingItem]]: """載入待翻譯的 Markdown 文件。""" data = read_json(path) @@ -103,6 +109,7 @@ def load_pending_doc(path: Path) -> Tuple[Dict[str, Any], List[PendingItem]]: ) return data, items + def compute_out_json_path( src_json: Path, in_pending_root: Path, out_root: Path ) -> Path: @@ -110,6 +117,7 @@ def compute_out_json_path( rel = src_json.relative_to(in_pending_root) return out_root / "LM翻譯後" / rel + def translate_md_pending( *, pending_dir: str | Path, @@ -118,9 +126,7 @@ def translate_md_pending( dry_run: bool = False, session=None, ) -> Dict[str, Any]: - """ - - """ + """ """ validate_api_keys() start_time = time.perf_counter() @@ -252,7 +258,10 @@ def translate_md_pending( # A) 待翻譯 preview(list) p1 = write_dry_run_preview( out_root, - [{k: v for k, v in it.items() if k != "_shielded"} for it in items_to_translate], + [ + {k: v for k, v in it.items() if k != "_shielded"} + for it in items_to_translate + ], meta=meta, filename="_md_dry_run_preview.json", ) @@ -260,7 +269,10 @@ def translate_md_pending( # B) cache hit preview(list) p2 = write_cache_hit_preview( out_root, - [{k: v for k, v in it.items() if k != "_shielded"} for it in cached_items], + [ + {k: v for k, v in it.items() if k != "_shielded"} + for it in cached_items + ], meta=meta, filename="_md_dry_run_cache_hit_preview.json", ) @@ -481,6 +493,7 @@ def on_progress(p: float, msg: str, eta_sec: float) -> None: "out_dir": str(out_root), } + def main(): """MD 翻譯工具主入口。""" log_info("=== MD Pending Blocks -> LM 翻譯(md cache 全接 + content_hash 去重)===") @@ -509,5 +522,6 @@ def main(): for k, v in res.items(): log_info("%s: %s", k, v) + if __name__ == "__main__": main() diff --git a/translation_tool/utils/cache_loader.py b/translation_tool/utils/cache_loader.py index 88641404..880c3ca6 100644 --- a/translation_tool/utils/cache_loader.py +++ b/translation_tool/utils/cache_loader.py @@ -15,9 +15,10 @@ logger = logging.getLogger(__name__) + def load_shard_file(path: Path) -> dict[str, Any]: """載入並解析單一分片(Shard)的 JSON 檔案,將其轉換為記憶體中的快取物件。 - + 若 shard 檔案為空(0 bytes),會記錄警告並回傳空 dict。 """ try: @@ -31,6 +32,7 @@ def load_shard_file(path: Path) -> dict[str, Any]: logger.warning(f"載入分片失敗 {path}: {e}") return {} + def load_cache_type( cache_type: str, *, diff --git a/translation_tool/utils/cache_manager.py b/translation_tool/utils/cache_manager.py index 53077c7c..acbedc1d 100644 --- a/translation_tool/utils/cache_manager.py +++ b/translation_tool/utils/cache_manager.py @@ -54,16 +54,19 @@ "find_similar_translations", ] + def _state(): """取得或建立快取執行期狀態實例""" return cache_store.ensure_runtime_maps(CACHE_TYPES) + def _get_cache_root() -> Path: """從設定取得快取根目錄路徑""" translation_config = load_config().get("translator", {}) cache_dir_name = translation_config.get("cache_directory", _CACHE_DIR_NAME) return resolve_project_path(cache_dir_name) + def _load_cache_type(cache_type: str): """載入指定類型的快取。""" state = _state() @@ -77,6 +80,7 @@ def _load_cache_type(cache_type: str): logger=log, ) + def initialize_translation_cache(): """初始化翻譯快取系統。""" state = _state() @@ -90,10 +94,12 @@ def initialize_translation_cache(): except Exception as e: log.error(f"快取系統初始化失敗: {e}", exc_info=True) + def is_cache_initialized() -> bool: """檢查快取是否已初始化。""" return bool(_state().initialized) + def reload_translation_cache(): """重新載入翻譯快取。""" state = cache_store.get_runtime_state() @@ -109,11 +115,14 @@ def reload_translation_cache(): translation_cache=state.translation_cache, cache_file_path=state.cache_file_path, cache_root=_get_cache_root(), - parallel_workers=translation_config.get("parallel_execution_workers", 4), + parallel_workers=translation_config.get( + "parallel_execution_workers", 4 + ), logger=log, ) state.initialized = True + def reload_translation_cache_type(cache_type: str): """重新載入指定類型的翻譯快取。""" if cache_type not in CACHE_TYPES: @@ -126,6 +135,7 @@ def reload_translation_cache_type(cache_type: str): cache_store.clear_dirty(state.is_dirty, cache_type) _load_cache_type(cache_type) + def _save_entries_to_active_shards( cache_type: str, entries: dict, force_new_shard: bool = False ): @@ -142,6 +152,7 @@ def _save_entries_to_active_shards( logger=log, ) + def save_translation_cache(cache_type: str, write_new_shard: bool = True): """儲存翻譯快取。""" if not load_config().get("translator", {}).get("enable_cache_saving", True): @@ -169,6 +180,7 @@ def save_translation_cache(cache_type: str, write_new_shard: bool = True): except Exception as e: log.error(f"❌ 儲存 {cache_type} 失敗: {e}", exc_info=True) + def _get_active_shard_path(cache_type: str) -> Path: """取得目前作用中的分片檔案路徑""" state = _state() @@ -179,6 +191,7 @@ def _get_active_shard_path(cache_type: str) -> Path: active_shard_file=ACTIVE_SHARD_FILE, ) + def add_to_cache( cache_type: str, key: str, @@ -217,7 +230,7 @@ def add_to_cache_batch( paths: Optional[List[Optional[str]]] = None, ): """批次新增翻譯到快取(單次鎖獲取,減少鎖競爭)。 - + Args: cache_type: 快取類型 (lang, patchouli, ftbquests, kubejs, md) entries: List of (key, src, dst) tuples @@ -262,6 +275,7 @@ def get_from_cache(cache_type: str, key: str) -> Optional[str]: return None return cache_store.get_value(cache, key) + def get_cache_entry(cache_type: str, key: str) -> Optional[Dict[str, Any]]: """取得指定 key 的完整快取項目(包含 src、dst、mod、path)。""" state = _state() @@ -272,6 +286,7 @@ def get_cache_entry(cache_type: str, key: str) -> Optional[Dict[str, Any]]: return None return cache_store.get_entry(cache, key) + def get_cache_dict_ref(cache_type: str) -> Dict[str, Dict[str, Any]]: """取得指定類型的快取字典參照。""" state = _state() @@ -280,6 +295,7 @@ def get_cache_dict_ref(cache_type: str) -> Dict[str, Dict[str, Any]]: cache = state.translation_cache.get(cache_type) return cache if isinstance(cache, dict) else {} + def get_session_new_count(cache_type: str) -> int: """取得本次 session 新增的項目數""" state = _state() @@ -288,6 +304,7 @@ def get_session_new_count(cache_type: str) -> int: cache_store.get_session_entries(state.session_new_entries, cache_type) ) + def get_active_shard_id(cache_type: str) -> str: """取得指定快取類型的目前作用中分片 ID""" state = _state() @@ -295,6 +312,7 @@ def get_active_shard_id(cache_type: str) -> str: state.cache_file_path, cache_type, ACTIVE_SHARD_FILE ) + def get_cache_overview() -> Dict[str, Any]: """取得所有快取類型的概覽(包含項目數與狀態)""" initialize_translation_cache() @@ -314,6 +332,7 @@ def get_cache_overview() -> Dict[str, Any]: resolve_project_path=resolve_project_path, ) + def force_rotate_shard(cache_type: str) -> bool: """強制輪轉至下一個分片。""" initialize_translation_cache() @@ -332,6 +351,7 @@ def force_rotate_shard(cache_type: str) -> bool: except Exception: return False + def _get_search_facade() -> CacheSearchFacade: """取得或建立快取搜尋外觀(惰性初始化)""" global _search_facade @@ -341,10 +361,12 @@ def _get_search_facade() -> CacheSearchFacade: _search_facade = CacheSearchFacade(_get_cache_root, log) return _search_facade + def get_search_engine(): """取得快取查詢用的搜尋引擎實例""" return _get_search_facade().get_search_engine() + def rebuild_search_index(): """重建所有快取類型的搜尋索引。""" state = _state() @@ -352,6 +374,7 @@ def rebuild_search_index(): CACHE_TYPES, state.translation_cache ) + def rebuild_search_index_for_type(cache_type: str): """重建指定快取類型的搜尋索引""" state = _state() @@ -359,6 +382,7 @@ def rebuild_search_index_for_type(cache_type: str): cache_type, CACHE_TYPES, state.translation_cache ) + def search_cache( query: str, cache_type: str = None, limit: int = 50, use_fuzzy: bool = True ) -> list: @@ -367,6 +391,7 @@ def search_cache( query=query, cache_type=cache_type, limit=limit, use_fuzzy=use_fuzzy ) + def find_similar_translations( text: str, cache_type: str = None, threshold: float = 0.6, limit: int = 20 ) -> list: @@ -375,6 +400,7 @@ def find_similar_translations( text=text, cache_type=cache_type, threshold=threshold, limit=limit ) + initialize_translation_cache() _state_obj = _state() log.info( diff --git a/translation_tool/utils/cache_search.py b/translation_tool/utils/cache_search.py index 5d664976..8b0e8866 100644 --- a/translation_tool/utils/cache_search.py +++ b/translation_tool/utils/cache_search.py @@ -32,6 +32,7 @@ # 全文搜尋引擎 # ============================================================================= + class CacheSearchEngine: """快取全文搜尋引擎(使用 SQLite FTS5)""" @@ -53,14 +54,16 @@ def __init__(self, db_path: str | None = None): self.conn = sqlite3.connect(db_path, check_same_thread=False) self.conn.row_factory = sqlite3.Row # 讓結果可以用欄位名稱存取 self._lock = threading.RLock() - + # SQLite 效能優化(大幅提升大量寫入速度) with self._lock: self.conn.execute("PRAGMA journal_mode = WAL") - self.conn.execute("PRAGMA synchronous = NORMAL") # ATK-011: OFF 會導致 daemon crash 後 WAL recovery corruption + self.conn.execute( + "PRAGMA synchronous = NORMAL" + ) # ATK-011: OFF 會導致 daemon crash 後 WAL recovery corruption self.conn.execute("PRAGMA cache_size = 10000") self.conn.execute("PRAGMA temp_store = MEMORY") - + self._init_fts_table() def _init_fts_table(self): @@ -188,9 +191,9 @@ def index_batch(self, entries: List[dict], batch_size: int = 20000): """ if not entries: return - + t0 = time.time() - + data = [ ( e.get("key", ""), @@ -202,7 +205,7 @@ def index_batch(self, entries: List[dict], batch_size: int = 20000): ) for e in entries ] - + prepare_time = time.time() - t0 with self._lock: @@ -211,10 +214,10 @@ def index_batch(self, entries: List[dict], batch_size: int = 20000): self.conn.execute("PRAGMA recursive_triggers = OFF") except Exception as e: log_debug(f"設定 PRAGMA recursive_triggers 失敗: {e}") - + write_start = time.time() for i in range(0, len(data), batch_size): - batch = data[i:i + batch_size] + batch = data[i : i + batch_size] self.conn.executemany( """ INSERT INTO cache_fts (cache_key, source_text, translated_text, mod_name, file_path, cache_type) @@ -225,7 +228,7 @@ def index_batch(self, entries: List[dict], batch_size: int = 20000): except sqlite3.OperationalError: write_start = time.time() for i in range(0, len(data), batch_size): - batch = data[i:i + batch_size] + batch = data[i : i + batch_size] self.conn.executemany( """ INSERT INTO cache_basic (cache_key, source_text, translated_text, mod_name, file_path, cache_type) @@ -236,11 +239,15 @@ def index_batch(self, entries: List[dict], batch_size: int = 20000): write_time = time.time() - write_start self.conn.commit() - + total_time = time.time() - t0 - log_debug(f"index_batch: {len(entries)} entries, prepare={prepare_time:.2f}s, write={write_time:.2f}s, total={total_time:.2f}s") + log_debug( + f"index_batch: {len(entries)} entries, prepare={prepare_time:.2f}s, write={write_time:.2f}s, total={total_time:.2f}s" + ) - def search(self, query: str, limit: int = 50, cache_type: str | None = None) -> List[Dict]: + def search( + self, query: str, limit: int = 50, cache_type: str | None = None + ) -> List[Dict]: """搜尋快取(支援中英文、模糊比對) Args: @@ -368,10 +375,12 @@ def __exit__(self, exc_type, exc_val, exc_tb): """離開上下文時關閉搜尋引擎。""" self.close() + # ============================================================================= # 模糊比對器 # ============================================================================= + class FuzzyMatcher: """模糊比對器(相似度計算)""" @@ -454,10 +463,12 @@ def rank_results( # 按綜合分數排序 return sorted(scored, key=lambda x: x["combined_score"], reverse=True) + # ============================================================================= # 便利函式 # ============================================================================= + def search_cache( query: str, db_path: str = None, @@ -475,10 +486,12 @@ def search_cache( return results + # ============================================================================= # 搜尋協調輔助函式(PR12) # ============================================================================= + def _extract_path_from_composite_key(key: str, src: str = "") -> str: """從複合 key 拆出路徑段。 @@ -493,6 +506,7 @@ def _extract_path_from_composite_key(key: str, src: str = "") -> str: return key.split("|", 1)[0] return key + def _infer_search_path(cache_type: str, key: str, entry: Dict[str, Any] | None) -> str: """推導索引要寫入的 path 欄位。 @@ -513,6 +527,7 @@ def _infer_search_path(cache_type: str, key: str, entry: Dict[str, Any] | None) return _extract_path_from_composite_key(key, src) + def _infer_search_mod( cache_type: str, key: str, path: str, entry: Dict[str, Any] | None ) -> str: @@ -546,6 +561,7 @@ def _infer_search_mod( } return fallback.get(cache_type, "") + def _build_search_metadata( cache_type: str, key: str, entry: Dict[str, Any] | None ) -> Dict[str, str]: @@ -554,16 +570,19 @@ def _build_search_metadata( mod = _infer_search_mod(cache_type, key, path, entry) return {"mod": mod, "path": path} + def build_index_entries( cache_type: str, cache_dict: Dict[str, Any] ) -> List[Dict[str, Any]]: """把單一 cache_type 的記憶體字典轉成可批次索引的條目陣列(並行版本)。""" t0 = time.time() - items = [(key, entry) for key, entry in cache_dict.items() if isinstance(entry, dict)] - + items = [ + (key, entry) for key, entry in cache_dict.items() if isinstance(entry, dict) + ] + if not items: return [] - + # 使用多執行緒並行處理 metadata 建立 with ThreadPoolExecutor(max_workers=4) as executor: futures = { @@ -577,13 +596,18 @@ def build_index_entries( results[idx] = future.result() except Exception as e: log_debug(f"建構索引條目失敗: {e}") - + elapsed = time.time() - t0 - log_debug(f"build_index_entries({cache_type}): {len(items)} entries in {elapsed:.2f}s") - + log_debug( + f"build_index_entries({cache_type}): {len(items)} entries in {elapsed:.2f}s" + ) + return [r for r in results if r is not None] -def _build_single_entry(cache_type: str, key: str, entry: Dict[str, Any]) -> Optional[Dict[str, Any]]: + +def _build_single_entry( + cache_type: str, key: str, entry: Dict[str, Any] +) -> Optional[Dict[str, Any]]: """建立單筆索引條目(供並行呼叫)。""" return { "key": key, @@ -593,6 +617,7 @@ def _build_single_entry(cache_type: str, key: str, entry: Dict[str, Any]) -> Opt **_build_search_metadata(cache_type, key, entry), } + def rebuild_from_cache_dicts( engine: CacheSearchEngine, cache_types: List[str], @@ -600,12 +625,16 @@ def rebuild_from_cache_dicts( ) -> int: """依序重建多個類型的索引,回傳實際索引筆數(並行版本)。""" total_indexed = 0 - + # 先並行處理所有 cache_type 的 entries 建立 all_entries: Dict[str, List[Dict[str, Any]]] = {} with ThreadPoolExecutor(max_workers=len(cache_types)) as executor: futures = { - executor.submit(build_index_entries, cache_type, cache_store.get_cache_type_dict(cache_state, cache_type)): cache_type + executor.submit( + build_index_entries, + cache_type, + cache_store.get_cache_type_dict(cache_state, cache_type), + ): cache_type for cache_type in cache_types } for future in as_completed(futures): @@ -613,16 +642,17 @@ def rebuild_from_cache_dicts( entries = future.result() if entries: all_entries[cache_type] = entries - + # 再依序寫入 SQLite(保持原有寫入邏輯) for cache_type in cache_types: entries = all_entries.get(cache_type, []) if entries: engine.index_batch(entries) total_indexed += len(entries) - + return total_indexed + class SearchOrchestrator: """快取搜尋協調器。 @@ -665,6 +695,7 @@ def rebuild_search_index( except PermissionError: if attempt < max_retries - 1: import time + time.sleep(retry_delay) continue else: @@ -672,17 +703,13 @@ def rebuild_search_index( raise def _do_rebuild_search_index( - self, - db_path, - cache_types: List[str], - cache_state: Dict[str, Dict[str, Any]] + self, db_path, cache_types: List[str], cache_state: Dict[str, Dict[str, Any]] ) -> int: """執行實際的索引重建作業。""" tmp_engine: Optional[CacheSearchEngine] = None old_engine: Optional[CacheSearchEngine] = None total_indexed = 0 - - + try: # 直接寫入目標資料庫(不使用 tmp 檔案,避免 Windows 檔案鎖問題) # 先關閉舊引擎 @@ -691,11 +718,12 @@ def _do_rebuild_search_index( if old_engine is not None: old_engine.close() self._engine = None - + del old_engine import gc + gc.collect() - + # 清理 WAL/SHM 檔案 for suffix in ["-wal", "-shm"]: wal_file = db_path.with_name(db_path.name + suffix) @@ -704,24 +732,24 @@ def _do_rebuild_search_index( wal_file.unlink() except Exception as e: log_debug(f"刪除 WAL/SHM 檔案失敗: {e}") - + # 刪除舊資料庫重新建立 if db_path.exists(): try: db_path.unlink() except Exception as e: log_debug(f"刪除舊資料庫檔案失敗: {e}") - + # 建立新引擎並直接寫入 tmp_engine = CacheSearchEngine(str(db_path)) total_indexed = rebuild_from_cache_dicts( tmp_engine, cache_types, cache_state ) - + # 重新建立引擎 with self._lock: self._engine = CacheSearchEngine(str(db_path)) - + log_debug(f"索引重建完成: {total_indexed} 條") return total_indexed finally: diff --git a/workspace/patch_md_lmtranslator.py b/workspace/patch_md_lmtranslator.py index f865bd18..ca1feda5 100644 --- a/workspace/patch_md_lmtranslator.py +++ b/workspace/patch_md_lmtranslator.py @@ -1,26 +1,26 @@ from pathlib import Path import re -p = Path(r'translation_tool/plugins/md/md_lmtranslator.py') -text = p.read_text(encoding='utf-8') +p = Path(r"translation_tool/plugins/md/md_lmtranslator.py") +text = p.read_text(encoding="utf-8") pattern = re.compile( - r'for h, src in hash_to_src\.items\(\):\n' - r'\s+if is_already_zh\(src\):\n' - r'\s+already_zh_skipped \+= 1\n' - r'\s+continue\n' - r'\s+all_unique_items\.append\(\n' - r'\s+\{\n' + r"for h, src in hash_to_src\.items\(\):\n" + r"\s+if is_already_zh\(src\):\n" + r"\s+already_zh_skipped \+= 1\n" + r"\s+continue\n" + r"\s+all_unique_items\.append\(\n" + r"\s+\{\n" r'\s+"cache_type": "md",\n' r'\s+"file": "md_pending_blocks",\n' r'\s+"path": h, # .*?\n' r'\s+"source_text": src,\n' r'\s+"text": src,\n' - r'\s+\}\n' - r'\s+\)', + r"\s+\}\n" + r"\s+\)", re.S, ) -replacement = '''for h, src in hash_to_src.items(): +replacement = """for h, src in hash_to_src.items(): if is_already_zh(src): already_zh_skipped += 1 continue @@ -39,20 +39,20 @@ "text": translate_text, "_shielded": shielded, } - )''' + )""" text, n = pattern.subn(replacement, text, count=1) if n != 1: - raise SystemExit(f'pattern replace 1 failed: {n}') + raise SystemExit(f"pattern replace 1 failed: {n}") text = text.replace( -''' hash_to_dst: Dict[str, str] = {} + """ hash_to_dst: Dict[str, str] = {} for it in cached_items: h = str(it.get("path") or "") dst = str(it.get("text") or "") if h and dst: hash_to_dst[h] = dst -''', -''' hash_to_dst: Dict[str, str] = {} +""", + """ hash_to_dst: Dict[str, str] = {} for it in cached_items: h = str(it.get("path") or "") dst = str(it.get("text") or "") @@ -64,10 +64,11 @@ except Exception: pass hash_to_dst[h] = dst -''') +""", +) text = text.replace( -''' def on_translated_item(it: Dict[str, Any]) -> None: + ''' def on_translated_item(it: Dict[str, Any]) -> None: """處理翻譯結果。""" h = str(it.get("path") or "") dst = str(it.get("text") or "") @@ -80,7 +81,7 @@ pass hash_to_dst[h] = dst ''', -''' def on_translated_item(it: Dict[str, Any]) -> None: + ''' def on_translated_item(it: Dict[str, Any]) -> None: """處理翻譯結果。""" h = str(it.get("path") or "") dst = str(it.get("text") or "") @@ -99,7 +100,8 @@ except Exception: pass hash_to_dst[h] = dst -''') +''', +) -p.write_text(text, encoding='utf-8') -print('patched md_lmtranslator') +p.write_text(text, encoding="utf-8") +print("patched md_lmtranslator") From 8d9189477bfa295d6e95e457606e77c8890f5fc3 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Sun, 29 Mar 2026 01:02:55 +0800 Subject: [PATCH 30/33] fix(pr43): align gemini auth and shield dry-run flow --- tests/test_ftbquests_lmtranslator.py | 45 ++++++++++++---- tests/test_lm_api_client.py | 11 ++-- tests/test_md_lmtranslator.py | 54 ++++++++++--------- tests/test_plugins_shared_helpers.py | 26 +++++++-- translation_tool/core/lm_api_client.py | 2 +- .../ftbquests/ftbquests_lmtranslator.py | 29 ++++++++-- .../plugins/md/md_lmtranslator.py | 6 --- .../plugins/shared/rich_text_shield.py | 4 +- 8 files changed, 119 insertions(+), 58 deletions(-) diff --git a/tests/test_ftbquests_lmtranslator.py b/tests/test_ftbquests_lmtranslator.py index 37f8fd8f..c02d2be2 100644 --- a/tests/test_ftbquests_lmtranslator.py +++ b/tests/test_ftbquests_lmtranslator.py @@ -2,6 +2,7 @@ 用途:測試 ftbquests_lmtranslator 模組的功能。 """ + from __future__ import annotations import sys @@ -12,7 +13,7 @@ sys.path.insert(0, str(ROOT)) # 測試模組 -from translation_tool.plugins.ftbquests import ftbquests_lmtranslator +from translation_tool.plugins.ftbquests import ftbquests_lmtranslator # noqa: E402 def test_map_to_items_basic(tmp_path: Path) -> None: @@ -21,13 +22,13 @@ def test_map_to_items_basic(tmp_path: Path) -> None: "quest.1.title": "Hello", "quest.2.title": "World", } - + items = ftbquests_lmtranslator.map_to_items( mapping, cache_type="ftbquests", file_hint="config/ftbquests/quests/en_us/test.json", ) - + assert len(items) == 2 assert items[0]["cache_type"] == "ftbquests" assert items[0]["path"] == "quest.1.title" @@ -41,17 +42,39 @@ def test_map_to_items_filters_invalid(tmp_path: Path) -> None: 123: "Invalid key", # key 不是字串 "empty.value": "", # value 是空字串 } - + items = ftbquests_lmtranslator.map_to_items( mapping, cache_type="ftbquests", file_hint="config/ftbquests/quests/test.json", ) - + assert len(items) == 1 assert items[0]["path"] == "valid.key" +def test_map_to_items_shields_text_and_marks_skip_reason() -> None: + """FTB item 應先 shield,需要 skip 的項目要標記 skip_reason。""" + mapping = { + "quest.title": "Hello &aWorld", + "quest.url": "https://example.com", + } + + items = ftbquests_lmtranslator.map_to_items( + mapping, + cache_type="ftbquests", + file_hint="config/ftbquests/quests/test.json", + ) + + normal_item = next(it for it in items if it["path"] == "quest.title") + skip_item = next(it for it in items if it["path"] == "quest.url") + + assert normal_item["text"] != normal_item["source_text"] + assert getattr(normal_item["_shielded"], "shields", []) + assert skip_item["_skip_reason"] == "url" + assert skip_item["text"] == "https://example.com" + + def test_count_translatable_keys(tmp_path: Path) -> None: """測試 count_translatable_keys 計算可翻譯鍵數量。""" mapping = { @@ -60,9 +83,9 @@ def test_count_translatable_keys(tmp_path: Path) -> None: "key3": "", # 空值不計 "key4": " ", # 只有空白不計 } - + count = ftbquests_lmtranslator.count_translatable_keys(mapping) - + assert count == 2 @@ -73,16 +96,16 @@ def test_count_translatable_keys_non_string_values(tmp_path: Path) -> None: "key2": 123, # 不是字串 "key3": ["list"], # 不是字串 } - + count = ftbquests_lmtranslator.count_translatable_keys(mapping) - + assert count == 1 def test_dry_run_stats_default(tmp_path: Path) -> None: """測試 DryRunStats 預設值。""" stats = ftbquests_lmtranslator.DryRunStats() - + assert stats.files == 0 assert stats.total_keys == 0 assert stats.cache_hit == 0 @@ -98,7 +121,7 @@ def test_dry_run_stats_with_values(tmp_path: Path) -> None: cache_hit=30, cache_miss=70, ) - + assert stats.files == 5 assert stats.total_keys == 100 assert stats.cache_hit == 30 diff --git a/tests/test_lm_api_client.py b/tests/test_lm_api_client.py index 7f52f20b..5e52b060 100644 --- a/tests/test_lm_api_client.py +++ b/tests/test_lm_api_client.py @@ -117,7 +117,7 @@ def test_custom_timeout(self, mock_config, mock_post): @patch("translation_tool.core.lm_api_client.requests.post") @patch("translation_tool.core.lm_api_client.load_config") def test_api_key_not_in_url(self, mock_config, mock_post): - """測試 API Key 不出現在 URL 中,而是放在 Authorization: Bearer header。""" + """測試 API Key 不出現在 URL 中,而是放在 x-goog-api-key header。""" from translation_tool.core.lm_api_client import call_gemini_requests # 使用假的 API key(長度 35-45 字,以 AIza 開頭) @@ -147,12 +147,11 @@ def test_api_key_not_in_url(self, mock_config, mock_post): ) assert fake_api_key not in called_url, "API key 不應出現在 URL 中" - # 驗證 Authorization: Bearer header 存在 + # 驗證 x-goog-api-key header 存在(對照 Google 官方 REST 範例) headers = call_args.kwargs.get("headers", {}) - assert "Authorization" in headers, "Authorization header 必須存在" - assert headers["Authorization"] == f"Bearer {fake_api_key}", ( - "Authorization header 應為 Bearer {api_key} 格式" - ) + assert "x-goog-api-key" in headers, "x-goog-api-key header 必須存在" + assert headers["x-goog-api-key"] == fake_api_key + assert "Authorization" not in headers, "不應再使用 Authorization: Bearer" # 確保 URL 中沒有 key=... 之類的 query string assert "?" not in called_url or "key=" not in called_url, ( diff --git a/tests/test_md_lmtranslator.py b/tests/test_md_lmtranslator.py index d47a61a3..10899aeb 100644 --- a/tests/test_md_lmtranslator.py +++ b/tests/test_md_lmtranslator.py @@ -2,6 +2,7 @@ 用途:測試 md_lmtranslator 模組的功能。 """ + from __future__ import annotations import sys @@ -13,25 +14,25 @@ sys.path.insert(0, str(ROOT)) # 測試模組 -from translation_tool.plugins.md import md_lmtranslator +from translation_tool.plugins.md import md_lmtranslator # noqa: E402 def test_read_json(tmp_path: Path) -> None: """測試 read_json 讀取 JSON。""" json_file = tmp_path / "test.json" json_file.write_text(json.dumps({"key": "value"})) - + result = md_lmtranslator.read_json(json_file) - + assert result == {"key": "value"} def test_write_json(tmp_path: Path) -> None: """測試 write_json 寫入 JSON。""" json_file = tmp_path / "output.json" - + md_lmtranslator.write_json(json_file, {"key": "value"}) - + assert json_file.exists() assert json.loads(json_file.read_text()) == {"key": "value"} @@ -39,9 +40,9 @@ def test_write_json(tmp_path: Path) -> None: def test_write_json_creates_parent(tmp_path: Path) -> None: """測試 write_json 建立父目錄。""" json_file = tmp_path / "subdir" / "output.json" - + md_lmtranslator.write_json(json_file, {"data": 123}) - + assert json_file.exists() assert json_file.parent.exists() @@ -51,21 +52,21 @@ def test_collect_pending_json_files(tmp_path: Path) -> None: # 建立測試結構 pending_root = tmp_path / "pending" pending_root.mkdir() - + # 一般 JSON 檔案 (pending_root / "file1.json").write_text("{}") (pending_root / "file2.json").write_text("{}") - + # Manifest 檔案(應該被跳過) (pending_root / "_manifest.json").write_text("{}") - + # 子目錄中的 JSON subdir = pending_root / "subdir" subdir.mkdir() (subdir / "file3.json").write_text("{}") - + files = md_lmtranslator.collect_pending_json_files(pending_root) - + # 應該有 3 個檔案(排除 manifest) assert len(files) == 3 assert all("_manifest" not in str(f) for f in files) @@ -85,14 +86,14 @@ def test_load_pending_doc(tmp_path: Path) -> None: "start_line": 1, "end_line": 2, } - ] + ], } - + json_file = tmp_path / "test.json" json_file.write_text(json.dumps(json_data)) - + data, items = md_lmtranslator.load_pending_doc(json_file) - + assert data["schema"] == "md_pending_blocks_v1" assert len(items) == 1 assert items[0].text == "Hello" @@ -102,7 +103,7 @@ def test_load_pending_doc_invalid_schema(tmp_path: Path) -> None: """測試 load_pending_doc 無效 schema。""" json_file = tmp_path / "test.json" json_file.write_text(json.dumps({"schema": "unknown", "items": []})) - + try: md_lmtranslator.load_pending_doc(json_file) assert False, "應該拋出 ValueError" @@ -115,14 +116,12 @@ def test_compute_out_json_path(tmp_path: Path) -> None: src_json = tmp_path / "pending" / "test.json" src_json.parent.mkdir(parents=True) src_json.write_text("{}") - + in_pending_root = tmp_path / "pending" out_root = tmp_path / "output" - - result = md_lmtranslator.compute_out_json_path( - src_json, in_pending_root, out_root - ) - + + result = md_lmtranslator.compute_out_json_path(src_json, in_pending_root, out_root) + # 應該在 LM翻譯後 目錄下 assert "LM翻譯後" in result.parts assert result.name == "test.json" @@ -137,9 +136,16 @@ def test_pending_item_dataclass(tmp_path: Path) -> None: start_line=1, end_line=2, ) - + assert item.id == "test:1-2" assert item.text == "Hello World" assert item.content_hash == "abc123" assert item.start_line == 1 assert item.end_line == 2 + + +def test_md_skip_reason_item_stays_original_in_dry_run_inputs() -> None: + """skip_reason 項目在 MD 流程中應保留原文。""" + shielded = md_lmtranslator.shield_text("https://example.com") + assert shielded.skip_reason == "url" + assert shielded.clean == "https://example.com" diff --git a/tests/test_plugins_shared_helpers.py b/tests/test_plugins_shared_helpers.py index 84bd4bc0..f4d1d379 100644 --- a/tests/test_plugins_shared_helpers.py +++ b/tests/test_plugins_shared_helpers.py @@ -11,6 +11,7 @@ is_lang_code_segment, ) from translation_tool.plugins.shared.lang_text_rules import _strip_fmt, is_already_zh +from translation_tool.plugins.shared.rich_text_shield import shield_text def test_compute_output_path_renames_lang_folder_and_filename() -> None: @@ -63,10 +64,10 @@ def test_strip_fmt_samples(raw: str, expected: str) -> None: @pytest.mark.parametrize( ("text", "expected"), [ - ("這是中文內容", True), # 中文 -> True + ("這是中文內容", True), # 中文 -> True ("This is english", False), # 英文 -> False - ("獲得 3x Iron", False), # 邊界:中英混合(英文字母較多) - ("§a獲得 3x 鐵", True), # 邊界:有格式碼且主要為中文 + ("獲得 3x Iron", False), # 邊界:中英混合(英文字母較多) + ("§a獲得 3x 鐵", True), # 邊界:有格式碼且主要為中文 ], ) def test_is_already_zh_samples(text: str, expected: bool) -> None: @@ -83,10 +84,27 @@ def test_read_write_json_dict_roundtrip(tmp_path: Path) -> None: assert loaded == payload - def test_read_json_dict_raises_when_root_is_not_dict(tmp_path: Path) -> None: target = tmp_path / "list.json" target.write_text(json.dumps(["a", "b"]), encoding="utf-8") with pytest.raises(ValueError, match="object/dict"): read_json_dict(target) + + +@pytest.mark.parametrize( + ("raw", "expected_skip_reason"), + [ + ("https://example.com", "url"), + ("icon.png", "image"), + ("{@pagebreak}", "pagebreak"), + ], +) +def test_shield_text_skip_reason_samples(raw: str, expected_skip_reason: str) -> None: + shielded = shield_text(raw) + assert shielded.skip_reason == expected_skip_reason + + +def test_shield_text_supports_reset_code_r() -> None: + shielded = shield_text("Hello &rWorld") + assert any(piece.original == "&r" for piece in shielded.shields) diff --git a/translation_tool/core/lm_api_client.py b/translation_tool/core/lm_api_client.py index 7c953506..85121d0a 100644 --- a/translation_tool/core/lm_api_client.py +++ b/translation_tool/core/lm_api_client.py @@ -29,7 +29,7 @@ def call_gemini_requests( headers = { "Content-Type": "application/json", - "Authorization": f"Bearer {api_key}", + "x-goog-api-key": api_key, } data = { diff --git a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py index 98fee5d1..de9ddf77 100644 --- a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py +++ b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py @@ -93,6 +93,9 @@ def map_to_items( if not isinstance(v, str) or not v.strip(): continue + shielded = shield_text(v) + translate_text = v if shielded.skip_reason is not None else shielded.clean + items.append( { # 提供 smart translator 判斷用的檔案提示路徑 @@ -103,9 +106,11 @@ def map_to_items( # 原始文字(快取與比對用) "source_text": v, # 當前文字(會被翻譯器覆寫) - "text": v, + "text": translate_text, # 指定快取分類(對應 cache_rules) - "cache_type": cache_type, # 例如 "ftbquests" + "cache_type": cache_type, + "_shielded": shielded, + "_skip_reason": shielded.skip_reason, } ) @@ -278,6 +283,13 @@ def _count_one(src: Path) -> Tuple[Path, int, Dict[str, Any]]: is_valid_hit=_is_valid_hit, ) + skip_items = [it for it in items_to_translate if it.get("_skip_reason")] + if skip_items: + cached_items.extend(skip_items) + items_to_translate = [ + it for it in items_to_translate if not it.get("_skip_reason") + ] + # ✅ 中文/已翻譯:不送 LM,也不算 cache_miss already_zh_items = [] real_to_translate = [] @@ -352,6 +364,13 @@ def _writer(file_id: str) -> None: is_valid_hit=_is_valid_hit, ) + skip_items = [it for it in items_to_translate if it.get("_skip_reason")] + if skip_items: + cached_items.extend(skip_items) + items_to_translate = [ + it for it in items_to_translate if not it.get("_skip_reason") + ] + already_zh_items = [] real_to_translate = [] for it in items_to_translate: @@ -503,8 +522,10 @@ def on_translated_item(it: Dict[str, Any]) -> None: src_text = str(it.get("source_text") or "") if isinstance(p, str) and isinstance(t, str): try: - shielded_src = shield_text(src_text) - t = unshield_text(t, shielded_src.shields) + shielded_src = it.get("_shielded") or shield_text(src_text) + shields = getattr(shielded_src, "shields", []) + if shields: + t = unshield_text(t, shields) except Exception: pass out_map[p] = t diff --git a/translation_tool/plugins/md/md_lmtranslator.py b/translation_tool/plugins/md/md_lmtranslator.py index d0c05cd8..8212bce6 100644 --- a/translation_tool/plugins/md/md_lmtranslator.py +++ b/translation_tool/plugins/md/md_lmtranslator.py @@ -333,12 +333,6 @@ def on_translated_item(it: Dict[str, Any]) -> None: dst = unshield_text(dst, shielded.shields) except Exception as e: log_warning(f"[MD-LM] unshield 失敗: {e}") - else: - try: - shielded_src = shield_text(src_text) - dst = unshield_text(dst, shielded_src.shields) - except Exception as e: - log_warning(f"[MD-LM] unshield (else branch) 失敗: {e}") hash_to_dst[h] = dst # 這裡 recorder 的 cache_type 用 md(方便你日後 QC) try: diff --git a/translation_tool/plugins/shared/rich_text_shield.py b/translation_tool/plugins/shared/rich_text_shield.py index c2e7d557..1c119557 100644 --- a/translation_tool/plugins/shared/rich_text_shield.py +++ b/translation_tool/plugins/shared/rich_text_shield.py @@ -20,8 +20,8 @@ # 物品ID(#namespace:item 或 #namespace/item — 兩種都支援) ITEM_ID_PATTERN = re.compile(r"#[a-z0-9_.\-]+[:/][a-z0-9_.\-]+", re.IGNORECASE) -# 標準彩色碼:&a ~ &o(不含 k 的 16 進位格式碼) -COLOR_CODE_PATTERN = re.compile(r"&[a-f0-9k-o]", re.IGNORECASE) +# 標準彩色碼:支援 Minecraft 格式化代碼 0-9, a-f, k-o, r +COLOR_CODE_PATTERN = re.compile(r"&[0-9a-fk-or]", re.IGNORECASE) # &#RRGGBB 十六進位顏色 HEX_COLOR_PATTERN = re.compile(r"&#[0-9A-Fa-f]{6}", re.IGNORECASE) From 3aded2a15ce0e01aa220c2a34073d574ac812efc Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Sun, 29 Mar 2026 01:06:39 +0800 Subject: [PATCH 31/33] chore(security): stop tracking local config.json --- .gitignore | 3 ++ config.json | 124 ---------------------------------------------------- 2 files changed, 3 insertions(+), 124 deletions(-) delete mode 100644 config.json diff --git a/.gitignore b/.gitignore index a4b46184..589797b4 100644 --- a/.gitignore +++ b/.gitignore @@ -54,3 +54,6 @@ translation_tool/core/*.txt # === OS === Thumbs.db .DS_Store + +# === Local secrets === +config.json diff --git a/config.json b/config.json deleted file mode 100644 index 3f186d98..00000000 --- a/config.json +++ /dev/null @@ -1,124 +0,0 @@ -{ - "logging": { - "log_level": "INFO", - "log_format": "%(asctime)s - %(levelname)s - [%(name)s] - %(message)s", - "log_dir": "logs" - }, - "translator": { - "output_dir_name": "zh_tw_generated", - "replace_rules_path": "replace_rules.json", - "cache_directory": "快取資料", - "enable_cache_saving": true, - "parallel_execution_workers": 8, - "cjk_ratio_threshold": 0.7 - }, - "species_cache": { - "cache_directory": "學名資料庫", - "cache_filename": "species_cache.tsv", - "wikipedia_language": "zh", - "wikipedia_rate_limit_delay": 0.5 - }, - "lm_translator": { - "temperature": 0.3, - "lm_translate_folder_name": "LM翻譯後", - "initial_batch_size_patchouli": 100, - "initial_batch_size_lang": 300, - "initial_batch_size_ftb": 200, - "initial_batch_size_kubejs": 200, - "initial_batch_size_md": 100, - "min_batch_size": 50, - "batch_shrink_factor": 0.5, - "rate_limit": { - "timeout": 600, - "sleep_seconds": 600 - }, - "models": { - "gemini-3.1-flash-lite-preview": { - "enabled": true - }, - "gemini-2.5-flash": { - "enabled": false - } - }, - "keys": [ - "***REMOVED***" - ], - "patchouli_system_prompt": "你是專業的 Minecraft Patchouli 手冊翻譯員。\r\n\r\n你正在翻譯一個「ID → Value 對照表」。\r\n\r\n⚠️【極重要規則 — ID 不可變】⚠️\r\n- items[].id 是不可變的識別符號\r\n- id 不具有任何語意,也不對應任何 JSON 結構\r\n- id 只能被視為純文字索引\r\n- 絕對禁止:\r\n - 修改、重寫、補零、轉型、排序、重編任何 id\r\n - 新增或刪除任何 id\r\n - 嘗試推測 id 與內容的關聯\r\n\r\n📌 任務規則:\r\n1. 只允許修改 items[].value 的字串內容\r\n2. items[].id 必須與輸入完全一字不差\r\n3. items 的數量與順序必須與輸入完全一致\r\n4. 如果你不確定如何翻譯,請原樣回傳 value\r\n5. 回傳必須是合法 JSON,且格式與輸入完全一致\r\n6. 僅翻譯為繁體中文(台灣用語)\r\n7. 保留 §, %, {}, $(...) 等所有符號與格式\r\n8. 單位(mb、tick 等)請保留原文\r\n9. Minecraft 請保持原文,不要翻譯成「當個創世神」\r\n10. 每一筆 value 必須只根據該筆原文自身內容翻譯\r\n11. 只要 value 包含人類語言就必須翻譯\r\n12. 學名請翻譯為台灣常用語(如 Creeper → 苦力怕),(Spawn Egg-> 生怪蛋),(cobblestone->鵝卵石)", - "lang_system_prompt": "你正在翻譯 Minecraft 語言檔案(JSON 格式)。\r\n\r\n你收到的是一個「ID → value 對照表」。\r\n\r\n⚠️【極重要規則 — ID 不可變】⚠️\r\n- items[].id 是唯一識別符號\r\n- id 不具有任何語意\r\n- 絕對禁止:\r\n - 修改、轉型、補零、重排、推測或重寫任何 id\r\n - 新增或刪除任何 item\r\n\r\n📌 任務規則:\r\n1. 只允許修改 items[].value 的字串內容\r\n2. items[].id 必須與輸入完全一字不差\r\n3. items 的數量與順序必須與輸入完全一致\r\n4. 如果你不確定如何翻譯,請原樣回傳 value\r\n5. 回傳必須是合法 JSON,格式必須為 {\"items\":[{\"id\":...,\"value\":...}, ...]}\r\n6. 僅翻譯為繁體中文(台灣用語)\r\n7. 保留 §, %, {}, $(...) 等所有符號與格式\r\n8. 單位(mb、tick 等)請保留原文\r\n9. Minecraft 請保持原文\r\n10. 每一筆 value 只依該筆原文翻譯\r\n11. 只要 value 包含人類語言就必須翻譯\r\n12. 學名請翻譯為台灣常用語(如 Creeper → 苦力怕),(Spawn Egg-> 生怪蛋),(cobblestone->鵝卵石)", - "translator": { - "skip_terms": [ - "api documentation", - "api docs", - "documentation", - "discord", - "github", - "homepage", - "mod page", - "modpack", - "official website", - "patreon", - "Twitter", - "Modrinth", - "CurseForge", - "Crowdin", - "Twitch", - "Wiki", - "Minecraft", - "Forge", - "YouTube", - "Reddit", - "Ko-fi", - "Flattr" - ], - "translatable_keywords": [ - "text", - "name", - "title", - "description", - "subtitle", - "hover", - "note", - "warning", - "quote", - "paragraph", - "body", - "header", - "footer", - "heading", - "effects", - "category", - "link_text", - "pages.title" - ] - }, - "patchouli": { - "dir_names": [ - "patchouli_books", - "book", - "manual", - "guidebook" - ] - }, - "rpm_cooldown_sec": 12, - "overload_retry_sec": 12, - "key_rotation_buffer_sec": 5, - "request_interval_sec": 4 - }, - "output_bundler": { - "output_zip_name": "可使用翻譯.zip", - "source_folders": { - "assets": "zh_tw_generated/assets", - "root": "zh_tw_generated/pack_mcmeta" - } - }, - "lang_merger": { - "pending_folder_name": "待翻譯", - "pending_organized_folder_name": "待翻譯整理需翻譯", - "filtered_pending_min_count": 3, - "quarantine_folder_name": "問題檔案skipped_json", - "process_zh_cn_files": true, - "skip_zh_cn_when_only_process_lang": false, - "patchouli_skip_en_us_when_zh_cn_exists": false, - "patchouli_effective_translation_threshold": 0.5 - } -} \ No newline at end of file From 703858bd4e20c1c05ad1892d4dbbaf15113c2a43 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Sun, 29 Mar 2026 11:25:37 +0800 Subject: [PATCH 32/33] chore(security): stop tracking local config.json --- config.json | 124 ---------------------------------------------------- 1 file changed, 124 deletions(-) delete mode 100644 config.json diff --git a/config.json b/config.json deleted file mode 100644 index 3f186d98..00000000 --- a/config.json +++ /dev/null @@ -1,124 +0,0 @@ -{ - "logging": { - "log_level": "INFO", - "log_format": "%(asctime)s - %(levelname)s - [%(name)s] - %(message)s", - "log_dir": "logs" - }, - "translator": { - "output_dir_name": "zh_tw_generated", - "replace_rules_path": "replace_rules.json", - "cache_directory": "快取資料", - "enable_cache_saving": true, - "parallel_execution_workers": 8, - "cjk_ratio_threshold": 0.7 - }, - "species_cache": { - "cache_directory": "學名資料庫", - "cache_filename": "species_cache.tsv", - "wikipedia_language": "zh", - "wikipedia_rate_limit_delay": 0.5 - }, - "lm_translator": { - "temperature": 0.3, - "lm_translate_folder_name": "LM翻譯後", - "initial_batch_size_patchouli": 100, - "initial_batch_size_lang": 300, - "initial_batch_size_ftb": 200, - "initial_batch_size_kubejs": 200, - "initial_batch_size_md": 100, - "min_batch_size": 50, - "batch_shrink_factor": 0.5, - "rate_limit": { - "timeout": 600, - "sleep_seconds": 600 - }, - "models": { - "gemini-3.1-flash-lite-preview": { - "enabled": true - }, - "gemini-2.5-flash": { - "enabled": false - } - }, - "keys": [ - "***REMOVED***" - ], - "patchouli_system_prompt": "你是專業的 Minecraft Patchouli 手冊翻譯員。\r\n\r\n你正在翻譯一個「ID → Value 對照表」。\r\n\r\n⚠️【極重要規則 — ID 不可變】⚠️\r\n- items[].id 是不可變的識別符號\r\n- id 不具有任何語意,也不對應任何 JSON 結構\r\n- id 只能被視為純文字索引\r\n- 絕對禁止:\r\n - 修改、重寫、補零、轉型、排序、重編任何 id\r\n - 新增或刪除任何 id\r\n - 嘗試推測 id 與內容的關聯\r\n\r\n📌 任務規則:\r\n1. 只允許修改 items[].value 的字串內容\r\n2. items[].id 必須與輸入完全一字不差\r\n3. items 的數量與順序必須與輸入完全一致\r\n4. 如果你不確定如何翻譯,請原樣回傳 value\r\n5. 回傳必須是合法 JSON,且格式與輸入完全一致\r\n6. 僅翻譯為繁體中文(台灣用語)\r\n7. 保留 §, %, {}, $(...) 等所有符號與格式\r\n8. 單位(mb、tick 等)請保留原文\r\n9. Minecraft 請保持原文,不要翻譯成「當個創世神」\r\n10. 每一筆 value 必須只根據該筆原文自身內容翻譯\r\n11. 只要 value 包含人類語言就必須翻譯\r\n12. 學名請翻譯為台灣常用語(如 Creeper → 苦力怕),(Spawn Egg-> 生怪蛋),(cobblestone->鵝卵石)", - "lang_system_prompt": "你正在翻譯 Minecraft 語言檔案(JSON 格式)。\r\n\r\n你收到的是一個「ID → value 對照表」。\r\n\r\n⚠️【極重要規則 — ID 不可變】⚠️\r\n- items[].id 是唯一識別符號\r\n- id 不具有任何語意\r\n- 絕對禁止:\r\n - 修改、轉型、補零、重排、推測或重寫任何 id\r\n - 新增或刪除任何 item\r\n\r\n📌 任務規則:\r\n1. 只允許修改 items[].value 的字串內容\r\n2. items[].id 必須與輸入完全一字不差\r\n3. items 的數量與順序必須與輸入完全一致\r\n4. 如果你不確定如何翻譯,請原樣回傳 value\r\n5. 回傳必須是合法 JSON,格式必須為 {\"items\":[{\"id\":...,\"value\":...}, ...]}\r\n6. 僅翻譯為繁體中文(台灣用語)\r\n7. 保留 §, %, {}, $(...) 等所有符號與格式\r\n8. 單位(mb、tick 等)請保留原文\r\n9. Minecraft 請保持原文\r\n10. 每一筆 value 只依該筆原文翻譯\r\n11. 只要 value 包含人類語言就必須翻譯\r\n12. 學名請翻譯為台灣常用語(如 Creeper → 苦力怕),(Spawn Egg-> 生怪蛋),(cobblestone->鵝卵石)", - "translator": { - "skip_terms": [ - "api documentation", - "api docs", - "documentation", - "discord", - "github", - "homepage", - "mod page", - "modpack", - "official website", - "patreon", - "Twitter", - "Modrinth", - "CurseForge", - "Crowdin", - "Twitch", - "Wiki", - "Minecraft", - "Forge", - "YouTube", - "Reddit", - "Ko-fi", - "Flattr" - ], - "translatable_keywords": [ - "text", - "name", - "title", - "description", - "subtitle", - "hover", - "note", - "warning", - "quote", - "paragraph", - "body", - "header", - "footer", - "heading", - "effects", - "category", - "link_text", - "pages.title" - ] - }, - "patchouli": { - "dir_names": [ - "patchouli_books", - "book", - "manual", - "guidebook" - ] - }, - "rpm_cooldown_sec": 12, - "overload_retry_sec": 12, - "key_rotation_buffer_sec": 5, - "request_interval_sec": 4 - }, - "output_bundler": { - "output_zip_name": "可使用翻譯.zip", - "source_folders": { - "assets": "zh_tw_generated/assets", - "root": "zh_tw_generated/pack_mcmeta" - } - }, - "lang_merger": { - "pending_folder_name": "待翻譯", - "pending_organized_folder_name": "待翻譯整理需翻譯", - "filtered_pending_min_count": 3, - "quarantine_folder_name": "問題檔案skipped_json", - "process_zh_cn_files": true, - "skip_zh_cn_when_only_process_lang": false, - "patchouli_skip_en_us_when_zh_cn_exists": false, - "patchouli_effective_translation_threshold": 0.5 - } -} \ No newline at end of file From 32c952f6a5a22c21c84c081da58d9366fdaa4493 Mon Sep 17 00:00:00 2001 From: jlin53882 Date: Sun, 29 Mar 2026 16:54:15 +0800 Subject: [PATCH 33/33] Merge origin/main into fix/all-code-review-issues (resolve remaining PR#43 conflicts) Auto-resolved 6 additional files: - test_lm_api_client.py: keep HEAD test (API key header test) - ftbquests_lmtranslator.py, kubejs_tooltip_lmtranslator.py, md_inject_qa.py, md_lmtranslator.py, rich_text_shield.py: take main's version (PR#41/42 updates) --- tests/test_ftbquests_lmtranslator.py | 16 ++++++ tests/test_kubejs_tooltip_lmtranslator.py | 44 ++++++++++---- .../core/kubejs_translator_clean.py | 14 +---- .../ftbquests/ftbquests_lmtranslator.py | 57 ++++++++++++++++++- .../kubejs/kubejs_tooltip_lmtranslator.py | 38 +++++++++++-- translation_tool/plugins/md/md_inject_qa.py | 3 + .../plugins/md/md_lmtranslator.py | 14 ++--- .../plugins/shared/rich_text_shield.py | 2 +- 8 files changed, 149 insertions(+), 39 deletions(-) diff --git a/tests/test_ftbquests_lmtranslator.py b/tests/test_ftbquests_lmtranslator.py index c02d2be2..35e7ae99 100644 --- a/tests/test_ftbquests_lmtranslator.py +++ b/tests/test_ftbquests_lmtranslator.py @@ -126,3 +126,19 @@ def test_dry_run_stats_with_values(tmp_path: Path) -> None: assert stats.total_keys == 100 assert stats.cache_hit == 30 assert stats.cache_miss == 70 + + +def test_ftb_dry_run_preview_items_can_drop_runtime_fields() -> None: + """dry-run preview 寫檔前應可移除 _shielded 等 runtime 欄位。""" + mapping = {"quest.title": "Hello &aWorld"} + items = ftbquests_lmtranslator.map_to_items( + mapping, + cache_type="ftbquests", + file_hint="config/ftbquests/quests/test.json", + ) + + sanitized = [{k: v for k, v in it.items() if k != "_shielded"} for it in items] + + assert "_shielded" not in sanitized[0] + assert sanitized[0]["path"] == "quest.title" + assert sanitized[0]["text"] != sanitized[0]["source_text"] diff --git a/tests/test_kubejs_tooltip_lmtranslator.py b/tests/test_kubejs_tooltip_lmtranslator.py index 9ddcbd17..85e78aef 100644 --- a/tests/test_kubejs_tooltip_lmtranslator.py +++ b/tests/test_kubejs_tooltip_lmtranslator.py @@ -2,6 +2,7 @@ 用途:測試 kubejs_tooltip_lmtranslator 模組的功能。 """ + from __future__ import annotations import sys @@ -12,7 +13,7 @@ sys.path.insert(0, str(ROOT)) # 測試模組 -from translation_tool.plugins.kubejs import kubejs_tooltip_lmtranslator +from translation_tool.plugins.kubejs import kubejs_tooltip_lmtranslator # noqa: E402 def test_collect_items_from_mapping_basic(tmp_path: Path) -> None: @@ -21,12 +22,12 @@ def test_collect_items_from_mapping_basic(tmp_path: Path) -> None: "tooltip.1": "Hello", "tooltip.2": "World", } - + items = kubejs_tooltip_lmtranslator.collect_items_from_mapping( mapping, file_hint="output/kubejs/test.json", ) - + assert len(items) == 2 assert items[0]["cache_type"] == "kubejs" assert items[0]["path"] == "tooltip.1" @@ -40,15 +41,36 @@ def test_collect_items_from_mapping_filters_invalid(tmp_path: Path) -> None: 123: "Invalid key", "empty.value": "", } - + items = kubejs_tooltip_lmtranslator.collect_items_from_mapping( mapping, file_hint="output/kubejs/test.json", ) - + assert len(items) == 1 +def test_collect_items_from_mapping_marks_skip_reason() -> None: + """URL / 圖片等 skip_reason 項目應被明確標記。""" + mapping = { + "tooltip.url": "https://example.com", + "tooltip.normal": "Hello &aWorld", + } + + items = kubejs_tooltip_lmtranslator.collect_items_from_mapping( + mapping, + file_hint="output/kubejs/test.json", + ) + + skip_item = next(it for it in items if it["path"] == "tooltip.url") + normal_item = next(it for it in items if it["path"] == "tooltip.normal") + + assert skip_item["_skip_reason"] == "url" + assert skip_item["text"] == "https://example.com" + assert normal_item.get("_skip_reason") is None + assert normal_item["text"] != normal_item["source_text"] + + def test_count_translatable_keys(tmp_path: Path) -> None: """測試 count_translatable_keys 計算數量。""" mapping = { @@ -57,9 +79,9 @@ def test_count_translatable_keys(tmp_path: Path) -> None: "key3": "", # 空值不計 "key4": " ", # 空白不計 } - + count = kubejs_tooltip_lmtranslator.count_translatable_keys(mapping) - + assert count == 2 @@ -70,16 +92,16 @@ def test_count_translatable_keys_non_string(tmp_path: Path) -> None: "key2": 123, "key3": None, } - + count = kubejs_tooltip_lmtranslator.count_translatable_keys(mapping) - + assert count == 1 def test_dry_run_stats_default(tmp_path: Path) -> None: """測試 DryRunStats 預設值。""" stats = kubejs_tooltip_lmtranslator.DryRunStats() - + assert stats.files == 0 assert stats.total_keys == 0 assert stats.cache_hit == 0 @@ -95,7 +117,7 @@ def test_dry_run_stats_with_values(tmp_path: Path) -> None: cache_hit=20, cache_miss=30, ) - + assert stats.files == 3 assert stats.total_keys == 50 assert stats.cache_hit == 20 diff --git a/translation_tool/core/kubejs_translator_clean.py b/translation_tool/core/kubejs_translator_clean.py index 600cceb5..fb57a371 100644 --- a/translation_tool/core/kubejs_translator_clean.py +++ b/translation_tool/core/kubejs_translator_clean.py @@ -77,16 +77,6 @@ def _shielded_convert(text: str, convert_fn: Callable[[str], str]) -> str: 用於 OpenCC s2t 轉換時,保護 KubeJS 格式標記(彩色碼、物品ID 等) 不被轉換破壞。 """ - # 注意:本函式依賴 translation_tool.plugins.shared.rich_text_shield。 - # 若 rich_text_shield 尚未啟用,此函式退化成直接轉換。 - try: - from translation_tool.plugins.shared.rich_text_shield import ( - shield_text, - unshield_text, - ) - except ImportError: - return convert_fn(text) - shielded = shield_text(text) if shielded.skip_reason is not None: return text @@ -292,7 +282,7 @@ def clean_kubejs_from_raw_impl( # 表示該英文原文已有翻譯,不需要再送 pending。 # 建立 reverse_index:{英文文字: [key1, key2, ...]} if pending_en and final_root_p.exists(): - # 從 final/zh_tw.json 建立 final_tw_lookup(key → 翻譯值) + # 從 final/zh_tw.json 建立 final_tw_lookup(key → 原文) final_tw_lookup: dict[str, str] = {} for tw_file in final_root_p.rglob("zh_tw.json"): tw_data = read_json_dict_fn(tw_file) @@ -300,7 +290,7 @@ def clean_kubejs_from_raw_impl( final_tw_lookup.update(tw_data) if final_tw_lookup: - # 使用確定性 reverse_index 建構 + cross-namespace dedup + # 使用 PR #40 的乾淨去重實作 reverse_index = _build_reverse_index_impl(final_tw_lookup) pending_en = _dedup_pending_en_impl(pending_en, reverse_index) # ── 雙軌去重 end ─────────────────────────────────────────────── diff --git a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py index de9ddf77..fe52943a 100644 --- a/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py +++ b/translation_tool/plugins/ftbquests/ftbquests_lmtranslator.py @@ -492,7 +492,48 @@ def _writer(file_id: str) -> None: ) continue - # ✅ Issue #8 修復:將 callbacks 定義在迴圈外部,改用工廠函式 + # shared while-loop(includes add_to_cache + save_translation_cache + safe slicing) + def on_translated_item(it: Dict[str, Any]) -> None: + """處理翻譯結果並寫入映射。""" + p = it.get("path") + t = it.get("text") + src_text = str(it.get("source_text") or "") + if isinstance(p, str) and isinstance(t, str): + try: + shielded_src = it.get("_shielded") or shield_text(src_text) + shields = getattr(shielded_src, "shields", []) + if shields: + t = unshield_text(t, shields) + except Exception: + pass + out_map[p] = t + try: + rec.record( + cache_type="ftbquests", + file_id=rel_src, + path=p, + src=src_text, + dst=t, + cache_hit=False, + extra={"dst_file": dst.relative_to(out_dir).as_posix()}, + ) + except Exception: + pass + + # ✅ 確保此檔案在翻譯路徑也有 file_id + file_id = dst.as_posix() + _file_write_table[file_id] = (dst, out_map) + + # 在這之前先確保 file_id/_file_write_table 設定好了(下面會說加在哪) + def on_batch_flushed() -> None: + """批量寫入翻譯結果。""" + try: + touch.touch(file_id) + touch.flush(_writer) # 最小改動:每批也照樣寫,避免中斷損失 + except Exception: + # fallback + write_json_dict(dst, out_map) + def _fmt_eta(sec: float) -> str: """格式化剩餘時間。""" if sec <= 0: @@ -609,6 +650,16 @@ def on_batch_flushed() -> None: # ---- Dry-run 結尾摘要 ---- if dry_run: + + def _strip_runtime_fields(items: List[Dict[str, Any]]) -> List[Dict[str, Any]]: + """移除 dry-run preview 中不可 JSON 序列化的暫存欄位。""" + sanitized: List[Dict[str, Any]] = [] + for it in items: + sanitized.append( + {k: v for k, v in it.items() if k not in {"_shielded"}} + ) + return sanitized + batch_size = _get_default_batch_size("ftbquests", None) est_batches = ( math.ceil(global_total_to_translate / batch_size) @@ -627,7 +678,7 @@ def on_batch_flushed() -> None: # 原本:待翻譯 preview write_dry_run_preview( out_dir, - dry_preview_items, + _strip_runtime_fields(dry_preview_items), meta=meta, filename="_ftbquests_dry_run_preview.json", # 可選:明確檔名 ) @@ -635,7 +686,7 @@ def on_batch_flushed() -> None: # ✅ NEW:cache hit preview write_cache_hit_preview( out_dir, - all_cached_items, + _strip_runtime_fields(all_cached_items), filename="_ftbquests_dry_run_cache_hit_preview.json", meta=meta, ) diff --git a/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py b/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py index 5299f55f..e67daf2e 100644 --- a/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py +++ b/translation_tool/plugins/kubejs/kubejs_tooltip_lmtranslator.py @@ -85,15 +85,15 @@ def collect_items_from_mapping( shielded = shield_text(v) if shielded.skip_reason is not None: - # 不應翻譯(圖片/URL/事件/空白),直接寫入原文不經翻譯管線 + # 不應翻譯(圖片/URL/事件/空白),直接保留原文並標記 skip。 items.append( { "file": file_hint, "path": k, "source_text": v, - "text": v, # 保持原文 + "text": v, "cache_type": "kubejs", - "_shielded": shielded, # 供 unshield 回查(此情境無需還原) + "_shielded": shielded, "_skip_reason": shielded.skip_reason, } ) @@ -294,6 +294,12 @@ def _count_one(src: Path) -> Tuple[Path, int]: cache_rules=cache_rules, is_valid_hit=_is_valid_hit, ) + skip_items = [it for it in items_to_translate if it.get("_skip_reason")] + if skip_items: + cached_items.extend(skip_items) + items_to_translate = [ + it for it in items_to_translate if not it.get("_skip_reason") + ] # ✅ 跳過值已經是繁體中文的 items(節省 API + 避免簡體當英文翻) _split_off_tw_items(cached_items, items_to_translate) global_total_hit += len(cached_items) @@ -347,6 +353,13 @@ def _writer(file_id: str) -> None: is_valid_hit=_is_valid_hit, ) + skip_items = [it for it in items_to_translate if it.get("_skip_reason")] + if skip_items: + cached_items.extend(skip_items) + items_to_translate = [ + it for it in items_to_translate if not it.get("_skip_reason") + ] + # ✅ 跳過值已經是繁體中文的 items(從翻譯清單移至 cache hit) _split_off_tw_items(cached_items, items_to_translate) @@ -424,7 +437,22 @@ def _writer(file_id: str) -> None: # Dry-run: preview only (no API) # ------------------------- if dry_run: - dry_preview_items = all_miss_items[:2000] if all_miss_items else [] + + def _strip_runtime_fields(items: List[Dict[str, Any]]) -> List[Dict[str, Any]]: + """移除 dry-run preview 中不可 JSON 序列化的暫存欄位。""" + sanitized: List[Dict[str, Any]] = [] + for it in items: + sanitized.append( + {k: v for k, v in it.items() if k not in {"_shielded"}} + ) + return sanitized + + dry_preview_items = ( + _strip_runtime_fields(all_miss_items[:2000]) if all_miss_items else [] + ) + dry_hit_items = ( + _strip_runtime_fields(all_hit_items[:2000]) if all_hit_items else [] + ) preview_path = None try: @@ -446,7 +474,7 @@ def _writer(file_id: str) -> None: # ✅ NEW:cache hit preview hit_preview_path = write_cache_hit_preview( out_dir, - all_hit_items[:2000], # 或不切片 + dry_hit_items, filename="_kubejs_dry_run_cache_hit_preview.json", meta=meta, ) diff --git a/translation_tool/plugins/md/md_inject_qa.py b/translation_tool/plugins/md/md_inject_qa.py index fa5b3cc6..7726297f 100644 --- a/translation_tool/plugins/md/md_inject_qa.py +++ b/translation_tool/plugins/md/md_inject_qa.py @@ -86,6 +86,9 @@ def map_lang_in_rel_path( # ======== 語言資料夾段落映射:允許 en_us -> zh_tw,也允許來源是 zh_tw ======== +RE_LANG_SEG = re.compile(r"^(_?)([a-z]{2}_[a-z]{2})$", re.IGNORECASE) + + def map_lang_in_rel_path_allow_zh( rel_path: str, src_lang: str = "en_us", dst_lang: str = "zh_tw" ) -> tuple[str, str]: diff --git a/translation_tool/plugins/md/md_lmtranslator.py b/translation_tool/plugins/md/md_lmtranslator.py index 8212bce6..53a808b1 100644 --- a/translation_tool/plugins/md/md_lmtranslator.py +++ b/translation_tool/plugins/md/md_lmtranslator.py @@ -192,7 +192,7 @@ def translate_md_pending( all_unique_items: List[Dict[str, Any]] = [] already_zh_skipped = 0 - skip_skipped = 0 # 新增:統計因 skip_reason 跳過的項目 + skip_skipped = 0 for h, src in hash_to_src.items(): if is_already_zh(src): @@ -209,7 +209,7 @@ def translate_md_pending( "file": "md_pending_blocks", "path": h, "source_text": src, - "text": src, # 保持原文 + "text": src, "_shielded": shielded, "_skip_reason": shielded.skip_reason, } @@ -224,7 +224,7 @@ def translate_md_pending( "file": "md_pending_blocks", "path": h, # ✅ 用 content_hash 當 path(去重 + 快取 key 的一部分) "source_text": src, - "text": translate_text, + "text": shielded.clean, "_shielded": shielded, } ) @@ -309,8 +309,8 @@ def translate_md_pending( if shielded is not None and getattr(shielded, "shields", None): try: dst = unshield_text(dst, shielded.shields) - except Exception as e: - log_warning(f"[MD-LM] unshield 失敗: {e}") + except Exception: + pass hash_to_dst[h] = dst rec = TranslationRecorder() @@ -331,8 +331,8 @@ def on_translated_item(it: Dict[str, Any]) -> None: if shielded is not None and getattr(shielded, "shields", None): try: dst = unshield_text(dst, shielded.shields) - except Exception as e: - log_warning(f"[MD-LM] unshield 失敗: {e}") + except Exception: + pass hash_to_dst[h] = dst # 這裡 recorder 的 cache_type 用 md(方便你日後 QC) try: diff --git a/translation_tool/plugins/shared/rich_text_shield.py b/translation_tool/plugins/shared/rich_text_shield.py index 1c119557..05efd001 100644 --- a/translation_tool/plugins/shared/rich_text_shield.py +++ b/translation_tool/plugins/shared/rich_text_shield.py @@ -89,7 +89,7 @@ class ShieldedText: _counter_color: int = 0 _counter_item: int = 0 _counter_escaped: int = 0 -_counter_lock = threading.Lock() # 計數器執行緒安全鎖 +_counter_lock = threading.Lock() def _next_color_placeholder() -> str: