From 403f014ac0a6374cde93e6307a4c81b2e33da50b Mon Sep 17 00:00:00 2001 From: Belugary <53219544+Belugary@users.noreply.github.com> Date: Wed, 13 May 2026 13:00:19 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E6=96=B0=E5=A2=9E=20decode-images=20?= =?UTF-8?q?=E5=AD=90=E5=91=BD=E4=BB=A4(=E6=89=B9=E9=87=8F=E8=A7=A3?= =?UTF-8?q?=E5=AF=86=20.dat=20=E5=9B=BE=E7=89=87=E5=88=B0=E6=98=8E?= =?UTF-8?q?=E6=96=87=E5=9B=BE=E7=89=87=E6=A0=91)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## 问题 \`decode_image.py\` 目前只有 \`decrypt_dat_file()\` 单文件 API,以及 \`monitor_web\` 在收到新消息时\"按需解一张\"的路径。**没有\"一次性扫 attach 目录、产出明文图片树到固定路径\"的批量入口**。结果是任何想把微信图片做下游消费(数据分析、搜索索引、归档、第三方 viewer)的用户都得各自写一遍 walk + decrypt 的 wrapper,且各自约定输出布局,生态不收敛。 ## 修复 - \`decode_image.py\` 新增 \`decode_all_dats(attach_dir, out_dir, aes_key, xor_key, force, on_file)\` 函数,扫描 \`///Img/*.dat\` 并镜像产出 \`///.\`。 - \`main.py\` 新增 \`decode-images\` 子命令(早路由,跳过 \`check_wechat_running\` 和 \`ensure_keys\` —— 这条路径只读 \`.dat\` 文件,既不需要微信进程也不需要 DB 密钥)。 设计选择: - **输出布局 1:1 镜像 attach**,只做最小 path massage(去 \`Img/\`、去 \`_t/_h\` 缩略图后缀、换扩展名),不发明新结构。下游能用 \`md5(username)\` 反推路径,无需读 mapping 文件。 - **幂等性 = 按 basename 存在性 skip**,不做 mtime 比较 —— \`.dat\` 是 content-hash 命名(\`file_md5 = 文件内容 md5\`),实际上 write-once。\`--force\` 强制重解。 - **原子写**:解密先写 \`..tmp\`(同目录),\`os.replace\` 到正式路径。中断不留半个 jpg。残留 \`.tmp\` 不会被 skip 误判(glob 显式排除)。 - **错误隔离**:单文件失败计入 \`failed\` 继续下一个,stderr 打 \`[WARN]\` 指出相对路径。退出码 2 表示\"部分失败,产物部分可用\"。 - **V2 无 key**:计入 \`skipped_no_key\` 而非 \`failed\` —— 这是可恢复状态(跑 \`find_image_key_macos.py\` / \`find_image_key.py\` 后重跑即可),跟\"真失败\"区分对待。V1 / 老 XOR 不依赖 \`image_aes_key\`。 - **wxgf 容器**只产 \`.hevc\` 裸流,**不**做 mp4 转换:上游不引入 ffmpeg subprocess 依赖,转换是消费层职责。 - **CLI override**:\`--attach-dir\` / \`--decoded-dir\` / \`--aes-key\` / \`--xor-key\` / \`--force\` 都可覆盖 \`config.json\`,适合 CI / 多账号 / 容器化场景。 ## 测试 新文件 \`tests/test_decode_images_batch.py\`,13 个新测试: - \`PathParsingTests\` (4):glob 命中 / \`_t\` 后缀剥离 / \`_h\` 后缀剥离 / chat_hash + YYYY-MM 镜像 - \`IdempotentTests\` (3):已存在跳过 / \`--force\` 覆写 / 残留 \`.tmp\` 不误判 - \`AtomicWriteTests\` (3):成功路径无 \`.tmp\` / decrypt 返回 None 无 \`.tmp\` / decrypt 抛异常无 \`.tmp\` - \`V2NoKeyTests\` (2):V2 + 无 key → skipped_no_key / V1 + 无 key 仍解码 - \`CallbackTests\` (1):\`on_file\` 回调每文件触发 基线 183 → 196 通过(+13 新增),0 回归。\`decrypt_dat_file\` 用 mock 隔离(避免依赖真实加密图片);\`is_v2_format\` 走真实 magic 检测路径。 ## 范围 - \`decode_image.py\`:新增 \`decode_all_dats\` 函数,134 行,纯加,不改任何现有 API。 - \`main.py\`:新增 \`_run_decode_images\` helper + 早路由 + 用法 hint,104 行加 2 行删。无 backward-compat 影响。 - \`tests/test_decode_images_batch.py\`:新增,295 行。合成 fixture(假 V1/V2 magic + mock decrypt_dat_file),不依赖真实加密素材。 --- decode_image.py | 134 ++++++++++++++ main.py | 104 ++++++++++- tests/test_decode_images_batch.py | 295 ++++++++++++++++++++++++++++++ 3 files changed, 531 insertions(+), 2 deletions(-) create mode 100644 tests/test_decode_images_batch.py diff --git a/decode_image.py b/decode_image.py index e7cb0fa..4a4442b 100644 --- a/decode_image.py +++ b/decode_image.py @@ -277,6 +277,140 @@ def decrypt_dat_file(dat_path, out_path=None, aes_key=None, xor_key=0x88): return xor_decrypt_file(dat_path, out_path) +def decode_all_dats(attach_dir, out_dir, aes_key=None, xor_key=0x88, + force=False, progress_every=200, on_file=None): + """批量解密 attach_dir 下所有 .dat 图片到 out_dir 的镜像目录树。 + + 输入路径形态(微信本地约定): + ///Img/[_t|_h].dat + + 其中 chat_hash = md5(username).hexdigest(),username 是 wxid 或 + @chatroom;_t/_h 分别是缩略图 / 高清缩略图后缀。 + + 输出路径形态(镜像 + 移除 _t/_h 缩略图后缀,平铺到原图 basename): + ///. + + 其中 由 magic 自动检测(jpg / png / gif / webp / hevc 等)。 + wxgf 容器输出 .hevc;不在 upstream 做 mp4 转换(scope 留给下游)。 + + 幂等性:目标存在(任何扩展名,基于 basename)时跳过,无需 mtime 比较 —— + .dat 是 content-hash 命名,实际上 write-once。force=True 强制重解。 + + 原子写:解密先写到 ..tmp(同目录),`os.replace` 重命名 + 到最终路径,中断不留半文件。 + + 错误隔离:单文件失败不阻塞批次。V2 文件遇到 aes_key=None 计入 + skipped_no_key(可恢复:跑 find_image_key_macos.py 提取 key 后重跑)。 + + Args: + attach_dir: 微信 msg/attach 根目录(含 chat_hash 子目录) + out_dir: 输出根目录 + aes_key: V2 AES key(16 字节 str/bytes);V1 / 老 XOR 不需要 + xor_key: V2 XOR key(默认 0x88) + force: True 时忽略已存在目标重新解密 + progress_every: 每解 N 个文件打一行进度到 stderr;None 关闭(测试用) + on_file: 可选回调 (i, total, dat_path, status, fmt) 每文件调用一次, + status ∈ {"decoded", "skipped", "skipped_no_key", "failed"} + + Returns: + dict {decoded, skipped, skipped_no_key, failed, total, formats} + formats: dict[ext, count] + """ + pattern = os.path.join(attach_dir, "*", "*", "Img", "*.dat") + dat_files = sorted(glob.glob(pattern)) + + decoded = 0 + skipped = 0 + skipped_no_key = 0 + failed = 0 + formats = {} + + for i, dat_path in enumerate(dat_files): + rel = os.path.relpath(dat_path, attach_dir) + parts = rel.split(os.sep) + if len(parts) != 4 or parts[2] != "Img": + failed += 1 + print(f"[WARN] 跳过非标准路径: {rel}", file=sys.stderr) + if on_file: + on_file(i, len(dat_files), dat_path, "failed", None) + continue + chat_hash, ym, _img, fname = parts + basename = os.path.splitext(fname)[0] # 去 .dat + for suffix in ("_t", "_h"): + if basename.endswith(suffix): + basename = basename[:-len(suffix)] + break + + target_dir = os.path.join(out_dir, chat_hash, ym) + + # 幂等性:目标 basename 已存在(任何 ext,排除 .tmp) + if not force: + existing = [ + p for p in glob.glob(os.path.join(target_dir, f"{basename}.*")) + if not p.endswith(".tmp") + ] + if existing: + skipped += 1 + if on_file: + on_file(i, len(dat_files), dat_path, "skipped", None) + continue + + # V2 文件需要 key;无 key 时计入 skipped_no_key + if is_v2_format(dat_path) and aes_key is None: + skipped_no_key += 1 + if on_file: + on_file(i, len(dat_files), dat_path, "skipped_no_key", None) + if progress_every and (i + 1) % progress_every == 0: + print( + f" ...扫描 {i+1}/{len(dat_files)} (解码 {decoded}, 跳过 {skipped}, " + f"无 key {skipped_no_key}, 失败 {failed})", + file=sys.stderr, + ) + continue + + os.makedirs(target_dir, exist_ok=True) + tmp_path = os.path.join(target_dir, f"{basename}.unknown.tmp") + fmt = None + try: + result_path, fmt = decrypt_dat_file(dat_path, tmp_path, aes_key, xor_key) + if result_path is None or fmt is None: + failed += 1 + if os.path.exists(tmp_path): + try: os.remove(tmp_path) + except OSError: pass + else: + final_path = os.path.join(target_dir, f"{basename}.{fmt}") + os.replace(result_path, final_path) + decoded += 1 + formats[fmt] = formats.get(fmt, 0) + 1 + except Exception as e: + failed += 1 + if os.path.exists(tmp_path): + try: os.remove(tmp_path) + except OSError: pass + print(f"[WARN] {rel}: {e}", file=sys.stderr) + + if on_file: + status = "decoded" if fmt else "failed" + on_file(i, len(dat_files), dat_path, status, fmt) + + if progress_every and (i + 1) % progress_every == 0: + print( + f" ...扫描 {i+1}/{len(dat_files)} (解码 {decoded}, 跳过 {skipped}, " + f"无 key {skipped_no_key}, 失败 {failed})", + file=sys.stderr, + ) + + return { + "decoded": decoded, + "skipped": skipped, + "skipped_no_key": skipped_no_key, + "failed": failed, + "total": len(dat_files), + "formats": formats, + } + + def extract_md5_from_packed_info(blob): """从 message_resource.db 的 packed_info (protobuf) 中提取文件 MD5 diff --git a/main.py b/main.py index 9ee522c..4376523 100644 --- a/main.py +++ b/main.py @@ -28,6 +28,97 @@ def check_wechat_running(): return False +def _run_decode_images(cfg, argv): + """`decode-images` 子命令:批量把 .dat 图片解密成明文图片树。 + + 与 decrypt 不同,decode-images **不需要** 微信进程在运行,也不需要 DB 密钥 + (只读已存在的 .dat 文件;V2 文件用 config.json 里的 image_aes_key)。 + """ + import argparse + from decode_image import decode_all_dats + + parser = argparse.ArgumentParser( + prog="main.py decode-images", + description=( + "批量解密微信本地 .dat 图片到明文图片树。" + "区别于 decode_image.py 单文件 CLI,本子命令扫描 attach_dir 下" + "全部 .dat,镜像目录结构产出明文(jpg / png / gif / webp / hevc)。" + ), + ) + default_base = cfg.get("wechat_base_dir") or os.path.dirname(cfg["db_dir"]) + default_attach = os.path.join(default_base, "msg", "attach") + default_out = cfg.get("decoded_image_dir", "decoded_images") + parser.add_argument( + "--attach-dir", default=None, + help=f"微信 msg/attach 根目录,覆盖默认推断(默认: {default_attach})", + ) + parser.add_argument( + "--decoded-dir", default=None, + help=f"明文图片输出根目录,覆盖 config.json 的 decoded_image_dir(默认: {default_out})", + ) + parser.add_argument( + "--aes-key", default=None, + help="V2 AES key(16 字节 ASCII 字符串),覆盖 config.json 的 image_aes_key", + ) + parser.add_argument( + "--xor-key", default=None, + help="V2 XOR key(可十进制或 0x 十六进制),覆盖 config.json 的 image_xor_key(默认: 0x88)", + ) + parser.add_argument( + "--force", action="store_true", + help="忽略已存在目标重新解密(默认按 basename 跳过)", + ) + args = parser.parse_args(argv) + + attach_dir = args.attach_dir or default_attach + out_dir = args.decoded_dir or default_out + aes_key = args.aes_key if args.aes_key is not None else cfg.get("image_aes_key") + xor_key_raw = args.xor_key if args.xor_key is not None else cfg.get("image_xor_key", 0x88) + if isinstance(xor_key_raw, str): + xor_key = int(xor_key_raw, 0) + else: + xor_key = xor_key_raw + + if not os.path.isdir(attach_dir): + print(f"[ERROR] attach 目录不存在: {attach_dir}", file=sys.stderr) + sys.exit(1) + + if aes_key is None: + print( + "[NOTE] 未配置 image_aes_key,V2 加密图片将被跳过(计入 skipped_no_key);" + "V1 / 老 XOR 图片不受影响。提取 V2 key 见 README 的图片解密章节。", + file=sys.stderr, + ) + + print(f" attach_dir = {attach_dir}") + print(f" out_dir = {out_dir}") + print(f" aes_key = {'已配置' if aes_key else '未配置'}") + print(f" xor_key = 0x{xor_key:02x}") + print(f" force = {args.force}") + print() + + stats = decode_all_dats( + attach_dir=attach_dir, + out_dir=out_dir, + aes_key=aes_key, + xor_key=xor_key, + force=args.force, + ) + + print() + print("=" * 60) + print(f"扫描 {stats['total']} 个 .dat 文件") + print(f" 解码: {stats['decoded']} 跳过(已存在): {stats['skipped']} " + f"无 key 跳过: {stats['skipped_no_key']} 失败: {stats['failed']}") + if stats["formats"]: + fmt_summary = ", ".join(f"{ext}={n}" for ext, n in sorted(stats["formats"].items())) + print(f" 按格式: {fmt_summary}") + print(f"输出在: {out_dir}") + + if stats["failed"] > 0: + sys.exit(2) + + def ensure_keys(keys_file, db_dir): """确保密钥文件存在且匹配当前 db_dir,否则重新提取""" if os.path.exists(keys_file): @@ -84,6 +175,13 @@ def main(): from config import load_config cfg = load_config() + # 早路由:decode-images 不需要微信进程在运行,也不需要 DB 密钥 + if len(sys.argv) > 1 and sys.argv[1] == "decode-images": + print("[*] 批量解密图片...") + print() + _run_decode_images(cfg, sys.argv[2:]) + return + # 2. 检查微信进程 if not check_wechat_running(): print(f"[!] 未检测到微信进程 ({cfg.get('wechat_process', 'WeChat')})") @@ -111,8 +209,10 @@ def main(): print(f"[!] 未知命令: {cmd}") print() print("用法:") - print(" python main.py 启动实时消息监听 (Web UI)") - print(" python main.py decrypt 解密全部数据库到 decrypted/") + print(" python main.py 启动实时消息监听 (Web UI)") + print(" python main.py decrypt 解密全部数据库到 decrypted/") + print(" python main.py decode-images 批量解密 .dat 图片到 decoded_image_dir/") + print(" python main.py decode-images --help 查看 decode-images 全部选项") sys.exit(1) diff --git a/tests/test_decode_images_batch.py b/tests/test_decode_images_batch.py new file mode 100644 index 0000000..6bb4180 --- /dev/null +++ b/tests/test_decode_images_batch.py @@ -0,0 +1,295 @@ +"""decode_image.decode_all_dats() batch CLI 行为测试。 + +覆盖: +- 路径扫描:glob 命中 attach///Img/*.dat +- 路径解析:chat_hash / YYYY-MM 提取,_t / _h 后缀移除归并到原图 basename +- 幂等性:目标 basename 已存在(任何扩展名)时跳过;--force 强制重解 +- 原子写:写到 tmp 再 os.replace;失败/异常路径不留 .tmp +- V2 无 key:计入 skipped_no_key 而非 failed +- 错误隔离:单文件异常不阻塞批次;返回失败计数 + +decrypt_dat_file 用 mock 隔离(避免依赖真实加密图片);is_v2_format +单独覆盖真实 magic 检测路径。 +""" +import os +import struct +import tempfile +import unittest +from contextlib import redirect_stderr +import io +from unittest.mock import patch + +import decode_image + + +def _write(path, data): + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, "wb") as f: + f.write(data) + + +def _v2_magic_bytes(): + # 仅用于让 is_v2_format() 返回 True + return decode_image.V2_MAGIC_FULL + struct.pack("