From e6272ae087e4ef68f59c497ecdc67a6d97f95983 Mon Sep 17 00:00:00 2001 From: Aldrich Chen <109075336+Chen17-sq@users.noreply.github.com> Date: Mon, 27 Apr 2026 20:27:11 +0800 Subject: [PATCH 1/8] fix(keys/macos): report correct key count after init The C binary find_all_keys_macos writes JSON in the schema {"db.path": {"enc_key": "..."}} (without a salt field), but the Python wrapper at scanner_macos.py was filtering on both enc_key and salt, so every entry was rejected and key_map ended up empty. init.py then echoed 'Extracted 0 keys' even though all_keys.json was correctly written and downstream queries worked. Build key_map keyed by rel_path instead, matching the C binary's actual output and how core/key_utils.get_key_info already does lookups. Fixes #2 --- wechat_cli/keys/scanner_macos.py | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/wechat_cli/keys/scanner_macos.py b/wechat_cli/keys/scanner_macos.py index ea316ee..f095e2f 100644 --- a/wechat_cli/keys/scanner_macos.py +++ b/wechat_cli/keys/scanner_macos.py @@ -127,7 +127,7 @@ def extract_keys(db_dir, output_path, pid=None): pid: 未使用(C 二进制自动检测进程) Returns: - dict: salt_hex -> enc_key_hex 映射 + dict: rel_path -> enc_key_hex 映射(与 C 二进制输出 schema 一致) """ import json @@ -212,11 +212,14 @@ def extract_keys(db_dir, output_path, pid=None): if os.path.abspath(c_output) != os.path.abspath(output_path): os.remove(c_output) - # 构建 salt -> key 映射 - key_map = {} - for rel, info in keys_data.items(): - if isinstance(info, dict) and "enc_key" in info and "salt" in info: - key_map[info["salt"]] = info["enc_key"] + # 构建 rel_path -> enc_key 映射。C 二进制输出 schema 是 + # {"db.path": {"enc_key": "..."}}(不含 salt),下游 get_key_info + # 也按 rel_path 查询,所以这里保持同样的索引方式。 + key_map = { + rel: info["enc_key"] + for rel, info in keys_data.items() + if isinstance(info, dict) and "enc_key" in info + } print(f"\n[+] 提取到 {len(key_map)} 个密钥,保存到: {output_path}") return key_map From 8ac0c52cac2be3d372f799b93a135bb7be76e5a7 Mon Sep 17 00:00:00 2001 From: Jack Song Date: Fri, 18 Sep 2026 08:37:11 +1000 Subject: [PATCH 2/8] feat: expose contact labels, phone, and stable message IDs (ROAD-354) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Decode contact.extra_buffer protobuf blob: field 30 (label IDs, joined to contact_label for names) and field 14->2->1 (mobile number) via one shared varint/tag parser. Surfaced as labels/phone in contacts list JSON and contacts --detail (labels, label_ids, phone). - Add server_id to the message query and return structured entries from collect_chat_history; history JSON now emits local_id/server_id/timestamp/time/sender/text per message, enabling wechat_message_id-based dedup. Text output and export unchanged. - Cherry-pick huohuoer/wechat-cli#4: fix key_map indexing so init no longer misreports "提取到 0 个密钥" after a successful scan. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- wechat_cli/commands/contacts.py | 6 ++ wechat_cli/commands/export.py | 5 +- wechat_cli/commands/history.py | 16 +++-- wechat_cli/core/contacts.py | 120 ++++++++++++++++++++++++++++++-- wechat_cli/core/messages.py | 27 ++++--- 5 files changed, 153 insertions(+), 21 deletions(-) diff --git a/wechat_cli/commands/contacts.py b/wechat_cli/commands/contacts.py index fd2206b..b568941 100644 --- a/wechat_cli/commands/contacts.py +++ b/wechat_cli/commands/contacts.py @@ -50,6 +50,8 @@ def contacts(ctx, query, detail, limit, fmt): line = f"{display} ({c['username']})" if c['remark']: line += f" 备注: {c['remark']}" + if c.get('labels'): + line += f" 标签: {'、'.join(c['labels'])}" lines.append(line) output(header + "\n\n" + "\n".join(lines), 'text') @@ -80,6 +82,10 @@ def _show_detail(app, name_or_id, fmt): lines.append(f"wxid: {info['username']}") if info['description']: lines.append(f"个性签名: {info['description']}") + if info.get('labels'): + lines.append(f"标签: {'、'.join(info['labels'])}") + if info.get('phone'): + lines.append(f"手机号: {info['phone']}") if info['is_group']: lines.append("类型: 群聊") elif info['is_subscription']: diff --git a/wechat_cli/commands/export.py b/wechat_cli/commands/export.py index c1be606..80a6b7f 100644 --- a/wechat_cli/commands/export.py +++ b/wechat_cli/commands/export.py @@ -48,15 +48,16 @@ def export(ctx, chat_name, fmt, output_path, start_time, end_time, limit): ctx.exit(1) names = get_contact_names(app.cache, app.decrypted_dir) - lines, failures = collect_chat_history( + entries, failures = collect_chat_history( chat_ctx, names, app.display_name_fn, start_ts=start_ts, end_ts=end_ts, limit=limit, offset=0, ) - if not lines: + if not entries: click.echo(f"{chat_ctx['display_name']} 无消息记录", err=True) ctx.exit(0) + lines = [e['line'] for e in entries] now = datetime.now().strftime('%Y-%m-%d %H:%M') chat_type = "群聊" if chat_ctx['is_group'] else "私聊" time_range = f"{start_time or '最早'} ~ {end_time or '最新'}" diff --git a/wechat_cli/commands/history.py b/wechat_cli/commands/history.py index 3e88452..c1f8d21 100644 --- a/wechat_cli/commands/history.py +++ b/wechat_cli/commands/history.py @@ -53,35 +53,39 @@ def history(ctx, chat_name, limit, offset, start_time, end_time, fmt, msg_type, names = get_contact_names(app.cache, app.decrypted_dir) type_filter = MSG_TYPE_FILTERS[msg_type] if msg_type else None - lines, failures = collect_chat_history( + entries, failures = collect_chat_history( chat_ctx, names, app.display_name_fn, start_ts=start_ts, end_ts=end_ts, limit=limit, offset=offset, msg_type_filter=type_filter, resolve_media=media, db_dir=app.db_dir, ) if fmt == 'json': + messages = [ + {k: e[k] for k in ('local_id', 'server_id', 'timestamp', 'time', 'sender', 'text')} + for e in entries + ] output({ 'chat': chat_ctx['display_name'], 'username': chat_ctx['username'], 'is_group': chat_ctx['is_group'], - 'count': len(lines), + 'count': len(messages), 'offset': offset, 'limit': limit, 'start_time': start_time or None, 'end_time': end_time or None, 'type': msg_type or None, - 'messages': lines, + 'messages': messages, 'failures': failures if failures else None, }, 'json') else: - header = f"{chat_ctx['display_name']} 的消息记录(返回 {len(lines)} 条,offset={offset}, limit={limit})" + header = f"{chat_ctx['display_name']} 的消息记录(返回 {len(entries)} 条,offset={offset}, limit={limit})" if chat_ctx['is_group']: header += " [群聊]" if start_time or end_time: header += f"\n时间范围: {start_time or '最早'} ~ {end_time or '最新'}" if failures: header += "\n查询失败: " + ";".join(failures) - if lines: - output(header + ":\n\n" + "\n".join(lines), 'text') + if entries: + output(header + ":\n\n" + "\n".join(e['line'] for e in entries), 'text') else: output(f"{chat_ctx['display_name']} 无消息记录", 'text') diff --git a/wechat_cli/core/contacts.py b/wechat_cli/core/contacts.py index ed1cd51..e9422eb 100644 --- a/wechat_cli/core/contacts.py +++ b/wechat_cli/core/contacts.py @@ -10,16 +10,121 @@ _self_username = None +# ---- extra_buffer protobuf 解码 ---- +# contact.extra_buffer 是 protobuf BLOB,多个字段共用一列,靠 field number 区分: +# field 30 → 逗号分隔的 contact_label.label_id_ 列表(标签) +# field 14 → 2 → 1 → 手机号(嵌套子消息,纯字符串,无国家码) + +def _read_varint(buf, i): + result = 0 + shift = 0 + while True: + b = buf[i] + i += 1 + result |= (b & 0x7f) << shift + if not (b & 0x80): + break + shift += 7 + return result, i + + +def _parse_protobuf_fields(data): + """通用 protobuf 解析。返回 [(field_no, wire_type, value)],wt=0→int, wt=2→bytes。""" + i = 0 + fields = [] + while i < len(data): + tag, i = _read_varint(data, i) + fno, wt = tag >> 3, tag & 7 + if wt == 0: + val, i = _read_varint(data, i) + elif wt == 2: + ln, i = _read_varint(data, i) + val = data[i:i + ln] + i += ln + else: + break + fields.append((fno, wt, val)) + return fields + + +def _split_label_ids(raw): + ids = [] + for part in re.split(r'[,,;;|\s]+', raw): + if part.isdigit(): + ids.append(int(part)) + return ids + + +def _decode_extra_labels(extra_buffer, label_names=None): + """extra_buffer field 30 → (label_ids, label_names)。""" + try: + fields = _parse_protobuf_fields(extra_buffer) + except Exception: + return [], [] + for fno, wt, val in fields: + if fno == 30 and wt == 2: + raw = val.decode('utf-8', errors='replace') + ids = _split_label_ids(raw) + names = [label_names.get(i) for i in ids] if label_names else [] + return ids, [n for n in names if n] + return [], [] + + +def _decode_extra_phone(extra_buffer): + """extra_buffer field 14 → field 2 → field 1 → 手机号字符串。""" + try: + for fno, wt, val in _parse_protobuf_fields(extra_buffer): + if fno != 14 or wt != 2: + continue + for fno2, wt2, val2 in _parse_protobuf_fields(val): + if fno2 != 2 or wt2 != 2: + continue + for fno3, wt3, val3 in _parse_protobuf_fields(val2): + if fno3 == 1 and wt3 == 2: + return val3.decode('utf-8', errors='replace') + except Exception: + pass + return '' + + +def _decode_extra_buffer(extra_buffer, label_names=None): + """返回 (label_ids, labels, phone)。""" + if not extra_buffer: + return [], [], '' + label_ids, labels = _decode_extra_labels(extra_buffer, label_names) + phone = _decode_extra_phone(extra_buffer) + return label_ids, labels, phone + + +def _load_label_names(conn): + try: + return {lid: name for lid, name in conn.execute( + "SELECT label_id_, label_name_ FROM contact_label" + ).fetchall()} + except sqlite3.Error: + return {} + + def _load_contacts_from(db_path): names = {} full = [] conn = sqlite3.connect(db_path) try: - for r in conn.execute("SELECT username, nick_name, remark FROM contact").fetchall(): - uname, nick, remark = r + label_names = _load_label_names(conn) + for r in conn.execute( + "SELECT username, nick_name, remark, extra_buffer FROM contact" + ).fetchall(): + uname, nick, remark, extra_buffer = r display = remark if remark else nick if nick else uname names[uname] = display - full.append({'username': uname, 'nick_name': nick or '', 'remark': remark or ''}) + label_ids, labels, phone = _decode_extra_buffer(extra_buffer, label_names) + full.append({ + 'username': uname, + 'nick_name': nick or '', + 'remark': remark or '', + 'labels': labels, + 'phone': phone, + }) finally: conn.close() return names, full @@ -168,15 +273,17 @@ def get_contact_detail(username, cache, decrypted_dir): conn = sqlite3.connect(db_path) try: + label_names = _load_label_names(conn) row = conn.execute( "SELECT username, nick_name, remark, alias, description, " - "small_head_url, big_head_url, verify_flag, local_type " + "small_head_url, big_head_url, verify_flag, local_type, extra_buffer " "FROM contact WHERE username = ?", (username,) ).fetchone() if not row: return None - uname, nick, remark, alias, desc, small_url, big_url, verify, ltype = row + uname, nick, remark, alias, desc, small_url, big_url, verify, ltype, extra_buffer = row + label_ids, labels, phone = _decode_extra_buffer(extra_buffer, label_names) return { 'username': uname, 'nick_name': nick or '', @@ -186,6 +293,9 @@ def get_contact_detail(username, cache, decrypted_dir): 'avatar': small_url or big_url or '', 'verify_flag': verify or 0, 'local_type': ltype, + 'labels': labels, + 'label_ids': label_ids, + 'phone': phone, 'is_group': '@chatroom' in uname, 'is_subscription': uname.startswith('gh_'), } diff --git a/wechat_cli/core/messages.py b/wechat_cli/core/messages.py index d62ef33..63af074 100644 --- a/wechat_cli/core/messages.py +++ b/wechat_cli/core/messages.py @@ -410,7 +410,7 @@ def _query_messages(conn, table_name, start_ts=None, end_ts=None, keyword='', li clauses, params = _build_message_filters(start_ts, end_ts, keyword, msg_type_filter) where_sql = f"WHERE {' AND '.join(clauses)}" if clauses else '' sql = f""" - SELECT local_id, local_type, create_time, real_sender_id, message_content, + SELECT local_id, server_id, local_type, create_time, real_sender_id, message_content, WCDB_CT_message_content FROM [{table_name}] {where_sql} @@ -510,8 +510,8 @@ def _page_ranked_entries(entries, limit, offset): # ---- 构建行 ---- -def _build_history_line(row, ctx, names, id_to_username, display_name_fn, resolve_media=False, db_dir=None): - local_id, local_type, create_time, real_sender_id, content, ct = row +def _build_history_entry(row, ctx, names, id_to_username, display_name_fn, resolve_media=False, db_dir=None): + local_id, server_id, local_type, create_time, real_sender_id, content, ct = row time_str = datetime.fromtimestamp(create_time).strftime('%Y-%m-%d %H:%M') content = decompress_content(content, ct) if content is None: @@ -524,12 +524,23 @@ def _build_history_line(row, ctx, names, id_to_username, display_name_fn, resolv real_sender_id, sender, ctx['is_group'], ctx['username'], ctx['display_name'], names, id_to_username, display_name_fn ) if sender_label: - return create_time, f'[{time_str}] {sender_label}: {text}' - return create_time, f'[{time_str}] {text}' + line = f'[{time_str}] {sender_label}: {text}' + else: + line = f'[{time_str}] {text}' + entry = { + 'local_id': local_id, + 'server_id': server_id, + 'timestamp': create_time, + 'time': time_str, + 'sender': sender_label, + 'text': text, + 'line': line, + } + return create_time, entry def _build_search_entry(row, ctx, names, id_to_username, display_name_fn, resolve_media=False, db_dir=None): - local_id, local_type, create_time, real_sender_id, content, ct = row + local_id, server_id, local_type, create_time, real_sender_id, content, ct = row content = decompress_content(content, ct) if content is None: return None @@ -571,7 +582,7 @@ def collect_chat_history(ctx, names, display_name_fn, start_ts=None, end_ts=None fetch_offset += len(rows) for row in rows: try: - collected.append(_build_history_line(row, table_ctx, names, id_to_username, display_name_fn, resolve_media=resolve_media, db_dir=db_dir)) + collected.append(_build_history_entry(row, table_ctx, names, id_to_username, display_name_fn, resolve_media=resolve_media, db_dir=db_dir)) except Exception as e: failures.append(f"local_id={row[0]}: {e}") if len(collected) - before >= candidate_limit: @@ -582,7 +593,7 @@ def collect_chat_history(ctx, names, display_name_fn, start_ts=None, end_ts=None failures.append(f"{table_ctx['db_path']}: {e}") paged = _page_ranked_entries(collected, limit, offset) - return [line for _, line in paged], failures + return [entry for _, entry in paged], failures # ---- 搜索查询 ---- From 3c39942be8efba4eda39eee6083a27774fe6653a Mon Sep 17 00:00:00 2001 From: Jack Song Date: Fri, 18 Sep 2026 08:42:47 +1000 Subject: [PATCH 3/8] fix: don't cache torn decrypts or clobber configured db_dir MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - db_cache: validate decrypted output with PRAGMA integrity_check and retry before caching — a torn read (WeChat checkpointing mid-decrypt) was previously cached by mtime and served forever, breaking all contact/message queries until manual cache removal. - init: reuse existing config db_dir before auto-detect, so init --force cannot silently switch accounts when multiple xwechat_files db_storage dirs exist. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- wechat_cli/commands/init.py | 10 +++++++- wechat_cli/core/db_cache.py | 47 +++++++++++++++++++++++++++++++++---- 2 files changed, 51 insertions(+), 6 deletions(-) diff --git a/wechat_cli/commands/init.py b/wechat_cli/commands/init.py index 5396a4f..5905984 100644 --- a/wechat_cli/commands/init.py +++ b/wechat_cli/commands/init.py @@ -26,7 +26,15 @@ def init(db_dir, force): # 2. 创建状态目录 os.makedirs(STATE_DIR, exist_ok=True) - # 3. 确定 db_dir + # 3. 确定 db_dir(优先沿用已有配置,避免多账号时自动检测选错) + if db_dir is None and os.path.exists(CONFIG_FILE): + try: + with open(CONFIG_FILE, encoding="utf-8") as f: + existing = json.load(f).get("db_dir") + if existing and os.path.isdir(existing): + db_dir = existing + except (json.JSONDecodeError, OSError): + pass if db_dir is None: db_dir = auto_detect_db_dir() if db_dir is None: diff --git a/wechat_cli/core/db_cache.py b/wechat_cli/core/db_cache.py index 2cf5f64..98e09e4 100644 --- a/wechat_cli/core/db_cache.py +++ b/wechat_cli/core/db_cache.py @@ -3,11 +3,39 @@ import hashlib import json import os +import sqlite3 import tempfile +import time -from .crypto import full_decrypt, decrypt_wal +from .crypto import full_decrypt, decrypt_wal, SQLITE_HDR from .key_utils import get_key_info +_DECRYPT_ATTEMPTS = 3 +_DECRYPT_RETRY_DELAY = 0.3 + + +def _has_sqlite_header(path): + try: + with open(path, 'rb') as f: + return f.read(len(SQLITE_HDR)) == SQLITE_HDR + except OSError: + return False + + +def _is_valid_sqlite(path): + """校验解密结果完整性(源库可能被微信实时写入,解密可能读到撕裂页)。""" + if not _has_sqlite_header(path): + return False + try: + conn = sqlite3.connect(path) + try: + row = conn.execute("PRAGMA integrity_check").fetchone() + return bool(row) and row[0] == 'ok' + finally: + conn.close() + except (OSError, sqlite3.Error): + return False + class DBCache: CACHE_DIR = os.path.join(tempfile.gettempdir(), "wechat_cli_cache") @@ -75,14 +103,23 @@ def get(self, rel_key): if rel_key in self._cache: c_db_mt, c_wal_mt, c_path = self._cache[rel_key] - if c_db_mt == db_mtime and c_wal_mt == wal_mtime and os.path.exists(c_path): + if c_db_mt == db_mtime and c_wal_mt == wal_mtime and _has_sqlite_header(c_path): return c_path tmp_path = self._cache_path(rel_key) enc_key = bytes.fromhex(key_info["enc_key"]) - full_decrypt(db_path, tmp_path, enc_key) - if os.path.exists(wal_path): - decrypt_wal(wal_path, tmp_path, enc_key) + for attempt in range(_DECRYPT_ATTEMPTS): + full_decrypt(db_path, tmp_path, enc_key) + if os.path.exists(wal_path): + decrypt_wal(wal_path, tmp_path, enc_key) + if _is_valid_sqlite(tmp_path): + break + if attempt < _DECRYPT_ATTEMPTS - 1: + time.sleep(_DECRYPT_RETRY_DELAY) + else: + self._cache.pop(rel_key, None) + self._save_persistent_cache() + return None self._cache[rel_key] = (db_mtime, wal_mtime, tmp_path) self._save_persistent_cache() return tmp_path From 72ad3c03a7996e3df32c77ba39307bbfe58ce5cb Mon Sep 17 00:00:00 2001 From: Jack Song Date: Fri, 18 Sep 2026 08:43:27 +1000 Subject: [PATCH 4/8] chore: gitignore wechat_ent.plist re-sign artifact Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .gitignore | 1 + 1 file changed, 1 insertion(+) diff --git a/.gitignore b/.gitignore index 24eecf2..00e7fb8 100644 --- a/.gitignore +++ b/.gitignore @@ -33,6 +33,7 @@ Thumbs.db *.db *.db-wal *.db-shm +wechat_ent.plist # Sensitive data — NEVER commit *.json From e06a83f289a626f6e7cdf5f2de902afbd5e77167 Mon Sep 17 00:00:00 2001 From: Jack Song Date: Fri, 18 Sep 2026 08:44:17 +1000 Subject: [PATCH 5/8] docs: add AGENTS.md (+ CLAUDE.md import) for agent consumers Captures the read-only security posture, init/re-sign setup, layout, data-model gotchas (extra_buffer fields, Msg_ table naming, live-DB decrypt validation), and verification commands learned implementing ROAD-354. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- AGENTS.md | 60 +++++++++++++++++++++++++++++++++++++++++++++++++++++++ CLAUDE.md | 1 + 2 files changed, 61 insertions(+) create mode 100644 AGENTS.md create mode 100644 CLAUDE.md diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..6b12da8 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,60 @@ +# AGENTS.md + +Read-only WeChat data query CLI (fork of `huohuoer/wechat-cli`) — decrypts local +WeChat 4.x databases and exposes messages, contacts, sessions, etc. as JSON for +LLM/agent consumption. + +## Hard rules + +- **Read-only by design.** No send/write capability — do not add one (security + posture per LANDIT ROAD-336 review). UI automation of WeChat is out of scope. +- **Sensitive data everywhere.** `~/.wechat-cli/all_keys.json` holds SQLCipher + keys; decrypted DBs land in `$TMPDIR/wechat_cli_cache` and + `~/.wechat-cli/decrypted`. Never commit keys, `*.db*`, or `*.json` output dumps + (already gitignored). Don't echo key material or bulk personal data into logs. +- **Chinese comments/docstrings** are the codebase convention — match them. + +## Setup + +```bash +python3 -m venv .venv && .venv/bin/pip install -e . +.venv/bin/wechat-cli init # one-time: extract keys (WeChat must be running + logged in) +.venv/bin/wechat-cli sessions # smoke test +``` + +`init` runs a bundled C binary (`wechat_cli/bin/find_all_keys_macos.*`) that +reads WeChat process memory via `task_for_pid`. If blocked, it re-signs +WeChat preserving entitlements (adds `get-task-allow`), then you must restart +WeChat and re-run `init`. Works without sudo once re-signed. +`wechat_ent.plist` produced during re-sign is a temp artifact — delete it. + +## Layout + +- `wechat_cli/commands/` — one click command per file, registered in `main.py` +- `wechat_cli/core/` — `context` (AppContext singleton), `config` (`~/.wechat-cli`), + `db_cache` (mtime-keyed decrypt cache), `crypto` (SQLCipher AES-256-CBC + page/WAL decrypt), `contacts`, `messages`, `key_utils` +- `wechat_cli/keys/` — platform key scanners; `output/formatter.py` — `output(data, fmt)` + +## Data model gotchas + +- `contact.db`: `contact` table + `contact_label` (label id→name only; membership + is **not** a table — it's `contact.extra_buffer` protobuf field 30). +- `extra_buffer` protobuf: field 30 = comma-separated label_ids; field 14→2→1 = + mobile number. Shared decoder lives in `core/contacts.py` — extend there, don't + fork a second parser. +- Message tables: `Msg_` across `message/message_*.db`; + `Name2Id` maps `real_sender_id`→username. Message identity = `local_id`/`server_id`. +- Live DBs are read while WeChat writes — `db_cache` validates decrypts with + `PRAGMA integrity_check` and retries; don't bypass it by copying DB files. +- `init` reuses `config.json`'s `db_dir`; machines may have several + `xwechat_files/*/db_storage` accounts — never auto-switch. + +## Verify + +No test suite. Verify against the live DB: + +```bash +.venv/bin/wechat-cli contacts --detail "" # labels/phone +.venv/bin/wechat-cli history "" --limit 3 # local_id/server_id in JSON +``` diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 0000000..43c994c --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1 @@ +@AGENTS.md From 30740a91a8c6e9bcf5237d753459609e1337abd4 Mon Sep 17 00:00:00 2001 From: Jack Song Date: Fri, 18 Sep 2026 08:48:19 +1000 Subject: [PATCH 6/8] test: hermetic pytest suite with synthetic fixture DBs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 26 tests covering the extra_buffer protobuf decoder, contact loading/detail, history message IDs, db_cache torn-read poisoning, and init db_dir preservation. Regression-verified: the init and cache-poisoning tests fail against pre-fix code. Runs without WeChat or real data — protects future changes from reintroducing the two bugs found during ROAD-354 live testing. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- AGENTS.md | 10 +++- pyproject.toml | 5 ++ tests/conftest.py | 100 ++++++++++++++++++++++++++++++++++ tests/test_contacts.py | 41 ++++++++++++++ tests/test_db_cache.py | 106 +++++++++++++++++++++++++++++++++++++ tests/test_extra_buffer.py | 91 +++++++++++++++++++++++++++++++ tests/test_history_ids.py | 74 ++++++++++++++++++++++++++ tests/test_init.py | 57 ++++++++++++++++++++ 8 files changed, 483 insertions(+), 1 deletion(-) create mode 100644 tests/conftest.py create mode 100644 tests/test_contacts.py create mode 100644 tests/test_db_cache.py create mode 100644 tests/test_extra_buffer.py create mode 100644 tests/test_history_ids.py create mode 100644 tests/test_init.py diff --git a/AGENTS.md b/AGENTS.md index 6b12da8..4e98022 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -52,7 +52,15 @@ WeChat and re-run `init`. Works without sudo once re-signed. ## Verify -No test suite. Verify against the live DB: +Hermetic suite (synthetic fixture DBs — never touches real WeChat data): + +```bash +.venv/bin/pip install -e ".[dev]" && .venv/bin/python -m pytest tests/ +``` + +Covers the `extra_buffer` decoder, contact loading/detail, history message IDs, +db_cache torn-read poisoning, and init `db_dir` preservation. For end-to-end +checks against the live DB: ```bash .venv/bin/wechat-cli contacts --detail "" # labels/phone diff --git a/pyproject.toml b/pyproject.toml index 032555e..7a4a8d9 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -13,6 +13,11 @@ dependencies = [ "zstandard>=0.22,<1", ] +[project.optional-dependencies] +dev = [ + "pytest>=8,<9", +] + [project.scripts] wechat-cli = "wechat_cli.main:cli" diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..028359d --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,100 @@ +"""测试辅助 — 合成 protobuf 编码与 SQLite fixture 库(不触碰真实微信数据)""" + +import hashlib +import sqlite3 + +import pytest + + +# ---- protobuf 编码辅助 ---- + +def encode_varint(value): + out = bytearray() + while True: + b = value & 0x7F + value >>= 7 + if value: + out.append(b | 0x80) + else: + out.append(b) + return bytes(out) + + +def field_bytes(fno, data): + return encode_varint((fno << 3) | 2) + encode_varint(len(data)) + data + + +def field_varint(fno, value): + return encode_varint(fno << 3) + encode_varint(value) + + +def make_extra_buffer(labels_raw=None, phone=None): + """构造 contact.extra_buffer:field 30 = 标签 ID 字符串,field 14→2→1 = 手机号。""" + buf = b"" + if phone is not None: + inner = field_bytes(1, phone.encode()) + buf += field_bytes(14, field_varint(1, 1) + field_bytes(2, inner)) + if labels_raw is not None: + buf += field_bytes(30, labels_raw.encode()) + return buf + + +# ---- SQLite fixture ---- + +CONTACT_SCHEMA = """ +CREATE TABLE contact( + id INTEGER PRIMARY KEY, username TEXT, local_type INTEGER, alias TEXT, + encrypt_username TEXT, flag INTEGER, delete_flag INTEGER, verify_flag INTEGER, + remark TEXT, remark_quan_pin TEXT, remark_pin_yin_initial TEXT, nick_name TEXT, + pin_yin_initial TEXT, quan_pin TEXT, big_head_url TEXT, small_head_url TEXT, + head_img_md5 TEXT, chat_room_notify INTEGER, is_in_chat_room INTEGER, + description TEXT, extra_buffer BLOB, chat_room_type INTEGER); +CREATE TABLE contact_label(label_id_ INTEGER PRIMARY KEY, label_name_ TEXT, sort_order_ INTEGER); +""" + + +def msg_table_name(username): + return "Msg_" + hashlib.md5(username.encode()).hexdigest() + + +MSG_SCHEMA = """ +CREATE TABLE {table}( + local_id INTEGER PRIMARY KEY, server_id INTEGER, local_type INTEGER, + sort_seq INTEGER, real_sender_id INTEGER, create_time INTEGER, status INTEGER, + upload_status INTEGER, download_status INTEGER, server_seq INTEGER, + origin_source INTEGER, source TEXT, message_content TEXT, + compress_content TEXT, packed_info_data BLOB, + WCDB_CT_message_content INTEGER, WCDB_CT_source INTEGER); +CREATE TABLE Name2Id(user_name TEXT); +""" + + +def insert_contact(conn, username, nick_name="", remark="", extra_buffer=None, **kw): + cols = {"username": username, "nick_name": nick_name, "remark": remark, + "extra_buffer": extra_buffer} + cols.update(kw) + keys = ", ".join(cols) + conn.execute( + f"INSERT INTO contact({keys}) VALUES ({', '.join('?' * len(cols))})", + list(cols.values()), + ) + + +@pytest.fixture +def contact_db_path(tmp_path): + """带真实 schema 的 contact.db,预置 2 个标签 + 2 个联系人。""" + path = tmp_path / "contact.db" + conn = sqlite3.connect(path) + conn.executescript(CONTACT_SCHEMA) + conn.executemany( + "INSERT INTO contact_label VALUES (?, ?, ?)", + [(1, "客户", 0), (5, "Sydney", 1)], + ) + insert_contact( + conn, "wxid_alice", nick_name="Alice", remark="爱丽丝", + extra_buffer=make_extra_buffer(labels_raw="1,5", phone="0412345678"), + ) + insert_contact(conn, "wxid_bob", nick_name="Bob") + conn.commit() + conn.close() + return path diff --git a/tests/test_contacts.py b/tests/test_contacts.py new file mode 100644 index 0000000..bc0c2eb --- /dev/null +++ b/tests/test_contacts.py @@ -0,0 +1,41 @@ +"""contacts 加载/详情测试 — 使用合成 contact.db,不触碰真实微信数据""" + +from wechat_cli.core.contacts import _load_contacts_from, get_contact_detail + + +class _NullCache: + def get(self, rel_key): + return None + + +def test_load_contacts_labels_and_phone(contact_db_path): + names, full = _load_contacts_from(str(contact_db_path)) + assert names["wxid_alice"] == "爱丽丝" + alice = next(c for c in full if c["username"] == "wxid_alice") + assert alice["labels"] == ["客户", "Sydney"] + assert alice["phone"] == "0412345678" + bob = next(c for c in full if c["username"] == "wxid_bob") + assert bob["labels"] == [] + assert bob["phone"] == "" + + +def test_get_contact_detail(contact_db_path, tmp_path): + # get_contact_detail 优先读 decrypted_dir/contact/contact.db + decrypted_dir = tmp_path / "decrypted" + (decrypted_dir / "contact").mkdir(parents=True) + import shutil + shutil.copy(contact_db_path, decrypted_dir / "contact" / "contact.db") + + info = get_contact_detail("wxid_alice", _NullCache(), str(decrypted_dir)) + assert info["labels"] == ["客户", "Sydney"] + assert info["label_ids"] == [1, 5] + assert info["phone"] == "0412345678" + assert info["is_group"] is False + + +def test_get_contact_detail_missing(contact_db_path, tmp_path): + decrypted_dir = tmp_path / "decrypted" + (decrypted_dir / "contact").mkdir(parents=True) + import shutil + shutil.copy(contact_db_path, decrypted_dir / "contact" / "contact.db") + assert get_contact_detail("wxid_nobody", _NullCache(), str(decrypted_dir)) is None diff --git a/tests/test_db_cache.py b/tests/test_db_cache.py new file mode 100644 index 0000000..05f5b1e --- /dev/null +++ b/tests/test_db_cache.py @@ -0,0 +1,106 @@ +"""DBCache 回归测试 — 防止撕裂解密结果被缓存(曾导致全部查询失败的 bug)""" + +import json +import os +import sqlite3 + +import pytest + +import wechat_cli.core.db_cache as db_cache_mod +from wechat_cli.core.db_cache import DBCache, _has_sqlite_header, _is_valid_sqlite + + +REL_KEY = "contact/contact.db" +ENC_KEY_HEX = "ab" * 32 + + +def _make_valid_db(path): + if os.path.exists(path): + os.remove(path) + conn = sqlite3.connect(path) + conn.executescript( + "DROP TABLE IF EXISTS t;" + "CREATE TABLE t(id INTEGER PRIMARY KEY, v TEXT);" + "INSERT INTO t(v) VALUES ('x');" + ) + conn.commit() + conn.close() + + +@pytest.fixture +def env(tmp_path, monkeypatch): + """独立 db_dir + cache 目录,full_decrypt/decrypt_wal 可注入。""" + cache_dir = tmp_path / "cache" + monkeypatch.setattr(DBCache, "CACHE_DIR", str(cache_dir)) + monkeypatch.setattr(DBCache, "MTIME_FILE", str(cache_dir / "_mtimes.json")) + monkeypatch.setattr(db_cache_mod, "_DECRYPT_RETRY_DELAY", 0) + + db_dir = tmp_path / "db_storage" + (db_dir / "contact").mkdir(parents=True) + (db_dir / "contact" / "contact.db").write_bytes(b"encrypted") + + all_keys = {REL_KEY: {"enc_key": ENC_KEY_HEX}} + cache = DBCache(all_keys, str(db_dir)) + calls = {"decrypt": 0} + + def set_decrypt_result(valid): + def fake_full_decrypt(db_path, out_path, enc_key): + calls["decrypt"] += 1 + if valid: + _make_valid_db(out_path) + else: + with open(out_path, "wb") as f: + f.write(b"\x00" * 8192) + monkeypatch.setattr(db_cache_mod, "full_decrypt", fake_full_decrypt) + + monkeypatch.setattr(db_cache_mod, "decrypt_wal", lambda *a: 0) + return cache, set_decrypt_result, calls, db_dir + + +def test_get_decrypts_and_caches(env): + cache, set_decrypt, calls, _ = env + set_decrypt(True) + p = cache.get(REL_KEY) + assert p and _is_valid_sqlite(p) + # 第二次命中缓存,不再解密 + assert cache.get(REL_KEY) == p + assert calls["decrypt"] == 1 + + +def test_torn_decrypt_not_cached(env): + """撕裂读取(微信写入中)不得进入缓存 — 曾导致缓存中毒的核心回归。""" + cache, set_decrypt, calls, _ = env + set_decrypt(False) + assert cache.get(REL_KEY) is None + assert calls["decrypt"] == db_cache_mod._DECRYPT_ATTEMPTS + # 失败结果不写入持久缓存 + assert not os.path.exists(DBCache.MTIME_FILE) or REL_KEY not in json.load( + open(DBCache.MTIME_FILE) + ) + + +def test_poisoned_cache_file_not_served(env): + """已存在的损坏缓存文件不得被直接返回。""" + cache, set_decrypt, calls, db_dir = env + set_decrypt(True) + good = cache.get(REL_KEY) + assert _is_valid_sqlite(good) + + # 模拟中毒:缓存文件被撕裂内容覆盖(mtime 记录不变) + with open(good, "wb") as f: + f.write(b"\x00" * 8192) + + p2 = cache.get(REL_KEY) + assert p2 is not None + assert _is_valid_sqlite(p2) + + +def test_is_valid_sqlite(tmp_path): + good = tmp_path / "good.db" + _make_valid_db(str(good)) + assert _is_valid_sqlite(str(good)) + bad = tmp_path / "bad.db" + bad.write_bytes(b"\x00" * 8192) + assert not _is_valid_sqlite(str(bad)) + assert not _has_sqlite_header(str(bad)) + assert not _is_valid_sqlite(str(tmp_path / "nonexistent.db")) diff --git a/tests/test_extra_buffer.py b/tests/test_extra_buffer.py new file mode 100644 index 0000000..1b14af2 --- /dev/null +++ b/tests/test_extra_buffer.py @@ -0,0 +1,91 @@ +"""extra_buffer protobuf 解码器单元测试""" + +from wechat_cli.core.contacts import ( + _decode_extra_buffer, + _decode_extra_labels, + _decode_extra_phone, + _parse_protobuf_fields, + _read_varint, + _split_label_ids, +) +from conftest import encode_varint, field_bytes, field_varint, make_extra_buffer + + +def test_read_varint_single_byte(): + assert _read_varint(b"\x08", 0) == (8, 1) + + +def test_read_varint_multi_byte(): + assert _read_varint(encode_varint(300), 0) == (300, 2) + + +def test_parse_protobuf_fields_mixed(): + blob = field_varint(1, 42) + field_bytes(30, b"1,5") + fields = _parse_protobuf_fields(blob) + assert fields == [(1, 0, 42), (30, 2, b"1,5")] + + +def test_parse_protobuf_fields_truncated(): + # 截断的 len-delimited 字段不应崩溃 + blob = field_bytes(4, b"abcdef")[:4] + _parse_protobuf_fields(blob) + + +def test_split_label_ids_ascii(): + assert _split_label_ids("226,251") == [226, 251] + + +def test_split_label_ids_cjk_and_misc_delimiters(): + assert _split_label_ids("1,2;3;4|5 6") == [1, 2, 3, 4, 5, 6] + + +def test_split_label_ids_ignores_non_numeric(): + assert _split_label_ids("1,abc,,3") == [1, 3] + + +def test_decode_labels_with_names(): + blob = make_extra_buffer(labels_raw="1,5") + ids, names = _decode_extra_labels(blob, {1: "客户", 5: "Sydney"}) + assert ids == [1, 5] + assert names == ["客户", "Sydney"] + + +def test_decode_labels_missing_name_dropped(): + blob = make_extra_buffer(labels_raw="1,99") + ids, names = _decode_extra_labels(blob, {1: "客户"}) + assert ids == [1, 99] + assert names == ["客户"] + + +def test_decode_labels_empty_blob(): + assert _decode_extra_labels(b"") == ([], []) + assert _decode_extra_labels(make_extra_buffer(labels_raw="")) == ([], []) + + +def test_decode_phone_nested(): + blob = make_extra_buffer(phone="+61411225796") + assert _decode_extra_phone(blob) == "+61411225796" + + +def test_decode_phone_flag_only(): + # field 14 仅有 "phone present" flag、无 field 2 → 无号码 + blob = field_bytes(14, field_varint(1, 0)) + assert _decode_extra_phone(blob) == "" + + +def test_decode_phone_absent(): + assert _decode_extra_phone(make_extra_buffer(labels_raw="1")) == "" + assert _decode_extra_phone(b"") == "" + + +def test_decode_extra_buffer_full(): + blob = make_extra_buffer(labels_raw="1", phone="0451122734") + ids, names, phone = _decode_extra_buffer(blob, {1: "客户"}) + assert ids == [1] + assert names == ["客户"] + assert phone == "0451122734" + + +def test_decode_extra_buffer_none_and_garbage(): + assert _decode_extra_buffer(None) == ([], [], "") + assert _decode_extra_buffer(b"\xff\xff\xff\xff") == ([], [], "") diff --git a/tests/test_history_ids.py b/tests/test_history_ids.py new file mode 100644 index 0000000..097cf16 --- /dev/null +++ b/tests/test_history_ids.py @@ -0,0 +1,74 @@ +"""history 消息 ID 测试 — 合成 Msg_ 表验证 local_id/server_id 暴露""" + +import sqlite3 + +import pytest + +from wechat_cli.core.messages import ( + _query_messages, + collect_chat_history, +) +from conftest import MSG_SCHEMA, msg_table_name + + +@pytest.fixture +def msg_db(tmp_path): + """单表消息库:Msg_,3 条消息。""" + path = tmp_path / "message_0.db" + table = msg_table_name("friend") + conn = sqlite3.connect(path) + conn.executescript(MSG_SCHEMA.format(table=table)) + conn.execute("INSERT INTO Name2Id(rowid, user_name) VALUES (1, 'friend')") + conn.execute("INSERT INTO Name2Id(rowid, user_name) VALUES (2, 'me_wxid')") + rows = [ + (1, 111111, 1, 0, 1, 1700000000, "hello"), + (2, 222222, 1, 0, 2, 1700000060, "hi back"), + (3, 333333, 1, 0, 1, 1700000120, "see you"), + ] + conn.executemany( + f"INSERT INTO [{table}](local_id, server_id, local_type, sort_seq, " + f"real_sender_id, create_time, message_content) VALUES (?,?,?,?,?,?,?)", + rows, + ) + conn.commit() + conn.close() + return str(path), table + + +def _ctx(db_path, table): + return { + "query": "friend", "username": "friend", "display_name": "Friend", + "db_path": db_path, "table_name": table, + "message_tables": [{"db_path": db_path, "table_name": table}], + "is_group": False, + } + + +def test_query_messages_returns_server_id(msg_db): + db_path, table = msg_db + conn = sqlite3.connect(db_path) + rows = _query_messages(conn, table, limit=10) + conn.close() + assert len(rows[0]) == 7 # local_id, server_id, local_type, create_time, sender, content, ct + local_ids = [r[0] for r in rows] + server_ids = [r[1] for r in rows] + assert sorted(local_ids) == [1, 2, 3] + assert sorted(server_ids) == [111111, 222222, 333333] + + +def test_collect_chat_history_entries_have_ids(msg_db): + db_path, table = msg_db + names = {"friend": "Friend", "me_wxid": "me"} + entries, failures = collect_chat_history( + _ctx(db_path, table), names, lambda u, n: n.get(u, u), limit=10, + ) + assert not failures + assert len(entries) == 3 + for e in entries: + assert set(e) >= {"local_id", "server_id", "timestamp", "time", "sender", "text", "line"} + by_local = {e["local_id"]: e for e in entries} + assert by_local[1]["server_id"] == 111111 + assert by_local[1]["text"] == "hello" + assert by_local[2]["sender"] == "me" + # 时间升序输出 + assert [e["local_id"] for e in entries] == [1, 2, 3] diff --git a/tests/test_init.py b/tests/test_init.py new file mode 100644 index 0000000..7dd3ca0 --- /dev/null +++ b/tests/test_init.py @@ -0,0 +1,57 @@ +"""init 回归测试 — init --force 不得覆盖已配置的 db_dir(多账号场景)""" + +import json + +import pytest +from click.testing import CliRunner + +import wechat_cli.commands.init as init_mod +import wechat_cli.keys as keys_mod +from wechat_cli.commands.init import init + + +@pytest.fixture +def env(tmp_path, monkeypatch): + state_dir = tmp_path / ".wechat-cli" + state_dir.mkdir() + config_file = state_dir / "config.json" + keys_file = state_dir / "all_keys.json" + + monkeypatch.setattr(init_mod, "STATE_DIR", str(state_dir)) + monkeypatch.setattr(init_mod, "CONFIG_FILE", str(config_file)) + monkeypatch.setattr(init_mod, "KEYS_FILE", str(keys_file)) + return state_dir, config_file, keys_file + + +def test_init_force_preserves_configured_db_dir(env, tmp_path, monkeypatch): + """多账号机器上 --force 不得因自动检测选错而覆盖 db_dir。""" + _, config_file, keys_file = env + good_dir = tmp_path / "xwechat_files" / "account_a" / "db_storage" + good_dir.mkdir(parents=True) + wrong_dir = tmp_path / "xwechat_files" / "account_b" / "db_storage" + wrong_dir.mkdir(parents=True) + + config_file.write_text(json.dumps({"db_dir": str(good_dir)})) + keys_file.write_text("{}") + + def _auto_detect_should_not_run(): + raise AssertionError("auto_detect_db_dir 不应被调用 — 已有配置应优先") + monkeypatch.setattr(init_mod, "auto_detect_db_dir", _auto_detect_should_not_run) + monkeypatch.setattr(keys_mod, "extract_keys", lambda *a, **k: {"k1": "v1"}) + + result = CliRunner().invoke(init, ["--force"]) + assert result.exit_code == 0, result.output + assert json.loads(config_file.read_text())["db_dir"] == str(good_dir) + assert "提取到 1 个数据库密钥" in result.output + + +def test_init_force_autodetect_when_no_config(env, tmp_path, monkeypatch): + state_dir, config_file, keys_file = env + detected = tmp_path / "detected" / "db_storage" + detected.mkdir(parents=True) + monkeypatch.setattr(init_mod, "auto_detect_db_dir", lambda: str(detected)) + monkeypatch.setattr(keys_mod, "extract_keys", lambda *a, **k: {"k1": "v1"}) + + result = CliRunner().invoke(init, ["--force"]) + assert result.exit_code == 0, result.output + assert json.loads(config_file.read_text())["db_dir"] == str(detected) From 72b39644a32ed0a9f1189ac2e97fbdcaa2738973 Mon Sep 17 00:00:00 2001 From: Jack Song Date: Fri, 18 Sep 2026 08:48:56 +1000 Subject: [PATCH 7/8] chore: add exports/ dir for chat history dumps, gitignored MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Designated place for personal-data exports on disk — dir contents ignored except .gitkeep (JSON was already covered by the *.json rule; this also covers md/txt exports and makes the convention explicit). Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .gitignore | 4 ++++ AGENTS.md | 3 +++ exports/.gitkeep | 0 3 files changed, 7 insertions(+) create mode 100644 exports/.gitkeep diff --git a/.gitignore b/.gitignore index 00e7fb8..76e560b 100644 --- a/.gitignore +++ b/.gitignore @@ -35,6 +35,10 @@ Thumbs.db *.db-shm wechat_ent.plist +# Chat history exports (personal data — never commit) +exports/* +!exports/.gitkeep + # Sensitive data — NEVER commit *.json !pyproject.toml diff --git a/AGENTS.md b/AGENTS.md index 4e98022..d5fa136 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -12,6 +12,9 @@ LLM/agent consumption. keys; decrypted DBs land in `$TMPDIR/wechat_cli_cache` and `~/.wechat-cli/decrypted`. Never commit keys, `*.db*`, or `*.json` output dumps (already gitignored). Don't echo key material or bulk personal data into logs. +- **Exports go in `exports/`.** Dump query/export output there — the dir is + gitignored (except `.gitkeep`) and is the designated place for personal data + on disk, e.g. `wechat-cli history "X" --limit 500 > exports/x.json`. - **Chinese comments/docstrings** are the codebase convention — match them. ## Setup diff --git a/exports/.gitkeep b/exports/.gitkeep new file mode 100644 index 0000000..e69de29 From 77c07e4f47f3695e24a367e197feac648c148c0e Mon Sep 17 00:00:00 2001 From: Jack Song Date: Fri, 18 Sep 2026 08:52:51 +1000 Subject: [PATCH 8/8] fix: skip fixed32/64 wire types in protobuf field parser MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Parser previously stopped at wt=1/5 fields — a fixed-width field appearing before field 30/14 would silently drop labels/phone. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/test_extra_buffer.py | 11 +++++++++++ wechat_cli/core/contacts.py | 9 ++++++++- 2 files changed, 19 insertions(+), 1 deletion(-) diff --git a/tests/test_extra_buffer.py b/tests/test_extra_buffer.py index 1b14af2..918c0ad 100644 --- a/tests/test_extra_buffer.py +++ b/tests/test_extra_buffer.py @@ -31,6 +31,17 @@ def test_parse_protobuf_fields_truncated(): _parse_protobuf_fields(blob) +def test_parse_protobuf_fields_skips_fixed_width(): + # wt=1 (fixed64) / wt=5 (fixed32) 字段应跳过而非中断解析 + blob = ( + encode_varint((7 << 3) | 1) + b"\x00" * 8 + + encode_varint((8 << 3) | 5) + b"\x00" * 4 + + field_bytes(30, b"1") + ) + fields = _parse_protobuf_fields(blob) + assert fields == [(30, 2, b"1")] + + def test_split_label_ids_ascii(): assert _split_label_ids("226,251") == [226, 251] diff --git a/wechat_cli/core/contacts.py b/wechat_cli/core/contacts.py index e9422eb..296a0d0 100644 --- a/wechat_cli/core/contacts.py +++ b/wechat_cli/core/contacts.py @@ -29,7 +29,8 @@ def _read_varint(buf, i): def _parse_protobuf_fields(data): - """通用 protobuf 解析。返回 [(field_no, wire_type, value)],wt=0→int, wt=2→bytes。""" + """通用 protobuf 解析。返回 [(field_no, wire_type, value)],wt=0→int, wt=2→bytes。 + wt=1/5 (fixed64/fixed32) 跳过其定长字节继续解析;group (wt=3/4) 不支持,停止。""" i = 0 fields = [] while i < len(data): @@ -41,6 +42,12 @@ def _parse_protobuf_fields(data): ln, i = _read_varint(data, i) val = data[i:i + ln] i += ln + elif wt == 1: + i += 8 + continue + elif wt == 5: + i += 4 + continue else: break fields.append((fno, wt, val))