Files
OpenMesh/patch_pyz.py
zhaolei 6f402ffcee
Some checks failed
CI / pytest (push) Has been cancelled
CI / gui-unit (push) Has been cancelled
CI / gui-e2e (push) Has been cancelled
feat: OpenMesh 基础平台与 MD/PDF 转换技能
- 后端: coworker 智能体框架, WS API, 文件上传, 附件处理
- 前端: Open WebUI, 文件全量走 upload API (含 MD/TXT/JSON 等文本类)
- 技能: md-to-office (pandoc + wkhtmltopdf)
- 修复: 上传文件路径丢失, Agent 搜索浪费, 输出文件跑到 uploads/
- 打包: PyInstaller one-dir, 预打包 pandoc/wkhtmltopdf/chromium
2026-09-13 23:41:04 +08:00

272 lines
9.4 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
PYZ 热补丁脚本 v3 - 原地替换 PYZ 数据,不重建 CArchive
原理:新的 PYZ 比原来小,直接把新数据写入原来的位置,后面补 0。
只修改 TOC 中的 uncompressed_length不改变 CArchive 结构。
"""
import sys
import os
import marshal
import zlib
import struct
import shutil
import tempfile
from pathlib import Path
sys.path.insert(0, r'd:\project\senmeshworker-main\.venv\Lib\site-packages')
from PyInstaller.archive.readers import CArchiveReader, ZlibArchiveReader
def extract_pyz(exe_path):
"""从 exe 中提取 PYZ 数据(已解压)"""
arch = CArchiveReader(exe_path)
pyz_info = arch.toc['PYZ.pyz']
offset, csize, usize, is_compressed, typ = pyz_info
with open(exe_path, 'rb') as f:
f.seek(arch._start_offset + offset)
data = f.read(csize)
if is_compressed:
data = zlib.decompress(data)
print(f"PYZ extracted: {len(data)} bytes (compressed: {csize})")
return data, arch, offset, csize
def patch_pyz(pyz_data, source_py_root, module_prefixes=('coworker',)):
"""用 .py 源码文件替换 PYZ 中的模块,返回新的 PYZ 数据"""
# 用临时文件来读 PYZZlibArchiveReader 需要文件路径)
with tempfile.NamedTemporaryFile(suffix='.pyz', delete=False) as f:
f.write(pyz_data)
tmp_path = f.name
try:
pyz = ZlibArchiveReader(tmp_path)
toc = dict(pyz.toc) # {name: (is_pkg, offset, size)}
finally:
os.unlink(tmp_path)
print(f"Original modules: {len(toc)}")
# 解析 header
pyz_magic = pyz_data[:4]
py_magic = pyz_data[4:8]
toc_offset_old = struct.unpack('!i', pyz_data[8:12])[0]
print(f"PYZ magic: {pyz_magic.hex()}")
print(f"Python magic: {py_magic.hex()}")
# 找到所有需要替换的模块
source_root_parent = Path(source_py_root).parent
replacements = {} # module_name -> (is_pkg, compiled_code_bytes)
for py_file in source_root_parent.rglob('*.py'):
try:
rel = py_file.relative_to(source_root_parent)
except ValueError:
continue
parts = list(rel.parts)
parts[-1] = parts[-1][:-3]
if parts[-1] == '__init__':
parts = parts[:-1]
module_name = '.'.join(parts)
if not any(module_name.startswith(p) for p in module_prefixes):
continue
source_code = py_file.read_text(encoding='utf-8')
is_pkg = py_file.name == '__init__.py'
try:
code_obj = compile(source_code, module_name, 'exec')
code_bytes = marshal.dumps(code_obj)
replacements[module_name] = (is_pkg, code_bytes)
print(f" Compiled: {module_name} ({len(code_bytes)} bytes)")
except SyntaxError as e:
print(f" ERROR compiling {module_name}: {e}")
if not replacements:
print("No modules to replace!")
return pyz_data
# 收集所有模块数据
module_data = {} # name -> (is_pkg, compressed_bytes)
# 原有模块(跳过要替换的)
for name, (is_pkg, offset, size) in toc.items():
if name in replacements:
continue
raw_data = pyz_data[offset:offset + size]
module_data[name] = (is_pkg, raw_data)
# 替换模块
for name, (is_pkg, code_bytes) in replacements.items():
compressed = zlib.compress(code_bytes, 9)
module_data[name] = (int(is_pkg), compressed)
# 构建新 PYZ
new_pyz = bytearray()
new_pyz.extend(pyz_magic) # [0:4] PYZ magic
new_pyz.extend(py_magic) # [4:8] Python magic
new_pyz.extend(b'\x00\x00\x00\x00') # [8:12] TOC offset placeholder
# 模块数据(按名字排序,保持一致)
new_toc = {}
for name in sorted(module_data.keys()):
is_pkg, data = module_data[name]
offset = len(new_pyz)
new_pyz.extend(data)
new_toc[name] = (is_pkg, offset, len(data))
# TOC 数据
toc_start = len(new_pyz)
toc_marshalled = marshal.dumps(new_toc)
new_pyz.extend(toc_marshalled)
# 回填 TOC offset
struct.pack_into('!i', new_pyz, 8, toc_start)
print(f"New PYZ: {len(new_pyz)} bytes (original: {len(pyz_data)} bytes)")
print(f"Modules: {len(module_data)}")
return bytes(new_pyz)
def replace_pyz_inplace(exe_path, new_pyz_data):
"""原地替换 PYZ 数据。
由于新 PYZ 更小,我们直接写入原位置,剩余空间补 0。
更新 PYZ 条目的 uncompressed_length但保持 csize 不变(因为我们补了 0
等等,这样不行——读取时会按 csize 读,解压会失败。
正确做法:
- PYZ 在 CArchive 中是 type='z'compression_flag=0不压缩
- 所以 csize == usize
- 我们需要保持 csize 和 usize 都和原来一样大
- 新数据小,后面补 0 填充到原大小
- 但 PYZ 内部有自己的 TOC 和大小,补 0 不影响 PYZ 的读取PYZ 用自己的 TOC 找模块)
"""
arch = CArchiveReader(exe_path)
pyz_info = arch.toc['PYZ.pyz']
offset, csize, usize, is_compressed, typ = pyz_info
print(f"Original PYZ in CArchive: offset={offset}, csize={csize}, usize={usize}, compressed={is_compressed}, type={typ}")
if is_compressed:
raise Exception("PYZ is compressed in CArchive! Expected uncompressed (type 'z').")
if len(new_pyz_data) > csize:
raise Exception(f"New PYZ ({len(new_pyz_data)}) is larger than original ({csize})!")
# 读取整个 exe
with open(exe_path, 'rb') as f:
all_data = bytearray(f.read())
start_offset = arch._start_offset
pyz_start = start_offset + offset
# 写入新 PYZ 数据
all_data[pyz_start:pyz_start + len(new_pyz_data)] = new_pyz_data
# 剩余空间补 0
remaining = csize - len(new_pyz_data)
if remaining > 0:
all_data[pyz_start + len(new_pyz_data):pyz_start + csize] = b'\x00' * remaining
# 更新 TOC 中的 uncompressed_length
# TOC 格式: entry_length(4) + offset(4) + length(4) + uncompressed_length(4) + compression_flag(1) + typecode(1) + name(padded to 16)
toc_offset = arch._toc_offset
toc_length = arch._toc_length
print(f"TOC offset: {toc_offset}, length: {toc_length}")
# 找到 PYZ.pyz 的 TOC 条目并更新 uncompressed_length
# 解析 TOC找到 PYZ.pyz修改其 usize然后写回
toc_data = all_data[start_offset + toc_offset : start_offset + toc_offset + toc_length]
# 解析 TOC
TOC_ENTRY_FORMAT = '!IIIIBc'
TOC_ENTRY_LENGTH = struct.calcsize(TOC_ENTRY_FORMAT)
cur_pos = 0
new_toc_data = bytearray()
found = False
while cur_pos < len(toc_data):
entry_start = cur_pos
entry_length, entry_offset, data_length, uncompressed_length, compression_flag, typecode = \
struct.unpack(TOC_ENTRY_FORMAT, toc_data[cur_pos:cur_pos + TOC_ENTRY_LENGTH])
cur_pos += TOC_ENTRY_LENGTH
name_length = entry_length - TOC_ENTRY_LENGTH
name = toc_data[cur_pos:cur_pos + name_length].rstrip(b'\0').decode('utf-8')
cur_pos += name_length
if name == 'PYZ.pyz':
# 更新 uncompressed_length
uncompressed_length = len(new_pyz_data)
found = True
print(f"Updated PYZ.pyz uncompressed_length: {data_length} -> {len(new_pyz_data)}")
# 重建条目
entry_data = struct.pack(TOC_ENTRY_FORMAT, entry_length, entry_offset, data_length, uncompressed_length, compression_flag, typecode)
name_bytes = toc_data[entry_start + TOC_ENTRY_LENGTH : entry_start + entry_length]
new_toc_data.extend(entry_data)
new_toc_data.extend(name_bytes)
if not found:
print("WARNING: PYZ.pyz not found in TOC!")
# 写回 TOC大小不变直接覆盖
all_data[start_offset + toc_offset : start_offset + toc_offset + toc_length] = bytes(new_toc_data)
# 写回文件
with open(exe_path, 'wb') as f:
f.write(bytes(all_data))
print(f"EXE updated: {len(all_data)} bytes")
return True
def main():
exe_path = r'd:\project\senmeshworker-main\dist\openmesh\openmesh-server.exe'
source_root = r'd:\project\senmeshworker-main\coworker'
# 1. 备份
backup_path = exe_path + '.bak_pyz'
if not os.path.exists(backup_path):
shutil.copy2(exe_path, backup_path)
print(f"Backup: {backup_path}")
# 2. 提取 PYZ
pyz_data, arch, pyz_offset, pyz_csize = extract_pyz(exe_path)
# 3. 打补丁
new_pyz_data = patch_pyz(pyz_data, source_root, module_prefixes=('coworker',))
# 4. 验证新 PYZ
with tempfile.NamedTemporaryFile(suffix='.pyz', delete=False) as f:
f.write(new_pyz_data)
tmp_path = f.name
try:
test_reader = ZlibArchiveReader(tmp_path)
print(f"\nVerification: new PYZ has {len(test_reader.toc)} modules")
assert 'coworker.file_upload' in test_reader.toc
assert 'coworker.attachments' in test_reader.toc
assert 'coworker.server.app' in test_reader.toc
print("✅ New PYZ is valid!")
finally:
os.unlink(tmp_path)
# 5. 原地替换 PYZ
replace_pyz_inplace(exe_path, new_pyz_data)
size_mb = os.path.getsize(exe_path) / 1024 / 1024
print(f"\n✅ Done! EXE size: {size_mb:.1f} MB")
if __name__ == '__main__':
main()