362 lines
12 KiB
Python
362 lines
12 KiB
Python
import t, c
|
||
from stdint import *
|
||
import memhub
|
||
import string
|
||
import stdio
|
||
import stdlib
|
||
import w32.win32file as win32file
|
||
import w32.win32base as win32base
|
||
import w32.fileio as fileio
|
||
import hashlib
|
||
import viperlib
|
||
|
||
|
||
# ============================================================
|
||
# IncludesScanner - includes 目录递归扫描
|
||
#
|
||
# 递归扫描 includes 目录,收集所有 .py 文件路径和 SHA1,
|
||
# 为 Phase1 stub 生成和 Phase2 stub 合并提供基础数据。
|
||
#
|
||
# 使用 Win32 FindFirstFileA/FindNextFileA 实现目录遍历。
|
||
# ============================================================
|
||
|
||
# 全局 mbuddy 指针
|
||
_mbuddy: memhub.MemManager | t.CPtr
|
||
|
||
# 文件路径最大长度
|
||
MAX_PATH_LEN: t.CDefine = 512
|
||
# 单次扫描最大文件数
|
||
MAX_FILES: t.CDefine = 256
|
||
|
||
|
||
# ============================================================
|
||
# FileEntry - 文件条目(路径 + SHA1)
|
||
# ============================================================
|
||
@t.NoVTable
|
||
class FileEntry:
|
||
"""扫描到的文件条目"""
|
||
Path: str # 文件绝对路径
|
||
Sha1: str # SHA1 前16字符(16字节+null)
|
||
RelPath: str # 相对于 includes 根目录的路径
|
||
ModuleName: str # 模块名(如 "ast.parser")
|
||
|
||
|
||
# ============================================================
|
||
# ScanResult - 扫描结果
|
||
# ============================================================
|
||
@t.NoVTable
|
||
class ScanResult:
|
||
"""扫描结果集合"""
|
||
Entries: FileEntry | t.CPtr # FileEntry 数组
|
||
Count: t.CInt
|
||
Capacity: t.CInt
|
||
|
||
|
||
# ============================================================
|
||
# create_scan_result - 创建扫描结果容器
|
||
# ============================================================
|
||
def create_scan_result(pool: memhub.MemBuddy | t.CPtr) -> ScanResult | t.CPtr:
|
||
"""创建扫描结果容器,预分配 MAX_FILES 个槽位"""
|
||
if pool is None:
|
||
return None
|
||
size: t.CSizeT = MAX_FILES * FileEntry.__sizeof__()
|
||
# 使用 stdlib.malloc 避免 mbuddy 池耗尽(与 StubMerger 保持一致)
|
||
entries: FileEntry | t.CPtr = stdlib.malloc(size)
|
||
if entries is None:
|
||
return None
|
||
string.memset(entries, 0, size)
|
||
|
||
result: ScanResult | t.CPtr = stdlib.malloc(ScanResult.__sizeof__())
|
||
if result is None:
|
||
return None
|
||
string.memset(result, 0, ScanResult.__sizeof__())
|
||
result.Entries = entries
|
||
result.Count = 0
|
||
result.Capacity = MAX_FILES
|
||
return result
|
||
|
||
|
||
# ============================================================
|
||
# add_file_entry - 向扫描结果添加文件条目
|
||
# ============================================================
|
||
def add_file_entry(result: ScanResult | t.CPtr,
|
||
pool: memhub.MemBuddy | t.CPtr,
|
||
abs_path: str, rel_path: str,
|
||
sha1: str) -> int:
|
||
"""添加文件条目到扫描结果,返回 0 成功"""
|
||
if result is None or pool is None:
|
||
return 1
|
||
if result.Count >= result.Capacity:
|
||
return 1
|
||
|
||
# 计算模块名:rel_path 中的 / 替换为 .,去掉 .py 扩展名
|
||
rel_len: t.CSizeT = string.strlen(rel_path)
|
||
mod_buf: str = stdlib.malloc(rel_len + 1)
|
||
if mod_buf is None:
|
||
return 1
|
||
string.strcpy(mod_buf, rel_path)
|
||
# 替换 / 为 .
|
||
for i in range(rel_len):
|
||
ch: t.CChar = mod_buf[i]
|
||
if ch == '/' or ch == '\\':
|
||
mod_buf[i] = '.'
|
||
# 去掉 .py 扩展名
|
||
if rel_len >= 3:
|
||
if mod_buf[rel_len - 3] == '.' and mod_buf[rel_len - 2] == 'p' and mod_buf[rel_len - 1] == 'y':
|
||
mod_buf[rel_len - 3] = '\0'
|
||
|
||
# 获取条目地址
|
||
entry_size: t.CSizeT = FileEntry.__sizeof__()
|
||
entry_addr: t.CUInt64T = t.CUInt64T(result.Entries) + result.Count * entry_size
|
||
entry: FileEntry | t.CPtr = (FileEntry | t.CPtr)(t.CVoid(entry_addr, t.CPtr))
|
||
if entry is None:
|
||
return 1
|
||
|
||
# 复制路径字符串
|
||
abs_len: t.CSizeT = string.strlen(abs_path)
|
||
abs_buf: str = stdlib.malloc(abs_len + 1)
|
||
if abs_buf is None:
|
||
return 1
|
||
string.strcpy(abs_buf, abs_path)
|
||
entry.Path = abs_buf
|
||
|
||
# 复制 SHA1
|
||
sha1_buf: str = stdlib.malloc(17)
|
||
if sha1_buf is None:
|
||
return 1
|
||
string.strcpy(sha1_buf, sha1)
|
||
entry.Sha1 = sha1_buf
|
||
|
||
# 复制相对路径
|
||
rel_buf: str = stdlib.malloc(rel_len + 1)
|
||
if rel_buf is None:
|
||
return 1
|
||
string.strcpy(rel_buf, rel_path)
|
||
entry.RelPath = rel_buf
|
||
|
||
# 复制模块名
|
||
entry.ModuleName = mod_buf
|
||
|
||
result.Count += 1
|
||
return 0
|
||
|
||
|
||
# ============================================================
|
||
# compute_file_sha1 - 读取文件内容并计算 SHA1
|
||
#
|
||
# 与 Projectrans.py 一致:CRLF → LF 转换后计算 SHA1。
|
||
# Projectrans.py 用 Python 文本模式读取(自动 CRLF→LF),
|
||
# IncludesScanner 用二进制模式读取,需手动去除 \r。
|
||
# ============================================================
|
||
def compute_file_sha1(pool: memhub.MemBuddy | t.CPtr,
|
||
file_path: str) -> str:
|
||
"""读取文件内容并计算 SHA1 前16字符(CRLF→LF 后计算,与 Projectrans.py 一致)"""
|
||
if pool is None or file_path is None:
|
||
return None
|
||
|
||
# 打开文件
|
||
f: fileio.File | t.CPtr = fileio.File(file_path, fileio.MODE.R)
|
||
if f.closed:
|
||
return None
|
||
|
||
# 分配读取缓冲区(128KB,足够大多数 .py 文件)
|
||
# 使用 stdlib.malloc 避免 mbuddy 池耗尽(每文件 128KB,60+ 文件会耗尽 16MB 池)
|
||
BUF_SIZE: t.CSizeT = 131072
|
||
buf: bytes = stdlib.malloc(BUF_SIZE)
|
||
if buf is None:
|
||
f.close()
|
||
return None
|
||
|
||
# 读取文件内容
|
||
bytes_read: t.CInt64T = f.read_all(buf, BUF_SIZE)
|
||
f.close()
|
||
if bytes_read <= 0:
|
||
return None
|
||
|
||
# 原地去除 \r(CRLF → LF),与 Projectrans.py 文本模式读取一致
|
||
write_pos: t.CSizeT = 0
|
||
read_pos: t.CSizeT = 0
|
||
while read_pos < bytes_read:
|
||
ch: t.CChar = buf[read_pos]
|
||
if ch != '\r':
|
||
buf[write_pos] = ch
|
||
write_pos += 1
|
||
read_pos += 1
|
||
|
||
# 添加 null 终止符
|
||
buf[write_pos] = 0
|
||
|
||
# 计算 SHA1
|
||
ctx: hashlib.sha1 | t.CPtr = hashlib.sha1()
|
||
if ctx is None:
|
||
return None
|
||
ctx.update(buf)
|
||
|
||
digest: bytes = stdlib.malloc(hashlib.SHA1_DIGEST_LEN)
|
||
if digest is None:
|
||
return None
|
||
ctx.final(digest)
|
||
|
||
# 转为十六进制字符串(取前 8 字节 = 16 个十六进制字符)
|
||
hex_buf: str = stdlib.malloc(17)
|
||
if hex_buf is None:
|
||
return None
|
||
for i in range(8):
|
||
hi: int = (digest[i] >> 4) & 0xF
|
||
lo: int = digest[i] & 0xF
|
||
if hi < 10:
|
||
hex_buf[i * 2] = '0' + hi
|
||
else:
|
||
hex_buf[i * 2] = 'a' + (hi - 10)
|
||
if lo < 10:
|
||
hex_buf[i * 2 + 1] = '0' + lo
|
||
else:
|
||
hex_buf[i * 2 + 1] = 'a' + (lo - 10)
|
||
hex_buf[16] = '\0'
|
||
return hex_buf
|
||
|
||
|
||
# ============================================================
|
||
# scan_directory_recursive - 递归扫描目录
|
||
#
|
||
# 使用 FindFirstFileA/FindNextFileA 遍历目录树,
|
||
# 对每个 .py 文件计算 SHA1 并添加到结果中。
|
||
# ============================================================
|
||
def scan_directory_recursive(pool: memhub.MemBuddy | t.CPtr,
|
||
root_dir: str,
|
||
rel_prefix: str,
|
||
result: ScanResult | t.CPtr) -> int:
|
||
"""递归扫描目录,收集 .py 文件"""
|
||
if pool is None or root_dir is None or result is None:
|
||
return 1
|
||
|
||
# 构造搜索模式: root_dir/*
|
||
root_len: t.CSizeT = string.strlen(root_dir)
|
||
pattern: bytes = stdlib.malloc(root_len + 4)
|
||
if pattern is None:
|
||
return 1
|
||
viperlib.snprintf(pattern, root_len + 4, "%s/*", root_dir)
|
||
|
||
# 使用 FindFirstFileA 开始搜索
|
||
find_data: win32file.WIN32_FIND_DATAA | t.CPtr = stdlib.malloc(win32file.WIN32_FIND_DATAA.__sizeof__())
|
||
if find_data is None:
|
||
return 1
|
||
string.memset(find_data, 0, win32file.WIN32_FIND_DATAA.__sizeof__())
|
||
|
||
handle: win32base.HANDLE = win32file.FindFirstFileA(pattern, find_data)
|
||
if handle == win32base.INVALID_HANDLE_VALUE:
|
||
return 1
|
||
|
||
# 遍历所有文件和子目录
|
||
while True:
|
||
# 跳过 . 和 ..
|
||
fname: str = find_data.cFileName
|
||
if fname is not None:
|
||
fname0: t.CChar = fname[0]
|
||
if fname0 == '.':
|
||
fname1: t.CChar = fname[1]
|
||
if fname1 == '\0':
|
||
# "."
|
||
if win32file.FindNextFileA(handle, find_data) == 0:
|
||
break
|
||
continue
|
||
elif fname1 == '.':
|
||
fname2: t.CChar = fname[2]
|
||
if fname2 == '\0':
|
||
# ".."
|
||
if win32file.FindNextFileA(handle, find_data) == 0:
|
||
break
|
||
continue
|
||
|
||
# 检查是否为目录
|
||
is_dir: int = find_data.dwFileAttributes & win32file.FILE_ATTRIBUTE_DIRECTORY
|
||
|
||
# 构造完整路径
|
||
fname_len: t.CSizeT = string.strlen(fname)
|
||
full_path: bytes = stdlib.malloc(root_len + fname_len + 2)
|
||
if full_path is None:
|
||
break
|
||
viperlib.snprintf(full_path, root_len + fname_len + 2, "%s/%s", root_dir, fname)
|
||
|
||
# 构造相对路径
|
||
prefix_len: t.CSizeT = 0
|
||
if rel_prefix is not None:
|
||
prefix_len = string.strlen(rel_prefix)
|
||
rel_path: bytes = stdlib.malloc(prefix_len + fname_len + 2)
|
||
if rel_path is None:
|
||
break
|
||
if rel_prefix is not None and prefix_len > 0:
|
||
viperlib.snprintf(rel_path, prefix_len + fname_len + 2, "%s/%s", rel_prefix, fname)
|
||
else:
|
||
string.strcpy(rel_path, fname)
|
||
|
||
if is_dir != 0:
|
||
# 递归扫描子目录
|
||
scan_directory_recursive(pool, full_path, rel_path, result)
|
||
else:
|
||
# 检查是否为 .py 文件
|
||
is_py: int = 0
|
||
if fname_len >= 3:
|
||
if fname[fname_len - 3] == '.' and fname[fname_len - 2] == 'p' and fname[fname_len - 1] == 'y':
|
||
is_py = 1
|
||
|
||
if is_py != 0:
|
||
# 跳过 __pycache__ 目录下的文件
|
||
is_pycache: int = 0
|
||
if rel_prefix is not None:
|
||
if string.strcmp(rel_prefix, "__pycache__") == 0:
|
||
is_pycache = 1
|
||
|
||
if is_pycache == 0:
|
||
# 计算 SHA1
|
||
sha1: str = compute_file_sha1(pool, full_path)
|
||
if sha1 is not None:
|
||
add_file_entry(result, pool, full_path, rel_path, sha1)
|
||
stdio.printf(" [scan] %s -> %s\n", rel_path, sha1)
|
||
|
||
# 继续搜索下一个文件
|
||
if win32file.FindNextFileA(handle, find_data) == 0:
|
||
break
|
||
|
||
win32file.FindClose(handle)
|
||
return 0
|
||
|
||
|
||
# ============================================================
|
||
# scan_includes - 扫描 includes 目录入口
|
||
#
|
||
# 扫描指定的 includes 目录,返回所有 .py 文件的路径和 SHA1。
|
||
# ============================================================
|
||
def scan_includes(pool: memhub.MemBuddy | t.CPtr,
|
||
includes_dir: str) -> ScanResult | t.CPtr:
|
||
"""扫描 includes 目录,返回所有 .py 文件的扫描结果"""
|
||
if pool is None or includes_dir is None:
|
||
return None
|
||
|
||
result: ScanResult | t.CPtr = create_scan_result(pool)
|
||
if result is None:
|
||
return None
|
||
|
||
scan_directory_recursive(pool, includes_dir, None, result)
|
||
|
||
return result
|
||
|
||
|
||
# ============================================================
|
||
# find_entry_by_module - 按模块名查找文件条目
|
||
# ============================================================
|
||
def find_entry_by_module(result: ScanResult | t.CPtr,
|
||
module_name: str) -> FileEntry | t.CPtr:
|
||
"""按模块名查找文件条目"""
|
||
if result is None or module_name is None:
|
||
return None
|
||
entry_size: t.CSizeT = FileEntry.__sizeof__()
|
||
for i in range(result.Count):
|
||
entry_addr: t.CUInt64T = t.CUInt64T(result.Entries) + i * entry_size
|
||
entry: FileEntry | t.CPtr = (FileEntry | t.CPtr)(t.CVoid(entry_addr, t.CPtr))
|
||
if entry is not None:
|
||
if entry.ModuleName is not None:
|
||
if entry.ModuleName == module_name:
|
||
return entry
|
||
return None
|