Initial import of TransPyV
This commit is contained in:
364
App/lib/core/IncludesScanner.py
Normal file
364
App/lib/core/IncludesScanner.py
Normal file
@@ -0,0 +1,364 @@
|
||||
import t, c
|
||||
from stdint import *
|
||||
import memhub
|
||||
import string
|
||||
import stdio
|
||||
import stdlib
|
||||
import w32.win32file as win32file
|
||||
import w32.win32base as win32base
|
||||
import w32.fileio as fileio
|
||||
import hashlib
|
||||
import viperlib
|
||||
|
||||
|
||||
# ============================================================
|
||||
# IncludesScanner - includes 目录递归扫描
|
||||
#
|
||||
# 递归扫描 includes 目录,收集所有 .py 文件路径和 SHA1,
|
||||
# 为 Phase1 stub 生成和 Phase2 stub 合并提供基础数据。
|
||||
#
|
||||
# 使用 Win32 FindFirstFileA/FindNextFileA 实现目录遍历。
|
||||
# ============================================================
|
||||
|
||||
# 全局 mbuddy 指针
|
||||
_mbuddy: memhub.MemManager | t.CPtr
|
||||
|
||||
# 文件路径最大长度
|
||||
MAX_PATH_LEN: t.CDefine = 512
|
||||
# 单次扫描最大文件数
|
||||
MAX_FILES: t.CDefine = 256
|
||||
|
||||
|
||||
# ============================================================
|
||||
# FileEntry - 文件条目(路径 + SHA1)
|
||||
# ============================================================
|
||||
@t.NoVTable
|
||||
class FileEntry:
|
||||
"""扫描到的文件条目"""
|
||||
Path: str # 文件绝对路径
|
||||
Sha1: str # SHA1 前16字符(16字节+null)
|
||||
RelPath: str # 相对于 includes 根目录的路径
|
||||
ModuleName: str # 模块名(如 "ast.parser")
|
||||
|
||||
|
||||
# ============================================================
|
||||
# ScanResult - 扫描结果
|
||||
# ============================================================
|
||||
@t.NoVTable
|
||||
class ScanResult:
|
||||
"""扫描结果集合"""
|
||||
Entries: FileEntry | t.CPtr # FileEntry 数组
|
||||
Count: t.CInt
|
||||
Capacity: t.CInt
|
||||
|
||||
|
||||
# ============================================================
|
||||
# create_scan_result - 创建扫描结果容器
|
||||
# ============================================================
|
||||
def create_scan_result(pool: memhub.MemBuddy | t.CPtr) -> ScanResult | t.CPtr:
|
||||
"""创建扫描结果容器,预分配 MAX_FILES 个槽位"""
|
||||
if pool is None:
|
||||
return None
|
||||
size: t.CSizeT = MAX_FILES * FileEntry.__sizeof__()
|
||||
# 使用 stdlib.malloc 避免 mbuddy 池耗尽(与 StubMerger 保持一致)
|
||||
entries: FileEntry | t.CPtr = stdlib.malloc(size)
|
||||
if entries is None:
|
||||
return None
|
||||
string.memset(entries, 0, size)
|
||||
|
||||
result: ScanResult | t.CPtr = stdlib.malloc(ScanResult.__sizeof__())
|
||||
if result is None:
|
||||
return None
|
||||
string.memset(result, 0, ScanResult.__sizeof__())
|
||||
result.Entries = entries
|
||||
result.Count = 0
|
||||
result.Capacity = MAX_FILES
|
||||
return result
|
||||
|
||||
|
||||
# ============================================================
|
||||
# add_file_entry - 向扫描结果添加文件条目
|
||||
# ============================================================
|
||||
def add_file_entry(result: ScanResult | t.CPtr,
|
||||
pool: memhub.MemBuddy | t.CPtr,
|
||||
abs_path: str, rel_path: str,
|
||||
sha1: str) -> int:
|
||||
"""添加文件条目到扫描结果,返回 0 成功"""
|
||||
if result is None or pool is None:
|
||||
return 1
|
||||
if result.Count >= result.Capacity:
|
||||
return 1
|
||||
|
||||
# 计算模块名:rel_path 中的 / 替换为 .,去掉 .py 扩展名
|
||||
rel_len: t.CSizeT = string.strlen(rel_path)
|
||||
mod_buf: str = stdlib.malloc(rel_len + 1)
|
||||
if mod_buf is None:
|
||||
return 1
|
||||
string.strcpy(mod_buf, rel_path)
|
||||
# 替换 / 为 .
|
||||
for i in range(rel_len):
|
||||
ch: t.CChar = mod_buf[i]
|
||||
if ch == '/' or ch == '\\':
|
||||
mod_buf[i] = '.'
|
||||
# 去掉 .py 扩展名
|
||||
if rel_len >= 3:
|
||||
if mod_buf[rel_len - 3] == '.' and mod_buf[rel_len - 2] == 'p' and mod_buf[rel_len - 1] == 'y':
|
||||
mod_buf[rel_len - 3] = '\0'
|
||||
|
||||
# 获取条目地址
|
||||
entry_size: t.CSizeT = FileEntry.__sizeof__()
|
||||
entry_addr: t.CUInt64T = t.CUInt64T(result.Entries) + result.Count * entry_size
|
||||
entry: FileEntry | t.CPtr = (FileEntry | t.CPtr)(t.CVoid(entry_addr, t.CPtr))
|
||||
if entry is None:
|
||||
return 1
|
||||
|
||||
# 复制路径字符串
|
||||
abs_len: t.CSizeT = string.strlen(abs_path)
|
||||
abs_buf: str = stdlib.malloc(abs_len + 1)
|
||||
if abs_buf is None:
|
||||
return 1
|
||||
string.strcpy(abs_buf, abs_path)
|
||||
entry.Path = abs_buf
|
||||
|
||||
# 复制 SHA1
|
||||
sha1_buf: str = stdlib.malloc(17)
|
||||
if sha1_buf is None:
|
||||
return 1
|
||||
string.strcpy(sha1_buf, sha1)
|
||||
entry.Sha1 = sha1_buf
|
||||
|
||||
# 复制相对路径
|
||||
rel_buf: str = stdlib.malloc(rel_len + 1)
|
||||
if rel_buf is None:
|
||||
return 1
|
||||
string.strcpy(rel_buf, rel_path)
|
||||
entry.RelPath = rel_buf
|
||||
|
||||
# 复制模块名
|
||||
entry.ModuleName = mod_buf
|
||||
|
||||
result.Count += 1
|
||||
return 0
|
||||
|
||||
|
||||
# ============================================================
|
||||
# compute_file_sha1 - 读取文件内容并计算 SHA1
|
||||
#
|
||||
# 与 Projectrans.py 一致:CRLF → LF 转换后计算 SHA1。
|
||||
# Projectrans.py 用 Python 文本模式读取(自动 CRLF→LF),
|
||||
# IncludesScanner 用二进制模式读取,需手动去除 \r。
|
||||
# ============================================================
|
||||
def compute_file_sha1(pool: memhub.MemBuddy | t.CPtr,
|
||||
file_path: str) -> str:
|
||||
"""读取文件内容并计算 SHA1 前16字符(CRLF→LF 后计算,与 Projectrans.py 一致)"""
|
||||
if pool is None or file_path is None:
|
||||
return None
|
||||
|
||||
# 打开文件
|
||||
f: fileio.File | t.CPtr = fileio.File(file_path, fileio.MODE.R)
|
||||
if f.closed:
|
||||
return None
|
||||
|
||||
# 分配读取缓冲区(128KB,足够大多数 .py 文件)
|
||||
# 使用 stdlib.malloc 避免 mbuddy 池耗尽(每文件 128KB,60+ 文件会耗尽 16MB 池)
|
||||
BUF_SIZE: t.CSizeT = 131072
|
||||
buf: bytes = stdlib.malloc(BUF_SIZE)
|
||||
if buf is None:
|
||||
f.close()
|
||||
return None
|
||||
|
||||
# 读取文件内容
|
||||
bytes_read: t.CInt64T = f.read_all(buf, BUF_SIZE)
|
||||
f.close()
|
||||
if bytes_read <= 0:
|
||||
return None
|
||||
|
||||
# 原地去除 \r(CRLF → LF),与 Projectrans.py 文本模式读取一致
|
||||
write_pos: t.CSizeT = 0
|
||||
read_pos: t.CSizeT = 0
|
||||
while read_pos < bytes_read:
|
||||
ch: t.CChar = buf[read_pos]
|
||||
if ch != '\r':
|
||||
buf[write_pos] = ch
|
||||
write_pos += 1
|
||||
read_pos += 1
|
||||
|
||||
# 添加 null 终止符
|
||||
buf[write_pos] = 0
|
||||
|
||||
# 计算 SHA1
|
||||
ctx: hashlib.sha1 | t.CPtr = hashlib.sha1()
|
||||
if ctx is None:
|
||||
return None
|
||||
ctx.update(buf)
|
||||
|
||||
digest: bytes = stdlib.malloc(hashlib.SHA1_DIGEST_LEN)
|
||||
if digest is None:
|
||||
return None
|
||||
ctx.final(digest)
|
||||
|
||||
# 转为十六进制字符串(取前 8 字节 = 16 个十六进制字符)
|
||||
hex_buf: str = stdlib.malloc(17)
|
||||
if hex_buf is None:
|
||||
return None
|
||||
for i in range(8):
|
||||
hi: int = (digest[i] >> 4) & 0xF
|
||||
lo: int = digest[i] & 0xF
|
||||
if hi < 10:
|
||||
hex_buf[i * 2] = '0' + hi
|
||||
else:
|
||||
hex_buf[i * 2] = 'a' + (hi - 10)
|
||||
if lo < 10:
|
||||
hex_buf[i * 2 + 1] = '0' + lo
|
||||
else:
|
||||
hex_buf[i * 2 + 1] = 'a' + (lo - 10)
|
||||
hex_buf[16] = '\0'
|
||||
return hex_buf
|
||||
|
||||
|
||||
# ============================================================
|
||||
# scan_directory_recursive - 递归扫描目录
|
||||
#
|
||||
# 使用 FindFirstFileA/FindNextFileA 遍历目录树,
|
||||
# 对每个 .py 文件计算 SHA1 并添加到结果中。
|
||||
# ============================================================
|
||||
def scan_directory_recursive(pool: memhub.MemBuddy | t.CPtr,
|
||||
root_dir: str,
|
||||
rel_prefix: str,
|
||||
result: ScanResult | t.CPtr) -> int:
|
||||
"""递归扫描目录,收集 .py 文件"""
|
||||
if pool is None or root_dir is None or result is None:
|
||||
return 1
|
||||
|
||||
# 构造搜索模式: root_dir/*
|
||||
root_len: t.CSizeT = string.strlen(root_dir)
|
||||
pattern: bytes = stdlib.malloc(root_len + 4)
|
||||
if pattern is None:
|
||||
return 1
|
||||
viperlib.snprintf(pattern, root_len + 4, "%s/*", root_dir)
|
||||
|
||||
# 使用 FindFirstFileA 开始搜索
|
||||
find_data: win32file.WIN32_FIND_DATAA | t.CPtr = stdlib.malloc(win32file.WIN32_FIND_DATAA.__sizeof__())
|
||||
if find_data is None:
|
||||
return 1
|
||||
string.memset(find_data, 0, win32file.WIN32_FIND_DATAA.__sizeof__())
|
||||
|
||||
handle: win32base.HANDLE = win32file.FindFirstFileA(pattern, find_data)
|
||||
if handle == win32base.INVALID_HANDLE_VALUE:
|
||||
return 1
|
||||
|
||||
# 遍历所有文件和子目录
|
||||
while True:
|
||||
# 跳过 . 和 ..
|
||||
fname: str = find_data.cFileName
|
||||
if fname is not None:
|
||||
fname0: t.CChar = fname[0]
|
||||
if fname0 == '.':
|
||||
fname1: t.CChar = fname[1]
|
||||
if fname1 == '\0':
|
||||
# "."
|
||||
if win32file.FindNextFileA(handle, find_data) == 0:
|
||||
break
|
||||
continue
|
||||
elif fname1 == '.':
|
||||
fname2: t.CChar = fname[2]
|
||||
if fname2 == '\0':
|
||||
# ".."
|
||||
if win32file.FindNextFileA(handle, find_data) == 0:
|
||||
break
|
||||
continue
|
||||
|
||||
# 检查是否为目录
|
||||
is_dir: int = find_data.dwFileAttributes & win32file.FILE_ATTRIBUTE_DIRECTORY
|
||||
|
||||
# 构造完整路径
|
||||
fname_len: t.CSizeT = string.strlen(fname)
|
||||
full_path: bytes = stdlib.malloc(root_len + fname_len + 2)
|
||||
if full_path is None:
|
||||
break
|
||||
viperlib.snprintf(full_path, root_len + fname_len + 2, "%s/%s", root_dir, fname)
|
||||
|
||||
# 构造相对路径
|
||||
prefix_len: t.CSizeT = 0
|
||||
if rel_prefix is not None:
|
||||
prefix_len = string.strlen(rel_prefix)
|
||||
rel_path: bytes = stdlib.malloc(prefix_len + fname_len + 2)
|
||||
if rel_path is None:
|
||||
break
|
||||
if rel_prefix is not None and prefix_len > 0:
|
||||
viperlib.snprintf(rel_path, prefix_len + fname_len + 2, "%s/%s", rel_prefix, fname)
|
||||
else:
|
||||
string.strcpy(rel_path, fname)
|
||||
|
||||
if is_dir != 0:
|
||||
# 递归扫描子目录
|
||||
scan_directory_recursive(pool, full_path, rel_path, result)
|
||||
else:
|
||||
# 检查是否为 .py 文件
|
||||
is_py: int = 0
|
||||
if fname_len >= 3:
|
||||
if fname[fname_len - 3] == '.' and fname[fname_len - 2] == 'p' and fname[fname_len - 1] == 'y':
|
||||
is_py = 1
|
||||
|
||||
if is_py != 0:
|
||||
# 跳过 __pycache__ 目录下的文件
|
||||
is_pycache: int = 0
|
||||
if rel_prefix is not None:
|
||||
if string.strcmp(rel_prefix, "__pycache__") == 0:
|
||||
is_pycache = 1
|
||||
|
||||
if is_pycache == 0:
|
||||
# 计算 SHA1
|
||||
sha1: str = compute_file_sha1(pool, full_path)
|
||||
if sha1 is not None:
|
||||
add_file_entry(result, pool, full_path, rel_path, sha1)
|
||||
stdio.printf(" [scan] %s -> %s\n", rel_path, sha1)
|
||||
|
||||
# 继续搜索下一个文件
|
||||
if win32file.FindNextFileA(handle, find_data) == 0:
|
||||
break
|
||||
|
||||
win32file.FindClose(handle)
|
||||
return 0
|
||||
|
||||
|
||||
# ============================================================
|
||||
# scan_includes - 扫描 includes 目录入口
|
||||
#
|
||||
# 扫描指定的 includes 目录,返回所有 .py 文件的路径和 SHA1。
|
||||
# ============================================================
|
||||
def scan_includes(pool: memhub.MemBuddy | t.CPtr,
|
||||
includes_dir: str) -> ScanResult | t.CPtr:
|
||||
"""扫描 includes 目录,返回所有 .py 文件的扫描结果"""
|
||||
if pool is None or includes_dir is None:
|
||||
return None
|
||||
|
||||
stdio.printf("[Phase1] 扫描 includes 目录: %s\n", includes_dir)
|
||||
|
||||
result: ScanResult | t.CPtr = create_scan_result(pool)
|
||||
if result is None:
|
||||
return None
|
||||
|
||||
scan_directory_recursive(pool, includes_dir, None, result)
|
||||
|
||||
stdio.printf("[Phase1] 扫描完成: %d 个 .py 文件\n", result.Count)
|
||||
return result
|
||||
|
||||
|
||||
# ============================================================
|
||||
# find_entry_by_module - 按模块名查找文件条目
|
||||
# ============================================================
|
||||
def find_entry_by_module(result: ScanResult | t.CPtr,
|
||||
module_name: str) -> FileEntry | t.CPtr:
|
||||
"""按模块名查找文件条目"""
|
||||
if result is None or module_name is None:
|
||||
return None
|
||||
entry_size: t.CSizeT = FileEntry.__sizeof__()
|
||||
for i in range(result.Count):
|
||||
entry_addr: t.CUInt64T = t.CUInt64T(result.Entries) + i * entry_size
|
||||
entry: FileEntry | t.CPtr = (FileEntry | t.CPtr)(t.CVoid(entry_addr, t.CPtr))
|
||||
if entry is not None:
|
||||
if entry.ModuleName is not None:
|
||||
if entry.ModuleName == module_name:
|
||||
return entry
|
||||
return None
|
||||
Reference in New Issue
Block a user