from __future__ import annotations import ast import cProfile import functools import os import pstats import re import sys import time from dataclasses import dataclass from engine.schema import FlameNode, FunctionNode from engine.harness import _build_user_globals, _load_code, _restore_argv, _scrub_argv_for_user_code, load_source @dataclass class StructureResult: functions: list flame: FlameNode # ────────────────────────────────────────────────────────────────────────────── # 内部测试代码识别 + 量化噪声过滤 # ────────────────────────────────────────────────────────────────────────────── # # 用户原话:「软件内部的测试部分默认百分百过滤掉,不在统计范围内容。完全不显示。」 # —— 所以测试代码 + tottime=0 的帧默认不进 result.json / functions[] / flame。 # # 这是 v5 才加的过滤。在 App 入口(filterAnalysis)也有一份,这里再加一份是为了 # 「不打开 UI 也想看干净数据」的场景:用户拿 result.json 跑自己的聚合脚本时, # 拿到的就是过滤后的数据,而不是 774 帧 + 90% 是噪声。 # # 跟 App 层契约一致:只匹配 CONTEXT(文件路径 / 模块路径) + 引擎内部硬编码白名单, # 完全不看函数名 —— TS 端 isInternalTest 的同样设计原则,误伤 = bug。 import re as _re_noise # 文件名是 test_*.py / *_test.py —— 任意位置都算测试代码(pytest 文件命名约定) _INTERNAL_TEST_FILE_RE = _re_noise.compile(r'[\\/](?:test_[^\\/]+|[^\\/]+_test)\.py$') # 在 tests/ / test/ / __tests__/ **直接下面**的文件 —— pytest 目录约定。 # 收紧到要求「tests/ 后面紧接一个文件名,不能再有 / 子目录」: # * tests/foo.py → 算 (foo.py 直接在 tests/ 下) # * tests/fixtures/foo.py → 不算 (fixtures 是子目录,foo.py 不直接挂在 tests/ 下) # 之前用 `[\\/](tests?|__tests__)[\\/]` 任何含 `/tests/` 的路径都中招,把 fixtures 这类 # 测试数据夹具也当成测试代码误伤 —— 用户脚本只要住在带 tests/ 子目录的路径下, # 函数表直接被过滤成 0 行。 _INTERNAL_TEST_DIR_RE = _re_noise.compile(r'[\\/](tests?|__tests__)[\\/][^\\/]+$') _INTERNAL_TEST_MODULE_RE = _re_noise.compile(r'^(tests?|__tests__)([._]|$)') _INTERNAL_NAMES: frozenset = frozenset(["_pyrof_calib"]) def _is_internal_test(name: str, file: str, module: str) -> bool: """Python 版的 isInternalTest —— 命中即视作「噪声帧」,不进 functions[]。 设计原则(同 TS 版 utils/origin.ts): - 只看语境(文件 / 模块 / 引擎白名单),不看函数名 —— 函数名匹配太容易误伤 (test_helper() / TestCase.test_login() 都是合法业务函数) - 引擎内部硬编码白名单(_pyrof_calib)是兜底防御,任何情况下都不该出现在用户 stats 里 """ if name in _INTERNAL_NAMES: return True # 文件名是 test_*.py / *_test.py —— 任意位置 if _INTERNAL_TEST_FILE_RE.search(file): return True # 在 tests/ / test/ / __tests__/ 直接下面的文件(不能是子目录) if _INTERNAL_TEST_DIR_RE.search(file): return True # module 一定非空(_top_module 不会返回空字符串),不用做 truthy 兜底 if _INTERNAL_TEST_MODULE_RE.match(module): return True return False def _extract_user_imports(src: str) -> set: """从用户脚本源码里抽出顶层显式 import 的模块名集合。 只看模块级 Import / ImportFrom —— 函数/类内部的 import 是延迟副作用,不是 「我需要分析的目标」。返回集合里保留顶级包名(numpy / json / os 等),不展开 as 后的别名(as np → numpy 也在集合里,别名不进集合)。 解析失败(syntax error 等)时返回空集合 —— 调用方已经独立做了 compile 预检, 走到 profile_and_measure 的代码一定是合法 Python,这里只是防御。 """ try: tree = ast.parse(src) except SyntaxError: return set() modules: set = set() for node in tree.body: if isinstance(node, ast.Import): for alias in node.names: # 'import a.b.c' → 'a' 就够了;子包自然跟着顶级包一起保留。 modules.add(alias.name.split(".")[0]) elif isinstance(node, ast.ImportFrom): # 'from . import x' 的 level > 0 是相对导入,跳过 —— 它们依附于某个 # 已知包(用户脚本 / 已导入的第三方),不展开根模块。 # node.level 是 int(0/1/2/...),0 是 falsy,这里只需 if node.level。 if node.level: continue if node.module: modules.add(node.module.split(".")[0]) return modules def _fid(func): file, line, name = func return f"{file}:{line}:{name}" @functools.lru_cache(maxsize=4096) def _norm(path): """规范化路径 —— Windows NTFS 大小写不敏感 + 跨斜杠风格统一。 包 lru_cache:典型 profile 5000 帧 / 50 unique file path —— 之前每个 frame 都跑 abspath + normcase,scope=user 的 hot loop 里反复调; 缓存后只有 50 次真正的 abspath。同一路径跨多次 _top_module / _classify_origin / _make_frame_filter 复用同一结果。 maxsize=4096 覆盖任何真实 profile 看不到的 unique file 数。 """ try: return os.path.normcase(os.path.abspath(path)) except Exception: return path def _make_frame_filter(script_path, scope="user"): """构造帧过滤器。 scope=user(默认):只保留用户脚本里的函数 — 历史上一直如此。 scope=all:保留所有非 cProfile 内部帧(包含标准库和第三方包)— 让耗时 可以"进入到 import 的包里",否则 `pandas.read_csv` 永远是一个黑盒, 看到 3s 也不知道是序列化慢、IO 慢还是解析慢。 `~` 帧的过滤有讲究 —— 之前一刀切 file == '~' 都丢,实际上: - cProfile 用 `~` 作为**所有 C 函数帧**的文件名(time.sleep / numpy C 核 / json 加速 / re 等),丢这些直接打瞎 scope=all - 真正的 cProfile 内部帧靠 name 区分:`'_lsprof.Profiler'` / `'Profiler' / ''` 之类 只剔除 name 含 '_lsprof.Profiler' 的帧,其它 C 函数保留 —— schema 里 origin='builtin' / module='' 也终于能命中真实 builtin 帧。 """ target = _norm(script_path) def _is_kept(func): file, _, name = func # cProfile 内部帧 —— file='~',name 含 "_lsprof.Profiler" 子串 # (实际形态:'Profiler' / '_lsprof.Profiler' / # "") if file == "~" and ("_lsprof.Profiler" in name or name == "Profiler"): return False if scope == "all": return True # 虚拟路径( / / 等)不可能是用户脚本, # 直接 False 省一次 _norm —— _norm("") 会跑 abspath 拼成 # "/" 然后 normcase,跟用户脚本路径比必然不等,但白做一次 # 文件系统查询。~ 在 cProfile 内部已上面短路,这里同等处理。 if file.startswith("<") or file == "~": return False return _norm(file) == target return _is_kept def _top_module(file_path: str, user_script_path: str, user_script_norm: str | None = None) -> str: """把 cProfile 给的 file 路径归类成「顶层模块名」,给 UI 做分类用。 输出样例: - 用户脚本(==user_script_path) → "" - "/usr/lib/python3.11/json/decoder.py" → "json" - "C:\\Python39\\Lib\\json\\decoder.py" → "json" - "C:\\Python39\\Lib\\functools.py" → "functools" ← 顶层 .py 必须剥掉后缀 - "/.../site-packages/numpy/core/array.py" → "numpy" - "" → "" - "" → "" - 其他 → 取倒数第二个目录段作 fallback(基本不会走到) user_script_norm: 调用方预计算的 `_norm(user_script_path)` —— 同一进程内 同一脚本会查几百次,提到外面省 abspath;不传则本函数内现算(保持单点调用方兼容)。 """ if file_path == "~": # C 函数(time.sleep / numpy 加速 / json C decoder 等): # file==~ + name 是 "" # 之前丢光后这里 dead code,现在 _make_frame_filter 不再丢 C 帧, # module 字段需要给出有意义分类 —— 用 "" 跟 origin 字段对齐。 return "" if _norm(file_path) == (user_script_norm if user_script_norm is not None else _norm(user_script_path)): return "" # frozen / built-in / 这种「虚拟」文件:整段作为标签 if file_path.startswith("<"): if file_path.startswith("" if file_path.startswith("" return file_path # 路径规范化:跨平台 + 跨斜杠 norm = file_path.replace("\\", "/") parts = [p for p in norm.split("/") if p] # site-packages / dist-packages:标记之后的第一个目录段就是包名 for marker in ("site-packages", "dist-packages"): if marker in parts: idx = parts.index(marker) + 1 if idx < len(parts): return parts[idx] # 标准库(Windows 安装布局):C:\Python39\Lib\\... 或顶层 .py # `Lib\\functools.py` → "functools"(不是 "functools.py"),用 _strip_py 兜底 if "Lib" in parts: idx = parts.index("Lib") + 1 if idx < len(parts): return _strip_py(parts[idx]) # 标准库(Linux / macOS 安装布局):/usr/lib/python3.X/\... # 之前用 p.startswith("python") + p[6:7].isdigit() 太松散 —— "python3-extra" # 这种目录会被误识别;收紧到严格的 `pythonX(.Y)?` 形式。 for i, p in enumerate(parts): if re.fullmatch(r"python\d+(\.\d+)?", p): if i + 1 < len(parts): return _strip_py(parts[i + 1]) # 兜底:取倒数第二个目录段(例如 ".../myproj/src/utils/helper.py" → "utils") if len(parts) >= 2: return _strip_py(parts[-2]) return _strip_py(file_path) def _strip_py(name: str) -> str: """顶层 .py 文件剥掉扩展名 —— `functools.py` → `functools`。 只剥 `.py` 后缀;其它段('__init__'、'site-packages' 等)原样保留。 非顶层文件不会被这个函数触碰 —— _top_module 把它包在 `parts[idx]` 之外的位置时 返回的就是 `parts[-2]` 这种目录段,永远不带 `.py`;只有顶层 `.py` 才走到这里。""" if name.endswith(".py"): return name[:-3] return name # sys.stdlib_module_names 是 3.10+ 才有的;3.9 及之前要走路径兜底。 # 提前 frozen 一次 —— 同一进程内不变,反复判 in 走 frozenset 是 O(1)。 _STDLIB_MODULES: frozenset | None try: _STDLIB_MODULES = frozenset(getattr(sys, "stdlib_module_names", set())) except Exception: _STDLIB_MODULES = None # 路径兜底"是不是真的在 stdlib 根下"用 sysconfig —— sysconfig.get_paths() 是 # Python 官方给出的 stdlib 根解析工具,远比手算 + 'Lib' / '/lib/pythonX.Y' # 靠谱(venv / embed / framework 几种安装布局都覆盖)。 # # M5 fix:之前只看路径里是否含 "/Lib/" 或 "/lib/pythonX.Y/" —— 用户项目 # 里有 `myproject/Lib/foo.py` 这种就会被错认成 stdlib。现在锚到 sysconfig # 算出的真 stdlib 根上:不是真正的 Python 安装根下面的,一律不算 stdlib。 # # 一次性算好缓存,frozen 之后 hot path 上 O(1) prefix 比对。 def _stdlib_roots() -> tuple: roots: list = [] try: import sysconfig stdlib_path = sysconfig.get_paths().get("stdlib", "") if stdlib_path: roots.append(_norm(stdlib_path)) except Exception: pass # 兜底再放 sys.prefix/Lib —— 有些 embedded 安装 sysconfig 拿不到 fallback = os.path.join(sys.prefix, "Lib") roots.append(_norm(fallback)) return tuple(roots) _STDLIB_ROOTS_NORM: tuple = _stdlib_roots() def _is_under_real_stdlib(file_path: str) -> bool: """判断 file_path 是否在真正的 Python stdlib 根下。 只走"路径兜底"分支 (3.9 / 未知模块名 兜底),已有 _STDLIB_MODULES 命中时不调用本函数,无谓开销。 """ fp = _norm(file_path) for root in _STDLIB_ROOTS_NORM: if fp == root or fp.startswith(root + os.sep): return True return False def _classify_origin(file_path: str, user_script_path: str, module_name: str, user_script_norm: str | None = None) -> str: """给一帧函数归类来源(v3 新增字段 origin)。 返回值(即 JSON 里的字面量,UI 端按这个 group): "user" 用户脚本(路径与 user_script_path 一致) "frozen" 等冻结帧 "builtin" 等 C 实现的 builtin "stdlib" Python 标准库(按 sys.stdlib_module_names 校准;3.10+ 才生效,老版本退化为路径启发式) "third_party" site-packages / dist-packages 下的第三方包 "other" 兜底 —— 例如 、未匹配任何已知布局的奇怪路径 顺序很关键: 1) user / frozen / builtin 用文件路径前缀直接判,O(1) 2) stdlib 先查 sys.stdlib_module_names(权威),命中即返回 3) 路径里出现 site-packages / dist-packages → third_party 4) 路径里出现 /Lib/ 或 /lib/pythonX.Y/ → stdlib(启发式兜底) 5) 其它 → other user_script_norm: 调用方预计算的 `_norm(user_script_path)` —— 同 _top_module, 热路径上几百次调用,提到外面省一次 abspath。 """ # C 扩展函数(time.sleep / numpy C 核 / json C 加速器 等)cProfile 把 file 标成 "~"。 # _top_module 已经把 module 字段定为 ""(与 origin 字段对齐的契约见那里), # 这里也要走 builtin 分支,否则 origin = "other" 与 module = "" 错位,UI # 端按 origin 分组时这条帧会落到别的桶里 —— 之前一直漏到这里。 if file_path == "~": return "builtin" # 虚拟文件路径( / / 等)优先短路 —— 之前 # 先 _norm 再判 < 是浪费 abspath,而且 "<..." 这种路径跟用户脚本路径无论如何 # 都不可能相等,白调一次 norm。顺序调成「<... 优先」后 hot path 上少 100+ # 次 _norm 调用(典型 scope=all 的 profile 里 / 帧占大头)。 if file_path.startswith("<"): if file_path.startswith(" tuple: """v4 单跑架构:一次 exec(code) under cProfile,同时拿到函数归因和 instrumented wall-time。 替代 v3 的两阶段执行(先裸跑测 wall_time 再 cProfile 跑): - 用户脚本只 exec 一次 → plot / print / file-write 等副作用只发生一次 - 返回的 instrumented_wall 含 cProfile 自身开销(典型 1.5~3x 膨胀) - 调用方需要配合 calibrate_cprofile_overhead 折算:wall_time = instrumented / ratio scope: "user"(默认)只归因用户脚本里的函数;"all" 包含所有非 cProfile 内部帧 (标准库 + 第三方包 + 用户代码),让用户能下钻到 import 的包里。 hide_internal (v5 新增,默认 True):过滤掉测试代码 + tottime=0 的量化噪声帧。 - 测试代码:tests/ / test_*.py / _pyrof_calib 等(见 _is_internal_test) - tottime=0:scope=all 时 typing / inspect / functools 等内部展开常被 cProfile 量化精度截到 0, 这些不是优化目标,默认剔除能让 result.json 干净到「只剩真正在跑的代码」 - 设为 False 时不过滤 —— 给想排查调用栈 / 自定义聚合的用户留一条后路 code / src:可选的预读 code object 和源码文本。runner.py 已经在做 syntax precheck 时 read + compile 过一份,这里直接复用 —— 避免重复 IO(原来 profile_and_measure 自己又 read 两次 + AST parse 一次)。这两个参数 给 None 时回退到「自己 load_source + compile」,供老调用方 / 单测继续工作。 NOTE: 不在 cProfile exec 周围禁用 GC —— cProfile 应该看到真实的执行环境 (包括 GC 暂停),折算后的 wall-time 才能反映真实耗时。calibration 那两次 tight-loop 跑各跑各的 GC 策略(详见 harness.calibrate_cprofile_overhead)。 """ # 复用 caller 读好的 code / src —— runner.py 的 syntax precheck 已经 read+compile 过一次, # 再读一次等于把同样的字节流从磁盘捞 2 次 + AST parse 一次。None 时退回到旧的「自己读」路径。 if code is None: code = _load_code(script_path) if src is None and hide_internal: src = load_source(script_path) g = _build_user_globals(script_path) saved_argv = _scrub_argv_for_user_code(script_path) pr = cProfile.Profile() pr.enable() try: t0 = time.perf_counter() exec(code, g) instrumented = time.perf_counter() - t0 finally: pr.disable() _restore_argv(saved_argv) stats = pstats.Stats(pr) is_kept = _make_frame_filter(script_path, scope) # 规范化一次:每个函数帧都会把 file_path 与 script_path 比对;规范化结果 # 与具体帧无关 —— 提到循环外,几百行的 functions 表能省几百次 abspath 调用。 script_norm = _norm(script_path) # 解析用户脚本里显式 import 的模块集合。stdlib 内部帧如果来自「用户没 # 显式 import 的包」(typing/inspect/functools/re/_py_warnings/...), # 一律视为 numpy/pandas 这类第三方包触发的「间接调用链」—— 用户无法优化, # 默认剔除。结果:scope=all 时 result.json 也只剩用户代码 + 显式导入的 # 第三方包 + 真正大头的 stdlib 模块(json/os/etc.,用户写了 `import json` # 就看 json,否则不看)。 # 之前无条件 ast.parse 整个 src —— 即便用户脚本 module 全是 ""(scope=user) # 根本进不到这条 stdlib 过滤分支,几百行的 fixture 也走一遍 AST。改成只在真正会 # 消费 user_imports 的组合里算(scope=all + hide_internal=True)。 if hide_internal and scope == "all": user_imports = _extract_user_imports(src) else: user_imports = set() # file → (module, origin) 缓存:同一文件的多个函数帧(numpy 几百帧共享一个 file) # 只算一次 module + origin。_top_module 和 _classify_origin 各自又会再调一次 # _norm(file),加 cache 后这两个调用也都省了 —— 典型 profile 5000 帧 / 50 文件, # _top_module 从 5000 次降到 50 次,_classify_origin 同。 file_info_cache: dict[str, tuple] = {} def _classify_file(file_path: str) -> tuple: cached = file_info_cache.get(file_path) if cached is not None: return cached module = _top_module(file_path, script_path, script_norm) origin = _classify_origin(file_path, script_path, module, script_norm) cached = (module, origin) file_info_cache[file_path] = cached return cached # ── v6 「用户直接调用」过滤 ── # 用户原话:「只要代码的本身和import 调用的耗时统计,其他的不需要」/「目前好像 # 仍然统计到内部测试的代码了,不合理」—— 之前虽然过滤掉了 stdlib 内部噪声帧 # (typing/inspect/functools 等),但 487 帧里仍有: # - numpy 内部 250+ 帧(np.array 调用的 _core.fromnumeric 等)→ 间接 # - importlib._bootstrap 90 帧 → 间接导入机制 # - C 函数 131 帧(len / numpy C 核 / _warnings / dict.keys 等) # - _distutils_hack 2 帧、mkl 7 帧 → setup machinery,非用户调用 # 用户其实只要: # 1) 自己写的函数(模块名 == "") # 2) 自己「直接调用」的 import 入口(np.array / np.mean / json.dumps 等) # 实现:cProfile 的 callers 字段自带调用方信息。「某帧的 caller 含用户脚本 # 里的帧」=「用户直接调用」。再加 origin 闸门:仅 third_party / 用户显式 import # 的 stdlib 才算「import 调用」—— builtin / frozen / other / 没显式 import 的 # stdlib 一律不保留,即使技术上确实被用户代码调用到(len / print / numpy C 核)。 # ────────────────────────────────────────────────────────────────────────────── functions = [] if hide_internal: # Pass 1:收集用户帧 + 把通过基础过滤的帧的 (module, origin) 缓存下来。 # 注意:这里不再做 tt <= 0 过滤 —— 用户代码 + import 入口里常有纯 C 分派的 # 薄包装(np.random.rand / numpy.__getattr__ 等),cumtime 远大于 0 但 tottime # 恰好压在 cProfile 量化精度地板上,被滤掉就把"用户调用了哪个 API"这条信息丢了。 # 用户代码 + import 入口自然就少,即便有 tt=0 也只是干净 user-function 占位, # 全保留就行。 user_frames: set = set() func_info_cache: dict = {} for func, (cc, nc, tt, ct, callers) in stats.stats.items(): if not is_kept(func): continue file = func[0] module_name, origin = _classify_file(file) if _is_internal_test(func[2], file, module_name): continue func_info_cache[func] = (module_name, origin) if module_name == "": user_frames.add(func) # Pass 2:边判断「是不是 import_callee」边构造 FunctionNode(原版是分两个独立 # pass 跑 import_callees 再跑 functions,合一遍能省掉一次 stats.stats 全量迭代。 # import_callees 设单独 set 也不必要 —— 这里用 include 标志位本地决定,跳出本 # 帧判断后立即 append / continue。 for func, (cc, nc, tt, ct, callers) in stats.stats.items(): info = func_info_cache.get(func) if info is None: continue # pass 1 已过滤 module_name, origin = info if func in user_frames: include = True else: include = False for caller in callers: if caller in user_frames: if origin == "third_party" or ( origin == "stdlib" and module_name in user_imports ): # 用户代码确实调到了这个 import 入口 —— 但还要看 ct: # 低于 cProfile 量化精度(1µs)的「被调到的帧」(典型: # numpy._mean_dispatcher 这种注册期被触发的辅助分发器, # cProfile 把调用方记成 但实际不干活)是噪声。 # 用户代码写的空函数另算(user_frames 不受这条约束)。 include = ct >= 1e-6 break if not include: continue file = func[0] functions.append( FunctionNode( id=_fid(func), file=file, line=func[1], name=func[2], cumtime=ct, tottime=tt, ncalls=nc, percallTot=(tt / nc if nc else 0.0), module=module_name, origin=origin, ) ) else: # hide_internal=False:用户要的是全量原始数据 —— 不做 import_callee / 测试代码 / 零耗时过滤, # 直接把 is_kept 通过的帧全收下来。 for func, (cc, nc, tt, ct, callers) in stats.stats.items(): if not is_kept(func): continue file = func[0] module_name, origin = _classify_file(file) functions.append( FunctionNode( id=_fid(func), file=file, line=func[1], name=func[2], cumtime=ct, tottime=tt, ncalls=nc, percallTot=(tt / nc if nc else 0.0), module=module_name, origin=origin, ) ) functions.sort(key=lambda f: f.tottime, reverse=True) total = sum(f.tottime for f in functions) flame = _build_flame(functions, total) return StructureResult(functions=functions, flame=flame), instrumented def profile_structure( script_path: str, scope: str = "user", hide_internal: bool = True, ) -> StructureResult: """Backward-compat shim:v3 时期暴露的「只拿归因、不读 wall-time」接口。 v4 起实际工作在 profile_and_measure 里完成;保留这个包装是为了不破坏直接 import engine.structure.profile_structure 的测试 / 旧调用方。语义跟 v3 一样: 只跑一次 cProfile exec、返回 functions + flame。 """ result, _ = profile_and_measure(script_path, scope=scope, hide_internal=hide_internal) return result def _build_flame(functions: list, total: float) -> FlameNode: """构造火焰图根节点。 火焰图第一层有两个 layout 选项: - 单模块(scope=user 或刚好只 import 一个包):保持扁平(函数列表), 和 v2 之前完全一致,向后兼容。 - 多模块(scope=all 且命中 ≥2 个不同的顶层模块):按 module 聚合—— 用户问「时间花在了哪个包」时第一眼就能看到;点模块 tile 下钻看内部函数。 """ # 单模块时维持扁平 —— 同名兄弟不会被 module 节点挤占,截断阈值不变 modules = {f.module for f in functions} if len(modules) <= 1: return FlameNode( name="root", value=total, children=[FlameNode(name=f.name, value=f.tottime) for f in functions], ) # 多模块:按 module 聚合;模块自身 value 是该模块下所有函数 tottime 之和 by_module: dict = {} for f in functions: node = by_module.get(f.module) if node is None: node = FlameNode(name=f.module, value=0.0, children=[]) by_module[f.module] = node node.value += f.tottime node.children.append(FlameNode(name=f.name, value=f.tottime)) # 模块按总 tottime 降序,让最贵的包排最左(icicle 布局的视觉约定) sorted_modules = sorted(by_module.values(), key=lambda m: m.value, reverse=True) return FlameNode(name="root", value=total, children=sorted_modules)