| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298 |
- #!/usr/bin/env python3
- # -*- coding: utf-8 -*-
- r"""配置审计 —— 把"配置目录/文件统一"变成机器每天能查的事 (2026-09-17 用户令 2)。
- ## 查什么 (每条都能指到具体文件/代码行)
- R1 顶层约定 configs/ 顶层只许放 `src/paths.py::TOP_LEVEL_CONFIGS` 里的运行期单件配置 +
- 已登记的域目录 + README.md/registry.yaml。散件一律报出来。
- R2 备份垃圾 任何 `*.bak*` `*~` `*.old` `*.orig` `*.tmp` `*.rej` 都是垃圾 (实测 configs/ 里
- 躺过一个 `serve.json.bak-bomfix`)。
- R3 命名 配置名不得含空格/中文; 全大写视为"命名不统一"(提示级, 因为场名/机型名有历史约定)。
- R4 内容绝对路径 配置**内容**里不得出现本机绝对路径 (`C:\` `D:\` `F:\` `/Users/…` `/Volumes/…` `/home/<人名>`)。
- R5 引用闭合 代码里 `P.config('x')` / `P.config_dir('x')` 指向的文件/目录必须存在; 不存在且**未登记**
- 在 `configs/registry.yaml::known_missing` 的, 按"悬空引用"报错 (登记过的记 `i`)。
- R6 手拼路径 代码里不许再拼 `"configs" / …` 字符串 (白名单: src/paths.py 自身、审计脚本、注释/文档串)。
- R7 无读者 configs/ 下每个文件要么被登记表的 consumers 覆盖, 要么所在域声明了 `no_reader_ok` + 理由。
- R8 场配置 `app_ETL/configs/farms/` 下每个文件要么是**能加载的场定义**(必需键齐), 要么在登记表里
- 声明为"机型/物理约束 profile"或"CSV 数据表" —— 否则就是"配了看不见"的东西。
- ## 退出码 (给 check / rebuild_all 用)
- 0 全部通过 · 5 结构问题(散件/垃圾/未登记域/未登记文件) · 6 悬空引用 · 7 内容含本机绝对路径 · 8 手拼路径 · 9 场配置不合约定
- ## 用法
- python scripts/config_audit.py # 表 + 检查
- python scripts/config_audit.py --list # 只打配置目录表
- python scripts/config_audit.py --json
- """
- from __future__ import annotations
- try:
- from app_common.app_common_guanlan.api import install_root as _install_root
- except ImportError: # 理论不可达;包结构异常时回退到按位置上跳
- from pathlib import Path as _P
- def _install_root(_f): return _P(_f).resolve().parents[4]
- import argparse
- import fnmatch
- import json
- import os
- import pathlib
- import re
- import sys
- ROOT = _install_root(__file__)
- sys.path.insert(0, str(ROOT))
- from src import paths as P # noqa: E402
- REGISTRY = P.config('registry.yaml')
- CONFIG_EXTS = ('.yaml', '.yml', '.json', '.csv', '.example', '.md')
- JUNK = ('*.bak', '*.bak-*', '*~', '*.old', '*.orig', '*.tmp', '*.rej', '*.swp')
- RE_JUNK = re.compile(r'\.(?:bak|old|orig|tmp|rej|swp)(?:$|[-.])|~$')
- RE_BADNAME = re.compile(r'[\s]') # 空格会切断脚本参数, 一律不许
- RE_CN_NAME = re.compile(r'[\u4e00-\u9fff]') # 中文文件名: 项目里大量存在, 只提示不判错
- RE_ABS = re.compile(r'(?<![\w:/])(?:[A-Za-z]:[\\/](?![\\/])|/Users/[A-Za-z0-9_.\-]+|/Volumes/[A-Za-z0-9_\- ]+|/home/[a-z][A-Za-z0-9_.\-]*)')
- # 配置调用的参数可能不止一个: P.config('value_assumptions', f'{farm}.yaml') —— 要把字面量参数全抓下来拼成相对路径
- RE_CONFIG_CALL = re.compile(r'P\.config(?:_dir)?\(([^)]*)\)')
- RE_STR_ARG = re.compile(r'[\'"]([^\'"]+)[\'"]')
- RE_HANDMADE = re.compile(r'''(?:ROOT|root|REPO|here|dst)\s*/\s*['"]configs['"]|['"]configs/[\w.-]+['"]''')
- # 说明性字段 (记录"当时数据在哪台机器上") 里的本机路径算**溯源信息**, 不算 bug —— 但只限注释行与这些键
- RE_DESCRIPTIVE_KEY = re.compile(r'^\s*#|^\s*(?:data_location|data_source|source_root|source_note|note|comment|说明|来源)\s*:')
- SKIP_WALK = {'.venv', '.git', '.github', '__pycache__', 'node_modules', 'wheels', 'vendor',
- 'outputs', 'data', 'logs', 'run', 'release', 'resources', 'reference'}
- def load_registry():
- import yaml
- if not REGISTRY.is_file():
- raise SystemExit(f'缺配置登记表 {P.rel(REGISTRY)}')
- return yaml.safe_load(REGISTRY.read_text(encoding='utf-8')) or {}
- def cfg_roots():
- """配置的根(P9 双根):安装根 configs/(跨模块共用单件)+ 各模块 configs/(模块专属域)。"""
- return (P.CONFIGS,) + tuple(getattr(P, 'MODULE_CONFIG_ROOTS', ()))
- def cfg_files():
- """→ [(文件, 所属根)];两处根都扫(P9 起)。"""
- out = []
- for root in cfg_roots():
- if not root.is_dir():
- continue
- for dp, dn, fns in os.walk(root):
- dn[:] = [d for d in dn if d not in {'__pycache__'}]
- for f in fns:
- out.append((pathlib.Path(dp) / f, root))
- return sorted(out, key=lambda x: str(x[0]))
- def owned_root(domain: str, reg: dict):
- """域 → 它**应当**所在的根(登记表的 owner 字段;未登记 owner 返回 None)。"""
- for d in reg.get('domains') or []:
- if d.get('dir') == domain:
- owner = d.get('owner')
- if not owner or owner == 'root':
- return P.CONFIGS
- return P.ROOT / owner / 'configs'
- return None
- def py_files():
- out = []
- for dp, dn, fns in os.walk(ROOT):
- dn[:] = [d for d in dn if d not in SKIP_WALK]
- out += [pathlib.Path(dp) / f for f in fns if f.endswith('.py')]
- return out
- def audit():
- """→ (rc, results, info)。results = [(level, item, note, rc)]。"""
- reg = load_registry()
- res: list[tuple[str, str, str, int]] = []
- top_ok = {t['file'] for t in reg.get('top_level') or []}
- domain_dirs = {d['dir'] for d in reg.get('domains') or []}
- known_missing = set()
- known_why = {}
- for m in reg.get('known_missing') or []:
- for name in [m['file']] + list(m.get('also_missing') or []):
- known_missing.add(name)
- known_why[name] = m.get('impact', '')
- # ── R1/R2/R3/R4: 文件层(P9 双根:安装根 = 共用单件;模块 configs/ = 模块专属域)
- for f, root in cfg_files():
- rel = f.relative_to(root).as_posix()
- at_install_root = (root == P.CONFIGS)
- if f.name == 'registry.yaml' or f.suffix == '.md':
- continue
- if '/' not in rel:
- if at_install_root:
- if f.name not in top_ok:
- res.append(('X', rel, 'configs/ 顶层只许放已登记的运行期单件配置 (见登记表 top_level)', 5))
- else:
- res.append(('X', f'{P.rel(f)}',
- '模块 configs/ 顶层只许放"域名目录"与 README (单件配置属共用, 放安装根)', 5))
- else:
- d = rel.split('/')[0]
- if d not in domain_dirs:
- res.append(('X', P.rel(f), f'所在域 {d}/ 未在登记表登记 (新域要写明 kind + consumers + owner)', 5))
- else:
- want = owned_root(d, reg)
- if want is not None and root != want:
- res.append(('X', P.rel(f),
- f'域 {d}/ 归属 {want.relative_to(P.ROOT).as_posix()}/,不在该根 ⇒ 配置必须单源', 5))
- if RE_JUNK.search(f.name) or any(fnmatch.fnmatch(f.name, j) for j in JUNK):
- res.append(('X', rel, '备份/垃圾文件 (配置文件只留正本; 改历史靠 git)', 5))
- if RE_BADNAME.search(f.name):
- res.append(('X', rel, '文件名含空格 —— 空格会切断脚本参数, 配置名一律用 [a-z0-9_.-]', 5))
- elif RE_CN_NAME.search(f.name):
- res.append(('?', rel, '文件名含中文 (项目里常见, 不是错; 注意某些工具/编码环境会踩)', 0))
- elif re.search(r'[A-Z]', f.stem) and '/' in rel:
- res.append(('?', rel, '文件名含大写 (历史约定, 不是错; 新文件建议小写)', 0))
- if f.suffix.lower() in ('.yaml', '.yml', '.json', '.csv'):
- try:
- body = f.read_text(encoding='utf-8-sig', errors='replace')
- except OSError:
- body = ''
- hit_desc = hit_code = None
- for ln in body.splitlines():
- m = RE_ABS.search(ln)
- if not m:
- continue
- if RE_DESCRIPTIVE_KEY.match(ln):
- hit_desc = hit_desc or m.group(0)
- else:
- hit_code = hit_code or m.group(0)
- if hit_code:
- res.append(('X', rel, f'内容含本机绝对路径 {hit_code[:24]!r} (在**被读取的字段**里 ⇒ 换机必失配)', 7))
- elif hit_desc:
- res.append(('?', rel, f'说明性字段/注释里记了本机路径 {hit_desc[:24]!r} —— 属溯源信息(当时数据在哪台机器), 不算失配', 0))
- # ── R5/R6: 代码层
- seen_conf = set()
- for p in py_files():
- t = p.read_text(encoding='utf-8', errors='replace')
- if p.name != 'config_audit.py': # 本脚本的说明文字里就有 P.config('x') 之类的例子, 别自证其罪
- for args in RE_CONFIG_CALL.findall(t):
- parts = RE_STR_ARG.findall(args)
- if not parts:
- continue
- rel_call = '/'.join(parts).replace('\\', '/')
- # f-string 参数 (P.config(f'analysis_lock_{farm}.yaml')) 在源码里是字面量带花括号:
- # 把 {…} 当通配符, 只要有任一匹配文件就算"在位", 否则报缺。
- if '{' in rel_call:
- import glob as _glob
- pats = [str(r / re.sub(r'\{[^}]*\}', '*', rel_call)) for r in cfg_roots()]
- if any(_glob.glob(x) for x in pats):
- continue
- wildcard = rel_call
- known = known_missing | {k.replace('<场>', '*') for k in known_missing}
- if any(fnmatch.fnmatch(wildcard, k) for k in known):
- res.append(('i', f'{P.rel(p)} → configs/{wildcard}',
- '悬空引用(已知): per-场配置未随包 (见登记表 known_missing)', 0))
- else:
- res.append(('X', f'{P.rel(p)} → configs/{wildcard}',
- '代码引用的 per-场配置一个都不存在, 且未登记 ⇒ 补文件或删引用', 6))
- continue
- seen_conf.add(rel_call)
- if any((r / rel_call).exists() for r in cfg_roots()):
- continue
- if rel_call in known_missing or any(rel_call.startswith(k.replace('<场>', '')) for k in known_missing):
- res.append(('i', f'{P.rel(p)} → configs/{rel_call}',
- f'悬空引用(已知): {known_why.get(rel_call, "见登记表 known_missing")[:80]}', 0))
- else:
- res.append(('X', f'{P.rel(p)} → configs/{rel_call}',
- '代码引用的配置不存在, 且未登记在 known_missing ⇒ 补文件或删引用', 6))
- for i, ln in enumerate(t.splitlines(), 1):
- s = ln.strip()
- # 豁免"配置取用口自己的定义文件": 原先按旧路径 src/paths.py 判, P1 迁移后它搬到
- # app_common/app_common_guanlan/paths.py —— 改成**按取用口所在文件**判, 位置无关。
- import sys as _s
- _accessor = _s.modules.get(P.config.__module__)
- _accessor_file = pathlib.Path(_accessor.__file__).resolve() if getattr(_accessor, '__file__', None) else None
- if s.startswith('#') or p.name in ('config_audit.py',) or p == P.SRC / 'paths.py' \
- or (_accessor_file is not None and p.resolve() == _accessor_file):
- continue
- if 'config-path-allow' in ln or 'portability-allow' in ln:
- continue
- if RE_HANDMADE.search(ln):
- res.append(('!', f'{P.rel(p)}:{i}', f'手拼 configs 路径: {s[:70]} ⇒ 改用 P.config()/P.config_dir()', 8))
- # ── R7: 无读者
- declared = json.dumps(reg, ensure_ascii=False)
- for d in reg.get('domains') or []:
- if (d.get('consumers') or []) or d.get('no_reader_ok'):
- continue
- res.append(('X', f'{d["dir"]}/', '未声明 consumers, 也没写 no_reader_ok + 理由', 5))
- for f, root in cfg_files():
- rel = f.relative_to(root).as_posix()
- if '/' not in rel or f.suffix == '.md' or f.name in top_ok:
- continue
- # 逐件"有没有人提过它"太吵 (170 件里绝大多数靠域级 consumers 覆盖) —— 只查**域级声明**。
- # 域级没声明 consumers 又没写 no_reader_ok 的, 由上面的域检查报出来。
- _ = rel
- # ── R8: 场配置
- from src.windscada import config as WC
- for f, keys in WC.foreign_farm_files():
- res.append(('i', f'app_ETL/configs/farms/{f.name}', f'非场定义 (顶层键 {keys}) ⇒ 不参与场加载, 已登记为机型/物理约束 profile', 0))
- farm_dom = next((d for d in reg.get('domains') or [] if d['dir'] == 'farms'), {})
- n_prof = sum(1 for f, _ in WC.foreign_farm_files())
- declared_prof = next((s for s in (farm_dom.get('subkinds') or []) if 'profile' in s.get('name', '')), {})
- if declared_prof and int(declared_prof.get('present', -1)) not in (-1, n_prof):
- res.append(('X', 'app_ETL/configs/farms/', f'登记表写 profile {declared_prof.get("present")} 个, 实际 {n_prof} 个', 9))
- for name, src in WC.available().items():
- if src != '内置' and not (P.ROOT / src).is_file():
- res.append(('X', f'app_ETL/configs/farms/{name}', f'available() 列出了 {src}, 但文件不在', 9))
- info = dict(config_files=len(cfg_files()), config_roots=[P.rel(r) for r in cfg_roots()],
- domains=len(domain_dirs),
- top_level=sorted(top_ok),
- known_missing=sorted(known_missing),
- farm_defs=list(WC.available()),
- farm_profiles=n_prof)
- rc_map = {r[3] for r in res if r[0] not in ('OK', 'i', '?') and r[3]}
- return (max(rc_map) if rc_map else 0), res, info
- def main() -> int:
- ap = argparse.ArgumentParser()
- ap.add_argument('--list', action='store_true', help='只打配置目录表')
- ap.add_argument('--json', action='store_true')
- a = ap.parse_args()
- rc, res, info = audit()
- if a.json:
- print(json.dumps(dict(rc=rc, results=[dict(level=r[0], item=r[1], note=r[2], rc=r[3]) for r in res],
- info=info), ensure_ascii=False, indent=1))
- return rc
- if a.list or True:
- print('== 配置目录约定 (configs/registry.yaml) ==')
- print(f' 运行期单件: {", ".join(info["top_level"])}')
- for d in load_registry().get('domains') or []:
- print(f' {d["dir"]:20s} {d.get("kind", "")[:52]}')
- print(f' 配置根: {" · ".join(info.get("config_roots") or [])}')
- print(f' 配置文件合计 {info["config_files"]} 件 · 域 {info["domains"]} 个 · '
- f'场定义 {len(info["farm_defs"])} 个 · 机型 profile {info["farm_profiles"]} 件')
- if not a.list:
- lvl = {}
- for level, item, note, r in res:
- lvl[level] = lvl.get(level, 0) + 1
- print(f'\n== 检查: 不一致 {lvl.get("X", 0)} · 待处理 {lvl.get("!", 0)} · 提示 {lvl.get("?", 0)} · '
- f'已知缺口 {lvl.get("i", 0)} ==')
- for level, item, note, r in res:
- if level in ('X', '!'):
- print(f' [{level}] {item}: {note}')
- seen = set()
- for level, item, note, r in res:
- if level == 'i' and note not in seen:
- seen.add(note)
- print(f' [i] {item}: {note}')
- print(f'结论: {"全部符合约定" if rc == 0 else "见上"}; 退出码 {rc}')
- return rc
- if __name__ == '__main__':
- from src import console
- console.soft()
- sys.exit(main())
|