#!/usr/bin/env python3 # -*- coding: utf-8 -*- r"""日志审计 —— 把"日志目录/命名/格式统一"变成机器每天能查的事 (2026-09-17 用户令 2)。 ## 五条规则 L1 只在 logs/ 下 全仓的 `*.log` / `*.jsonl` 只许出现在 `logs/` (白名单: 产物里的 `analyze_stdout.log` 属**产物附件**, 与运行日志不是一回事, 见登记理由)。 L2 命名 `logs/<组件>.log`(组件 = configs/serve.json 的服务键 + gateway/serve/start_hidden)、 `logs/ops/<动作>_.log`、`logs/audit/<名字>.jsonl`。别的名字报出来。 L3 行格式 每一行必须匹配 `YYYY-MM-DD HH:MM:SS LEVEL 组件 消息` (src/logfile.py::LINE_RE)。 **按文件末尾 200 行判**: 启用统一格式之前的旧行不算数 (这批已归档到 logs/legacy/)。 L4 行尾与颜色 logs/ 下的文本日志不许有 CRLF、不许有 ANSI 颜色码 (重定向到文件后颜色码是垃圾字节)。 L5 保留策略 `logs/ops/` 只留最近 20 份 / 30 天; 超出即报 (并可用 --prune 就地清理)。 `logs/audit/*.jsonl` 每行必须是合法 JSON。 ## 退出码 0 全部通过 · 5 日志跑到 logs/ 外面 · 6 命名不合口径 · 7 行格式不合规 · 8 行尾/颜色问题 · 9 保留策略超限 ## 用法 python scripts/log_audit.py # 检查 python scripts/log_audit.py --prune # 检查并就地清理超限的动作日志 python scripts/log_audit.py --json """ from __future__ import annotations import argparse import json import os import pathlib import re import sys ROOT = pathlib.Path(__file__).resolve().parents[1] sys.path.insert(0, str(ROOT)) from src import logfile as L # noqa: E402 from src import paths as P # noqa: E402 SKIP_WALK = {'.venv', '.git', '.github', '__pycache__', 'node_modules', 'wheels', 'vendor', '.git-*'} # logs/ 之外的日志白名单: (正则, 理由)。两类: # ① 产物附件: CMS 插件把"这次分析的 stdout"当产物附件留在产物目录; # ② 第三方工具的缓存/配置库: npm 缓存、Puppeteer(Chromium) 的 QA profile —— 它们的 000003.log # 是 LevelDB 内部文件, 不是本系统的运行日志 (删不删是另一件事, 见 docs §11 的清理建议)。 OUTSIDE_OK = [ (re.compile(r'^outputs/.*/windcms/.*analyze_stdout\.log$'), 'CMS 插件把"这次分析的 stdout"当作产物附件留在产物目录 (属产物, 不是运行日志)'), (re.compile(r'^(?:.*/)?(?:\.npm-cache|\.qa-profile)/'), '第三方工具 (npm 缓存 / Chromium QA profile) 自带的状态文件, 非本系统日志'), (re.compile(r'^(?:.*/)?node_modules/'), 'node_modules 里的第三方包自带日志'), ] TAIL_LINES = 200 def scan_outside() -> list[pathlib.Path]: found = [] for dp, dn, fns in os.walk(ROOT): dn[:] = [d for d in dn if d not in SKIP_WALK and not d.startswith('.git')] rel_dir = pathlib.Path(dp).relative_to(ROOT).as_posix() if rel_dir == 'logs' or rel_dir.startswith('logs/'): dn[:] = [] continue for f in fns: if f.endswith(('.log', '.jsonl')): found.append(pathlib.Path(dp) / f) return found def tail_lines(p: pathlib.Path, n: int = TAIL_LINES) -> list[str]: try: data = p.read_bytes() except OSError: return [] txt = data.decode('utf-8', 'replace') lines = txt.replace('\r\n', '\n').split('\n') return lines[-n:] def migrate_product_logs(dry: bool = True) -> list: """把产物目录里的日志搬进 `logs/build/`, 并同步台账 (用户令 2: 日志不许留在产物里)。 为什么要搬: 交付包里 39 个 `.log` 躺在 `outputs/<场>/…` 下, 还被 `_provenance.json` 登记成 `shipped` 随包件 —— 产物台账里混着日志, 排障时也找不到。搬迁后: · 文件去 `logs/build/<场>/<原相对路径>`; · 台账里对应条目**删掉并重算计数**, 并写明"因日志归位而移出" (只搬文件不改台账 = 台账失真)。 返回 (moved, ledger_notes)。 """ moved, notes = [], [] out = P.ROOT / 'outputs' if not out.is_dir(): return moved, notes for f in sorted(out.rglob('*.log')): rel = f.relative_to(out) # 如 rudong/windscada/temp_build.log dst = L.LOGS / 'build' / rel moved.append((f, dst)) if dry: continue dst.parent.mkdir(parents=True, exist_ok=True) if dst.exists(): dst = dst.with_name(f'{dst.stem}_{int(dst.stat().st_mtime)}{dst.suffix}') f.replace(dst) if dry or not moved: return moved, notes for farm_dir in sorted(x for x in out.iterdir() if x.is_dir()): prov_p = farm_dir / '_provenance.json' if not prov_p.is_file(): continue prov = json.loads(prov_p.read_text(encoding='utf-8')) files = prov.get('files') or {} drop = [k for k in files if str(k).endswith('.log')] for k in drop: files.pop(k, None) if drop: counts = {} for v in files.values(): s = v.get('source') if isinstance(v, dict) else '?' counts[s] = counts.get(s, 0) + 1 prov['counts'] = counts prov['files'] = files prov['note'] = (str(prov.get('note', '')) + f' | 2026-09-17 因「日志归位」(用户令 2) 移出 {len(drop)} 个 .log 条目: ' f'那批是构建日志, 现位于 logs/build/{farm_dir.name}/ 下') prov_p.write_text(json.dumps(prov, ensure_ascii=False, indent=1), encoding='utf-8') notes.append(f'{farm_dir.name}/_provenance.json: 台账移出 {len(drop)} 个 .log 条目并重算计数') return moved, notes def audit() -> tuple[int, list, dict]: res: list[tuple[str, str, str, int]] = [] L.ensure_dirs() n_svc = n_ops = n_audit = n_build = 0 # L1 只在 logs/ 下 for f in scan_outside(): rel = f.relative_to(ROOT).as_posix() if any(r.match(rel) for r, _ in OUTSIDE_OK): why = next(w for r, w in OUTSIDE_OK if r.match(rel)) res.append(('i', rel, f'白名单: {why}', 0)) else: res.append(('X', rel, '日志跑到 logs/ 外面了 (用户令 2: 运行日志只在 logs/)', 5)) # L2/L3/L4/L5 for f in sorted(L.LOGS.rglob('*')): if not f.is_file(): continue rel = f.relative_to(ROOT).as_posix() if 'legacy' in f.parts: continue if f.suffix == '.jsonl': n_audit += 1 for i, ln in enumerate(tail_lines(f), 1): if not ln.strip(): continue try: json.loads(ln) except Exception: res.append(('X', rel, f'审计流水第 {i} 行不是合法 JSON (jsonl 一行一条)', 7)) break if f.parent != L.AUDIT_DIR: res.append(('!', rel, '审计流水应放 logs/audit/ 下', 6)) continue if f.suffix != '.log': if f.name.endswith('.json'): n_audit += 1 if f.parent != L.AUDIT_DIR: res.append(('!', rel, '审计/报告类 JSON 应放 logs/audit/ 下', 6)) continue if f.parent == L.OPS_DIR: n_ops += 1 if not re.match(r'^[\w\-]+_\d{8}-\d{6}\.log$', f.name): res.append(('!', rel, '动作日志命名应为 logs/ops/<动作>_.log', 6)) elif L.BUILD_DIR in f.parents: n_build += 1 # 构建日志: 允许按产物子路径归档 (logs/build/<场>/<原相对路径>.log)。 # 这些是**从产物目录归位过来的历史文件**, 由未随包的构建器写成, 无法追溯重排格式 ⇒ 只提示。 lines = [ln for ln in tail_lines(f) if ln.strip()] bad = [ln for ln in lines if not L.LINE_RE.match(ln)] if bad: res.append(('?', rel, f'历史构建日志 {len(bad)}/{len(lines)} 行不是统一格式 ' f'(写它的人不在本包; 新构建器请用 logfile.build_log() + logfile.line())', 0)) continue # 历史构建日志不再套"服务日志"那套严格检查 (没法追溯重排) elif f.parent == L.LOGS: n_svc += 1 if not re.match(r'^[\w\u4e00-\u9fff.\-]+\.log$', f.name): res.append(('!', rel, '服务日志命名应为 logs/<组件>.log', 6)) else: res.append(('!', rel, 'logs/ 下只允许 服务日志(顶层)、ops/、audit/、build/ 四处', 6)) raw = f.read_bytes() if b'\r\n' in raw: res.append(('?', rel, f'含 CRLF ({raw.count(bytes([13, 10]))} 处) —— 新写日志统一 LF', 0)) if L.ANSI_RE.search(raw.decode('utf-8', 'replace')): res.append(('?', rel, '含 ANSI 颜色码 (历史行; 重定向到文件后是垃圾字节)', 0)) lines = [ln for ln in tail_lines(f) if ln.strip()] if lines: bad = [ln for ln in lines if not L.LINE_RE.match(ln)] if len(bad) == len(lines): res.append(('X', rel, f'末尾 {len(lines)} 行都不符合统一格式 (应为 "时间戳 级别 组件 消息")', 7)) elif bad: res.append(('?', rel, f'末尾 {len(lines)} 行里 {len(bad)} 行不符格式 (多为启用前的旧行)', 0)) # L5 保留策略 over = L.prune_actions(dry=True) if over: res.append(('X', f'{P.rel(L.OPS_DIR)}', f'{len(over)} 份动作日志超出保留策略 (最近 {L.ACTION_KEEP} 份 / {L.ACTION_DAYS} 天)', 9)) info = dict(service_logs=n_svc, ops_logs=n_ops, audit_files=n_audit, build_logs=n_build, ops_over_quota=len(over), logs_dir=str(L.LOGS.relative_to(ROOT))) rc_map = {r[3] for r in res if r[0] not in ('OK', 'i', '?') and r[3]} return (max(rc_map) if rc_map else 0), res, info def main() -> int: ap = argparse.ArgumentParser() ap.add_argument('--prune', action='store_true', help='就地清理超限的动作日志') ap.add_argument('--migrate', action='store_true', help='把产物目录里的日志搬到 logs/build/ 并同步台账') ap.add_argument('--yes', action='store_true', help='--migrate 时真正执行 (默认只预演)') ap.add_argument('--json', action='store_true') a = ap.parse_args() rc, res, info = audit() if a.migrate: moved, notes = migrate_product_logs(dry=not a.yes) print(f'{"预演" if not a.yes else "已执行"}日志归位: {len(moved)} 个文件 ' f'(产物目录 → logs/build/)') for src, dst in moved[:5]: print(f' {src.relative_to(ROOT).as_posix()} → {dst.relative_to(ROOT).as_posix()}') if len(moved) > 5: print(f' … 另有 {len(moved) - 5} 个') for n in notes: print(f' 台账: {n}') if not a.yes: print(' (要真搬请加 --yes)') rc, res, info = audit() if a.prune: gone = L.prune_actions() print(f'已清理动作日志 {len(gone)} 份: {[g.name for g in gone[:5]]}{" …" if len(gone) > 5 else ""}') rc, res, info = audit() if a.json: print(json.dumps(dict(rc=rc, results=[dict(level=r[0], item=r[1], note=r[2], rc=r[3]) for r in res], info=info), ensure_ascii=False, indent=1)) return rc print('== 日志口径 (src/logfile.py, 用户令 2) ==') print(f' 服务日志 {info["service_logs"]} 份 (logs/<组件>.log) · 动作日志 {info["ops_logs"]} 份 ' f'(logs/ops/, 保留 {L.ACTION_KEEP} 份/{L.ACTION_DAYS} 天) · 审计流水 {info["audit_files"]} 份 (logs/audit/) · ' f'构建日志 {info["build_logs"]} 份 (logs/build/<场>/…)') print(f' 行格式 {L.line("INFO", "示例", "一行长这样")}') lvl = {} for level, item, note, r in res: lvl[level] = lvl.get(level, 0) + 1 print(f'\n== 检查: 不一致 {lvl.get("X", 0)} · 待处理 {lvl.get("!", 0)} · 提示 {lvl.get("?", 0)} · 白名单 {lvl.get("i", 0)} ==') for level, item, note, r in res: if level in ('X', '!'): print(f' [{level}] {item}: {note}') seen = set() for level, item, note, r in res: if level == '?' and note not in seen: seen.add(note) print(f' [?] {item}: {note}') print(f'结论: {"全部符合口径" if rc == 0 else "见上"}; 退出码 {rc}') return rc if __name__ == '__main__': from src import console console.soft() sys.exit(main())