#!/usr/bin/env python3 # -*- coding: utf-8 -*- """交付前安全核查 (出"安全包"用). 扫**将要发出去的文件**, 报四类: S1 密钥/凭据: API key / token / 私钥 / .env / 口令 S2 个人信息: 邮箱 / 手机号 / 身份证 / 家目录用户名 S3 D3 敏感: 业主与集团名 / 经济数字(¥·万元) / OEM 原件 S4 内部件: 内部文档、日志、测试快照、开发机绝对路径、git 元数据 用法: python scripts/security_scan.py [--root .] [--zip <包>] [--out logs/security_scan.json] [--strict] --strict: S1 任一命中, 或 S3 在**对外面孔**(门户/报告/契约)命中 → exit 2。 判据说明: 本系统是给业主自己用的内网软件, 所以"场名/台号/业主名"在**包内产物**里是正常的 (那就是他们的数据); 真正不能出的是 S1, 以及把 S3 放进**要上公网或转发第三方**的面孔。两者分开报, 不混。""" from __future__ import annotations import argparse, json, re, sys, time, zipfile from pathlib import Path ROOT_DEFAULT = Path(__file__).resolve().parents[1] PAT = { # AIza 真 key = 39 位且两侧不是 base64 字符 (2026-09-08 实逮: 内联 base64 里 104 位的巧合子串 = 假阳性) "S1.api_key": re.compile(r"(? 50_000_000: big.append((i.filename, i.file_size)) if Path(i.filename).suffix.lower() in TEXT_EXT and i.file_size < 8_000_000: n_text += 1; scan_text(i.filename, z.read(i).decode("utf-8", "ignore"), found) else: for rel, p in iter_files(root): n_files += 1 if p.stat().st_size > 50_000_000: big.append((rel, p.stat().st_size)) if p.suffix.lower() in TEXT_EXT and p.stat().st_size < 8_000_000: n_text += 1; scan_text(rel, p.read_text(encoding="utf-8", errors="ignore"), found) by = {} for f in found: by.setdefault(f["rule"], []).append(f) s1 = [f for f in found if f["rule"].startswith("S1")] s3_face = [f for f in found if f["rule"].startswith("S3") and f["public_face"]] rep = dict(ts=time.strftime("%Y-%m-%dT%H:%M:%S"), target=a.zip or str(root), n_files=n_files, n_text_scanned=n_text, by_rule={k: dict(files=len({x["file"] for x in v}), hits=sum(x["n"] for x in v)) for k, v in sorted(by.items())}, blockers=dict(S1_secrets=len(s1), S3_on_public_face=len(s3_face)), big_files=[dict(file=f, mb=round(s / 1e6, 1)) for f, s in sorted(big, key=lambda x: -x[1])[:10]], findings=found) o = Path(a.out) if Path(a.out).is_absolute() else root / a.out; o.parent.mkdir(parents=True, exist_ok=True) o.write_text(json.dumps(rep, ensure_ascii=False, indent=1), encoding="utf-8") print(f"扫 {n_files} 文件 (文本 {n_text}) → {o}") for k, v in rep["by_rule"].items(): print(f" {k:20s} 文件 {v['files']:4d} 命中 {v['hits']:6d}") print(f" 阻断项: 密钥 {len(s1)} · 对外面孔上的 D3 {len(s3_face)}") for f in (s1 + s3_face)[:8]: print(f" ✗ {f['rule']} {f['file']} {f['samples'][:2]}") if a.strict and (s1 or s3_face): return 2 return 0 if __name__ == "__main__": # 控制台可能是 GBK(中文 Windows 代码页 936): 正文里的 ✔ ✗ ✅ ⚠ 这类字符编不出来会抛 # UnicodeEncodeError, 脚本干成了事却以退出码 1 结束(同类坑见 src/console.py)。降级为 '?' 而不是崩; # 不用 import 是为了兼顾 python -m 与直接当脚本跑两种启动方式。 import sys as _sys for _s in (_sys.stdout, _sys.stderr): try: _s.reconfigure(errors='replace') except Exception: pass sys.exit(main())