security_scan.py 6.7 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100
  1. #!/usr/bin/env python3
  2. # -*- coding: utf-8 -*-
  3. """交付前安全核查 (出"安全包"用). 扫**将要发出去的文件**, 报四类:
  4. S1 密钥/凭据: API key / token / 私钥 / .env / 口令
  5. S2 个人信息: 邮箱 / 手机号 / 身份证 / 家目录用户名
  6. S3 D3 敏感: 业主与集团名 / 经济数字(¥·万元) / OEM 原件
  7. S4 内部件: 内部文档、日志、测试快照、开发机绝对路径、git 元数据
  8. 用法: python scripts/security_scan.py [--root .] [--zip <包>] [--out logs/security_scan.json] [--strict]
  9. --strict: S1 任一命中, 或 S3 在**对外面孔**(门户/报告/契约)命中 → exit 2。
  10. 判据说明: 本系统是给业主自己用的内网软件, 所以"场名/台号/业主名"在**包内产物**里是正常的 (那就是他们的数据);
  11. 真正不能出的是 S1, 以及把 S3 放进**要上公网或转发第三方**的面孔。两者分开报, 不混。"""
  12. from __future__ import annotations
  13. import argparse, json, re, sys, time, zipfile
  14. from pathlib import Path
  15. ROOT_DEFAULT = Path(__file__).resolve().parents[1]
  16. PAT = {
  17. # AIza 真 key = 39 位且两侧不是 base64 字符 (2026-09-08 实逮: 内联 base64 里 104 位的巧合子串 = 假阳性)
  18. "S1.api_key": re.compile(r"(?<![A-Za-z0-9+/=])(sk-[A-Za-z0-9]{20,}|ghp_[A-Za-z0-9]{20,}|AKIA[0-9A-Z]{16}|AIza[0-9A-Za-z_\-]{35})(?![A-Za-z0-9+/=])"),
  19. "S1.secret_kv": re.compile(r"(?i)\b(api[_-]?key|secret|token|passwd|password|access[_-]?key)\b\s*[:=]\s*['\"]?[A-Za-z0-9_\-]{12,}"),
  20. "S1.private_key": re.compile(r"-----BEGIN (RSA |EC |OPENSSH |PGP )?PRIVATE KEY-----"),
  21. "S2.email": re.compile(r"[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}"),
  22. # 手机/身份证: 只在**像人写的文本**里算 (前后有中文或常见标签词); 纯数字数据列 (时间戳/ID/哈希) 不算
  23. "S2.phone_cn": re.compile(r"(电话|手机|联系方式|Tel|Phone)\D{0,6}(1[3-9]\d{9})(?!\d)"),
  24. "S2.idcard_cn": re.compile(r"(身份证|证件号|ID ?No)\D{0,6}(\d{17}[\dXx])(?!\d)"),
  25. "S2.home_path": re.compile(r"/Users/[a-z][a-z0-9_\-]*|C:\\Users\\[A-Za-z][A-Za-z0-9_\-]*|/home/[a-z][a-z0-9_\-]*"),
  26. "S3.owner": re.compile(r"中广核|华电|国电投|国电|大唐|华能|三峡|龙源|中节能|京能|华润"),
  27. # 真金额: ¥ 或 万元/亿元/元每千瓦; **不含** 万kWh/万行/万m³ (那是电量与计数单位, 2026-09-09 实逮误判)
  28. "S3.money": re.compile(r"[¥¥]\s?\d|\d[\d,.]*\s?(万元|亿元|元/kW|元/千瓦)"),
  29. "S4.internal_doc": re.compile(r"TASKS\.md|SESSION_LOG|violations_log|CLAUDE\.md|AGENTS\.md"),
  30. }
  31. TEXT_EXT = {".py", ".js", ".json", ".yaml", ".yml", ".md", ".txt", ".html", ".htm", ".css", ".csv", ".sh", ".ps1", ".bat", ".cfg", ".ini", ".toml"}
  32. # 对外面孔: 会被转发/上公网的东西 (门户/报告/事实契约/说明书)
  33. PUBLIC_FACE = re.compile(r"release/portal\.html|report.*\.(html|md|docx)|facts_contract|说明书|docs/.*\.md")
  34. SKIP = re.compile(r"(^|/)(\.git|__pycache__|\.venv|wheels|vendor|runtime|dist|node_modules)(/|$)")
  35. # 白名单: 脱敏词表本身当然含业主名 (它就是用来拦这些词的), 不算泄露
  36. WHITELIST = re.compile(r"scripts/guanlan_facts_contract\.py|scripts/security_scan\.py|src/windscada/deid.*\.py|configs/terms/")
  37. def scan_text(name: str, text: str, out: list, cap=200):
  38. if WHITELIST.search(name): return
  39. face = bool(PUBLIC_FACE.search(name))
  40. for k, rx in PAT.items():
  41. hits = rx.findall(text)
  42. if not hits: continue
  43. flat = [h if isinstance(h, str) else next((x for x in h if x), "") for h in hits]
  44. uniq = sorted({str(x)[:60] for x in flat if x})[:5]
  45. out.append(dict(file=name, rule=k, n=len(hits), samples=uniq, public_face=face))
  46. def iter_files(root: Path):
  47. for p in sorted(root.rglob("*")):
  48. rel = p.relative_to(root).as_posix()
  49. if p.is_dir() or SKIP.search(rel): continue
  50. yield rel, p
  51. def main():
  52. ap = argparse.ArgumentParser(); ap.add_argument("--root", default=str(ROOT_DEFAULT)); ap.add_argument("--zip"); ap.add_argument("--out", default="logs/security_scan.json"); ap.add_argument("--strict", action="store_true"); a = ap.parse_args()
  53. root = Path(a.root).resolve(); found = []; n_files = n_text = 0; big = []
  54. if a.zip:
  55. with zipfile.ZipFile(a.zip) as z:
  56. for i in z.infolist():
  57. if i.is_dir() or SKIP.search(i.filename): continue
  58. n_files += 1
  59. if i.file_size > 50_000_000: big.append((i.filename, i.file_size))
  60. if Path(i.filename).suffix.lower() in TEXT_EXT and i.file_size < 8_000_000:
  61. n_text += 1; scan_text(i.filename, z.read(i).decode("utf-8", "ignore"), found)
  62. else:
  63. for rel, p in iter_files(root):
  64. n_files += 1
  65. if p.stat().st_size > 50_000_000: big.append((rel, p.stat().st_size))
  66. if p.suffix.lower() in TEXT_EXT and p.stat().st_size < 8_000_000:
  67. n_text += 1; scan_text(rel, p.read_text(encoding="utf-8", errors="ignore"), found)
  68. by = {}
  69. for f in found: by.setdefault(f["rule"], []).append(f)
  70. s1 = [f for f in found if f["rule"].startswith("S1")]
  71. s3_face = [f for f in found if f["rule"].startswith("S3") and f["public_face"]]
  72. rep = dict(ts=time.strftime("%Y-%m-%dT%H:%M:%S"), target=a.zip or str(root), n_files=n_files, n_text_scanned=n_text,
  73. by_rule={k: dict(files=len({x["file"] for x in v}), hits=sum(x["n"] for x in v)) for k, v in sorted(by.items())},
  74. blockers=dict(S1_secrets=len(s1), S3_on_public_face=len(s3_face)),
  75. big_files=[dict(file=f, mb=round(s / 1e6, 1)) for f, s in sorted(big, key=lambda x: -x[1])[:10]], findings=found)
  76. o = Path(a.out) if Path(a.out).is_absolute() else root / a.out; o.parent.mkdir(parents=True, exist_ok=True)
  77. o.write_text(json.dumps(rep, ensure_ascii=False, indent=1), encoding="utf-8")
  78. print(f"扫 {n_files} 文件 (文本 {n_text}) → {o}")
  79. for k, v in rep["by_rule"].items(): print(f" {k:20s} 文件 {v['files']:4d} 命中 {v['hits']:6d}")
  80. print(f" 阻断项: 密钥 {len(s1)} · 对外面孔上的 D3 {len(s3_face)}")
  81. for f in (s1 + s3_face)[:8]: print(f" ✗ {f['rule']} {f['file']} {f['samples'][:2]}")
  82. if a.strict and (s1 or s3_face): return 2
  83. return 0
  84. if __name__ == "__main__":
  85. # 控制台可能是 GBK(中文 Windows 代码页 936): 正文里的 ✔ ✗ ✅ ⚠ 这类字符编不出来会抛
  86. # UnicodeEncodeError, 脚本干成了事却以退出码 1 结束(同类坑见 src/console.py)。降级为 '?' 而不是崩;
  87. # 不用 import 是为了兼顾 python -m 与直接当脚本跑两种启动方式。
  88. import sys as _sys
  89. for _s in (_sys.stdout, _sys.stderr):
  90. try: _s.reconfigure(errors='replace')
  91. except Exception: pass
  92. sys.exit(main())