scan_portal_links.py 1.5 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950
  1. #!/usr/bin/env python3
  2. # -*- coding: utf-8 -*-
  3. """全量核查 portal.html 里指向本机静态件的引用 (含相对写法), 找出同类"交付件没随包"的坏链。"""
  4. import pathlib
  5. import re
  6. import urllib.error
  7. import urllib.parse
  8. import urllib.request
  9. REL = pathlib.Path(r"F:\temp\guanlan-rudong-v2_0.2.0\release")
  10. html = (REL / "portal.html").read_text(encoding="utf-8", errors="replace")
  11. # 所有 href/src 里含 如东 或 release 的
  12. cand = set()
  13. for m in re.finditer(r'(?:href|src)="([^"]+)"', html):
  14. u = m.group(1)
  15. if ("如东" in u or "/release" in u or "release/" in u) and not u.startswith(("mailto:", "javascript:")):
  16. cand.add(u)
  17. print(f"候选引用 {len(cand)} 个\n")
  18. def check(u: str):
  19. if u.startswith("http"):
  20. full = u
  21. else:
  22. base = "http://127.0.0.1:28084/" if u.startswith("/") else "http://127.0.0.1:28084/"
  23. full = base + u.lstrip("/")
  24. p = urllib.parse.urlsplit(full)
  25. full = urllib.parse.urlunsplit((p.scheme, p.netloc, urllib.parse.quote(p.path), p.query, ""))
  26. try:
  27. with urllib.request.urlopen(full, timeout=30) as r:
  28. r.read(1)
  29. return r.status
  30. except urllib.error.HTTPError as e:
  31. return e.code
  32. except Exception as e:
  33. return f"{e.__class__.__name__}"
  34. bad = []
  35. for u in sorted(cand):
  36. s = check(u)
  37. tag = "ok " if s == 200 else "BAD "
  38. if s != 200:
  39. bad.append(u)
  40. print(f" [{s}] {tag} {u}")
  41. print(f"\n坏链 {len(bad)} 个")
  42. for u in bad:
  43. print(" - " + u)