verify_bug2.py 1.6 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344
  1. #!/usr/bin/env python3
  2. # -*- coding: utf-8 -*-
  3. """bug-2 验收: 走网关取那个 iframe URL, 并把它自身的相对引用也逐条取一遍。"""
  4. import re
  5. import urllib.parse
  6. import urllib.request
  7. URL = ("http://127.0.0.1:28084/如东/如东治理清单_交付_20260901/"
  8. "02_治理清单/如东_液压系统治理清单_v1.2_2026-09-01.html")
  9. def get(u):
  10. req = urllib.request.Request(u, headers={'User-Agent': 'verify'})
  11. try:
  12. with urllib.request.urlopen(req, timeout=30) as r:
  13. return r.status, r.headers.get('Content-Type'), r.read()
  14. except urllib.error.HTTPError as e:
  15. return e.code, e.headers.get('Content-Type'), e.read()
  16. def q(u):
  17. """URL 里中文需编码, 网关会 unquote。"""
  18. p = urllib.parse.urlsplit(u)
  19. return urllib.parse.urlunsplit((p.scheme, p.netloc,
  20. urllib.parse.quote(p.path), p.query, p.fragment))
  21. st, ct, body = get(q(URL))
  22. print(f"iframe URL : [{st}] {ct} {len(body)} 字节")
  23. txt = body.decode('utf-8', 'replace')
  24. print(f"首行 : {txt.splitlines()[0][:120] if txt.strip() else '(空)'}")
  25. print(f"含 not found: {'not found' in txt[:400] and st != 200}")
  26. refs = sorted({m for m in re.findall(r'(?:href|src)="([^"]+)"', txt)
  27. if not m.startswith(('http', '#', 'data:', 'mailto:', 'javascript:'))})
  28. print(f"\n文中相对引用 {len(refs)} 个, 逐条取:")
  29. bad = 0
  30. for r in refs:
  31. u = urllib.parse.urljoin(URL, r)
  32. s, c, b = get(q(u))
  33. if s != 200:
  34. bad += 1
  35. print(f" [{s}] {len(b):8d} 字节 {r}")
  36. print(f"\n结论: {'全部 200' if not bad else f'{bad} 个引用取不到'}")