| 1234567891011121314151617181920212223242526272829303132333435363738394041424344 |
- #!/usr/bin/env python3
- # -*- coding: utf-8 -*-
- """bug-2 验收: 走网关取那个 iframe URL, 并把它自身的相对引用也逐条取一遍。"""
- import re
- import urllib.parse
- import urllib.request
- URL = ("http://127.0.0.1:28084/如东/如东治理清单_交付_20260901/"
- "02_治理清单/如东_液压系统治理清单_v1.2_2026-09-01.html")
- def get(u):
- req = urllib.request.Request(u, headers={'User-Agent': 'verify'})
- try:
- with urllib.request.urlopen(req, timeout=30) as r:
- return r.status, r.headers.get('Content-Type'), r.read()
- except urllib.error.HTTPError as e:
- return e.code, e.headers.get('Content-Type'), e.read()
- def q(u):
- """URL 里中文需编码, 网关会 unquote。"""
- p = urllib.parse.urlsplit(u)
- return urllib.parse.urlunsplit((p.scheme, p.netloc,
- urllib.parse.quote(p.path), p.query, p.fragment))
- st, ct, body = get(q(URL))
- print(f"iframe URL : [{st}] {ct} {len(body)} 字节")
- txt = body.decode('utf-8', 'replace')
- print(f"首行 : {txt.splitlines()[0][:120] if txt.strip() else '(空)'}")
- print(f"含 not found: {'not found' in txt[:400] and st != 200}")
- refs = sorted({m for m in re.findall(r'(?:href|src)="([^"]+)"', txt)
- if not m.startswith(('http', '#', 'data:', 'mailto:', 'javascript:'))})
- print(f"\n文中相对引用 {len(refs)} 个, 逐条取:")
- bad = 0
- for r in refs:
- u = urllib.parse.urljoin(URL, r)
- s, c, b = get(q(u))
- if s != 200:
- bad += 1
- print(f" [{s}] {len(b):8d} 字节 {r}")
- print(f"\n结论: {'全部 200' if not bad else f'{bad} 个引用取不到'}")
|