#!/usr/bin/env python3 # -*- coding: utf-8 -*- """bug-2 验收: 走网关取那个 iframe URL, 并把它自身的相对引用也逐条取一遍。""" import re import urllib.parse import urllib.request URL = ("http://127.0.0.1:28084/如东/如东治理清单_交付_20260901/" "02_治理清单/如东_液压系统治理清单_v1.2_2026-09-01.html") def get(u): req = urllib.request.Request(u, headers={'User-Agent': 'verify'}) try: with urllib.request.urlopen(req, timeout=30) as r: return r.status, r.headers.get('Content-Type'), r.read() except urllib.error.HTTPError as e: return e.code, e.headers.get('Content-Type'), e.read() def q(u): """URL 里中文需编码, 网关会 unquote。""" p = urllib.parse.urlsplit(u) return urllib.parse.urlunsplit((p.scheme, p.netloc, urllib.parse.quote(p.path), p.query, p.fragment)) st, ct, body = get(q(URL)) print(f"iframe URL : [{st}] {ct} {len(body)} 字节") txt = body.decode('utf-8', 'replace') print(f"首行 : {txt.splitlines()[0][:120] if txt.strip() else '(空)'}") print(f"含 not found: {'not found' in txt[:400] and st != 200}") refs = sorted({m for m in re.findall(r'(?:href|src)="([^"]+)"', txt) if not m.startswith(('http', '#', 'data:', 'mailto:', 'javascript:'))}) print(f"\n文中相对引用 {len(refs)} 个, 逐条取:") bad = 0 for r in refs: u = urllib.parse.urljoin(URL, r) s, c, b = get(q(u)) if s != 200: bad += 1 print(f" [{s}] {len(b):8d} 字节 {r}") print(f"\n结论: {'全部 200' if not bad else f'{bad} 个引用取不到'}")