| 1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950 |
- #!/usr/bin/env python3
- # -*- coding: utf-8 -*-
- """全量核查 portal.html 里指向本机静态件的引用 (含相对写法), 找出同类"交付件没随包"的坏链。"""
- import pathlib
- import re
- import urllib.error
- import urllib.parse
- import urllib.request
- REL = pathlib.Path(r"F:\temp\guanlan-rudong-v2_0.2.0\release")
- html = (REL / "portal.html").read_text(encoding="utf-8", errors="replace")
- # 所有 href/src 里含 如东 或 release 的
- cand = set()
- for m in re.finditer(r'(?:href|src)="([^"]+)"', html):
- u = m.group(1)
- if ("如东" in u or "/release" in u or "release/" in u) and not u.startswith(("mailto:", "javascript:")):
- cand.add(u)
- print(f"候选引用 {len(cand)} 个\n")
- def check(u: str):
- if u.startswith("http"):
- full = u
- else:
- base = "http://127.0.0.1:28084/" if u.startswith("/") else "http://127.0.0.1:28084/"
- full = base + u.lstrip("/")
- p = urllib.parse.urlsplit(full)
- full = urllib.parse.urlunsplit((p.scheme, p.netloc, urllib.parse.quote(p.path), p.query, ""))
- try:
- with urllib.request.urlopen(full, timeout=30) as r:
- r.read(1)
- return r.status
- except urllib.error.HTTPError as e:
- return e.code
- except Exception as e:
- return f"{e.__class__.__name__}"
- bad = []
- for u in sorted(cand):
- s = check(u)
- tag = "ok " if s == 200 else "BAD "
- if s != 200:
- bad.append(u)
- print(f" [{s}] {tag} {u}")
- print(f"\n坏链 {len(bad)} 个")
- for u in bad:
- print(" - " + u)
|