pages_sweep.py 15 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303
  1. #!/usr/bin/env python3
  2. # -*- coding: utf-8 -*-
  3. r"""观澜各页面「有没有输出」体检(用户令 2026-09-19)。
  4. 背景: 清过产物 + 从 `data/raw` 重算之后, 页面出现"没有输出"的现象。本器**逐页面(含嵌套页)**走一遍
  5. 网关入口(默认 http://127.0.0.1:28084), 对每一页给出:
  6. HTTP 码 · 字节数 · 判定(有输出 / 空 / 缺件 / 降级 / 报错 / 静态资源)· 证据(页面里那句话)· 疑似原因
  7. 覆盖面:
  8. · 门户 `/`(含内嵌面板是否有内容)· `/healthz` `/api/version` `/ops`
  9. · `/release/*` 本体发布层(随包件, 清产物后必缺)
  10. · `/detail/*` 综合详细分析工作台(v2 的 10 个页签取数接口 + 经典嵌套页 `/turbine/<台>` `/problem/<台>/<系统>`)
  11. · `/cms/*` CMS 振动诊断(总览/逐台页/报告页)
  12. · `/sim/` `/sim/sys/` `/viewer/` 三个静态模块的入口
  13. · `release/` 下**随包静态页**(取数单等, 按入口页里出现的链接发现)
  14. 判定口径(不猜): 先看 HTTP 与字节数; 再看**页面自己认账的那句话**(缺件/无产物/加载失败/…);
  15. JSON 则解析顶层 `err/error/no_products/no_report` 与空数组。每条判定都带证据片段, 便于人工复核。
  16. 用法:
  17. python scripts/pages_sweep.py # 控制台摘要 + 写 UTF-8 报告
  18. python scripts/pages_sweep.py --base http://127.0.0.1:28084
  19. python scripts/pages_sweep.py --write docs/页面输出体检_v0.1.md
  20. """
  21. from __future__ import annotations
  22. import argparse
  23. import json
  24. import pathlib
  25. import re
  26. import sys
  27. import urllib.error
  28. import urllib.parse
  29. import urllib.request
  30. from concurrent.futures import ThreadPoolExecutor
  31. ROOT = pathlib.Path(__file__).resolve().parents[1]
  32. sys.path.insert(0, str(ROOT))
  33. sys.path.insert(0, str(ROOT / 'scripts'))
  34. from src import paths as P # noqa: E402
  35. # ── 页面清单 ────────────────────────────────────────────────────────────────────
  36. # kind: page=人看的页面 · api=取数接口 · static=静态资源
  37. # why : 这一页的产物依赖(用于把"空"归因到"缺哪件")
  38. PAGES: list[dict] = [
  39. dict(path='/', kind='page', why='门户单文件 (release/portal.html), 内嵌结论段由产物渲染',
  40. dep='guanlan/derived/portal_claims.json'),
  41. dict(path='/healthz', kind='api', why='模块探活'),
  42. dict(path='/api/version', kind='api', why='版本/契约指纹'),
  43. dict(path='/ops', kind='page', why='运维控制台 (网关自提供)'),
  44. dict(path='/release/', kind='api', why='本体发布层 manifest (随包件 ontology/release_*)'),
  45. dict(path='/release/r1/manifest.json', kind='api', why='本体发布层 r1', dep='ontology/release_r1/manifest.json'),
  46. dict(path='/release/r2/manifest.json', kind='api', why='本体发布层 r2', dep='ontology/release_r2/manifest.json'),
  47. # /detail 工作台: v2 页签底数 + 各自取数
  48. dict(path='/detail/', kind='page', why='v2 工作台外壳 (10 页签)'),
  49. dict(path='/detail/api/fleet?win=2026%E5%B9%B4', kind='api', why='全部页签底数: windscada/*.parquet + ontology', dep='windscada/temp_monthly.parquet'),
  50. dict(path='/detail/api/curves', kind='api', why='特性曲线页', dep='windscada/curve_lenses.parquet'),
  51. dict(path='/detail/api/vibcms', kind='api', why='振动评估页: windcms/报告_CMS*.md', dep='windcms/报告_CMS振动状态评估报告_2026-09-19.md'),
  52. dict(path='/detail/api/maint_survey', kind='api', why='系统自查页: ontology/objects.json', dep='ontology/objects.json'),
  53. dict(path='/detail/api/ont_chain?kind=tree&system=%E9%BD%BF%E8%BD%AE%E7%AE%B1', kind='api', why='决策链页', dep='ontology/objects.json'),
  54. dict(path='/detail/api/facts', kind='api', why='问答/报告页: guanlan/ 事实契约 (人裁底稿派生)', dep='guanlan/derived/detail_cards.json'),
  55. dict(path='/detail/api/ask_status', kind='api', why='问答页: 本机模型'),
  56. # 经典嵌套页(/detail/<rest> → 组件根路径)
  57. dict(path='/detail/turbine/WTG01', kind='page', why='逐台页 (经典)'),
  58. dict(path='/detail/problem/WTG01/%E5%8F%98%E6%A1%A8%E7%B3%BB%E7%BB%9F', kind='page', why='问题页 (经典)'),
  59. dict(path='/detail/api/turbine?t=WTG01&win=2026%E5%B9%B4', kind='api', why='逐台下钻浮层'),
  60. dict(path='/detail/api/problem?t=WTG01&sys=%E5%8F%98%E6%A1%A8%E7%B3%BB%E7%BB%9F&win=2026%E5%B9%B4',
  61. kind='api', why='问题下钻浮层'),
  62. # /cms CMS 振动诊断
  63. dict(path='/cms/', kind='page', why='CMS 首页 (windcms/index.html)', dep='windcms/index.html'),
  64. dict(path='/cms/overview.html', kind='page', why='CMS 总览', dep='windcms/overview.html'),
  65. dict(path='/cms/turbines/WTG01.html', kind='page', why='CMS 逐台页', dep='windcms/turbines/WTG01.html'),
  66. dict(path='/cms/report.md', kind='page', why='CMS 结构化层报告', dep='windcms/report.md'),
  67. # 静态模块入口
  68. dict(path='/sim/', kind='page', why='仿真与回放 (随包资产)'),
  69. dict(path='/sim/sys/', kind='page', why='仿真·四系统合页 (随包资产)'),
  70. dict(path='/viewer/', kind='page', why='三维拆装工作台 (随包资产)'),
  71. ]
  72. # ★2026-09-19 收紧: 只认"这一页确实没数据"的**具体话术**。首版把 不可用/缺件/无数据 这类通用词也算命中,
  73. # 结果门户图例、i18n 词表、运维页说明文字都被算成"缺件" ⇒ 噪声淹掉真问题。
  74. EMPTY_MARKERS = [
  75. ('无产物', '页面自报"无产物"'), ('产物未生成', '页面自报"产物未生成"'),
  76. ('还没有生成', '页面自报"还没有生成"'), ('本体库缺失', '本体库缺失'),
  77. ('报告解析失败', '报告解析失败'), ('模块不可用: ', '网关的下线页'),
  78. ('加载失败', '页面自报加载失败'), ('读取失败', '页面自报读取失败'),
  79. ('Traceback', '页面里带 traceback'), ('no_products', '接口回 no_products'),
  80. ('not found', '404/未找到'),
  81. ]
  82. SPA_SHELLS = {'/', '/detail/'} # 单页应用壳: 正文由 JS 取数渲染, 不能按"可见文本"判空
  83. def fetch(base: str, path: str, timeout: int = 60):
  84. url = base + path
  85. try:
  86. with urllib.request.urlopen(url, timeout=timeout) as r:
  87. return r.status, r.headers.get('Content-Type', ''), r.read()
  88. except urllib.error.HTTPError as e:
  89. return e.code, e.headers.get('Content-Type', '') if e.headers else '', e.read()
  90. except Exception as e:
  91. return None, f'{type(e).__name__}', str(e).encode('utf-8')
  92. _TAG = re.compile(r'<[^>]+>')
  93. _SCRIPT = re.compile(r'<(script|style)\b.*?</\1>', re.S | re.I)
  94. _COMMENT = re.compile(r'<!--.*?-->', re.S)
  95. def visible_text(html: str) -> str:
  96. """页面**可见文本**(去掉 script/style/注释/标签)。
  97. ★为什么必须先去脚本: 首版直接在整页文本里找"不可用/undefined/缺件"这类词, 命中的全是 JS 里的
  98. i18n 词表与 CSS 类名(例如逐台页里 `'监测不可用':'var(--unknown)'`)⇒ **好页面被误判成缺件**。
  99. 判定只许看人能看到的那部分。
  100. """
  101. t = _SCRIPT.sub(' ', html)
  102. t = _COMMENT.sub(' ', t)
  103. t = _TAG.sub(' ', t)
  104. return re.sub(r'\s+', ' ', t).strip()
  105. def judge(path: str, kind: str, code, ctype: str, body: bytes) -> dict:
  106. raw = body.decode('utf-8', 'replace')
  107. n = len(body)
  108. is_html = ('html' in (ctype or '').lower()) or raw.lstrip()[:1] == '<'
  109. txt = visible_text(raw) if is_html else raw
  110. hits = [why for mk, why in EMPTY_MARKERS if mk in txt]
  111. err_msg = ''
  112. if 'json' in (ctype or '') and n < 400000:
  113. try:
  114. o = json.loads(raw)
  115. if isinstance(o, dict):
  116. err_msg = str(o.get('err') or o.get('error') or '')
  117. if o.get('no_products') or o.get('no_report'):
  118. hits.append('接口自报无产物')
  119. empties = [k for k, v in o.items() if isinstance(v, list) and not v]
  120. if empties and len(empties) >= max(1, len(o) // 2):
  121. hits.append('顶层多数数组为空: ' + ','.join(empties[:4]))
  122. except Exception:
  123. pass
  124. if code is None:
  125. verdict = '连不上'
  126. elif code >= 500:
  127. verdict = '报错'
  128. elif code >= 400:
  129. verdict = '报错' if code != 404 else '未找到'
  130. elif kind == 'static':
  131. verdict = '静态资源'
  132. elif n < 200:
  133. verdict = '空'
  134. elif hits:
  135. verdict = '缺件/降级'
  136. elif path in SPA_SHELLS:
  137. verdict = '有输出(单页壳)' # 壳本身只有标题, 正文由 JS 从 /api/* 取 —— 判空要看它的接口
  138. elif is_html and len(txt) < 120 and n < 60000:
  139. # ★阈值要带字节数: 200 KB 的逐台/问题页可见文本本来就少(正文全是 JS 渲染的),
  140. # 只看"可见文本 <120 字"会把它们误判成空页(2026-09-19 实逮)。
  141. verdict = '空(可见文本极少)'
  142. else:
  143. verdict = '有输出'
  144. ev = err_msg[:90] or (hits[0] if hits else txt[:70])
  145. return dict(verdict=verdict, code=code, n=n, ctype=(ctype or '')[:40], hits=hits[:3],
  146. evidence=ev, visible=len(txt))
  147. CAUSE_RUNNING = '重算进行中(跑完自愈)'
  148. CAUSE_NOGEN = '该族无生成端(重算也补不出)'
  149. CAUSE_HUMAN = '依赖人裁底稿(非机械可算)'
  150. CAUSE_EXTERNAL = '外部组件(本机模型)'
  151. CAUSE_STATIC = '随包静态资产(与重算无关)'
  152. CAUSE_OK = '有输出'
  153. def _families() -> tuple:
  154. """→ (族表, 是否正在重算)。族表用来判"缺的这一族有没有生成端"; 在跑与否决定能不能自愈。"""
  155. fams = []
  156. try:
  157. from products_reverse_audit import FAMILIES
  158. fams = FAMILIES
  159. except Exception:
  160. pass
  161. running = False
  162. try:
  163. from src import opsjob
  164. running = bool(opsjob.current().get('running'))
  165. except Exception:
  166. pass
  167. return fams, running
  168. def family_of(fams, rel: str):
  169. import fnmatch
  170. for fam in fams:
  171. pats = [fam.get('glob')] if isinstance(fam.get('glob'), str) else list(fam.get('glob') or [])
  172. for g in pats:
  173. if '{' in g:
  174. head, tail = g.split('{', 1)
  175. opts, rest = tail.split('}', 1)
  176. exp = [head + o + rest for o in opts.split(',')]
  177. else:
  178. exp = [g]
  179. for pt in exp:
  180. if '**' not in pt and rel.count('/') != pt.count('/'):
  181. continue
  182. if rel == pt or fnmatch.fnmatch(rel, pt):
  183. return fam
  184. return None
  185. def cause_of(path: str, verdict: str, dep: str | None, fams, running: bool) -> str:
  186. """把"空/报错"翻成原因: 依赖族有没有生成端 → 会不会自愈。判据只有事实, 不猜。"""
  187. if verdict == '有输出':
  188. return CAUSE_OK
  189. if verdict == '静态资源':
  190. return CAUSE_STATIC
  191. if path.startswith('/local-ai') or '/ask' in path:
  192. return CAUSE_EXTERNAL
  193. f = family_of(fams, dep) if dep else None
  194. if f is None:
  195. return CAUSE_RUNNING if running else '依赖件不在位(放原始件后重算)'
  196. if f.get('kind') in ('shipped', 'human') and not f.get('gen'):
  197. # 本体发布层/带版本交付件 = 研发出件(不是人裁底稿,别把两类混成一句)
  198. if f.get('id') in ('ontology_releases', 'windscada_pages', 'tcm_replay'):
  199. return '包内无生成端(研发出件)'
  200. return CAUSE_HUMAN if f.get('id') in ('guanlan_contract', 'sop_workspace', 'paradigm_r1',
  201. 'human_deliverables') else CAUSE_NOGEN
  202. return CAUSE_RUNNING if running else '依赖件不在位(重算未覆盖到)'
  203. def discover_links(base: str, prefix: str, html: bytes, limit: int = 25) -> list[str]:
  204. """从页面里发现**同前缀**的嵌套页(href/src),用于"含嵌套页面"的覆盖。"""
  205. txt = html.decode('utf-8', 'replace')
  206. got, seen = [], set()
  207. for m in re.finditer(r'(?:href|src)=["\']([^"\'#?]+)["\']', txt):
  208. u = m.group(1)
  209. if not u.startswith('/') or not u.startswith(prefix):
  210. continue
  211. if u.endswith(('.js', '.css', '.png', '.jpg', '.svg', '.woff', '.woff2', '.ico', '.map')):
  212. continue
  213. if u in seen or '${' in u: # JS 模板字面量不是页面
  214. continue
  215. seen.add(u)
  216. got.append(u)
  217. if len(got) >= limit:
  218. break
  219. return got
  220. def main() -> int:
  221. ap = argparse.ArgumentParser()
  222. ap.add_argument('--base', default='http://127.0.0.1:28084')
  223. ap.add_argument('--write', default=None)
  224. ap.add_argument('--links', type=int, default=12, help='每个模块再抽几个嵌套页')
  225. a = ap.parse_args()
  226. tasks = list(PAGES)
  227. # 嵌套页: 先抓每个模块入口, 从中发现同前缀链接
  228. for prefix in ('/detail', '/cms', '/sim', '/sim/sys', '/viewer', '/ops'):
  229. code, ct, body = fetch(a.base, prefix + '/')
  230. if not body:
  231. continue
  232. for u in discover_links(a.base, prefix, body, a.links):
  233. tasks.append(dict(path=u, kind='page', why=f'{prefix} 内的嵌套页 (从入口页发现)'))
  234. fams, running = _families()
  235. with ThreadPoolExecutor(max_workers=8) as ex:
  236. res = list(ex.map(lambda t: (t, fetch(a.base, t['path'])), tasks))
  237. rows = []
  238. for t, (code, ctype, body) in res:
  239. j = judge(t['path'], t['kind'], code, ctype, body)
  240. j.update(path=t['path'], kind=t['kind'], why=t['why'], dep=t.get('dep'))
  241. j['cause'] = cause_of(t['path'], j['verdict'], t.get('dep'), fams, running)
  242. rows.append(j)
  243. order = {'报错': 0, '未找到': 1, '空': 2, '缺件/降级': 3, '连不上': 4, '有输出': 5, '静态资源': 6}
  244. rows.sort(key=lambda r: (order.get(r['verdict'], 9), r['path']))
  245. L = [f'# 观澜页面输出体检 · {a.base}', '',
  246. f'共 {len(rows)} 页(含从各模块入口发现的嵌套页)· 判定口径: HTTP + 字节数 + 页面自己认账的那句话',
  247. f'· 重算是否正在跑: **{"是(产物会随步骤长回来)" if running else "否"}**', '',
  248. '| 判定 | 路径 | 码 | 字节 | 原因 | 证据 | 产物依赖 |', '|---|---|---:|---:|---|---|---|']
  249. for r in rows:
  250. L.append(f"| {r['verdict']} | `{r['path']}` | {r['code']} | {r['n']:,} | {r['cause']} | "
  251. f"{str(r['evidence'])[:60]} | {r['why']} |")
  252. rep = '\n'.join(L)
  253. if a.write:
  254. pathlib.Path(a.write).write_text(rep + '\n', encoding='utf-8')
  255. from collections import Counter
  256. cnt = Counter(r['verdict'] for r in rows)
  257. print('判定分布: ' + ' · '.join(f'{k}={v}' for k, v in cnt.most_common()))
  258. for r in rows:
  259. if r['verdict'] not in ('有输出', '静态资源'):
  260. print(f" [{r['verdict']}] {r['path']} HTTP {r['code']} {r['n']:,}B {str(r['evidence'])[:70]}")
  261. if a.write:
  262. print(f'已写 {a.write}')
  263. return 0
  264. if __name__ == '__main__':
  265. sys.exit(main())