#!/usr/bin/env python3 # -*- coding: utf-8 -*- r"""观澜各页面「有没有输出」体检(用户令 2026-09-19)。 背景: 清过产物 + 从 `data/raw` 重算之后, 页面出现"没有输出"的现象。本器**逐页面(含嵌套页)**走一遍 网关入口(默认 http://127.0.0.1:28084), 对每一页给出: HTTP 码 · 字节数 · 判定(有输出 / 空 / 缺件 / 降级 / 报错 / 静态资源)· 证据(页面里那句话)· 疑似原因 覆盖面: · 门户 `/`(含内嵌面板是否有内容)· `/healthz` `/api/version` `/ops` · `/release/*` 本体发布层(随包件, 清产物后必缺) · `/detail/*` 综合详细分析工作台(v2 的 10 个页签取数接口 + 经典嵌套页 `/turbine/<台>` `/problem/<台>/<系统>`) · `/cms/*` CMS 振动诊断(总览/逐台页/报告页) · `/sim/` `/sim/sys/` `/viewer/` 三个静态模块的入口 · `release/` 下**随包静态页**(取数单等, 按入口页里出现的链接发现) 判定口径(不猜): 先看 HTTP 与字节数; 再看**页面自己认账的那句话**(缺件/无产物/加载失败/…); JSON 则解析顶层 `err/error/no_products/no_report` 与空数组。每条判定都带证据片段, 便于人工复核。 用法: python scripts/pages_sweep.py # 控制台摘要 + 写 UTF-8 报告 python scripts/pages_sweep.py --base http://127.0.0.1:28084 python scripts/pages_sweep.py --write docs/页面输出体检_v0.1.md """ from __future__ import annotations import argparse import json import pathlib import re import sys import urllib.error import urllib.parse import urllib.request from concurrent.futures import ThreadPoolExecutor ROOT = pathlib.Path(__file__).resolve().parents[1] sys.path.insert(0, str(ROOT)) sys.path.insert(0, str(ROOT / 'scripts')) from src import paths as P # noqa: E402 # ── 页面清单 ──────────────────────────────────────────────────────────────────── # kind: page=人看的页面 · api=取数接口 · static=静态资源 # why : 这一页的产物依赖(用于把"空"归因到"缺哪件") PAGES: list[dict] = [ dict(path='/', kind='page', why='门户单文件 (release/portal.html), 内嵌结论段由产物渲染', dep='guanlan/derived/portal_claims.json'), dict(path='/healthz', kind='api', why='模块探活'), dict(path='/api/version', kind='api', why='版本/契约指纹'), dict(path='/ops', kind='page', why='运维控制台 (网关自提供)'), dict(path='/release/', kind='api', why='本体发布层 manifest (随包件 ontology/release_*)'), dict(path='/release/r1/manifest.json', kind='api', why='本体发布层 r1', dep='ontology/release_r1/manifest.json'), dict(path='/release/r2/manifest.json', kind='api', why='本体发布层 r2', dep='ontology/release_r2/manifest.json'), # /detail 工作台: v2 页签底数 + 各自取数 dict(path='/detail/', kind='page', why='v2 工作台外壳 (10 页签)'), dict(path='/detail/api/fleet?win=2026%E5%B9%B4', kind='api', why='全部页签底数: windscada/*.parquet + ontology', dep='windscada/temp_monthly.parquet'), dict(path='/detail/api/curves', kind='api', why='特性曲线页', dep='windscada/curve_lenses.parquet'), dict(path='/detail/api/vibcms', kind='api', why='振动评估页: windcms/报告_CMS*.md', dep='windcms/报告_CMS振动状态评估报告_2026-09-19.md'), dict(path='/detail/api/maint_survey', kind='api', why='系统自查页: ontology/objects.json', dep='ontology/objects.json'), dict(path='/detail/api/ont_chain?kind=tree&system=%E9%BD%BF%E8%BD%AE%E7%AE%B1', kind='api', why='决策链页', dep='ontology/objects.json'), dict(path='/detail/api/facts', kind='api', why='问答/报告页: guanlan/ 事实契约 (人裁底稿派生)', dep='guanlan/derived/detail_cards.json'), dict(path='/detail/api/ask_status', kind='api', why='问答页: 本机模型'), # 经典嵌套页(/detail/ → 组件根路径) dict(path='/detail/turbine/WTG01', kind='page', why='逐台页 (经典)'), dict(path='/detail/problem/WTG01/%E5%8F%98%E6%A1%A8%E7%B3%BB%E7%BB%9F', kind='page', why='问题页 (经典)'), dict(path='/detail/api/turbine?t=WTG01&win=2026%E5%B9%B4', kind='api', why='逐台下钻浮层'), dict(path='/detail/api/problem?t=WTG01&sys=%E5%8F%98%E6%A1%A8%E7%B3%BB%E7%BB%9F&win=2026%E5%B9%B4', kind='api', why='问题下钻浮层'), # /cms CMS 振动诊断 dict(path='/cms/', kind='page', why='CMS 首页 (windcms/index.html)', dep='windcms/index.html'), dict(path='/cms/overview.html', kind='page', why='CMS 总览', dep='windcms/overview.html'), dict(path='/cms/turbines/WTG01.html', kind='page', why='CMS 逐台页', dep='windcms/turbines/WTG01.html'), dict(path='/cms/report.md', kind='page', why='CMS 结构化层报告', dep='windcms/report.md'), # 静态模块入口 dict(path='/sim/', kind='page', why='仿真与回放 (随包资产)'), dict(path='/sim/sys/', kind='page', why='仿真·四系统合页 (随包资产)'), dict(path='/viewer/', kind='page', why='三维拆装工作台 (随包资产)'), ] # ★2026-09-19 收紧: 只认"这一页确实没数据"的**具体话术**。首版把 不可用/缺件/无数据 这类通用词也算命中, # 结果门户图例、i18n 词表、运维页说明文字都被算成"缺件" ⇒ 噪声淹掉真问题。 EMPTY_MARKERS = [ ('无产物', '页面自报"无产物"'), ('产物未生成', '页面自报"产物未生成"'), ('还没有生成', '页面自报"还没有生成"'), ('本体库缺失', '本体库缺失'), ('报告解析失败', '报告解析失败'), ('模块不可用: ', '网关的下线页'), ('加载失败', '页面自报加载失败'), ('读取失败', '页面自报读取失败'), ('Traceback', '页面里带 traceback'), ('no_products', '接口回 no_products'), ('not found', '404/未找到'), ] SPA_SHELLS = {'/', '/detail/'} # 单页应用壳: 正文由 JS 取数渲染, 不能按"可见文本"判空 def fetch(base: str, path: str, timeout: int = 60): url = base + path try: with urllib.request.urlopen(url, timeout=timeout) as r: return r.status, r.headers.get('Content-Type', ''), r.read() except urllib.error.HTTPError as e: return e.code, e.headers.get('Content-Type', '') if e.headers else '', e.read() except Exception as e: return None, f'{type(e).__name__}', str(e).encode('utf-8') _TAG = re.compile(r'<[^>]+>') _SCRIPT = re.compile(r'<(script|style)\b.*?', re.S | re.I) _COMMENT = re.compile(r'', re.S) def visible_text(html: str) -> str: """页面**可见文本**(去掉 script/style/注释/标签)。 ★为什么必须先去脚本: 首版直接在整页文本里找"不可用/undefined/缺件"这类词, 命中的全是 JS 里的 i18n 词表与 CSS 类名(例如逐台页里 `'监测不可用':'var(--unknown)'`)⇒ **好页面被误判成缺件**。 判定只许看人能看到的那部分。 """ t = _SCRIPT.sub(' ', html) t = _COMMENT.sub(' ', t) t = _TAG.sub(' ', t) return re.sub(r'\s+', ' ', t).strip() def judge(path: str, kind: str, code, ctype: str, body: bytes) -> dict: raw = body.decode('utf-8', 'replace') n = len(body) is_html = ('html' in (ctype or '').lower()) or raw.lstrip()[:1] == '<' txt = visible_text(raw) if is_html else raw hits = [why for mk, why in EMPTY_MARKERS if mk in txt] err_msg = '' if 'json' in (ctype or '') and n < 400000: try: o = json.loads(raw) if isinstance(o, dict): err_msg = str(o.get('err') or o.get('error') or '') if o.get('no_products') or o.get('no_report'): hits.append('接口自报无产物') empties = [k for k, v in o.items() if isinstance(v, list) and not v] if empties and len(empties) >= max(1, len(o) // 2): hits.append('顶层多数数组为空: ' + ','.join(empties[:4])) except Exception: pass if code is None: verdict = '连不上' elif code >= 500: verdict = '报错' elif code >= 400: verdict = '报错' if code != 404 else '未找到' elif kind == 'static': verdict = '静态资源' elif n < 200: verdict = '空' elif hits: verdict = '缺件/降级' elif path in SPA_SHELLS: verdict = '有输出(单页壳)' # 壳本身只有标题, 正文由 JS 从 /api/* 取 —— 判空要看它的接口 elif is_html and len(txt) < 120 and n < 60000: # ★阈值要带字节数: 200 KB 的逐台/问题页可见文本本来就少(正文全是 JS 渲染的), # 只看"可见文本 <120 字"会把它们误判成空页(2026-09-19 实逮)。 verdict = '空(可见文本极少)' else: verdict = '有输出' ev = err_msg[:90] or (hits[0] if hits else txt[:70]) return dict(verdict=verdict, code=code, n=n, ctype=(ctype or '')[:40], hits=hits[:3], evidence=ev, visible=len(txt)) CAUSE_RUNNING = '重算进行中(跑完自愈)' CAUSE_NOGEN = '该族无生成端(重算也补不出)' CAUSE_HUMAN = '依赖人裁底稿(非机械可算)' CAUSE_EXTERNAL = '外部组件(本机模型)' CAUSE_STATIC = '随包静态资产(与重算无关)' CAUSE_OK = '有输出' def _families() -> tuple: """→ (族表, 是否正在重算)。族表用来判"缺的这一族有没有生成端"; 在跑与否决定能不能自愈。""" fams = [] try: from products_reverse_audit import FAMILIES fams = FAMILIES except Exception: pass running = False try: from src import opsjob running = bool(opsjob.current().get('running')) except Exception: pass return fams, running def family_of(fams, rel: str): import fnmatch for fam in fams: pats = [fam.get('glob')] if isinstance(fam.get('glob'), str) else list(fam.get('glob') or []) for g in pats: if '{' in g: head, tail = g.split('{', 1) opts, rest = tail.split('}', 1) exp = [head + o + rest for o in opts.split(',')] else: exp = [g] for pt in exp: if '**' not in pt and rel.count('/') != pt.count('/'): continue if rel == pt or fnmatch.fnmatch(rel, pt): return fam return None def cause_of(path: str, verdict: str, dep: str | None, fams, running: bool) -> str: """把"空/报错"翻成原因: 依赖族有没有生成端 → 会不会自愈。判据只有事实, 不猜。""" if verdict == '有输出': return CAUSE_OK if verdict == '静态资源': return CAUSE_STATIC if path.startswith('/local-ai') or '/ask' in path: return CAUSE_EXTERNAL f = family_of(fams, dep) if dep else None if f is None: return CAUSE_RUNNING if running else '依赖件不在位(放原始件后重算)' if f.get('kind') in ('shipped', 'human') and not f.get('gen'): # 本体发布层/带版本交付件 = 研发出件(不是人裁底稿,别把两类混成一句) if f.get('id') in ('ontology_releases', 'windscada_pages', 'tcm_replay'): return '包内无生成端(研发出件)' return CAUSE_HUMAN if f.get('id') in ('guanlan_contract', 'sop_workspace', 'paradigm_r1', 'human_deliverables') else CAUSE_NOGEN return CAUSE_RUNNING if running else '依赖件不在位(重算未覆盖到)' def discover_links(base: str, prefix: str, html: bytes, limit: int = 25) -> list[str]: """从页面里发现**同前缀**的嵌套页(href/src),用于"含嵌套页面"的覆盖。""" txt = html.decode('utf-8', 'replace') got, seen = [], set() for m in re.finditer(r'(?:href|src)=["\']([^"\'#?]+)["\']', txt): u = m.group(1) if not u.startswith('/') or not u.startswith(prefix): continue if u.endswith(('.js', '.css', '.png', '.jpg', '.svg', '.woff', '.woff2', '.ico', '.map')): continue if u in seen or '${' in u: # JS 模板字面量不是页面 continue seen.add(u) got.append(u) if len(got) >= limit: break return got def main() -> int: ap = argparse.ArgumentParser() ap.add_argument('--base', default='http://127.0.0.1:28084') ap.add_argument('--write', default=None) ap.add_argument('--links', type=int, default=12, help='每个模块再抽几个嵌套页') a = ap.parse_args() tasks = list(PAGES) # 嵌套页: 先抓每个模块入口, 从中发现同前缀链接 for prefix in ('/detail', '/cms', '/sim', '/sim/sys', '/viewer', '/ops'): code, ct, body = fetch(a.base, prefix + '/') if not body: continue for u in discover_links(a.base, prefix, body, a.links): tasks.append(dict(path=u, kind='page', why=f'{prefix} 内的嵌套页 (从入口页发现)')) fams, running = _families() with ThreadPoolExecutor(max_workers=8) as ex: res = list(ex.map(lambda t: (t, fetch(a.base, t['path'])), tasks)) rows = [] for t, (code, ctype, body) in res: j = judge(t['path'], t['kind'], code, ctype, body) j.update(path=t['path'], kind=t['kind'], why=t['why'], dep=t.get('dep')) j['cause'] = cause_of(t['path'], j['verdict'], t.get('dep'), fams, running) rows.append(j) order = {'报错': 0, '未找到': 1, '空': 2, '缺件/降级': 3, '连不上': 4, '有输出': 5, '静态资源': 6} rows.sort(key=lambda r: (order.get(r['verdict'], 9), r['path'])) L = [f'# 观澜页面输出体检 · {a.base}', '', f'共 {len(rows)} 页(含从各模块入口发现的嵌套页)· 判定口径: HTTP + 字节数 + 页面自己认账的那句话', f'· 重算是否正在跑: **{"是(产物会随步骤长回来)" if running else "否"}**', '', '| 判定 | 路径 | 码 | 字节 | 原因 | 证据 | 产物依赖 |', '|---|---|---:|---:|---|---|---|'] for r in rows: L.append(f"| {r['verdict']} | `{r['path']}` | {r['code']} | {r['n']:,} | {r['cause']} | " f"{str(r['evidence'])[:60]} | {r['why']} |") rep = '\n'.join(L) if a.write: pathlib.Path(a.write).write_text(rep + '\n', encoding='utf-8') from collections import Counter cnt = Counter(r['verdict'] for r in rows) print('判定分布: ' + ' · '.join(f'{k}={v}' for k, v in cnt.most_common())) for r in rows: if r['verdict'] not in ('有输出', '静态资源'): print(f" [{r['verdict']}] {r['path']} HTTP {r['code']} {r['n']:,}B {str(r['evidence'])[:70]}") if a.write: print(f'已写 {a.write}') return 0 if __name__ == '__main__': sys.exit(main())