vib_confidence_wire.py 14 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286
  1. # -*- coding: utf-8 -*-
  2. r"""★① 置信度**接线**(用户令 2026-10-06):把 `vib_confidence` 的五轴评分接到现有产物上。
  3. 设计要点(刻意保守,防"凑分"):
  4. - **只读现有产物**:融合表行(`fusion_table`)+ 振动 handoff(`per_turbine` 列表)。**不新增数据源**。
  5. - **缺数的轴显式标 `[INS]`** 并计 0 分,同时在 `notes` 里写明缺什么 —— 不猜、不补、不做默认满分。
  6. - **只加不改**:产出 `置信度` 字段;**不参与**任何判级/等级/结论的生成(接线不影响现有裁决)。
  7. - 五轴取数(能取到什么就取什么):
  8. · A 内部证据强度:handoff 里若有逐台"闸/门"通过数则用之;否则 `[INS]`。
  9. · B 独立源共时确认:融合表行的 `温度判`(非 —)与 `油新鲜=True` 计为**同窗**独立源;两者都无 ⇒ `[INS]`。
  10. · C 实物锚:handoff 行里若有 `tier/anchor/实物` 字段则映射;否则 `[INS]`。
  11. · D 样本充分性:融合表 `窗区间` 的窗数 + handoff 行里的记录数(若有);无窗数 ⇒ 按 1 窗计并标注。
  12. · E 反证搜索:**当前无产物承载** ⇒ 一律 `[INS]`(这正是"没做过反证搜索 = 0 分"的口径)。
  13. """
  14. from __future__ import annotations
  15. import re
  16. from . import vib_confidence as VC
  17. WINDOW_CLAUSE_DEN = lambda n: max(4, -(-2 * int(n) // 3)) if n and n > 0 else 4 # ceil(2N/3) 起步 4
  18. # ── ★② 真输入装载(2026-10-06):A 轴 ← L6 过闸谱线表;C 轴 ← 实物锚登记 ───────────────
  19. # 说明:这两样都是**仓内既有产物**(非新增数据源):
  20. # A: outputs/rudong/m5_cms_tcm/model_run_l6.parquet(逐台逐线"过闸"结果, 列含 _闸/_依据/定级/绝对锚/_窗)
  21. # 口径映射(保守, 已在 notes 写明):某台有 ≥1 条 PASS 线 ⇒ 视为"该台走通了全闸"⇒ A 轴按 5 闸计;
  22. # 绝对锚:该台任一行"解封判据"命中 physical_in_window 或"绝对锚"列非「—」⇒ anchor_ok=True。
  23. # C: reference/rudong/positive_anchors.json 的 anchors(tier ∈ physical_in_window/closed_loop/physical_historical)
  24. # 同一台取**最高档**(20 > 12 > 8)。
  25. def load_axis_inputs(root=None) -> dict:
  26. """返回 {turbine: {"a":{"gates_passed":int,"anchor_ok":bool,"pass_lines":int,"level":str},
  27. "c":{"tier":str,"evidence":str}}}(容错: 缺件则空 dict)。"""
  28. import json
  29. import pathlib as _p
  30. base = _p.Path(root) if root else _p.Path(__file__).resolve().parents[3]
  31. out: dict = {}
  32. # A: L6 过闸谱线表
  33. try:
  34. import pandas as _pd
  35. pq = base / "outputs/rudong/m5_cms_tcm/model_run_l6.parquet"
  36. if pq.exists():
  37. df = _pd.read_parquet(pq)
  38. for tid, g in df.groupby("台"):
  39. key = _norm_tid(tid)
  40. passes = int((g["_闸"].astype(str).str.upper() == "PASS").sum())
  41. anchor = bool(g["解封判据"].astype(str).str.contains("physical_in_window").any()
  42. or (g.get("绝对锚") is not None
  43. and g["绝对锚"].astype(str).ne("—").any()))
  44. lv = "候选" if (g["定级"].astype(str) == "候选").any() else "参考"
  45. out.setdefault(key, {})["a"] = {"gates_passed": 5 if passes else 0,
  46. "anchor_ok": anchor, "pass_lines": passes, "level": lv}
  47. except Exception:
  48. pass
  49. # C: 实物锚登记(同台取最高档)
  50. try:
  51. ap = base / "reference/rudong/positive_anchors.json"
  52. if ap.exists():
  53. doc = json.loads(ap.read_text(encoding="utf-8"))
  54. rank = {"physical_in_window": 3, "closed_loop": 2, "physical_historical": 1}
  55. for a in (doc.get("anchors") or []):
  56. key = _norm_tid(a.get("turbine"))
  57. tier = str(a.get("tier") or "")
  58. cur = (out.setdefault(key, {}).get("c") or {}).get("tier")
  59. if rank.get(tier, 0) > rank.get(cur or "", 0):
  60. out[key]["c"] = {"tier": tier, "evidence": str(a.get("evidence") or "")[:80],
  61. "component_class": str(a.get("component_class") or "")}
  62. except Exception:
  63. pass
  64. # ★common_mode(2026-10-06, 5 窗数据到位后实现):同期"全场有几台同线通过" ⇒ 该线是否机群共有现象
  65. try:
  66. import pandas as _pd
  67. _pq = base / "outputs/rudong/m5_cms_tcm/model_run_l6.parquet"
  68. if _pq.exists():
  69. _d = _pd.read_parquet(_pq)
  70. if "线" in _d.columns and "台" in _d.columns:
  71. _g = _d.groupby("线")["台"].nunique()
  72. _n_turb = int(_d["台"].nunique()) or 1
  73. # 判据:同线在 >=3 台通过(或覆盖 >=30% 有结论的台)⇒ 视为机群共有现象(共模嫌疑)
  74. _shared = {str(k): int(v) for k, v in _g.items()
  75. if int(v) >= 3 or (int(v) / _n_turb) >= 0.30}
  76. out["__common_mode__"] = {"ran": True, "n_turbines": _n_turb, "shared_lines": _shared}
  77. except Exception:
  78. pass
  79. # E: 过闸 funnel ⇒ 已跑的"反证类别"(窗口级,非逐台;据实映射,未跑不补)
  80. try:
  81. import json as _json
  82. _sp = base / "outputs/rudong/m5_cms_tcm/model_run_summary.json"
  83. if _sp.exists():
  84. _doc = _json.loads(_sp.read_text(encoding="utf-8"))
  85. _map = {"G4_INHERENT_OR_BATCH": "intrinsic_line",
  86. "G8_NOT_FLEET_OUTLIER": "fleet_base_rate",
  87. "G6_SELECTIVITY": "alternative_explanation",
  88. "G7_PEAK_OFFSET": "alternative_explanation",
  89. "G5_INTEGER_ORDER": "alternative_explanation",
  90. "G2_NO_LINE": "alternative_explanation"}
  91. _ran: set = set()
  92. for _w, _g in (_doc.get("by_gate") or {}).items():
  93. for _gate, _n in (_g or {}).items():
  94. if _gate in _map and int(_n or 0) > 0:
  95. _ran.add(_map[_gate])
  96. out["__checks__"] = sorted(_ran)
  97. except Exception:
  98. pass
  99. return out
  100. def _n_win(frow: dict) -> int:
  101. """从融合表行的 窗区间 解析窗数(容错;解析不出按 1 窗)。"""
  102. wr = str(frow.get('窗区间') or '')
  103. try:
  104. return max(1, wr.count('[') // 2)
  105. except Exception:
  106. return 1
  107. def _pt_map(handoff: dict) -> dict:
  108. """handoff['per_turbine'] 是列表 ⇒ 归一成 {turbine: row}(容错)。"""
  109. out = {}
  110. pt = (handoff or {}).get('per_turbine') or []
  111. if isinstance(pt, dict):
  112. return dict(pt)
  113. for row in pt:
  114. if not isinstance(row, dict):
  115. continue
  116. tid = row.get('turbine') or row.get('id') or row.get('台') or row.get('WTG')
  117. if tid:
  118. out[str(tid)] = row
  119. return out
  120. def _norm_tid(x) -> str:
  121. s = str(x or '').strip()
  122. m = re.search(r'(\d+)', s)
  123. if not m:
  124. return s
  125. return 'WTG%02d' % int(m.group(1))
  126. def build_axes(frow: dict, prow: dict | None, axin: dict | None = None) -> dict:
  127. """按可用数据装配五轴;缺数轴标 [INS] 并计 0。axin = load_axis_inputs() 的逐台真输入。"""
  128. prow = prow or {}
  129. axin = axin or {}
  130. notes: list[str] = []
  131. ins: list[str] = []
  132. # ---- A 内部证据强度:优先用 L6 过闸谱线的真输入 ----
  133. _a = (axin.get('a') or {})
  134. if _a:
  135. A = VC.score_internal(int(_a.get('gates_passed') or 0), n_gates=5,
  136. anchor_ok=bool(_a.get('anchor_ok')),
  137. persist_windows=int(prow.get('persist_windows') or 0),
  138. n_windows=int(prow.get('n_windows') or _n_win(frow)))
  139. notes.append('A 来源=L6 过闸谱线: PASS 线 %s 条 · 绝对锚 %s · 定级 %s'
  140. % (_a.get('pass_lines'), _a.get('anchor_ok'), _a.get('level')))
  141. # 直接跳到 B(跳过下面的 handoff 取数)
  142. a = None if False else 'FROM_L6'
  143. else:
  144. a = None
  145. for k in ('gates_passed', '闸过', '闸', 'passed_gates', 'n_gates_passed'):
  146. if k in prow and isinstance(prow[k], (int, float)):
  147. a = float(prow[k])
  148. break
  149. if a is None:
  150. ins.append('A:本场无 L6 过闸记录且 handoff 无逐台闸过数')
  151. A = 0.0
  152. elif a != 'FROM_L6':
  153. A = VC.score_internal(int(a), n_gates=int(prow.get('n_gates') or 5),
  154. anchor_ok=bool(prow.get('anchor_ok') or False),
  155. persist_windows=int(prow.get('persist_windows') or 0),
  156. n_windows=int(prow.get('n_windows') or _n_win(frow)))
  157. # ---- B 独立源共时确认(同窗)----
  158. srcs = set()
  159. tjudge = str(frow.get('温度判') or '').strip()
  160. if tjudge and tjudge not in ('—', '-', ''):
  161. srcs.add('temperature')
  162. if str(frow.get('油新鲜')).lower() == 'true':
  163. srcs.add('oil')
  164. B = VC.score_independent(srcs) if srcs else 0.0
  165. if not srcs:
  166. ins.append('B(结构性):同窗独立源不在窗(温度轴判「—」/油样 stale) —— 字段级现实, 非逐台缺陷')
  167. # ---- C 实物锚 ----
  168. C = 0.0
  169. tier = None
  170. for k in ('tier', 'anchor_tier', '实物锚', 'physical_tier'):
  171. if prow.get(k):
  172. tier = str(prow[k])
  173. break
  174. if not tier:
  175. _c = (axin.get('c') or {})
  176. if _c.get('tier'):
  177. tier = str(_c['tier'])
  178. notes.append('C 来源=实物锚登记: %s(%s)' % (tier, _c.get('evidence', '')[:40]))
  179. if tier:
  180. C = VC.score_physical(tier)
  181. else:
  182. ins.append('C:本台无实物锚登记且 handoff 无该字段')
  183. # ---- D 样本充分性 ----
  184. nw = None
  185. wr = str(frow.get('窗区间') or '')
  186. if wr:
  187. nw = max(1, wr.count('[') // 2 or wr.count(',')) if wr else 1
  188. nrec = 0
  189. for k in ('n_records', '记录数', 'n'):
  190. if isinstance(prow.get(k), (int, float)):
  191. nrec = int(prow[k])
  192. break
  193. D = VC.score_sample(nrec, (nw or 1), False, n_windows=(nw or 1))
  194. if nw is None:
  195. ins.append('D:无窗区间解析结果')
  196. # ---- E 反证搜索:用过闸 funnel 的"排除类闸"据实计(未跑的不补)----
  197. # common_mode 未跑原因(2026-10-06 实测):L6 过闸 funnel 无共模闸;G4_INHERENT_OR_BATCH 是
  198. # 「固有线/批次」(fleet 共有谱峰),与「共模」(fleet 同期共同变化)不同义;且共模需 ≥2 窗时间序列。
  199. _checks = list(axin.get('__checks__') or [])
  200. _cm = (axin.get('__common_mode__') or {})
  201. _flag = None
  202. if _cm.get('ran'):
  203. _line = str(frow.get('线') or '')
  204. _shared = _cm.get('shared_lines') or {}
  205. if _line and _line in _shared:
  206. _flag = '本线同期在 %d 台通过 ⇒ **机群共有现象(共模嫌疑)**,个体性被削弱' % _shared[_line]
  207. _checks.append('common_mode')
  208. if _checks:
  209. E = VC.score_falsification(tuple(_checks))
  210. if _flag:
  211. E = max(0.0, E - 20.0 / max(len(_checks), 1)) # 该类被推翻 ⇒ 扣掉该类得分
  212. _miss = sorted(set(VC.__dict__.get('CHECKS_TOTAL', () ) or
  213. ('fleet_base_rate', 'common_mode', 'intrinsic_line', 'alternative_explanation')) - set(_checks))
  214. notes.append('E 来源=过闸 funnel%s: 已跑 %s;未跑 %s'
  215. % (' + common_mode(5 窗同期跨台)' if _cm.get('ran') else '', '/'.join(_checks), '/'.join(_miss) or '无'))
  216. if _flag:
  217. ins.append('E(共模): ' + _flag)
  218. else:
  219. E = 0.0
  220. ins.append('E:无过闸 funnel 产物(没人做过=0 分)')
  221. axes = {'A_internal': round(A, 1), 'B_independent': round(B, 1),
  222. 'C_physical': round(C, 1), 'D_sample': round(D, 1), 'E_falsification': round(E, 1)}
  223. return {'axes': axes, 'ins': ins, 'notes': notes}
  224. def score_turbine(frow: dict, prow: dict | None, axin: dict | None = None) -> dict:
  225. r = build_axes(frow, prow, axin)
  226. conf = VC.confidence(r['axes'])
  227. total = conf.get('total') if isinstance(conf, dict) else conf
  228. band = (conf or {}).get('band') if isinstance(conf, dict) else None
  229. # ★诊断门槛:A/B/C/E 只要能取到的轴缺件, 分数就**不可用于判断机组状态**
  230. # (低分反映的是"输入缺件"而不是"证据弱" —— 防止把缺数据读成"证据不足")
  231. hard = ('A:', 'C:', 'E:') # B 属结构性缺件, 单列
  232. missing = [s for s in r['ins'] if s.startswith(hard)]
  233. structural = [s for s in r['ins'] if s.startswith('B(')]
  234. usable = (len(missing) == 0)
  235. caveat = ('' if usable else
  236. '本分**不可用于判断机组状态**:输入端缺件 ' + str(len(missing)) + ' 项(' + ';'.join(missing) + ')。'
  237. '低分反映"缺少输入",不等于"证据弱";补件后重算才有意义。')
  238. return {'axes': r['axes'], 'total': total, 'band': band,
  239. 'usable': usable, 'missing_inputs': missing, 'structural': structural,
  240. 'caveat': caveat, 'ins': r['ins']}
  241. def attach_confidence(ftab, handoff: dict | None = None):
  242. """给融合表加一列 置信度(dict),不改任何既有列。
  243. """
  244. pm = _pt_map(handoff or {})
  245. axin = load_axis_inputs()
  246. out = []
  247. for _, row in ftab.iterrows():
  248. d = row.to_dict()
  249. tid = _norm_tid(d.get('turbine'))
  250. _ax = dict(axin.get(tid) or {})
  251. _ax['__checks__'] = axin.get('__checks__') or [] # ★E 轴(窗口级)随行传入
  252. _c = (pm.get(tid) or pm.get(str(d.get('turbine'))) or {})
  253. s = score_turbine(d, _c, _ax)
  254. d['置信度'] = s
  255. out.append(d)
  256. import pandas as pd
  257. return pd.DataFrame(out)