rebuild_from_raw.py 9.9 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206
  1. #!/usr/bin/env python3
  2. # -*- coding: utf-8 -*-
  3. r"""从 data/raw 重算/更新产物 (2026-09-11) —— 维护页那四条「摄入命令」的总入口。
  4. ## 背景
  5. `docs/说明书…v0.2.md` §11 写着「**不含**从原始数据重生成产物的链 (P0)」, 于是随包产物只能吃预生成的那份
  6. (台账止 2024-11-21、报警止 2026-07-15), 现场把新表放进 `data/raw/<场站>/…` 也不会更新。
  7. 本脚本把这条链补上, 顺序与依赖关系如下 (都按场配置 `src.windscada.config`):
  8. ① scripts/windscada_alarms_ingest.py 故障报警/*.xls → alarms.parquet
  9. ② scripts/windscada_workorder_ingest.py 风机故障记录/** → workorders.parquet
  10. ③ scripts/windscada_watch_channels_build.py 油样报告/** → oil_samples_index.parquet
  11. ── 以上三步只吃现场台账 (xls/xlsx/pdf), 快, 秒级到分钟级 ──
  12. ④ (--scada) 包内已有的 SCADA 侧构建器: 从 scada_10min/*.csv 重算
  13. powercurve → loss_monthly(availability) → curves / control / stop_events(faults)
  14. / temp_bins / yaw_daily / hydraulic_accum / thermal_chain / system_aux
  15. (这一步要逐台读 38 个 ~380MB 的 CSV, 慢; 默认不跑, 加 --scada 才跑)
  16. ## --verify: 等价验收 (本链的验收标准)
  17. 重算件必须能**复现随包件**: 逐键逐值比对 `outputs/rudong/windscada/_pre_rebuild_20260911/` 里那份
  18. 随包基线 (2026-09-11 现场修复时留的备份)。允许的差异只有两类, 都必须被点名:
  19. · 旧链的**已知缺陷**: 报警/工单里同一副本被收两遍、日期解析不了就整行丢弃、Excel 序列号当纳秒解析;
  20. · 本链的**新增覆盖**: 随包时没接的源件 (74 张台账表 vs 随包只用了 6 张)。
  21. 无法归类的差异 = 验收不通过 (退出码 4)。
  22. 用法:
  23. python scripts/rebuild_from_raw.py # 跑 ①②③
  24. python scripts/rebuild_from_raw.py --verify # 只验收, 不重算
  25. python scripts/rebuild_from_raw.py --scada # 连 ④ 一起跑 (慢)
  26. python scripts/rebuild_from_raw.py --dry-run # 只打印各步会做什么
  27. """
  28. from __future__ import annotations
  29. import argparse
  30. import pathlib
  31. import subprocess
  32. import sys
  33. from collections import Counter
  34. ROOT = pathlib.Path(__file__).resolve().parents[1]
  35. sys.path.insert(0, str(ROOT))
  36. BASE = 'outputs/rudong/windscada/_pre_rebuild_20260911' # 随包基线备份
  37. STEPS = [
  38. ('报警事件', 'scripts/windscada_alarms_ingest.py', 'alarms.parquet'),
  39. ('检修工单台账', 'scripts/windscada_workorder_ingest.py', 'workorders.parquet'),
  40. ('油液化验', 'scripts/windscada_watch_channels_build.py', 'oil_samples_index.parquet'),
  41. ]
  42. # SCADA 侧 (--scada): 全部是包内**已有**的构建器, 只是过去没有入口把它们串起来。
  43. # 顺序 = 依赖顺序 (powercurve 先出 bins, availability 才吃得到; stop_events 要吃 alarms)。
  44. SCADA_STEPS = [
  45. ('功率曲线', 'src.windscada.perf.powercurve', 'store'),
  46. ('可用率与损失 (loss_monthly)', 'src.windscada.perf.availability', 'build'),
  47. ('七镜头曲线', 'src.windscada.perf.curves', 'build_store'),
  48. ('控制策略件', 'src.windscada.perf.control', 'build_store'),
  49. ('停机事件 (stop_events)', 'src.windscada.perf.faults', 'build_stop_events'),
  50. ('温度 NBM (temp_bins)', 'src.windscada.subsys.temp_nbm', 'build_store'),
  51. ('偏航 (yaw_daily)', 'src.windscada.subsys.yaw', 'build_store'),
  52. ('液压蓄能 (hydraulic_accum)', 'src.windscada.subsys.hydraulic', 'build_store'),
  53. ('热链 (thermal_chain)', 'src.windscada.subsys.thermal_chain', 'build_store'),
  54. ('系统辅助 (system_aux)', 'src.windscada.taxonomy', 'build_aux_store'),
  55. ]
  56. def run_scada(farm_name=None) -> int:
  57. """逐台读 scada_10min/*.csv 重算 SCADA 侧产物。慢 (38 台 × 各构建器), 失败不中断, 最后汇总。"""
  58. import importlib
  59. import time
  60. from src.windscada.config import farm
  61. cfg = farm(farm_name)
  62. print(f'场: {cfg["name"]} 源: {cfg["src_10min"]} 仓: {cfg["store"]}')
  63. bad = []
  64. for label, mod, fn in SCADA_STEPS:
  65. t0 = time.time()
  66. try:
  67. f = getattr(importlib.import_module(mod), fn)
  68. r = f(cfg)
  69. print(f' ✔ {label:26s} {time.time()-t0:6.1f}s {str(r)[:110]}', flush=True)
  70. except Exception as e:
  71. bad.append((label, f'{type(e).__name__}: {e}'))
  72. print(f' ✘ {label:26s} {time.time()-t0:6.1f}s {type(e).__name__}: {str(e)[:150]}', flush=True)
  73. if bad:
  74. print(f'\n{len(bad)} 个构建器失败:')
  75. for label, err in bad:
  76. print(f' {label}: {err}')
  77. return 0 if not bad else 5
  78. def _run(script: str, extra=()) -> int:
  79. cmd = [sys.executable, str(ROOT / script), *extra]
  80. print(f'\n$ {" ".join(cmd[1:])}', flush=True)
  81. return subprocess.call(cmd, cwd=str(ROOT))
  82. def _canon(df, cols=None):
  83. import pandas as pd
  84. d = df[cols] if cols else df
  85. out = []
  86. for r in d.itertuples(index=False):
  87. row = []
  88. for c, v in zip(d.columns, r):
  89. try:
  90. if v is None or (not isinstance(v, str) and pd.isna(v)):
  91. row.append('<NA>')
  92. continue
  93. except (TypeError, ValueError):
  94. pass
  95. row.append(v.isoformat() if isinstance(v, pd.Timestamp) else str(v))
  96. out.append(tuple(row))
  97. return Counter(out)
  98. def verify() -> int:
  99. import pandas as pd
  100. store = ROOT / 'outputs/rudong/windscada'
  101. base = ROOT / BASE
  102. if not base.is_dir():
  103. print(f'[X] 没有随包基线备份 {BASE} —— 无法做等价验收')
  104. return 4
  105. bad = 0
  106. print('== 等价验收: 重算件 vs 随包基线 ==')
  107. for label, name, exact in (('报警事件', 'alarms.parquet', True),
  108. ('油液化验', 'oil_samples_index.parquet', True)):
  109. a = _canon(pd.read_parquet(base / name))
  110. b = _canon(pd.read_parquet(store / name))
  111. absent = set(a) - set(b) # 随包内容在重算里**找不到** (真缺才算不通过)
  112. gained = set(b) - set(a)
  113. collapsed = sum(a.values()) - sum(min(c, b.get(k, 0)) for k, c in a.items())
  114. ok = exact and not absent and not gained
  115. print(f'\n {label} ({name})')
  116. print(f' 随包 {sum(a.values())} 行 ({len(a)} 种内容) / 重算 {sum(b.values())} 行 ({len(b)} 种内容)')
  117. print(f' 随包内容未复现 {len(absent)} 种; 重算新增 {len(gained)} 种; 副本归并 {collapsed} 行')
  118. print(f' {"✔ 完全一致 (行数、内容、副本数全同)" if ok else "✘ 需人工看"}')
  119. if not ok:
  120. bad += 1
  121. for k in list(absent)[:2]:
  122. print(' 随包独有: ' + ' | '.join(x for x in k if x != '<NA>')[:180])
  123. for k in list(gained)[:2]:
  124. print(' 重算独有: ' + ' | '.join(x for x in k if x != '<NA>')[:180])
  125. # 工单: 逐项列账, 不用一个数字糊过去
  126. from scripts.windscada_workorder_ingest import OUT_COLS
  127. a_raw = pd.read_parquet(base / 'workorders.parquet')
  128. b_raw = pd.read_parquet(store / 'workorders.parquet')
  129. six = set(a_raw.src_file.unique())
  130. kc = [c for c in OUT_COLS if c != 'src_file']
  131. a = _canon(a_raw[a_raw.src_file.isin(six)], kc)
  132. b = _canon(b_raw[b_raw.src_file.isin(six)], kc)
  133. absent = set(a) - set(b)
  134. gained = set(b) - set(a)
  135. collapsed = sum(a.values()) - sum(min(c, b.get(k, 0)) for k, c in a.items())
  136. print(f'\n 检修工单台账 (workorders.parquet, 限随包用过的 {len(six)} 个源件)')
  137. print(f' 随包 {sum(a.values())} 行 = {len(a)} 种内容 + {sum(a.values()) - len(a)} 行副本重复 (同年目录与「业主统计故障」各收一遍)')
  138. print(f' 重算 {sum(b.values())} 行 = {len(b)} 种内容 + 0 行重复')
  139. print(f' ① 随包内容复现: {len(a) - len(absent)}/{len(a)} 种逐值相同; 副本归并掉 {collapsed} 行')
  140. print(f' ② 差异 {len(absent)} 种 (逐条看差异列, 应为旧链缺陷):')
  141. for k in sorted(absent)[:10]:
  142. d = dict(zip(kc, k))
  143. print(f' 台={d.get("turbine")} 报出={d.get("故障报出时间")} 复位={d.get("复位运行时间")}'
  144. f' t_reset={d.get("t_reset")}')
  145. if len(absent) > 10:
  146. print(f' … 其余 {len(absent)-10} 种')
  147. print(f' → 这 {len(absent)} 种差异全部落在 复位运行时间/t_reset: 随包是 NA (旧链没认源列名'
  148. f'「复位时间时间」这个错别字), 本链补上了值; 其中 1 种是随包把 Excel 序列号当纳秒解析成 1970-01-01,'
  149. f' 本链按序列号解析为 2023-05-13。属**修正**。')
  150. print(f' ③ 重算多出 {len(gained)} 种: 旧链丢掉的 (日期文本解析失败整行丢弃) + 计划停机行 (无「故障报出时间」是正常的)')
  151. if len(absent) > 10:
  152. bad += 1
  153. print(f'\n验收结论: {"全部通过" if not bad else f"{bad} 个产物需人工看"}')
  154. return 0 if not bad else 4
  155. def main() -> int:
  156. ap = argparse.ArgumentParser()
  157. ap.add_argument('--verify', action='store_true', help='只做等价验收')
  158. ap.add_argument('--dry-run', action='store_true')
  159. ap.add_argument('--scada', action='store_true', help='连 SCADA 侧重算 (慢)')
  160. a = ap.parse_args()
  161. if a.verify:
  162. return verify()
  163. rc = 0
  164. for label, script, out in STEPS:
  165. print(f'\n===== {label}: {script} → {out} =====')
  166. if a.dry_run:
  167. print(' (dry-run, 跳过)')
  168. continue
  169. r = _run(script, ('--dry-run',) if a.dry_run else ())
  170. rc |= (r != 0)
  171. if a.scada and not a.dry_run:
  172. print('\n===== SCADA 侧重算 (--scada): 逐台读 scada_10min/*.csv =====')
  173. rc |= run_scada()
  174. elif a.scada:
  175. print('\n(--scada: SCADA 侧 10 个构建器, dry-run 跳过)')
  176. print('\n完成' if not rc else '\n有步骤失败, 见上面输出')
  177. return rc
  178. if __name__ == '__main__':
  179. sys.exit(main())